From 437a95cfdbf9bec5a3b4b5812967a914e0c8fb30 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 21:40:59 +0200 Subject: [PATCH] Repair complete items in binary formats when parse_error() returns true (#3989) When the SAX parser asks to recover, the binary readers now repair an item whose end is known and read on after it, as RFC 8949, Section 5.3 describes for CBOR: - CBOR: tags are ignored, and simple values other than false, true, and null become null (RFC 8949, Section 6.1); a negative integer below the range of number_integer_t becomes the nearest floating-point number. - Strings that are not valid UTF-8 get U+FFFD for each ill-formed sequence, as in JSON text; so does a UBJSON/BJData char above 0x7F. - UBJSON/BJData high-precision numbers keep their longest valid beginning (via the lexer's recover_token()), or become infinity. - Members whose key is not a string are skipped (CBOR, MessagePack, BON8), like members without a key in JSON text. - BSON elements of types the library does not read (ObjectId, datetime, decimal128, ...) become null; a string without its terminator and a document whose size does not match are kept. Where the end of an item is unknown, reading stops as before, except that BSON skips to the end of the document, whose size it knows. The value read before such an error is now completed by the reader from its container stack, as the JSON parser does, instead of by a proxy SAX parser, which is removed. Like the parser, binary_reader gets an AllowRecovery template parameter, so that from_*() compile without the new code. Tests: a table of repairs, numbers out of range, errors that stop, and a sweep over changed and removed bytes of eight encodings that checks balanced events and that the first error is the one from_*() reports. All fuzzers now run a recovering checker; the binary ones also check that it reports an error exactly when from_*() fails. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/json_sax/parse_error.md | 6 +- .../docs/features/parsing/error_recovery.md | 37 +- .../nlohmann/detail/input/binary_reader.hpp | 1162 ++++++++++++- include/nlohmann/detail/input/json_sax.hpp | 175 +- include/nlohmann/detail/string_utils.hpp | 74 + include/nlohmann/json.hpp | 28 +- single_include/nlohmann/json.hpp | 1439 ++++++++++++++--- tests/src/fuzzer-parse_bjdata.cpp | 12 + tests/src/fuzzer-parse_bon8.cpp | 12 + tests/src/fuzzer-parse_bson.cpp | 12 + tests/src/fuzzer-parse_cbor.cpp | 12 + tests/src/fuzzer-parse_json.cpp | 134 +- tests/src/fuzzer-parse_msgpack.cpp | 12 + tests/src/fuzzer-parse_ubjson.cpp | 12 + tests/src/fuzzer-recovering_checker.hpp | 154 ++ tests/src/unit-regression2.cpp | 317 +++- 16 files changed, 2925 insertions(+), 673 deletions(-) create mode 100644 tests/src/fuzzer-recovering_checker.hpp diff --git a/docs/mkdocs/docs/api/json_sax/parse_error.md b/docs/mkdocs/docs/api/json_sax/parse_error.md index b2831234a..d04396bf4 100644 --- a/docs/mkdocs/docs/api/json_sax/parse_error.md +++ b/docs/mkdocs/docs/api/json_sax/parse_error.md @@ -24,9 +24,9 @@ A parse error occurred. Whether to recover from the error: - `#!cpp false` stops parsing. -- `#!cpp true` recovers from the error: JSON text is repaired and parsing continues; for the binary formats, the value - read so far is completed and parsing stops. See [error recovery](../../features/parsing/error_recovery.md) for how - errors are repaired. +- `#!cpp true` recovers from the error: the error is repaired and parsing continues. If that is not possible, which + happens in the binary formats when the end of the item with the error is unknown, the value read so far is completed + and parsing stops. See [error recovery](../../features/parsing/error_recovery.md) for how errors are repaired. Either way, [`sax_parse`](../basic_json/sax_parse.md) returns `#!cpp false`. diff --git a/docs/mkdocs/docs/features/parsing/error_recovery.md b/docs/mkdocs/docs/features/parsing/error_recovery.md index 7a2387c72..de41206ee 100644 --- a/docs/mkdocs/docs/features/parsing/error_recovery.md +++ b/docs/mkdocs/docs/features/parsing/error_recovery.md @@ -65,11 +65,36 @@ The input after the top-level value is not repaired: as without recovery, it is The binary formats ([BJData](../binary_formats/bjdata.md), [BON8](../binary_formats/bon8.md), [BSON](../binary_formats/bson.md), [CBOR](../binary_formats/cbor.md), [MessagePack](../binary_formats/messagepack.md), -and [UBJSON](../binary_formats/ubjson.md)) cannot be repaired: a value's size is stored before its content, and every -byte is a valid type marker, so after an error there is no way to tell where the next value begins. Parsing therefore -always stops at the first error. If `parse_error` returns `#!cpp true`, the value read so far is completed before -parsing stops: a key that waits for its value gets `#!json null`, and all open arrays and objects are closed. This keeps -everything before the error of an input that was cut off. +and [UBJSON](../binary_formats/ubjson.md)) have no delimiters to find the next value by. So what can be repaired depends +on whether the end of the item with the error is known, a distinction that +[RFC 8949, Section 5.3](https://www.rfc-editor.org/rfc/rfc8949.html#section-5.3) makes for CBOR, too. + +If the item is complete, but cannot be passed on as it is, it is replaced, and parsing continues after it: + +| Mistake | Formats | Repair | +|---------------------------------------------------------------------|-----------------------------------------|-------------------------------------------------------------------------| +| tag | CBOR | ignored | +| simple value other than `false`, `true`, and `null`, like undefined | CBOR | `#!json null` | +| negative integer below the range of `number_integer_t` | CBOR | the nearest floating-point number | +| string that is not valid UTF-8 | BJData, BSON, CBOR, MessagePack, UBJSON | each ill-formed sequence becomes U+FFFD | +| character (`C`) that is not ASCII | BJData, UBJSON | U+FFFD | +| invalid high-precision number (`H`) | BJData, UBJSON | the longest valid beginning is kept, as for JSON text, or `#!json null` | +| high-precision number too large | BJData, UBJSON | passed as infinity, together with its text | +| object key that is not a string | BON8, CBOR, MessagePack | the member is skipped | +| element of a type the library does not read, like ObjectId or date | BSON | `#!json null` | +| string without its terminator | BSON | kept | +| document whose size does not match its content | BSON | kept | + +CBOR tags and simple values are repaired as [RFC 8949, Section 6.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-6.1) +suggests for converting CBOR to JSON. Note that [`sax_parse`](../../api/basic_json/sax_parse.md) has no parameter for +CBOR tags, so every tag is an error there; when recovering, tags are ignored like with +[`cbor_tag_handler_t::ignore`](../../api/basic_json/cbor_tag_handler_t.md). + +After any other error, the end of the item is unknown: the input ended, a byte is not a valid type marker, or a size +cannot be right. Parsing then stops, and the value read so far is completed: a key that waits for its value gets +`#!json null`, and all open arrays and objects are closed. This keeps everything before the error of an input that was +cut off. The exception is BSON, which stores the size of every document: an element whose end is unknown gets +`#!json null`, the rest of its document is skipped, and parsing continues after the document. ## Limitations @@ -80,6 +105,8 @@ everything before the error of an input that was cut off. repair differs from the intention: `#!json {"a": {"b": [1, 2}, "c": 3}` is repaired to `#!json {"a": {"b": [1, 2], "c": 3}}`, although `#!json {"a": {"b": [1, 2]}, "c": 3}` may have been meant. - Keys without quotes, and strings in single quotes, are not supported; such members are skipped. +- In the binary formats, a member that is skipped because its key is not a string is lost, and so are the elements of a + BSON document after one whose end is unknown. - A number that is too large for `number_float_t` is passed as positive or negative infinity. The SAX parser's `number_float` also gets the number's text, but a JSON value cannot store it, and [`dump`](../../api/basic_json/dump.md) serializes infinity as `#!json null`. diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index 778ad5722..16f687d28 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -86,8 +86,13 @@ JSON_INLINE_VARIABLE constexpr std::size_t max_valueless_container_size = 1 << 2 /*! @brief deserialization of BJData, BON8, BSON, CBOR, MessagePack, and UBJSON values + +@tparam AllowRecovery whether the SAX parser may ask to recover from errors by + returning true from parse_error() (see #3989). The functions that read + into a JSON value use false, because their SAX parsers never do, and + then the code that recovers is not compiled. */ -template> +template, bool AllowRecovery = false> class binary_reader { using number_integer_t = typename BasicJsonType::number_integer_t; @@ -98,6 +103,9 @@ class binary_reader using json_sax_t = SAX; using char_type = typename InputAdapterType::char_type; using char_int_type = typename char_traits::int_type; + /// the result of @ref report_repairable_error, which is always false if + /// the code that recovers is not compiled + using repair_t = typename std::conditional::type; /// whether the input is a contiguous block of bytes that can be inspected /// and consumed in bulk (as in the lexer); used by @ref get_bon8_string_bulk @@ -128,7 +136,8 @@ class binary_reader @param[in] strict whether to expect the input to be consumed completed @param[in] tag_handler how to treat CBOR tags - @return whether parsing was successful + @return whether parsing was successful: the input was read without errors, + and no SAX event returned false */ JSON_HEDLEY_NON_NULL(3) bool sax_parse(const input_format_t format, @@ -139,6 +148,11 @@ class binary_reader sax = sax_; container_stack.clear(); bon8_pushback_size = 0; + close_requested = false; + error_repaired = false; + key_pending = false; + skip_requested = false; + ndarray_open = 0; bool result = false; switch (format) @@ -193,7 +207,12 @@ class binary_reader } } - return result; + if (!result) + { + close_open_containers(std::integral_constant {}); + } + + return result && !error_repaired; } private: @@ -287,20 +306,190 @@ class binary_reader plus the terminator); the equality also rejects those impossible sizes, since at least 5 bytes are always consumed. + When recovering from errors, a document whose size does not match is + accepted: its terminator was found, so everything in it has been read. + @param[in] document_start value of chars_read before the size prefix @param[in] document_size the declared document size - @return whether the declared size matches the number of bytes read + @return whether the declared size matches the number of bytes read, or + the mismatch is repaired */ bool check_bson_document_size(const std::size_t document_start, const std::int32_t document_size) { if (JSON_HEDLEY_UNLIKELY(document_size < 0 || static_cast(document_size) != chars_read - document_start)) { - return report_error(chars_read, get_token_string(), parse_error::create(112, chars_read, - exception_message(input_format_t::bson, concat("document size ", std::to_string(document_size), " does not match the number of bytes read (", std::to_string(chars_read - document_start), ")"), "document"), nullptr)); + return report_repairable_error(chars_read, get_token_string(), parse_error::create(112, chars_read, + exception_message(input_format_t::bson, concat("document size ", std::to_string(document_size), " does not match the number of bytes read (", std::to_string(chars_read - document_start), ")"), "document"), nullptr)); } return true; } + /*! + @brief whether the rest of the innermost BSON document can be skipped + + A BSON document declares its size, so the reader can continue after it + even if an element in it cannot be read. This requires a size that ends + the document after the current position. + */ + bool can_skip_to_bson_document_end() const noexcept + { + const container_frame& top = container_stack.back(); + return top.declared_size >= 5 && top.start_position + static_cast(top.declared_size) - 1 >= chars_read; + } + + /*! + @brief report an error that loses the end of a BSON element + + If the rest of the document can be skipped, the error is repairable, and + the SAX parser asks to recover, the reading loop passes null for the + element and skips to the end of the document (see @ref + skip_to_bson_document_end). Otherwise, reading stops. + + @return false, so that the caller stops reading the element + */ + template + bool report_bson_element_error(const std::size_t position, const std::string& last_token, const Exception& ex) + { + if (report_error_repairable_if(can_skip_to_bson_document_end(), position, last_token, ex)) + { + skip_requested = true; + } + return false; + } + + /// the code that recovers is not compiled: stop + std::false_type skip_to_bson_document_end(std::false_type /*allow_recovery*/) const noexcept + { + return {}; + } + + /*! + @brief skip the rest of a BSON document after an element that could not be + read + + Called after reading an element failed. If @ref report_bson_element_error + asked for it, passes null for the element and skips to the document's + terminator, which the reading loop reads next. The elements after the one + that could not be read are lost. + + @return whether reading continues + */ + bool skip_to_bson_document_end(std::true_type /*allow_recovery*/) + { + if (!skip_requested) + { + return false; + } + skip_requested = false; + + const container_frame& top = container_stack.back(); + const std::size_t terminator = top.start_position + static_cast(top.declared_size) - 1; + return skip_bytes(terminator - chars_read, "document") && sax->null(); + } + + /*! + @brief report a BSON element of a type the library does not read + + The BSON specification defines the size of the value of every element + type, so when recovering, the value is skipped and null passed instead. A + type the specification does not define, or a string length that cannot be + right, loses the end of the element (see @ref report_bson_element_error). + + @param[in] element_type the element's type + @param[in] element_type_parse_position where the type was read + @return whether the value was skipped and null passed instead + */ + bool skip_unsupported_bson_element(const char_int_type element_type, const std::size_t element_type_parse_position) + { + std::array cr{{}}; + static_cast((std::snprintf)(cr.data(), cr.size(), "%.2hhX", static_cast(element_type))); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) + const std::string cr_str{cr.data()}; + const auto error = parse_error::create(114, element_type_parse_position, concat("Unsupported BSON record type 0x", cr_str), nullptr); + + // the number of bytes to skip, -1 if the value is read differently, or + // -2 if the type is unknown + std::int64_t size = -1; + switch (element_type) + { + case 0x06: // undefined (deprecated) + case 0x7F: // max key + case 0xFF: // min key + size = 0; + break; + + case 0x07: // ObjectId + size = 12; + break; + + case 0x09: // UTC datetime + size = 8; + break; + + case 0x13: // 128-bit decimal floating point + size = 16; + break; + + case 0x0B: // regular expression: two C strings + case 0x0C: // DBPointer (deprecated): string and 12 bytes + case 0x0D: // JavaScript code: string + case 0x0E: // symbol (deprecated): string + case 0x0F: // JavaScript code with scope: size of it all, string, document + break; + + default: + size = -2; + break; + } + + // an element of an unknown type loses its end, like the elements + // reported with report_bson_element_error + const bool known = size != -2; + if (!report_error_repairable_if(known || can_skip_to_bson_document_end(), element_type_parse_position, cr_str, error)) + { + return false; + } + if (!known) + { + skip_requested = true; + return false; + } + + if (element_type == 0x0B) + { + string_t ignored; + return get_bson_cstr(ignored) && get_bson_cstr(ignored) && sax->null(); + } + + if (size < 0) + { + std::int32_t len{}; + if (!get_number(input_format_t::bson, len)) + { + return false; + } + // the size of code with scope counts the size itself + size = (element_type == 0x0F) ? static_cast(len) - 4 : len; + if (element_type == 0x0C) + { + size += 12; + } + if (JSON_HEDLEY_UNLIKELY(size < 0 || (element_type != 0x0F && len < 1))) + { + // already reported: skip to the end of the document if that + // is possible, or stop + if (!can_skip_to_bson_document_end()) + { + close_requested = true; + return false; + } + skip_requested = true; + return false; + } + } + + return skip_bytes(static_cast(size), "value") && sax->null(); + } + /*! @brief Reads in a BSON-object and passes it to the SAX-parser. @return whether a valid BSON-value was passed to the SAX parser @@ -398,7 +587,10 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_internal(element_type, element_type_parse_position))) { - return false; + if (!skip_to_bson_document_end(std::integral_constant {})) + { + return value_failed(); + } } } } @@ -490,8 +682,8 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(len < 1)) { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format_t::bson, concat("string length must be at least 1, is ", std::to_string(len)), "string"), nullptr)); + return report_bson_element_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bson, concat("string length must be at least 1, is ", std::to_string(len)), "string"), nullptr)); } if (JSON_HEDLEY_UNLIKELY(!get_string(input_format_t::bson, len - static_cast(1), result))) @@ -501,11 +693,13 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(get() != 0x00)) { + // when recovering, a byte in place of the terminator is dropped; + // the end of the input is not auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format_t::bson, - "BSON string is not null-terminated", - "string"), nullptr)); + return report_error_repairable_if(current != char_traits::eof(), chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bson, + "BSON string is not null-terminated", + "string"), nullptr)); } return true; @@ -526,8 +720,8 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(len < 0)) { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format_t::bson, concat("byte array length cannot be negative, is ", std::to_string(len)), "binary"), nullptr)); + return report_bson_element_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bson, concat("byte array length cannot be negative, is ", std::to_string(len)), "binary"), nullptr)); } // All BSON binary values have a subtype @@ -616,13 +810,7 @@ class binary_reader } default: // anything else is not supported (yet) - { - std::array cr{{}}; - static_cast((std::snprintf)(cr.data(), cr.size(), "%.2hhX", static_cast(element_type))); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) - const std::string cr_str{cr.data()}; - return report_error(element_type_parse_position, cr_str, - parse_error::create(114, element_type_parse_position, concat("Unsupported BSON record type 0x", cr_str), nullptr)); - } + return skip_unsupported_bson_element(element_type, element_type_parse_position); } } @@ -641,9 +829,14 @@ class binary_reader const auto max_val = static_cast((std::numeric_limits::max)()); if (number > max_val) { - return report_error(chars_read, get_token_string(), - parse_error::create(112, chars_read, - exception_message(input_format_t::cbor, "negative integer overflow", "value"), nullptr)); + if (!report_repairable_error(chars_read, get_token_string(), + parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, "negative integer overflow", "value"), nullptr))) + { + return false; + } + // too small for number_integer_t: pass the nearest floating-point number + return sax->number_float(static_cast(-1) - static_cast(number), ""); } return sax->number_integer(static_cast(-1) - static_cast(number)); } @@ -978,8 +1171,14 @@ class binary_reader case cbor_tag_handler_t::error: { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr)); + if (!report_repairable_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr))) + { + return false; + } + // when recovering, the tag is ignored, as RFC 8949, + // Section 6.1 suggests for converting to JSON + return parse_cbor_value(false, cbor_tag_handler_t::ignore, tag_pending, item_read); } case cbor_tag_handler_t::ignore: @@ -1177,8 +1376,18 @@ class binary_reader default: // anything else (0xFF is handled inside the other types) { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr)); + // the simple values other than false, true, and null (0xE0..0xF3, + // 0xF7 for undefined, and 0xF8 followed by a byte) are complete + const bool simple_value = (current >= 0xE0 && current <= 0xF3) || current == 0xF7 || current == 0xF8; + if (!report_error_repairable_if(simple_value, chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr))) + { + return false; + } + // when recovering, a simple value becomes null, as RFC 8949, + // Section 6.1 suggests for converting to JSON + std::uint8_t ignored{}; + return (current != 0xF8 || get_number(input_format_t::cbor, ignored)) && sax->null(); } } } @@ -1192,12 +1401,14 @@ class binary_reader into the same string. @param[out] result string the bytes are appended to + @param[in] whole_string whether the chunk is a string of its own rather + than a chunk of an indefinite-length string @return whether string creation completed @pre @a current is not EOF */ - bool get_cbor_string_chunk(string_t& result) + bool get_cbor_string_chunk(string_t& result, const bool whole_string) { switch (current) { @@ -1257,8 +1468,15 @@ class binary_reader default: { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(113, chars_read, - exception_message(input_format_t::cbor, concat("expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x", last_token), "string"), nullptr)); + // a string value begins with one of the bytes above, so only an + // object key can begin with another one; if it begins an item, + // the member is skipped when recovering (see skip_member) + if (report_error_repairable_if(whole_string && is_cbor_item_head(current), chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format_t::cbor, concat("expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x", last_token), "string"), nullptr))) + { + skip_requested = true; + } + return false; } } } @@ -1310,7 +1528,7 @@ class binary_reader continue; } - if (JSON_HEDLEY_UNLIKELY(!get_cbor_string_chunk(result))) + if (JSON_HEDLEY_UNLIKELY(!get_cbor_string_chunk(result, open == 0))) { return false; } @@ -1568,7 +1786,15 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key))) + { + if (!skip_member(std::integral_constant {})) + { + return false; + } + continue; + } + if (JSON_HEDLEY_UNLIKELY(!sax->key(key))) { return false; } @@ -1583,7 +1809,7 @@ class binary_reader { if (JSON_HEDLEY_UNLIKELY(!parse_cbor_value(fetch, tag_handler, tag_pending, item_read))) { - return false; + return value_failed(); } fetch = !item_read; } @@ -1598,6 +1824,137 @@ class binary_reader } } + /*! + @param[in] byte a byte + @return whether @a byte begins a well-formed CBOR data item (RFC 8949, + Section 3): its additional information is not reserved, and the + indefinite length is only used for strings, arrays, and maps + */ + static bool is_cbor_item_head(const char_int_type byte) noexcept + { + const auto major_type = static_cast(byte) >> 5u; + const auto additional_information = static_cast(byte) & 0x1Fu; + if (additional_information < 28) + { + return true; + } + return additional_information == 31 && major_type >= 2 && major_type <= 5; + } + + /*! + @brief skip the head of a CBOR data item, and its content unless it holds + other items (see @ref skip_items) + + @param[in] first whether the item's first byte has been read + @param[in] break_allowed whether the item may be a break stop code, which + ends an indefinite-length item + @param[out] children the number of items nested in the item, or npos if + they end at a break stop code + @param[out] is_break whether the item was a break stop code + + @return whether the item was read + */ + bool skip_cbor_item_head(const bool first, const bool break_allowed, std::size_t& children, bool& is_break) + { + if (!first) + { + get(); + } + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "value"))) + { + return false; + } + + if (current == 0xFF && break_allowed) + { + is_break = true; + return true; + } + + if (JSON_HEDLEY_UNLIKELY(!is_cbor_item_head(current))) + { + auto last_token = get_token_string(); + return report_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr)); + } + + const auto major_type = static_cast(current) >> 5u; + std::uint64_t argument = static_cast(current) & 0x1Fu; + switch (argument) + { + case 24: + { + std::uint8_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, number))) + { + return false; + } + argument = number; + break; + } + + case 25: + { + std::uint16_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, number))) + { + return false; + } + argument = number; + break; + } + + case 26: + { + std::uint32_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, number))) + { + return false; + } + argument = number; + break; + } + + case 27: + { + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, argument))) + { + return false; + } + break; + } + + case 31: // indefinite length: chunks or elements until a break + children = npos; + return true; + + default: + break; + } + + switch (major_type) + { + case 2: // byte string + case 3: // text string + return skip_bytes(argument, major_type == 2 ? "binary" : "string"); + + case 4: // array + children = item_count(argument, false); + return true; + + case 5: // map + children = item_count(argument, true); + return true; + + case 6: // tag: the tagged item follows + children = 1; + return true; + + default: // integers, simple values, and floats end with their argument + return true; + } + } + ///////////// // MsgPack // ///////////// @@ -2063,8 +2420,16 @@ class binary_reader default: { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(113, chars_read, - exception_message(input_format_t::msgpack, concat("expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0x", last_token), "string"), nullptr)); + // a string value begins with one of the bytes above, so only an + // object key can begin with another one; unless it is the + // unused byte 0xC1, it begins an item, and the member is + // skipped when recovering (see skip_member) + if (report_error_repairable_if(current != 0xC1, chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format_t::msgpack, concat("expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0x", last_token), "string"), nullptr))) + { + skip_requested = true; + } + return false; } } } @@ -2231,7 +2596,15 @@ class binary_reader { get(); key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key))) + { + if (!skip_member(std::integral_constant {})) + { + return false; + } + continue; + } + if (JSON_HEDLEY_UNLIKELY(!sax->key(key))) { return false; } @@ -2240,7 +2613,7 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(!parse_msgpack_value())) { - return false; + return value_failed(); } // a value that opened a container left it on the stack; one that @@ -2252,6 +2625,147 @@ class binary_reader } } + /*! + @brief skip the head of a MessagePack item, and its content unless it + holds other items (see @ref skip_items) + + @param[in] first whether the item's first byte has been read + @param[out] children the number of items nested in the item + + @return whether the item was read + */ + bool skip_msgpack_item_head(const bool first, std::size_t& children) + { + if (!first) + { + get(); + } + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::msgpack, "value"))) + { + return false; + } + + // positive and negative fixint + if (current <= 0x7F || current >= 0xE0) + { + return true; + } + + // fixmap, fixarray, and fixstr + if (current <= 0x8F) + { + children = item_count(static_cast(current) & 0x0Fu, true); + return true; + } + if (current <= 0x9F) + { + children = static_cast(current) & 0x0Fu; + return true; + } + if (current <= 0xBF) + { + return skip_bytes(static_cast(current) & 0x1Fu, "string"); + } + + const auto head = current; + const char* context = (head >= 0xD9) ? "string" : "binary"; + switch (head) + { + case 0xC0: // nil + case 0xC2: // false + case 0xC3: // true + return true; + + case 0xC4: // bin 8 + case 0xC7: // ext 8 + case 0xD9: // str 8 + { + std::uint8_t len{}; + return get_number(input_format_t::msgpack, len) && skip_bytes(len + (head == 0xC7 ? 1u : 0u), context); + } + + case 0xC5: // bin 16 + case 0xC8: // ext 16 + case 0xDA: // str 16 + { + std::uint16_t len{}; + return get_number(input_format_t::msgpack, len) && skip_bytes(len + (head == 0xC8 ? 1u : 0u), context); + } + + case 0xC6: // bin 32 + case 0xC9: // ext 32 + case 0xDB: // str 32 + { + std::uint32_t len{}; + return get_number(input_format_t::msgpack, len) && skip_bytes(static_cast(len) + (head == 0xC9 ? 1u : 0u), context); + } + + case 0xCC: // uint 8 + case 0xD0: // int 8 + return skip_bytes(1, "number"); + + case 0xCD: // uint 16 + case 0xD1: // int 16 + return skip_bytes(2, "number"); + + case 0xCA: // float 32 + case 0xCE: // uint 32 + case 0xD2: // int 32 + return skip_bytes(4, "number"); + + case 0xCB: // float 64 + case 0xCF: // uint 64 + case 0xD3: // int 64 + return skip_bytes(8, "number"); + + case 0xD4: // fixext 1 + return skip_bytes(2, "binary"); + + case 0xD5: // fixext 2 + return skip_bytes(3, "binary"); + + case 0xD6: // fixext 4 + return skip_bytes(5, "binary"); + + case 0xD7: // fixext 8 + return skip_bytes(9, "binary"); + + case 0xD8: // fixext 16 + return skip_bytes(17, "binary"); + + case 0xDC: // array 16 + case 0xDE: // map 16 + { + std::uint16_t len{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::msgpack, len))) + { + return false; + } + children = item_count(len, head == 0xDE); + return true; + } + + case 0xDD: // array 32 + case 0xDF: // map 32 + { + std::uint32_t len{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::msgpack, len))) + { + return false; + } + children = item_count(len, head == 0xDF); + return true; + } + + default: // 0xC1, which is never used + { + auto last_token = get_token_string(); + return report_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::msgpack, concat("invalid byte: 0x", last_token), "value"), nullptr)); + } + } + } + //////////// // UBJSON // //////////// @@ -2278,7 +2792,7 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(!get_ubjson_value(prefix))) { - return false; + return value_failed(); } // the value begun here is complete once it is not inside anything @@ -2740,6 +3254,7 @@ class binary_reader { return false; } + ndarray_open = 2; result = 1; for (auto i : dim) { @@ -2762,6 +3277,7 @@ class binary_reader } } is_ndarray = true; + ndarray_open = 1; return sax->end_array(); } result = 0; @@ -3030,8 +3546,16 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(current > 127)) { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(113, chars_read, - exception_message(input_format, concat("byte after 'C' must be in range 0x00..0x7F; last byte: 0x", last_token), "char"), nullptr)); + if (!report_repairable_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format, concat("byte after 'C' must be in range 0x00..0x7F; last byte: 0x", last_token), "char"), nullptr))) + { + return false; + } + // when recovering, the character becomes U+FFFD, as an + // invalid byte in a string does + string_t replacement; + append_replacement_character(replacement); + return sax->string(replacement); } string_t s(1, static_cast(current)); return sax->string(s); @@ -3101,6 +3625,7 @@ class binary_reader { return false; } + ndarray_open = 2; for (std::size_t i = 0; i < size_and_type.first; ++i) { @@ -3110,6 +3635,7 @@ class binary_reader } } + ndarray_open = 0; return (sax->end_array() && sax->end_object()); } @@ -3215,8 +3741,12 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(result_remainder != token_type::end_of_input)) { - return report_error(chars_read, number_string, parse_error::create(115, chars_read, - exception_message(input_format, concat("invalid number text: ", number_lexer.get_token_string()), "high-precision number"), nullptr)); + if (!report_repairable_error(chars_read, number_string, parse_error::create(115, chars_read, + exception_message(input_format, concat("invalid number text: ", number_lexer.get_token_string()), "high-precision number"), nullptr))) + { + return false; + } + return recover_high_precision_number(number_vector); } switch (result_number) @@ -3230,10 +3760,15 @@ class binary_reader const auto parsed_float = number_lexer.get_number_float(); if (JSON_HEDLEY_UNLIKELY(!std::isfinite(parsed_float))) { - return report_error( - chars_read, - number_string, - out_of_range::create(406, concat("number overflow parsing '", number_string, '\''), nullptr)); + // when recovering, the number is passed as infinity with + // its text, as it is in JSON text + if (!report_repairable_error( + chars_read, + number_string, + out_of_range::create(406, concat("number overflow parsing '", number_string, '\''), nullptr))) + { + return false; + } } // number_string is a std::string, while the SAX interface takes a // string_t; convert explicitly, as the two are only implicitly @@ -3255,8 +3790,60 @@ class binary_reader case token_type::end_of_input: case token_type::literal_or_value: default: - return report_error(chars_read, number_string, parse_error::create(115, chars_read, - exception_message(input_format, concat("invalid number text: ", number_lexer.get_token_string()), "high-precision number"), nullptr)); + if (!report_repairable_error(chars_read, number_string, parse_error::create(115, chars_read, + exception_message(input_format, concat("invalid number text: ", number_lexer.get_token_string()), "high-precision number"), nullptr))) + { + return false; + } + return recover_high_precision_number(number_vector); + } + } + + /*! + @brief pass what can be read of an invalid high-precision number + + Like the parser for JSON text when it recovers, keeps the longest beginning + of the text that is a number, or passes null if there is none. + + @param[in] number_vector the number's text + @return whether the SAX parser accepted the value + */ + bool recover_high_precision_number(const std::vector& number_vector) + { + using ia_type = decltype(detail::input_adapter(number_vector)); + auto number_lexer = detail::lexer(detail::input_adapter(number_vector), false); + using token_type = typename detail::lexer_base::token_type; + + auto token = number_lexer.scan(); + if (token == token_type::parse_error) + { + token = number_lexer.recover_token(); + } + + switch (token) + { + case token_type::value_integer: + return sax->number_integer(number_lexer.get_number_integer()); + case token_type::value_unsigned: + return sax->number_unsigned(number_lexer.get_number_unsigned()); + case token_type::value_float: + return sax->number_float(number_lexer.get_number_float(), number_lexer.get_string()); + case token_type::uninitialized: + case token_type::literal_true: + case token_type::literal_false: + case token_type::literal_null: + case token_type::value_string: + case token_type::begin_array: + case token_type::begin_object: + case token_type::end_array: + case token_type::end_object: + case token_type::name_separator: + case token_type::value_separator: + case token_type::parse_error: + case token_type::end_of_input: + case token_type::literal_or_value: + default: + return sax->null(); } } @@ -3322,10 +3909,24 @@ class binary_reader @return false */ bool bon8_error(const std::string& detail, const char* context) + { + return bon8_error_repairable_if(false, detail, context); + } + + /*! + @brief report a parse error at the last read byte that is repairable in + some cases (see @ref report_error_repairable_if) + + @param[in] repairable whether the error is repairable + @param[in] detail a detailed error message + @param[in] context further context information + @return whether the caller repairs the error and reads on + */ + repair_t bon8_error_repairable_if(const bool repairable, const std::string& detail, const char* context) { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format_t::bon8, concat(detail, ": 0x", last_token), context), nullptr)); + return report_error_repairable_if(repairable, chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bon8, concat(detail, ": 0x", last_token), context), nullptr)); } /*! @@ -3393,7 +3994,15 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_bon8_key(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_bon8_key(key))) + { + if (!skip_member(std::integral_constant {})) + { + return false; + } + continue; + } + if (JSON_HEDLEY_UNLIKELY(!sax->key(key))) { return false; } @@ -3402,7 +4011,7 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(!parse_bon8_value())) { - return false; + return value_failed(); } // a value that opened a container left it on the stack; one that @@ -3665,7 +4274,121 @@ class binary_reader current = byte; } - return bon8_error("expected a string; last byte", "key"); + // an end-of-container marker is no value; any other byte begins one, + // and the member is skipped when recovering (see skip_member) + if (bon8_error_repairable_if(byte != 0xFE, "expected a string; last byte", "key")) + { + skip_requested = true; + } + return false; + } + + /*! + @brief skip a BON8 value, except the elements of a container (see @ref + skip_items) + + @param[in] first whether the value's first byte has been read + @param[in] end_allowed whether the byte may be an end-of-container marker + @param[out] children the number of values nested in the value, or npos if + they end at an end-of-container marker + @param[out] is_end whether the byte was an end-of-container marker + + @return whether the value was read + */ + bool skip_bon8_item_head(const bool first, const bool end_allowed, std::size_t& children, bool& is_end) + { + const auto byte = first ? current : get_bon8(); + + if (byte == char_traits::eof()) + { + return unexpect_eof(input_format_t::bon8, "value"); + } + + if (byte == 0xFE && end_allowed) + { + is_end = true; + return true; + } + + // string: ASCII character + if (byte <= 0x7F) + { + string_t ignored; + unget_bon8(byte); + return get_bon8_string(ignored); + } + + // arrays and objects + if (byte <= 0x84) + { + children = static_cast(byte - 0x80); + return true; + } + if (byte == 0x85 || byte == 0x8B) + { + children = npos; + return true; + } + if (byte <= 0x8A) + { + children = item_count(static_cast(byte - 0x86), true); + return true; + } + + switch (byte) + { + case 0x8C: // int32 + case 0x8E: // binary32 + return skip_bon8_bytes(4); + + case 0x8D: // int64 + case 0x8F: // binary64 + return skip_bon8_bytes(8); + + case 0xFE: // end of container where a value is expected + return bon8_error("invalid byte", "value"); + + default: + break; + } + + // integers 0..39 and -1..-10, and the values 0xF8..0xFD and 0xFF + if (byte <= 0xC1 || byte >= 0xF8) + { + return true; + } + + // 0xC2..0xF7: a UTF-8 lead byte begins a string if a continuation + // byte follows and an integer of 2..4 bytes otherwise + const auto second = get_bon8(); + if (is_bon8_continuation(second)) + { + string_t ignored; + unget_bon8(second); + unget_bon8(byte); + return get_bon8_string(ignored); + } + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::bon8, "number"))) + { + return false; + } + return skip_bon8_bytes((byte <= 0xDF) ? 0 : ((byte <= 0xEF) ? 1 : 2)); + } + + /*! + @param[in] len the number of bytes to skip + @return whether the input had that many bytes + */ + bool skip_bon8_bytes(int len) + { + for (; len != 0; --len) + { + if (JSON_HEDLEY_UNLIKELY(get_bon8() == char_traits::eof())) + { + return unexpect_eof(input_format_t::bon8, "number"); + } + } + return true; } /*! @@ -3951,9 +4674,15 @@ class binary_reader // (which would defeat allow_exceptions=false / strict discarding). if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) { - return report_error(chars_read, get_token_string(), - parse_error::create(113, chars_read, - exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr)); + if (!report_repairable_error(chars_read, get_token_string(), + parse_error::create(113, chars_read, + exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr))) + { + return false; + } + // when recovering, each ill-formed sequence becomes U+FFFD, as it + // does in JSON text + replace_invalid_utf8(result, old_size); } return true; @@ -4041,31 +4770,312 @@ class binary_reader } /*! - @brief report an error to the SAX parser + @brief report an error after which the input cannot be read on - The binary formats cannot recover from an error: a value's size is given - before its payload, and every byte value is a valid type marker, so after - an error there is no way to find where the next value begins. Reading - therefore stops, whatever the SAX parser's parse_error() returns. That the - SAX parser may ask for the containers read so far to be closed is handled - by @ref json_sax_salvager, not here (see #3989). + After most errors, it is unknown where the item that was being read ends: + the input ended, a byte is not a valid type marker, or a size cannot be + right. The binary formats have no delimiters to find the next item by, so + reading stops, whatever the SAX parser's parse_error() returns. If it asks + to recover, the value read so far is completed before @ref sax_parse + returns (see @ref close_open_containers and #3989). @return false, so that the caller stops reading */ template - bool report_error(const std::size_t position, const std::string& last_token, const Exception& ex) const + bool report_error(const std::size_t position, const std::string& last_token, const Exception& ex) + { + close_requested = sax->parse_error(position, last_token, ex); + return false; + } + + /*! + @brief report an error in an item whose end is known + + Some items are complete, but cannot be passed on as they are: a CBOR tag or + simple value, a string that is not valid UTF-8, a BSON element of a type + the library does not read, or an object key that is not a string. If the + SAX parser's parse_error() returns true, the caller replaces the item and + reads on after it (RFC 8949, Section 5.3). + + @return whether the caller replaces the item and reads on + */ + template + repair_t report_repairable_error(const std::size_t position, const std::string& last_token, const Exception& ex) + { + return accept_repair(sax->parse_error(position, last_token, ex), std::integral_constant {}); + } + + /// the code that recovers is not compiled: stop + std::false_type accept_repair(const bool /*repair*/, std::false_type /*allow_recovery*/) const noexcept + { + return {}; + } + + /// remember that an error was repaired, so that @ref sax_parse returns false + bool accept_repair(const bool repair, std::true_type /*allow_recovery*/) noexcept + { + error_repaired = error_repaired || repair; + return repair; + } + + /*! + @brief report an error that is repairable in some cases + + Like @ref report_repairable_error if @a repairable is true, and like @ref + report_error otherwise. If the code that recovers is not compiled, the + error is reported in one place only. + + @return whether the caller repairs the item and reads on + */ + template + repair_t report_error_repairable_if(const bool repairable, const std::size_t position, const std::string& last_token, const Exception& ex) + { + return report_error_repairable_if(repairable, position, last_token, ex, std::integral_constant {}); + } + + /// the code that recovers is not compiled: stop + template + std::false_type report_error_repairable_if(const bool /*repairable*/, const std::size_t position, const std::string& last_token, const Exception& ex, std::false_type /*allow_recovery*/) { static_cast(sax->parse_error(position, last_token, ex)); + return {}; + } + + /// report the error as repairable or not + template + bool report_error_repairable_if(const bool repairable, const std::size_t position, const std::string& last_token, const Exception& ex, std::true_type /*allow_recovery*/) + { + return repairable ? report_repairable_error(position, last_token, ex) : report_error(position, last_token, ex); + } + + /*! + @brief stop after the value of an array element or object member could not + be read + + A value that could not be read has passed no event, except the object that + a BJData ndarray begins with, so the key of an object member still waits + for its value; @ref close_open_containers passes null for it. + + @return false, so that the caller stops reading + */ + bool value_failed() noexcept + { + return value_failed(std::integral_constant {}); + } + + /// the code that recovers is not compiled: nothing to remember + std::false_type value_failed(std::false_type /*allow_recovery*/) const noexcept + { + return {}; + } + + /// remember whether a key waits for its value + bool value_failed(std::true_type /*allow_recovery*/) noexcept + { + key_pending = !container_stack.empty() && container_stack.back().is_object && ndarray_open == 0; return false; } + /// the code that recovers is not compiled: nothing to complete + void close_open_containers(std::false_type /*allow_recovery*/) const noexcept {} + + /*! + @brief complete the value read before an error + + Does nothing unless the SAX parser's parse_error() asked to recover from the + error that stopped reading. Otherwise passes null for a key that waits for + its value and closes the arrays and objects that are still open, innermost + first, until an event returns false. + */ + void close_open_containers(std::true_type /*allow_recovery*/) + { + if (!close_requested) + { + return; + } + + if (key_pending && !sax->null()) + { + return; + } + + // the object of a BJData ndarray and the array inside it are not on + // the stack, as their elements are always read in one go + if (ndarray_open == 2 && !sax->end_array()) + { + return; + } + if (ndarray_open != 0 && !sax->end_object()) + { + return; + } + + while (!container_stack.empty()) + { + const bool is_object = container_stack.back().is_object; + container_stack.pop_back(); + if (is_object ? !sax->end_object() : !sax->end_array()) + { + return; + } + } + } + + /// the code that recovers is not compiled: stop + std::false_type skip_member(std::false_type /*allow_recovery*/) const noexcept + { + return {}; + } + + /*! + @brief skip an object member whose key is not a string + + Called after reading a key failed. If the key is a complete item of another + type, and the SAX parser asked to recover from the error, the key, whose + first byte has been read, and the value after it are skipped, like the + parser for JSON text skips a member without a key. + + @return whether the member was skipped and reading continues + */ + bool skip_member(std::true_type /*allow_recovery*/) + { + if (!skip_requested) + { + return false; + } + skip_requested = false; + return skip_items(2); + } + + /*! + @brief skip complete items without passing them to the SAX parser + + Reads the items with their nested items, keeping one count of items left to + skip per nesting level, so that deeply nested items cost heap rather than + native stack. + + @param[in] count the number of items to skip; the first byte of the first + one has been read + + @return whether the items were skipped + */ + bool skip_items(const std::size_t count) + { + // items left to skip on each level, or npos for a level that ends at + // a marker + std::vector levels(1, count); + bool first = true; + + while (!levels.empty()) + { + if (levels.back() == 0) + { + levels.pop_back(); + continue; + } + + // the number of items nested in the item, or npos if they end at + // a marker + std::size_t children = 0; + bool end_marker = false; + const bool marker_allowed = levels.back() == npos; + switch (input_format) + { + case input_format_t::cbor: + if (!skip_cbor_item_head(first, marker_allowed, children, end_marker)) + { + return false; + } + break; + + case input_format_t::msgpack: + if (!skip_msgpack_item_head(first, children)) + { + return false; + } + break; + + case input_format_t::bon8: + if (!skip_bon8_item_head(first, marker_allowed, children, end_marker)) + { + return false; + } + break; + + // the other formats have no object keys that are skipped + case input_format_t::json: // LCOV_EXCL_LINE + case input_format_t::bson: // LCOV_EXCL_LINE + case input_format_t::ubjson: // LCOV_EXCL_LINE + case input_format_t::bjdata: // LCOV_EXCL_LINE + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + return false; // LCOV_EXCL_LINE + } + first = false; + + if (end_marker) + { + levels.pop_back(); + continue; + } + + if (levels.back() != npos) + { + --levels.back(); + } + if (children != 0) + { + levels.push_back(children); + } + } + + return true; + } + + /*! + @brief the number of items a container of @a len elements holds + + @param[in] len the declared number of elements + @param[in] pairs whether the container is an object, whose elements are + pairs of items + @return the number of items, capped below npos, which marks a container + that ends at a marker; the input ends before a capped count is + reached + */ + static std::size_t item_count(const std::uint64_t len, const bool pairs) noexcept + { + const std::uint64_t max_len = conditional_static_cast(npos - 1) / (pairs ? 2u : 1u); + const std::uint64_t capped = (len < max_len) ? len : max_len; + return conditional_static_cast(pairs ? 2 * capped : capped); + } + + /*! + @brief skip bytes of an item that is not passed on + + @param[in] len the number of bytes to skip + @param[in] context further context information (for diagnostics) + @return whether the input had that many bytes + */ + bool skip_bytes(std::uint64_t len, const char* context) + { + for (; len != 0; --len) + { + get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, context))) + { + return false; + } + } + return true; + } + /*! @param[in] format the current format (for diagnostics) @param[in] context further context information (for diagnostics) @return whether the last read character is not EOF */ JSON_HEDLEY_NON_NULL(3) - bool unexpect_eof(const input_format_t format, const char* context) const + bool unexpect_eof(const input_format_t format, const char* context) { if (JSON_HEDLEY_UNLIKELY(current == char_traits::eof())) { @@ -4160,6 +5170,20 @@ class binary_reader /// BON8: number of bytes in @ref bon8_pushback std::size_t bon8_pushback_size = 0; + /// whether the SAX parser asked to recover from the error that stopped + /// reading, so that @ref close_open_containers completes the value + bool close_requested = false; + /// whether an error was repaired, so that @ref sax_parse returns false + bool error_repaired = false; + /// whether an object key waits for the value that could not be read + bool key_pending = false; + /// whether the item that could not be read is skipped: an object member + /// whose key is not a string, or the rest of a BSON document + bool skip_requested = false; + /// BJData: the containers of an ndarray's annotated array format that are + /// open: none, its object, or its object and an array inside it + std::uint8_t ndarray_open = 0; + // excluded markers in bjdata optimized type #define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \ make_array('F', 'H', 'N', 'S', 'T', 'Z', '[', '{') @@ -4195,8 +5219,8 @@ class binary_reader }; #ifndef JSON_HAS_CPP_17 - template - constexpr std::size_t binary_reader::npos; + template + constexpr std::size_t binary_reader::npos; #endif } // namespace detail diff --git a/include/nlohmann/detail/input/json_sax.hpp b/include/nlohmann/detail/input/json_sax.hpp index 7af6e1fe4..babe307a7 100644 --- a/include/nlohmann/detail/input/json_sax.hpp +++ b/include/nlohmann/detail/input/json_sax.hpp @@ -132,8 +132,8 @@ struct json_sax @param[in] last_token the last read token @param[in] ex an exception object describing the error @return whether to recover from the error: false stops parsing; true - repairs JSON text and continues, or, for the binary formats, stops - after closing the containers read so far + repairs the error and continues, or, if that is not possible, + stops after completing the value read so far */ virtual bool parse_error(std::size_t position, const std::string& last_token, @@ -1212,176 +1212,5 @@ class json_sax_acceptor } }; -/*! -@brief SAX proxy that lets the binary readers keep what was read before an error - -The binary formats cannot continue after an error: a value's size is given -before its payload, and every byte value is a valid type marker, so there is no -way to find where the next value begins. When the SAX parser's parse_error() -returns true to ask for error recovery, the best the binary readers can offer is -the value read up to the error. - -This proxy forwards every event to the SAX parser and records which containers -are open and whether a key still waits for its value. After an error the SAX -parser asked to recover from, @ref close_open_containers then completes the -value with null for a pending key and the missing end events, so the SAX parser -sees balanced events (see #3989). - -@tparam BasicJsonType the JSON type -@tparam SAX the SAX parser to forward the events to -*/ -template -class json_sax_salvager -{ - public: - using number_integer_t = typename BasicJsonType::number_integer_t; - using number_unsigned_t = typename BasicJsonType::number_unsigned_t; - using number_float_t = typename BasicJsonType::number_float_t; - using string_t = typename BasicJsonType::string_t; - using binary_t = typename BasicJsonType::binary_t; - - explicit json_sax_salvager(SAX* sax_) noexcept - : sax(sax_) - {} - - bool null() - { - key_pending = false; - return sax->null(); - } - - bool boolean(bool val) - { - key_pending = false; - return sax->boolean(val); - } - - bool number_integer(number_integer_t val) - { - key_pending = false; - return sax->number_integer(val); - } - - bool number_unsigned(number_unsigned_t val) - { - key_pending = false; - return sax->number_unsigned(val); - } - - bool number_float(number_float_t val, const string_t& s) - { - key_pending = false; - return sax->number_float(val, s); - } - - bool string(string_t& val) - { - key_pending = false; - return sax->string(val); - } - - bool binary(binary_t& val) - { - key_pending = false; - return sax->binary(val); - } - - bool start_object(std::size_t len) - { - key_pending = false; - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(len))) - { - return false; - } - open_containers.push_back(true); - return true; - } - - bool key(string_t& val) - { - key_pending = true; - return sax->key(val); - } - - bool end_object() - { - JSON_ASSERT(!open_containers.empty() && open_containers.back()); - open_containers.pop_back(); - return sax->end_object(); - } - - bool start_array(std::size_t len) - { - key_pending = false; - if (JSON_HEDLEY_UNLIKELY(!sax->start_array(len))) - { - return false; - } - open_containers.push_back(false); - return true; - } - - bool end_array() - { - JSON_ASSERT(!open_containers.empty() && !open_containers.back()); - open_containers.pop_back(); - return sax->end_array(); - } - - template - bool parse_error(std::size_t position, const std::string& last_token, - const Exception& ex) - { - recovery_requested = sax->parse_error(position, last_token, ex); - // the binary readers stop after an error anyway - return false; - } - - /*! - @brief complete the value read before an error - - Does nothing unless the SAX parser's parse_error() returned true. Otherwise - passes null for a key that waits for its value and closes the containers - that are still open, innermost first, until an event returns false. - */ - void close_open_containers() - { - if (!recovery_requested) - { - return; - } - recovery_requested = false; - - if (key_pending) - { - key_pending = false; - if (JSON_HEDLEY_UNLIKELY(!sax->null())) - { - return; - } - } - - while (!open_containers.empty()) - { - const bool is_object = open_containers.back(); - open_containers.pop_back(); - if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->end_object() : !sax->end_array())) - { - return; - } - } - } - - private: - /// the SAX parser the events are forwarded to - SAX* sax = nullptr; - /// the containers that are open, innermost last; true for an object - std::vector open_containers {}; // NOLINT(readability-redundant-member-init) - /// whether a key was passed whose value has not been passed yet - bool key_pending = false; - /// whether the SAX parser's parse_error() asked to recover from the error - bool recovery_requested = false; -}; - } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/string_utils.hpp b/include/nlohmann/detail/string_utils.hpp index 142943cd6..b07adbee8 100644 --- a/include/nlohmann/detail/string_utils.hpp +++ b/include/nlohmann/detail/string_utils.hpp @@ -12,6 +12,7 @@ #include // size_t #include // uint8_t, uint32_t #include // string, to_string +#include // move #include #include @@ -133,5 +134,78 @@ inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noex return state == UTF8_ACCEPT; } +/*! +@brief append U+FFFD REPLACEMENT CHARACTER, encoded in UTF-8 +@param[in,out] s the string to append to +*/ +template +inline void append_replacement_character(StringType& s) +{ + s.push_back(static_cast(0xEFu)); + s.push_back(static_cast(0xBFu)); + s.push_back(static_cast(0xBDu)); +} + +/*! +@brief replace ill-formed UTF-8 with U+FFFD REPLACEMENT CHARACTER + +Each maximal subpart of an ill-formed sequence becomes one U+FFFD, as the +Unicode Standard recommends (Section 3.9, "U+FFFD Substitution of Maximal +Subparts"), and as the parser for JSON text does when it recovers from errors. + +@param[in,out] s the string to repair +@param[in] first index of the first byte to repair; the bytes before it are + assumed to be valid UTF-8 that ends on a code point boundary +*/ +template +inline void replace_invalid_utf8(StringType& s, const std::size_t first = 0) +{ + StringType result = s; + result.resize(first); + + std::uint8_t state = UTF8_ACCEPT; + std::uint32_t codepoint = 0; + // the first byte of the sequence being decoded + std::size_t sequence_start = first; + + std::size_t i = first; + while (i < s.size()) + { + switch (decode(state, codepoint, static_cast(s[i]))) + { + case UTF8_ACCEPT: + for (++i; sequence_start < i; ++sequence_start) + { + result.push_back(s[sequence_start]); + } + break; + + case UTF8_REJECT: + append_replacement_character(result); + // the byte that made the sequence ill-formed begins the next + // one, unless it began this one + if (i == sequence_start) + { + ++i; + } + state = UTF8_ACCEPT; + sequence_start = i; + break; + + default: // in the middle of a sequence + ++i; + break; + } + } + + // a sequence that the string ends in the middle of + if (state != UTF8_ACCEPT) + { + append_replacement_character(result); + } + + s = std::move(result); +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index c144a87df..2418ff314 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -143,7 +143,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec friend class ::nlohmann::detail::iter_impl; template friend class ::nlohmann::detail::binary_writer; - template + template friend class ::nlohmann::detail::binary_reader; template friend class ::nlohmann::detail::json_sax_dom_parser; @@ -4979,26 +4979,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true); } - private: - /// read a binary format and pass it to a SAX parser; if the SAX parser - /// asks to recover from an error, the value read so far is completed - /// (see detail::json_sax_salvager and #3989) - template - static bool sax_parse_binary(InputAdapterType ia, SAX* sax, - const input_format_t format, const bool strict) - { - (void)detail::is_sax_static_asserts {}; - using salvager_t = detail::json_sax_salvager; - salvager_t salvager(sax); - const bool result = detail::binary_reader(std::move(ia), format).sax_parse(format, &salvager, strict); - if (!result) - { - salvager.close_open_containers(); - } - return result; - } - - public: /// @brief generate SAX events /// @sa https://json.nlohmann.me/api/basic_json/sax_parse/ template @@ -5012,7 +4992,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = detail::input_adapter(std::forward(i)); return format == input_format_t::json ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) - : sax_parse_binary(std::move(ia), sax, format, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); } /// @brief generate SAX events (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5029,7 +5009,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = detail::input_adapter(std::move(first), std::move(last)); return format == input_format_t::json ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) - : sax_parse_binary(std::move(ia), sax, format, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); } /// @brief generate SAX events @@ -5051,7 +5031,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - : sax_parse_binary(std::move(ia), sax, format, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); } #ifndef JSON_NO_IO /// @brief deserialize from stream diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 0ab2db096..4c87b1afd 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -6208,6 +6208,7 @@ NLOHMANN_JSON_NAMESPACE_END #include // size_t #include // uint8_t, uint32_t #include // string, to_string +#include // move // #include @@ -6331,6 +6332,79 @@ inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noex return state == UTF8_ACCEPT; } +/*! +@brief append U+FFFD REPLACEMENT CHARACTER, encoded in UTF-8 +@param[in,out] s the string to append to +*/ +template +inline void append_replacement_character(StringType& s) +{ + s.push_back(static_cast(0xEFu)); + s.push_back(static_cast(0xBFu)); + s.push_back(static_cast(0xBDu)); +} + +/*! +@brief replace ill-formed UTF-8 with U+FFFD REPLACEMENT CHARACTER + +Each maximal subpart of an ill-formed sequence becomes one U+FFFD, as the +Unicode Standard recommends (Section 3.9, "U+FFFD Substitution of Maximal +Subparts"), and as the parser for JSON text does when it recovers from errors. + +@param[in,out] s the string to repair +@param[in] first index of the first byte to repair; the bytes before it are + assumed to be valid UTF-8 that ends on a code point boundary +*/ +template +inline void replace_invalid_utf8(StringType& s, const std::size_t first = 0) +{ + StringType result = s; + result.resize(first); + + std::uint8_t state = UTF8_ACCEPT; + std::uint32_t codepoint = 0; + // the first byte of the sequence being decoded + std::size_t sequence_start = first; + + std::size_t i = first; + while (i < s.size()) + { + switch (decode(state, codepoint, static_cast(s[i]))) + { + case UTF8_ACCEPT: + for (++i; sequence_start < i; ++sequence_start) + { + result.push_back(s[sequence_start]); + } + break; + + case UTF8_REJECT: + append_replacement_character(result); + // the byte that made the sequence ill-formed begins the next + // one, unless it began this one + if (i == sequence_start) + { + ++i; + } + state = UTF8_ACCEPT; + sequence_start = i; + break; + + default: // in the middle of a sequence + ++i; + break; + } + } + + // a sequence that the string ends in the middle of + if (state != UTF8_ACCEPT) + { + append_replacement_character(result); + } + + s = std::move(result); +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -12082,8 +12156,8 @@ struct json_sax @param[in] last_token the last read token @param[in] ex an exception object describing the error @return whether to recover from the error: false stops parsing; true - repairs JSON text and continues, or, for the binary formats, stops - after closing the containers read so far + repairs the error and continues, or, if that is not possible, + stops after completing the value read so far */ virtual bool parse_error(std::size_t position, const std::string& last_token, @@ -13162,177 +13236,6 @@ class json_sax_acceptor } }; -/*! -@brief SAX proxy that lets the binary readers keep what was read before an error - -The binary formats cannot continue after an error: a value's size is given -before its payload, and every byte value is a valid type marker, so there is no -way to find where the next value begins. When the SAX parser's parse_error() -returns true to ask for error recovery, the best the binary readers can offer is -the value read up to the error. - -This proxy forwards every event to the SAX parser and records which containers -are open and whether a key still waits for its value. After an error the SAX -parser asked to recover from, @ref close_open_containers then completes the -value with null for a pending key and the missing end events, so the SAX parser -sees balanced events (see #3989). - -@tparam BasicJsonType the JSON type -@tparam SAX the SAX parser to forward the events to -*/ -template -class json_sax_salvager -{ - public: - using number_integer_t = typename BasicJsonType::number_integer_t; - using number_unsigned_t = typename BasicJsonType::number_unsigned_t; - using number_float_t = typename BasicJsonType::number_float_t; - using string_t = typename BasicJsonType::string_t; - using binary_t = typename BasicJsonType::binary_t; - - explicit json_sax_salvager(SAX* sax_) noexcept - : sax(sax_) - {} - - bool null() - { - key_pending = false; - return sax->null(); - } - - bool boolean(bool val) - { - key_pending = false; - return sax->boolean(val); - } - - bool number_integer(number_integer_t val) - { - key_pending = false; - return sax->number_integer(val); - } - - bool number_unsigned(number_unsigned_t val) - { - key_pending = false; - return sax->number_unsigned(val); - } - - bool number_float(number_float_t val, const string_t& s) - { - key_pending = false; - return sax->number_float(val, s); - } - - bool string(string_t& val) - { - key_pending = false; - return sax->string(val); - } - - bool binary(binary_t& val) - { - key_pending = false; - return sax->binary(val); - } - - bool start_object(std::size_t len) - { - key_pending = false; - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(len))) - { - return false; - } - open_containers.push_back(true); - return true; - } - - bool key(string_t& val) - { - key_pending = true; - return sax->key(val); - } - - bool end_object() - { - JSON_ASSERT(!open_containers.empty() && open_containers.back()); - open_containers.pop_back(); - return sax->end_object(); - } - - bool start_array(std::size_t len) - { - key_pending = false; - if (JSON_HEDLEY_UNLIKELY(!sax->start_array(len))) - { - return false; - } - open_containers.push_back(false); - return true; - } - - bool end_array() - { - JSON_ASSERT(!open_containers.empty() && !open_containers.back()); - open_containers.pop_back(); - return sax->end_array(); - } - - template - bool parse_error(std::size_t position, const std::string& last_token, - const Exception& ex) - { - recovery_requested = sax->parse_error(position, last_token, ex); - // the binary readers stop after an error anyway - return false; - } - - /*! - @brief complete the value read before an error - - Does nothing unless the SAX parser's parse_error() returned true. Otherwise - passes null for a key that waits for its value and closes the containers - that are still open, innermost first, until an event returns false. - */ - void close_open_containers() - { - if (!recovery_requested) - { - return; - } - recovery_requested = false; - - if (key_pending) - { - key_pending = false; - if (JSON_HEDLEY_UNLIKELY(!sax->null())) - { - return; - } - } - - while (!open_containers.empty()) - { - const bool is_object = open_containers.back(); - open_containers.pop_back(); - if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->end_object() : !sax->end_array())) - { - return; - } - } - } - - private: - /// the SAX parser the events are forwarded to - SAX* sax = nullptr; - /// the containers that are open, innermost last; true for an object - std::vector open_containers {}; // NOLINT(readability-redundant-member-init) - /// whether a key was passed whose value has not been passed yet - bool key_pending = false; - /// whether the SAX parser's parse_error() asked to recover from the error - bool recovery_requested = false; -}; - } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -13565,8 +13468,13 @@ JSON_INLINE_VARIABLE constexpr std::size_t max_valueless_container_size = 1 << 2 /*! @brief deserialization of BJData, BON8, BSON, CBOR, MessagePack, and UBJSON values + +@tparam AllowRecovery whether the SAX parser may ask to recover from errors by + returning true from parse_error() (see #3989). The functions that read + into a JSON value use false, because their SAX parsers never do, and + then the code that recovers is not compiled. */ -template> +template, bool AllowRecovery = false> class binary_reader { using number_integer_t = typename BasicJsonType::number_integer_t; @@ -13577,6 +13485,9 @@ class binary_reader using json_sax_t = SAX; using char_type = typename InputAdapterType::char_type; using char_int_type = typename char_traits::int_type; + /// the result of @ref report_repairable_error, which is always false if + /// the code that recovers is not compiled + using repair_t = typename std::conditional::type; /// whether the input is a contiguous block of bytes that can be inspected /// and consumed in bulk (as in the lexer); used by @ref get_bon8_string_bulk @@ -13607,7 +13518,8 @@ class binary_reader @param[in] strict whether to expect the input to be consumed completed @param[in] tag_handler how to treat CBOR tags - @return whether parsing was successful + @return whether parsing was successful: the input was read without errors, + and no SAX event returned false */ JSON_HEDLEY_NON_NULL(3) bool sax_parse(const input_format_t format, @@ -13618,6 +13530,11 @@ class binary_reader sax = sax_; container_stack.clear(); bon8_pushback_size = 0; + close_requested = false; + error_repaired = false; + key_pending = false; + skip_requested = false; + ndarray_open = 0; bool result = false; switch (format) @@ -13672,7 +13589,12 @@ class binary_reader } } - return result; + if (!result) + { + close_open_containers(std::integral_constant {}); + } + + return result && !error_repaired; } private: @@ -13766,20 +13688,190 @@ class binary_reader plus the terminator); the equality also rejects those impossible sizes, since at least 5 bytes are always consumed. + When recovering from errors, a document whose size does not match is + accepted: its terminator was found, so everything in it has been read. + @param[in] document_start value of chars_read before the size prefix @param[in] document_size the declared document size - @return whether the declared size matches the number of bytes read + @return whether the declared size matches the number of bytes read, or + the mismatch is repaired */ bool check_bson_document_size(const std::size_t document_start, const std::int32_t document_size) { if (JSON_HEDLEY_UNLIKELY(document_size < 0 || static_cast(document_size) != chars_read - document_start)) { - return report_error(chars_read, get_token_string(), parse_error::create(112, chars_read, - exception_message(input_format_t::bson, concat("document size ", std::to_string(document_size), " does not match the number of bytes read (", std::to_string(chars_read - document_start), ")"), "document"), nullptr)); + return report_repairable_error(chars_read, get_token_string(), parse_error::create(112, chars_read, + exception_message(input_format_t::bson, concat("document size ", std::to_string(document_size), " does not match the number of bytes read (", std::to_string(chars_read - document_start), ")"), "document"), nullptr)); } return true; } + /*! + @brief whether the rest of the innermost BSON document can be skipped + + A BSON document declares its size, so the reader can continue after it + even if an element in it cannot be read. This requires a size that ends + the document after the current position. + */ + bool can_skip_to_bson_document_end() const noexcept + { + const container_frame& top = container_stack.back(); + return top.declared_size >= 5 && top.start_position + static_cast(top.declared_size) - 1 >= chars_read; + } + + /*! + @brief report an error that loses the end of a BSON element + + If the rest of the document can be skipped, the error is repairable, and + the SAX parser asks to recover, the reading loop passes null for the + element and skips to the end of the document (see @ref + skip_to_bson_document_end). Otherwise, reading stops. + + @return false, so that the caller stops reading the element + */ + template + bool report_bson_element_error(const std::size_t position, const std::string& last_token, const Exception& ex) + { + if (report_error_repairable_if(can_skip_to_bson_document_end(), position, last_token, ex)) + { + skip_requested = true; + } + return false; + } + + /// the code that recovers is not compiled: stop + std::false_type skip_to_bson_document_end(std::false_type /*allow_recovery*/) const noexcept + { + return {}; + } + + /*! + @brief skip the rest of a BSON document after an element that could not be + read + + Called after reading an element failed. If @ref report_bson_element_error + asked for it, passes null for the element and skips to the document's + terminator, which the reading loop reads next. The elements after the one + that could not be read are lost. + + @return whether reading continues + */ + bool skip_to_bson_document_end(std::true_type /*allow_recovery*/) + { + if (!skip_requested) + { + return false; + } + skip_requested = false; + + const container_frame& top = container_stack.back(); + const std::size_t terminator = top.start_position + static_cast(top.declared_size) - 1; + return skip_bytes(terminator - chars_read, "document") && sax->null(); + } + + /*! + @brief report a BSON element of a type the library does not read + + The BSON specification defines the size of the value of every element + type, so when recovering, the value is skipped and null passed instead. A + type the specification does not define, or a string length that cannot be + right, loses the end of the element (see @ref report_bson_element_error). + + @param[in] element_type the element's type + @param[in] element_type_parse_position where the type was read + @return whether the value was skipped and null passed instead + */ + bool skip_unsupported_bson_element(const char_int_type element_type, const std::size_t element_type_parse_position) + { + std::array cr{{}}; + static_cast((std::snprintf)(cr.data(), cr.size(), "%.2hhX", static_cast(element_type))); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) + const std::string cr_str{cr.data()}; + const auto error = parse_error::create(114, element_type_parse_position, concat("Unsupported BSON record type 0x", cr_str), nullptr); + + // the number of bytes to skip, -1 if the value is read differently, or + // -2 if the type is unknown + std::int64_t size = -1; + switch (element_type) + { + case 0x06: // undefined (deprecated) + case 0x7F: // max key + case 0xFF: // min key + size = 0; + break; + + case 0x07: // ObjectId + size = 12; + break; + + case 0x09: // UTC datetime + size = 8; + break; + + case 0x13: // 128-bit decimal floating point + size = 16; + break; + + case 0x0B: // regular expression: two C strings + case 0x0C: // DBPointer (deprecated): string and 12 bytes + case 0x0D: // JavaScript code: string + case 0x0E: // symbol (deprecated): string + case 0x0F: // JavaScript code with scope: size of it all, string, document + break; + + default: + size = -2; + break; + } + + // an element of an unknown type loses its end, like the elements + // reported with report_bson_element_error + const bool known = size != -2; + if (!report_error_repairable_if(known || can_skip_to_bson_document_end(), element_type_parse_position, cr_str, error)) + { + return false; + } + if (!known) + { + skip_requested = true; + return false; + } + + if (element_type == 0x0B) + { + string_t ignored; + return get_bson_cstr(ignored) && get_bson_cstr(ignored) && sax->null(); + } + + if (size < 0) + { + std::int32_t len{}; + if (!get_number(input_format_t::bson, len)) + { + return false; + } + // the size of code with scope counts the size itself + size = (element_type == 0x0F) ? static_cast(len) - 4 : len; + if (element_type == 0x0C) + { + size += 12; + } + if (JSON_HEDLEY_UNLIKELY(size < 0 || (element_type != 0x0F && len < 1))) + { + // already reported: skip to the end of the document if that + // is possible, or stop + if (!can_skip_to_bson_document_end()) + { + close_requested = true; + return false; + } + skip_requested = true; + return false; + } + } + + return skip_bytes(static_cast(size), "value") && sax->null(); + } + /*! @brief Reads in a BSON-object and passes it to the SAX-parser. @return whether a valid BSON-value was passed to the SAX parser @@ -13877,7 +13969,10 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_internal(element_type, element_type_parse_position))) { - return false; + if (!skip_to_bson_document_end(std::integral_constant {})) + { + return value_failed(); + } } } } @@ -13969,8 +14064,8 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(len < 1)) { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format_t::bson, concat("string length must be at least 1, is ", std::to_string(len)), "string"), nullptr)); + return report_bson_element_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bson, concat("string length must be at least 1, is ", std::to_string(len)), "string"), nullptr)); } if (JSON_HEDLEY_UNLIKELY(!get_string(input_format_t::bson, len - static_cast(1), result))) @@ -13980,11 +14075,13 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(get() != 0x00)) { + // when recovering, a byte in place of the terminator is dropped; + // the end of the input is not auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format_t::bson, - "BSON string is not null-terminated", - "string"), nullptr)); + return report_error_repairable_if(current != char_traits::eof(), chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bson, + "BSON string is not null-terminated", + "string"), nullptr)); } return true; @@ -14005,8 +14102,8 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(len < 0)) { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format_t::bson, concat("byte array length cannot be negative, is ", std::to_string(len)), "binary"), nullptr)); + return report_bson_element_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bson, concat("byte array length cannot be negative, is ", std::to_string(len)), "binary"), nullptr)); } // All BSON binary values have a subtype @@ -14095,13 +14192,7 @@ class binary_reader } default: // anything else is not supported (yet) - { - std::array cr{{}}; - static_cast((std::snprintf)(cr.data(), cr.size(), "%.2hhX", static_cast(element_type))); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) - const std::string cr_str{cr.data()}; - return report_error(element_type_parse_position, cr_str, - parse_error::create(114, element_type_parse_position, concat("Unsupported BSON record type 0x", cr_str), nullptr)); - } + return skip_unsupported_bson_element(element_type, element_type_parse_position); } } @@ -14120,9 +14211,14 @@ class binary_reader const auto max_val = static_cast((std::numeric_limits::max)()); if (number > max_val) { - return report_error(chars_read, get_token_string(), - parse_error::create(112, chars_read, - exception_message(input_format_t::cbor, "negative integer overflow", "value"), nullptr)); + if (!report_repairable_error(chars_read, get_token_string(), + parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, "negative integer overflow", "value"), nullptr))) + { + return false; + } + // too small for number_integer_t: pass the nearest floating-point number + return sax->number_float(static_cast(-1) - static_cast(number), ""); } return sax->number_integer(static_cast(-1) - static_cast(number)); } @@ -14457,8 +14553,14 @@ class binary_reader case cbor_tag_handler_t::error: { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr)); + if (!report_repairable_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr))) + { + return false; + } + // when recovering, the tag is ignored, as RFC 8949, + // Section 6.1 suggests for converting to JSON + return parse_cbor_value(false, cbor_tag_handler_t::ignore, tag_pending, item_read); } case cbor_tag_handler_t::ignore: @@ -14656,8 +14758,18 @@ class binary_reader default: // anything else (0xFF is handled inside the other types) { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr)); + // the simple values other than false, true, and null (0xE0..0xF3, + // 0xF7 for undefined, and 0xF8 followed by a byte) are complete + const bool simple_value = (current >= 0xE0 && current <= 0xF3) || current == 0xF7 || current == 0xF8; + if (!report_error_repairable_if(simple_value, chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr))) + { + return false; + } + // when recovering, a simple value becomes null, as RFC 8949, + // Section 6.1 suggests for converting to JSON + std::uint8_t ignored{}; + return (current != 0xF8 || get_number(input_format_t::cbor, ignored)) && sax->null(); } } } @@ -14671,12 +14783,14 @@ class binary_reader into the same string. @param[out] result string the bytes are appended to + @param[in] whole_string whether the chunk is a string of its own rather + than a chunk of an indefinite-length string @return whether string creation completed @pre @a current is not EOF */ - bool get_cbor_string_chunk(string_t& result) + bool get_cbor_string_chunk(string_t& result, const bool whole_string) { switch (current) { @@ -14736,8 +14850,15 @@ class binary_reader default: { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(113, chars_read, - exception_message(input_format_t::cbor, concat("expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x", last_token), "string"), nullptr)); + // a string value begins with one of the bytes above, so only an + // object key can begin with another one; if it begins an item, + // the member is skipped when recovering (see skip_member) + if (report_error_repairable_if(whole_string && is_cbor_item_head(current), chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format_t::cbor, concat("expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x", last_token), "string"), nullptr))) + { + skip_requested = true; + } + return false; } } } @@ -14789,7 +14910,7 @@ class binary_reader continue; } - if (JSON_HEDLEY_UNLIKELY(!get_cbor_string_chunk(result))) + if (JSON_HEDLEY_UNLIKELY(!get_cbor_string_chunk(result, open == 0))) { return false; } @@ -15047,7 +15168,15 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key))) + { + if (!skip_member(std::integral_constant {})) + { + return false; + } + continue; + } + if (JSON_HEDLEY_UNLIKELY(!sax->key(key))) { return false; } @@ -15062,7 +15191,7 @@ class binary_reader { if (JSON_HEDLEY_UNLIKELY(!parse_cbor_value(fetch, tag_handler, tag_pending, item_read))) { - return false; + return value_failed(); } fetch = !item_read; } @@ -15077,6 +15206,137 @@ class binary_reader } } + /*! + @param[in] byte a byte + @return whether @a byte begins a well-formed CBOR data item (RFC 8949, + Section 3): its additional information is not reserved, and the + indefinite length is only used for strings, arrays, and maps + */ + static bool is_cbor_item_head(const char_int_type byte) noexcept + { + const auto major_type = static_cast(byte) >> 5u; + const auto additional_information = static_cast(byte) & 0x1Fu; + if (additional_information < 28) + { + return true; + } + return additional_information == 31 && major_type >= 2 && major_type <= 5; + } + + /*! + @brief skip the head of a CBOR data item, and its content unless it holds + other items (see @ref skip_items) + + @param[in] first whether the item's first byte has been read + @param[in] break_allowed whether the item may be a break stop code, which + ends an indefinite-length item + @param[out] children the number of items nested in the item, or npos if + they end at a break stop code + @param[out] is_break whether the item was a break stop code + + @return whether the item was read + */ + bool skip_cbor_item_head(const bool first, const bool break_allowed, std::size_t& children, bool& is_break) + { + if (!first) + { + get(); + } + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "value"))) + { + return false; + } + + if (current == 0xFF && break_allowed) + { + is_break = true; + return true; + } + + if (JSON_HEDLEY_UNLIKELY(!is_cbor_item_head(current))) + { + auto last_token = get_token_string(); + return report_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr)); + } + + const auto major_type = static_cast(current) >> 5u; + std::uint64_t argument = static_cast(current) & 0x1Fu; + switch (argument) + { + case 24: + { + std::uint8_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, number))) + { + return false; + } + argument = number; + break; + } + + case 25: + { + std::uint16_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, number))) + { + return false; + } + argument = number; + break; + } + + case 26: + { + std::uint32_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, number))) + { + return false; + } + argument = number; + break; + } + + case 27: + { + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, argument))) + { + return false; + } + break; + } + + case 31: // indefinite length: chunks or elements until a break + children = npos; + return true; + + default: + break; + } + + switch (major_type) + { + case 2: // byte string + case 3: // text string + return skip_bytes(argument, major_type == 2 ? "binary" : "string"); + + case 4: // array + children = item_count(argument, false); + return true; + + case 5: // map + children = item_count(argument, true); + return true; + + case 6: // tag: the tagged item follows + children = 1; + return true; + + default: // integers, simple values, and floats end with their argument + return true; + } + } + ///////////// // MsgPack // ///////////// @@ -15542,8 +15802,16 @@ class binary_reader default: { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(113, chars_read, - exception_message(input_format_t::msgpack, concat("expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0x", last_token), "string"), nullptr)); + // a string value begins with one of the bytes above, so only an + // object key can begin with another one; unless it is the + // unused byte 0xC1, it begins an item, and the member is + // skipped when recovering (see skip_member) + if (report_error_repairable_if(current != 0xC1, chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format_t::msgpack, concat("expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0x", last_token), "string"), nullptr))) + { + skip_requested = true; + } + return false; } } } @@ -15710,7 +15978,15 @@ class binary_reader { get(); key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key))) + { + if (!skip_member(std::integral_constant {})) + { + return false; + } + continue; + } + if (JSON_HEDLEY_UNLIKELY(!sax->key(key))) { return false; } @@ -15719,7 +15995,7 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(!parse_msgpack_value())) { - return false; + return value_failed(); } // a value that opened a container left it on the stack; one that @@ -15731,6 +16007,147 @@ class binary_reader } } + /*! + @brief skip the head of a MessagePack item, and its content unless it + holds other items (see @ref skip_items) + + @param[in] first whether the item's first byte has been read + @param[out] children the number of items nested in the item + + @return whether the item was read + */ + bool skip_msgpack_item_head(const bool first, std::size_t& children) + { + if (!first) + { + get(); + } + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::msgpack, "value"))) + { + return false; + } + + // positive and negative fixint + if (current <= 0x7F || current >= 0xE0) + { + return true; + } + + // fixmap, fixarray, and fixstr + if (current <= 0x8F) + { + children = item_count(static_cast(current) & 0x0Fu, true); + return true; + } + if (current <= 0x9F) + { + children = static_cast(current) & 0x0Fu; + return true; + } + if (current <= 0xBF) + { + return skip_bytes(static_cast(current) & 0x1Fu, "string"); + } + + const auto head = current; + const char* context = (head >= 0xD9) ? "string" : "binary"; + switch (head) + { + case 0xC0: // nil + case 0xC2: // false + case 0xC3: // true + return true; + + case 0xC4: // bin 8 + case 0xC7: // ext 8 + case 0xD9: // str 8 + { + std::uint8_t len{}; + return get_number(input_format_t::msgpack, len) && skip_bytes(len + (head == 0xC7 ? 1u : 0u), context); + } + + case 0xC5: // bin 16 + case 0xC8: // ext 16 + case 0xDA: // str 16 + { + std::uint16_t len{}; + return get_number(input_format_t::msgpack, len) && skip_bytes(len + (head == 0xC8 ? 1u : 0u), context); + } + + case 0xC6: // bin 32 + case 0xC9: // ext 32 + case 0xDB: // str 32 + { + std::uint32_t len{}; + return get_number(input_format_t::msgpack, len) && skip_bytes(static_cast(len) + (head == 0xC9 ? 1u : 0u), context); + } + + case 0xCC: // uint 8 + case 0xD0: // int 8 + return skip_bytes(1, "number"); + + case 0xCD: // uint 16 + case 0xD1: // int 16 + return skip_bytes(2, "number"); + + case 0xCA: // float 32 + case 0xCE: // uint 32 + case 0xD2: // int 32 + return skip_bytes(4, "number"); + + case 0xCB: // float 64 + case 0xCF: // uint 64 + case 0xD3: // int 64 + return skip_bytes(8, "number"); + + case 0xD4: // fixext 1 + return skip_bytes(2, "binary"); + + case 0xD5: // fixext 2 + return skip_bytes(3, "binary"); + + case 0xD6: // fixext 4 + return skip_bytes(5, "binary"); + + case 0xD7: // fixext 8 + return skip_bytes(9, "binary"); + + case 0xD8: // fixext 16 + return skip_bytes(17, "binary"); + + case 0xDC: // array 16 + case 0xDE: // map 16 + { + std::uint16_t len{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::msgpack, len))) + { + return false; + } + children = item_count(len, head == 0xDE); + return true; + } + + case 0xDD: // array 32 + case 0xDF: // map 32 + { + std::uint32_t len{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::msgpack, len))) + { + return false; + } + children = item_count(len, head == 0xDF); + return true; + } + + default: // 0xC1, which is never used + { + auto last_token = get_token_string(); + return report_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::msgpack, concat("invalid byte: 0x", last_token), "value"), nullptr)); + } + } + } + //////////// // UBJSON // //////////// @@ -15757,7 +16174,7 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(!get_ubjson_value(prefix))) { - return false; + return value_failed(); } // the value begun here is complete once it is not inside anything @@ -16219,6 +16636,7 @@ class binary_reader { return false; } + ndarray_open = 2; result = 1; for (auto i : dim) { @@ -16241,6 +16659,7 @@ class binary_reader } } is_ndarray = true; + ndarray_open = 1; return sax->end_array(); } result = 0; @@ -16509,8 +16928,16 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(current > 127)) { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(113, chars_read, - exception_message(input_format, concat("byte after 'C' must be in range 0x00..0x7F; last byte: 0x", last_token), "char"), nullptr)); + if (!report_repairable_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format, concat("byte after 'C' must be in range 0x00..0x7F; last byte: 0x", last_token), "char"), nullptr))) + { + return false; + } + // when recovering, the character becomes U+FFFD, as an + // invalid byte in a string does + string_t replacement; + append_replacement_character(replacement); + return sax->string(replacement); } string_t s(1, static_cast(current)); return sax->string(s); @@ -16580,6 +17007,7 @@ class binary_reader { return false; } + ndarray_open = 2; for (std::size_t i = 0; i < size_and_type.first; ++i) { @@ -16589,6 +17017,7 @@ class binary_reader } } + ndarray_open = 0; return (sax->end_array() && sax->end_object()); } @@ -16694,8 +17123,12 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(result_remainder != token_type::end_of_input)) { - return report_error(chars_read, number_string, parse_error::create(115, chars_read, - exception_message(input_format, concat("invalid number text: ", number_lexer.get_token_string()), "high-precision number"), nullptr)); + if (!report_repairable_error(chars_read, number_string, parse_error::create(115, chars_read, + exception_message(input_format, concat("invalid number text: ", number_lexer.get_token_string()), "high-precision number"), nullptr))) + { + return false; + } + return recover_high_precision_number(number_vector); } switch (result_number) @@ -16709,10 +17142,15 @@ class binary_reader const auto parsed_float = number_lexer.get_number_float(); if (JSON_HEDLEY_UNLIKELY(!std::isfinite(parsed_float))) { - return report_error( - chars_read, - number_string, - out_of_range::create(406, concat("number overflow parsing '", number_string, '\''), nullptr)); + // when recovering, the number is passed as infinity with + // its text, as it is in JSON text + if (!report_repairable_error( + chars_read, + number_string, + out_of_range::create(406, concat("number overflow parsing '", number_string, '\''), nullptr))) + { + return false; + } } // number_string is a std::string, while the SAX interface takes a // string_t; convert explicitly, as the two are only implicitly @@ -16734,8 +17172,60 @@ class binary_reader case token_type::end_of_input: case token_type::literal_or_value: default: - return report_error(chars_read, number_string, parse_error::create(115, chars_read, - exception_message(input_format, concat("invalid number text: ", number_lexer.get_token_string()), "high-precision number"), nullptr)); + if (!report_repairable_error(chars_read, number_string, parse_error::create(115, chars_read, + exception_message(input_format, concat("invalid number text: ", number_lexer.get_token_string()), "high-precision number"), nullptr))) + { + return false; + } + return recover_high_precision_number(number_vector); + } + } + + /*! + @brief pass what can be read of an invalid high-precision number + + Like the parser for JSON text when it recovers, keeps the longest beginning + of the text that is a number, or passes null if there is none. + + @param[in] number_vector the number's text + @return whether the SAX parser accepted the value + */ + bool recover_high_precision_number(const std::vector& number_vector) + { + using ia_type = decltype(detail::input_adapter(number_vector)); + auto number_lexer = detail::lexer(detail::input_adapter(number_vector), false); + using token_type = typename detail::lexer_base::token_type; + + auto token = number_lexer.scan(); + if (token == token_type::parse_error) + { + token = number_lexer.recover_token(); + } + + switch (token) + { + case token_type::value_integer: + return sax->number_integer(number_lexer.get_number_integer()); + case token_type::value_unsigned: + return sax->number_unsigned(number_lexer.get_number_unsigned()); + case token_type::value_float: + return sax->number_float(number_lexer.get_number_float(), number_lexer.get_string()); + case token_type::uninitialized: + case token_type::literal_true: + case token_type::literal_false: + case token_type::literal_null: + case token_type::value_string: + case token_type::begin_array: + case token_type::begin_object: + case token_type::end_array: + case token_type::end_object: + case token_type::name_separator: + case token_type::value_separator: + case token_type::parse_error: + case token_type::end_of_input: + case token_type::literal_or_value: + default: + return sax->null(); } } @@ -16801,10 +17291,24 @@ class binary_reader @return false */ bool bon8_error(const std::string& detail, const char* context) + { + return bon8_error_repairable_if(false, detail, context); + } + + /*! + @brief report a parse error at the last read byte that is repairable in + some cases (see @ref report_error_repairable_if) + + @param[in] repairable whether the error is repairable + @param[in] detail a detailed error message + @param[in] context further context information + @return whether the caller repairs the error and reads on + */ + repair_t bon8_error_repairable_if(const bool repairable, const std::string& detail, const char* context) { auto last_token = get_token_string(); - return report_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format_t::bon8, concat(detail, ": 0x", last_token), context), nullptr)); + return report_error_repairable_if(repairable, chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bon8, concat(detail, ": 0x", last_token), context), nullptr)); } /*! @@ -16872,7 +17376,15 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_bon8_key(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_bon8_key(key))) + { + if (!skip_member(std::integral_constant {})) + { + return false; + } + continue; + } + if (JSON_HEDLEY_UNLIKELY(!sax->key(key))) { return false; } @@ -16881,7 +17393,7 @@ class binary_reader if (JSON_HEDLEY_UNLIKELY(!parse_bon8_value())) { - return false; + return value_failed(); } // a value that opened a container left it on the stack; one that @@ -17144,7 +17656,121 @@ class binary_reader current = byte; } - return bon8_error("expected a string; last byte", "key"); + // an end-of-container marker is no value; any other byte begins one, + // and the member is skipped when recovering (see skip_member) + if (bon8_error_repairable_if(byte != 0xFE, "expected a string; last byte", "key")) + { + skip_requested = true; + } + return false; + } + + /*! + @brief skip a BON8 value, except the elements of a container (see @ref + skip_items) + + @param[in] first whether the value's first byte has been read + @param[in] end_allowed whether the byte may be an end-of-container marker + @param[out] children the number of values nested in the value, or npos if + they end at an end-of-container marker + @param[out] is_end whether the byte was an end-of-container marker + + @return whether the value was read + */ + bool skip_bon8_item_head(const bool first, const bool end_allowed, std::size_t& children, bool& is_end) + { + const auto byte = first ? current : get_bon8(); + + if (byte == char_traits::eof()) + { + return unexpect_eof(input_format_t::bon8, "value"); + } + + if (byte == 0xFE && end_allowed) + { + is_end = true; + return true; + } + + // string: ASCII character + if (byte <= 0x7F) + { + string_t ignored; + unget_bon8(byte); + return get_bon8_string(ignored); + } + + // arrays and objects + if (byte <= 0x84) + { + children = static_cast(byte - 0x80); + return true; + } + if (byte == 0x85 || byte == 0x8B) + { + children = npos; + return true; + } + if (byte <= 0x8A) + { + children = item_count(static_cast(byte - 0x86), true); + return true; + } + + switch (byte) + { + case 0x8C: // int32 + case 0x8E: // binary32 + return skip_bon8_bytes(4); + + case 0x8D: // int64 + case 0x8F: // binary64 + return skip_bon8_bytes(8); + + case 0xFE: // end of container where a value is expected + return bon8_error("invalid byte", "value"); + + default: + break; + } + + // integers 0..39 and -1..-10, and the values 0xF8..0xFD and 0xFF + if (byte <= 0xC1 || byte >= 0xF8) + { + return true; + } + + // 0xC2..0xF7: a UTF-8 lead byte begins a string if a continuation + // byte follows and an integer of 2..4 bytes otherwise + const auto second = get_bon8(); + if (is_bon8_continuation(second)) + { + string_t ignored; + unget_bon8(second); + unget_bon8(byte); + return get_bon8_string(ignored); + } + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::bon8, "number"))) + { + return false; + } + return skip_bon8_bytes((byte <= 0xDF) ? 0 : ((byte <= 0xEF) ? 1 : 2)); + } + + /*! + @param[in] len the number of bytes to skip + @return whether the input had that many bytes + */ + bool skip_bon8_bytes(int len) + { + for (; len != 0; --len) + { + if (JSON_HEDLEY_UNLIKELY(get_bon8() == char_traits::eof())) + { + return unexpect_eof(input_format_t::bon8, "number"); + } + } + return true; } /*! @@ -17430,9 +18056,15 @@ class binary_reader // (which would defeat allow_exceptions=false / strict discarding). if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) { - return report_error(chars_read, get_token_string(), - parse_error::create(113, chars_read, - exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr)); + if (!report_repairable_error(chars_read, get_token_string(), + parse_error::create(113, chars_read, + exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr))) + { + return false; + } + // when recovering, each ill-formed sequence becomes U+FFFD, as it + // does in JSON text + replace_invalid_utf8(result, old_size); } return true; @@ -17520,31 +18152,312 @@ class binary_reader } /*! - @brief report an error to the SAX parser + @brief report an error after which the input cannot be read on - The binary formats cannot recover from an error: a value's size is given - before its payload, and every byte value is a valid type marker, so after - an error there is no way to find where the next value begins. Reading - therefore stops, whatever the SAX parser's parse_error() returns. That the - SAX parser may ask for the containers read so far to be closed is handled - by @ref json_sax_salvager, not here (see #3989). + After most errors, it is unknown where the item that was being read ends: + the input ended, a byte is not a valid type marker, or a size cannot be + right. The binary formats have no delimiters to find the next item by, so + reading stops, whatever the SAX parser's parse_error() returns. If it asks + to recover, the value read so far is completed before @ref sax_parse + returns (see @ref close_open_containers and #3989). @return false, so that the caller stops reading */ template - bool report_error(const std::size_t position, const std::string& last_token, const Exception& ex) const + bool report_error(const std::size_t position, const std::string& last_token, const Exception& ex) + { + close_requested = sax->parse_error(position, last_token, ex); + return false; + } + + /*! + @brief report an error in an item whose end is known + + Some items are complete, but cannot be passed on as they are: a CBOR tag or + simple value, a string that is not valid UTF-8, a BSON element of a type + the library does not read, or an object key that is not a string. If the + SAX parser's parse_error() returns true, the caller replaces the item and + reads on after it (RFC 8949, Section 5.3). + + @return whether the caller replaces the item and reads on + */ + template + repair_t report_repairable_error(const std::size_t position, const std::string& last_token, const Exception& ex) + { + return accept_repair(sax->parse_error(position, last_token, ex), std::integral_constant {}); + } + + /// the code that recovers is not compiled: stop + std::false_type accept_repair(const bool /*repair*/, std::false_type /*allow_recovery*/) const noexcept + { + return {}; + } + + /// remember that an error was repaired, so that @ref sax_parse returns false + bool accept_repair(const bool repair, std::true_type /*allow_recovery*/) noexcept + { + error_repaired = error_repaired || repair; + return repair; + } + + /*! + @brief report an error that is repairable in some cases + + Like @ref report_repairable_error if @a repairable is true, and like @ref + report_error otherwise. If the code that recovers is not compiled, the + error is reported in one place only. + + @return whether the caller repairs the item and reads on + */ + template + repair_t report_error_repairable_if(const bool repairable, const std::size_t position, const std::string& last_token, const Exception& ex) + { + return report_error_repairable_if(repairable, position, last_token, ex, std::integral_constant {}); + } + + /// the code that recovers is not compiled: stop + template + std::false_type report_error_repairable_if(const bool /*repairable*/, const std::size_t position, const std::string& last_token, const Exception& ex, std::false_type /*allow_recovery*/) { static_cast(sax->parse_error(position, last_token, ex)); + return {}; + } + + /// report the error as repairable or not + template + bool report_error_repairable_if(const bool repairable, const std::size_t position, const std::string& last_token, const Exception& ex, std::true_type /*allow_recovery*/) + { + return repairable ? report_repairable_error(position, last_token, ex) : report_error(position, last_token, ex); + } + + /*! + @brief stop after the value of an array element or object member could not + be read + + A value that could not be read has passed no event, except the object that + a BJData ndarray begins with, so the key of an object member still waits + for its value; @ref close_open_containers passes null for it. + + @return false, so that the caller stops reading + */ + bool value_failed() noexcept + { + return value_failed(std::integral_constant {}); + } + + /// the code that recovers is not compiled: nothing to remember + std::false_type value_failed(std::false_type /*allow_recovery*/) const noexcept + { + return {}; + } + + /// remember whether a key waits for its value + bool value_failed(std::true_type /*allow_recovery*/) noexcept + { + key_pending = !container_stack.empty() && container_stack.back().is_object && ndarray_open == 0; return false; } + /// the code that recovers is not compiled: nothing to complete + void close_open_containers(std::false_type /*allow_recovery*/) const noexcept {} + + /*! + @brief complete the value read before an error + + Does nothing unless the SAX parser's parse_error() asked to recover from the + error that stopped reading. Otherwise passes null for a key that waits for + its value and closes the arrays and objects that are still open, innermost + first, until an event returns false. + */ + void close_open_containers(std::true_type /*allow_recovery*/) + { + if (!close_requested) + { + return; + } + + if (key_pending && !sax->null()) + { + return; + } + + // the object of a BJData ndarray and the array inside it are not on + // the stack, as their elements are always read in one go + if (ndarray_open == 2 && !sax->end_array()) + { + return; + } + if (ndarray_open != 0 && !sax->end_object()) + { + return; + } + + while (!container_stack.empty()) + { + const bool is_object = container_stack.back().is_object; + container_stack.pop_back(); + if (is_object ? !sax->end_object() : !sax->end_array()) + { + return; + } + } + } + + /// the code that recovers is not compiled: stop + std::false_type skip_member(std::false_type /*allow_recovery*/) const noexcept + { + return {}; + } + + /*! + @brief skip an object member whose key is not a string + + Called after reading a key failed. If the key is a complete item of another + type, and the SAX parser asked to recover from the error, the key, whose + first byte has been read, and the value after it are skipped, like the + parser for JSON text skips a member without a key. + + @return whether the member was skipped and reading continues + */ + bool skip_member(std::true_type /*allow_recovery*/) + { + if (!skip_requested) + { + return false; + } + skip_requested = false; + return skip_items(2); + } + + /*! + @brief skip complete items without passing them to the SAX parser + + Reads the items with their nested items, keeping one count of items left to + skip per nesting level, so that deeply nested items cost heap rather than + native stack. + + @param[in] count the number of items to skip; the first byte of the first + one has been read + + @return whether the items were skipped + */ + bool skip_items(const std::size_t count) + { + // items left to skip on each level, or npos for a level that ends at + // a marker + std::vector levels(1, count); + bool first = true; + + while (!levels.empty()) + { + if (levels.back() == 0) + { + levels.pop_back(); + continue; + } + + // the number of items nested in the item, or npos if they end at + // a marker + std::size_t children = 0; + bool end_marker = false; + const bool marker_allowed = levels.back() == npos; + switch (input_format) + { + case input_format_t::cbor: + if (!skip_cbor_item_head(first, marker_allowed, children, end_marker)) + { + return false; + } + break; + + case input_format_t::msgpack: + if (!skip_msgpack_item_head(first, children)) + { + return false; + } + break; + + case input_format_t::bon8: + if (!skip_bon8_item_head(first, marker_allowed, children, end_marker)) + { + return false; + } + break; + + // the other formats have no object keys that are skipped + case input_format_t::json: // LCOV_EXCL_LINE + case input_format_t::bson: // LCOV_EXCL_LINE + case input_format_t::ubjson: // LCOV_EXCL_LINE + case input_format_t::bjdata: // LCOV_EXCL_LINE + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + return false; // LCOV_EXCL_LINE + } + first = false; + + if (end_marker) + { + levels.pop_back(); + continue; + } + + if (levels.back() != npos) + { + --levels.back(); + } + if (children != 0) + { + levels.push_back(children); + } + } + + return true; + } + + /*! + @brief the number of items a container of @a len elements holds + + @param[in] len the declared number of elements + @param[in] pairs whether the container is an object, whose elements are + pairs of items + @return the number of items, capped below npos, which marks a container + that ends at a marker; the input ends before a capped count is + reached + */ + static std::size_t item_count(const std::uint64_t len, const bool pairs) noexcept + { + const std::uint64_t max_len = conditional_static_cast(npos - 1) / (pairs ? 2u : 1u); + const std::uint64_t capped = (len < max_len) ? len : max_len; + return conditional_static_cast(pairs ? 2 * capped : capped); + } + + /*! + @brief skip bytes of an item that is not passed on + + @param[in] len the number of bytes to skip + @param[in] context further context information (for diagnostics) + @return whether the input had that many bytes + */ + bool skip_bytes(std::uint64_t len, const char* context) + { + for (; len != 0; --len) + { + get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, context))) + { + return false; + } + } + return true; + } + /*! @param[in] format the current format (for diagnostics) @param[in] context further context information (for diagnostics) @return whether the last read character is not EOF */ JSON_HEDLEY_NON_NULL(3) - bool unexpect_eof(const input_format_t format, const char* context) const + bool unexpect_eof(const input_format_t format, const char* context) { if (JSON_HEDLEY_UNLIKELY(current == char_traits::eof())) { @@ -17639,6 +18552,20 @@ class binary_reader /// BON8: number of bytes in @ref bon8_pushback std::size_t bon8_pushback_size = 0; + /// whether the SAX parser asked to recover from the error that stopped + /// reading, so that @ref close_open_containers completes the value + bool close_requested = false; + /// whether an error was repaired, so that @ref sax_parse returns false + bool error_repaired = false; + /// whether an object key waits for the value that could not be read + bool key_pending = false; + /// whether the item that could not be read is skipped: an object member + /// whose key is not a string, or the rest of a BSON document + bool skip_requested = false; + /// BJData: the containers of an ndarray's annotated array format that are + /// open: none, its object, or its object and an array inside it + std::uint8_t ndarray_open = 0; + // excluded markers in bjdata optimized type #define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \ make_array('F', 'H', 'N', 'S', 'T', 'Z', '[', '{') @@ -17674,8 +18601,8 @@ class binary_reader }; #ifndef JSON_HAS_CPP_17 - template - constexpr std::size_t binary_reader::npos; + template + constexpr std::size_t binary_reader::npos; #endif } // namespace detail @@ -27400,7 +28327,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec friend class ::nlohmann::detail::iter_impl; template friend class ::nlohmann::detail::binary_writer; - template + template friend class ::nlohmann::detail::binary_reader; template friend class ::nlohmann::detail::json_sax_dom_parser; @@ -32236,26 +33163,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true); } - private: - /// read a binary format and pass it to a SAX parser; if the SAX parser - /// asks to recover from an error, the value read so far is completed - /// (see detail::json_sax_salvager and #3989) - template - static bool sax_parse_binary(InputAdapterType ia, SAX* sax, - const input_format_t format, const bool strict) - { - (void)detail::is_sax_static_asserts {}; - using salvager_t = detail::json_sax_salvager; - salvager_t salvager(sax); - const bool result = detail::binary_reader(std::move(ia), format).sax_parse(format, &salvager, strict); - if (!result) - { - salvager.close_open_containers(); - } - return result; - } - - public: /// @brief generate SAX events /// @sa https://json.nlohmann.me/api/basic_json/sax_parse/ template @@ -32269,7 +33176,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = detail::input_adapter(std::forward(i)); return format == input_format_t::json ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) - : sax_parse_binary(std::move(ia), sax, format, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); } /// @brief generate SAX events (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32286,7 +33193,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = detail::input_adapter(std::move(first), std::move(last)); return format == input_format_t::json ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) - : sax_parse_binary(std::move(ia), sax, format, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); } /// @brief generate SAX events @@ -32308,7 +33215,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - : sax_parse_binary(std::move(ia), sax, format, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); } #ifndef JSON_NO_IO /// @brief deserialize from stream diff --git a/tests/src/fuzzer-parse_bjdata.cpp b/tests/src/fuzzer-parse_bjdata.cpp index d3c9e7a33..5067de45f 100644 --- a/tests/src/fuzzer-parse_bjdata.cpp +++ b/tests/src/fuzzer-parse_bjdata.cpp @@ -45,6 +45,10 @@ dumps is stable under exactly the same values that break operator==. The unit tests run the same checks on a fixed corpus (see the "BJData round-trip invariants" test case), so keep both in sync. +Furthermore, it reads data with a SAX parser that recovers from every error +and checks that the events are balanced, that reading ends, and that it +reports an error exactly when from_bjdata() fails (see #3989). + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ @@ -59,6 +63,8 @@ drivers. #error "the fuzzer drivers must be built without NDEBUG" #endif +#include "fuzzer-recovering_checker.hpp" + using json = nlohmann::json; // value-stable comparison for the round-trip checks below; see the note @@ -71,11 +77,15 @@ static bool is_value_stable(const json& lhs, const json& rhs) // see http://llvm.org/docs/LibFuzzer.html extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + // step 0: recover from all errors, reading from memory and from a stream + const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bjdata).errors == 0; + try { // step 1: parse input std::vector const vec1(data, data + size); json const j1 = json::from_bjdata(vec1); + assert(recovered_without_errors); try { @@ -109,6 +119,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) catch (const json::parse_error&) { // parse errors are ok, because input may be random bytes + assert(!recovered_without_errors); } catch (const json::type_error&) { @@ -117,6 +128,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) catch (const json::out_of_range&) { // out of range errors may happen if provided sizes are excessive + assert(!recovered_without_errors); } // return 0 - non-zero return values are reserved for future use diff --git a/tests/src/fuzzer-parse_bon8.cpp b/tests/src/fuzzer-parse_bon8.cpp index e97d4f17b..871841f6d 100644 --- a/tests/src/fuzzer-parse_bon8.cpp +++ b/tests/src/fuzzer-parse_bon8.cpp @@ -19,6 +19,10 @@ It also checks that reading the data from a stream, which reads strings byte by byte, gives the same value or error as reading it from contiguous memory, which copies strings in bulk. +Furthermore, it reads data with a SAX parser that recovers from every error +and checks that the events are balanced, that reading ends, and that it +reports an error exactly when from_bon8() fails (see #3989). + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ @@ -33,6 +37,8 @@ drivers. #error "the fuzzer drivers must be built without NDEBUG" #endif +#include "fuzzer-recovering_checker.hpp" + using json = nlohmann::json; namespace @@ -56,6 +62,9 @@ std::string read_bon8(InputType&& input) // see http://llvm.org/docs/LibFuzzer.html extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + // step 0: recover from all errors, reading from memory and from a stream + const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bon8).errors == 0; + // contiguous and stream input must be read alike { std::istringstream stream(std::string(reinterpret_cast(data), size)); @@ -67,6 +76,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) // step 1: parse input std::vector const vec1(data, data + size); json const j1 = json::from_bon8(vec1); + assert(recovered_without_errors); try { @@ -88,6 +98,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) catch (const json::parse_error&) { // parse errors are ok, because input may be random bytes + assert(!recovered_without_errors); } catch (const json::type_error&) { @@ -96,6 +107,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) catch (const json::out_of_range&) { // out of range errors may happen if provided sizes are excessive + assert(!recovered_without_errors); } // return 0 - non-zero return values are reserved for future use diff --git a/tests/src/fuzzer-parse_bson.cpp b/tests/src/fuzzer-parse_bson.cpp index 16f36445b..cd64e4b26 100644 --- a/tests/src/fuzzer-parse_bson.cpp +++ b/tests/src/fuzzer-parse_bson.cpp @@ -15,6 +15,10 @@ array data, it performs the following steps: - j2 = from_bson(vec) - assert(j1 == j2) +Furthermore, it reads data with a SAX parser that recovers from every error +and checks that the events are balanced, that reading ends, and that it +reports an error exactly when from_bson() fails (see #3989). + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ @@ -29,16 +33,22 @@ drivers. #error "the fuzzer drivers must be built without NDEBUG" #endif +#include "fuzzer-recovering_checker.hpp" + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + // step 0: recover from all errors, reading from memory and from a stream + const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bson).errors == 0; + try { // step 1: parse input std::vector const vec1(data, data + size); json const j1 = json::from_bson(vec1); + assert(recovered_without_errors); if (j1.is_discarded()) { @@ -65,6 +75,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) catch (const json::parse_error&) { // parse errors are ok, because input may be random bytes + assert(!recovered_without_errors); } catch (const json::type_error&) { @@ -73,6 +84,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) catch (const json::out_of_range&) { // out of range errors can occur during parsing, too + assert(!recovered_without_errors); } // return 0 - non-zero return values are reserved for future use diff --git a/tests/src/fuzzer-parse_cbor.cpp b/tests/src/fuzzer-parse_cbor.cpp index 7d599abe2..ebfa586af 100644 --- a/tests/src/fuzzer-parse_cbor.cpp +++ b/tests/src/fuzzer-parse_cbor.cpp @@ -15,6 +15,10 @@ array data, it performs the following steps: - j2 = from_cbor(vec) - assert(j1 == j2) +Furthermore, it reads data with a SAX parser that recovers from every error +and checks that the events are balanced, that reading ends, and that it +reports an error exactly when from_cbor() fails (see #3989). + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ @@ -29,16 +33,22 @@ drivers. #error "the fuzzer drivers must be built without NDEBUG" #endif +#include "fuzzer-recovering_checker.hpp" + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + // step 0: recover from all errors, reading from memory and from a stream + const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::cbor).errors == 0; + try { // step 1: parse input std::vector const vec1(data, data + size); json const j1 = json::from_cbor(vec1); + assert(recovered_without_errors); try { @@ -60,6 +70,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) catch (const json::parse_error&) { // parse errors are ok, because input may be random bytes + assert(!recovered_without_errors); } catch (const json::type_error&) { @@ -68,6 +79,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) catch (const json::out_of_range&) { // out of range errors can occur during parsing, too + assert(!recovered_without_errors); } // return 0 - non-zero return values are reserved for future use diff --git a/tests/src/fuzzer-parse_json.cpp b/tests/src/fuzzer-parse_json.cpp index 4644fa969..42de4ae3c 100644 --- a/tests/src/fuzzer-parse_json.cpp +++ b/tests/src/fuzzer-parse_json.cpp @@ -28,7 +28,6 @@ drivers. #include #include #include -#include #include // the round-trip checks below are assertions; NDEBUG would compile them away @@ -36,143 +35,18 @@ drivers. #error "the fuzzer drivers must be built without NDEBUG" #endif +#include "fuzzer-recovering_checker.hpp" + using json = nlohmann::json; -namespace -{ -// a SAX parser that recovers from every error and checks that the events are -// balanced and that every key is followed by exactly one value -class recovering_checker : public nlohmann::json_sax -{ - public: - bool null() override - { - return value(); - } - - bool boolean(bool /*val*/) override - { - return value(); - } - - bool number_integer(number_integer_t /*val*/) override - { - return value(); - } - - bool number_unsigned(number_unsigned_t /*val*/) override - { - return value(); - } - - bool number_float(number_float_t /*val*/, const string_t& /*s*/) override - { - return value(); - } - - bool string(string_t& /*val*/) override - { - return value(); - } - - bool binary(binary_t& /*val*/) override - { - return value(); - } - - bool start_object(std::size_t /*elements*/) override - { - value(); - stack.push_back('o'); - return true; - } - - bool key(string_t& /*val*/) override - { - ++events; - assert(!stack.empty() && stack.back() == 'o'); - stack.back() = 'v'; - return true; - } - - bool end_object() override - { - ++events; - assert(!stack.empty() && stack.back() == 'o'); - stack.pop_back(); - return true; - } - - bool start_array(std::size_t /*elements*/) override - { - value(); - stack.push_back('a'); - return true; - } - - bool end_array() override - { - ++events; - assert(!stack.empty() && stack.back() == 'a'); - stack.pop_back(); - return true; - } - - bool parse_error(std::size_t /*position*/, const std::string& /*last_token*/, const nlohmann::detail::exception& /*ex*/) override - { - ++errors; - return true; - } - - bool complete() const - { - return stack.empty(); - } - - std::size_t events = 0; - std::size_t errors = 0; - - private: - bool value() - { - ++events; - if (!stack.empty()) - { - // an array element, or the value of a key - assert(stack.back() != 'o'); - if (stack.back() == 'v') - { - stack.back() = 'o'; - } - } - return true; - } - - // 'a' for an array, 'o' for an object that expects a key, 'v' for an - // object that expects the value of a key - std::vector stack; -}; -} // namespace - // see http://llvm.org/docs/LibFuzzer.html extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { // step 0: recover from all errors, reading from memory and from a stream { - recovering_checker checker; - const bool ok = json::sax_parse(data, data + size, &checker); - assert(checker.complete()); - assert(checker.errors <= size + 1); + const auto checker = check_recovering_parse(data, size, json::input_format_t::json); assert(checker.events <= (4 * size) + 4); - assert(ok == json::accept(data, data + size)); - assert(ok == (checker.errors == 0)); - - std::istringstream stream(std::string(reinterpret_cast(data), size)); - recovering_checker stream_checker; - assert(json::sax_parse(stream, &stream_checker) == ok); - assert(stream_checker.complete()); - assert(stream_checker.events == checker.events); - assert(stream_checker.errors == checker.errors); + assert((checker.errors == 0) == json::accept(data, data + size)); } try diff --git a/tests/src/fuzzer-parse_msgpack.cpp b/tests/src/fuzzer-parse_msgpack.cpp index df961b8d7..58e39be1e 100644 --- a/tests/src/fuzzer-parse_msgpack.cpp +++ b/tests/src/fuzzer-parse_msgpack.cpp @@ -15,6 +15,10 @@ array data, it performs the following steps: - j2 = from_msgpack(vec) - assert(j1 == j2) +Furthermore, it reads data with a SAX parser that recovers from every error +and checks that the events are balanced, that reading ends, and that it +reports an error exactly when from_msgpack() fails (see #3989). + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ @@ -29,16 +33,22 @@ drivers. #error "the fuzzer drivers must be built without NDEBUG" #endif +#include "fuzzer-recovering_checker.hpp" + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + // step 0: recover from all errors, reading from memory and from a stream + const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::msgpack).errors == 0; + try { // step 1: parse input std::vector const vec1(data, data + size); json const j1 = json::from_msgpack(vec1); + assert(recovered_without_errors); try { @@ -60,6 +70,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) catch (const json::parse_error&) { // parse errors are ok, because input may be random bytes + assert(!recovered_without_errors); } catch (const json::type_error&) { @@ -68,6 +79,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) catch (const json::out_of_range&) { // out of range errors may happen if provided sizes are excessive + assert(!recovered_without_errors); } // return 0 - non-zero return values are reserved for future use diff --git a/tests/src/fuzzer-parse_ubjson.cpp b/tests/src/fuzzer-parse_ubjson.cpp index ebf775b59..4862f0627 100644 --- a/tests/src/fuzzer-parse_ubjson.cpp +++ b/tests/src/fuzzer-parse_ubjson.cpp @@ -24,6 +24,10 @@ array data, it performs the following steps: The unit tests run the same checks on a fixed corpus (see the "UBJSON round-trip invariants" test case), so keep both in sync. +Furthermore, it reads data with a SAX parser that recovers from every error +and checks that the events are balanced, that reading ends, and that it +reports an error exactly when from_ubjson() fails (see #3989). + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ @@ -38,16 +42,22 @@ drivers. #error "the fuzzer drivers must be built without NDEBUG" #endif +#include "fuzzer-recovering_checker.hpp" + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + // step 0: recover from all errors, reading from memory and from a stream + const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::ubjson).errors == 0; + try { // step 1: parse input std::vector const vec1(data, data + size); json const j1 = json::from_ubjson(vec1); + assert(recovered_without_errors); try { @@ -79,6 +89,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) catch (const json::parse_error&) { // parse errors are ok, because input may be random bytes + assert(!recovered_without_errors); } catch (const json::type_error&) { @@ -87,6 +98,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) catch (const json::out_of_range&) { // out of range errors may happen if provided sizes are excessive + assert(!recovered_without_errors); } // return 0 - non-zero return values are reserved for future use diff --git a/tests/src/fuzzer-recovering_checker.hpp b/tests/src/fuzzer-recovering_checker.hpp new file mode 100644 index 000000000..1aeab0e63 --- /dev/null +++ b/tests/src/fuzzer-recovering_checker.hpp @@ -0,0 +1,154 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include +#include +#include +#include +#include +#include +#include + +namespace +{ +// a SAX parser that recovers from every error and checks that the events are +// balanced and that every key is followed by exactly one value +class recovering_checker : public nlohmann::json_sax +{ + public: + bool null() override + { + return value(); + } + + bool boolean(bool /*val*/) override + { + return value(); + } + + bool number_integer(number_integer_t /*val*/) override + { + return value(); + } + + bool number_unsigned(number_unsigned_t /*val*/) override + { + return value(); + } + + bool number_float(number_float_t /*val*/, const string_t& /*s*/) override + { + return value(); + } + + bool string(string_t& /*val*/) override + { + return value(); + } + + bool binary(binary_t& /*val*/) override + { + return value(); + } + + bool start_object(std::size_t /*elements*/) override + { + value(); + stack.push_back('o'); + return true; + } + + bool key(string_t& /*val*/) override + { + ++events; + assert(!stack.empty() && stack.back() == 'o'); + stack.back() = 'v'; + return true; + } + + bool end_object() override + { + ++events; + assert(!stack.empty() && stack.back() == 'o'); + stack.pop_back(); + return true; + } + + bool start_array(std::size_t /*elements*/) override + { + value(); + stack.push_back('a'); + return true; + } + + bool end_array() override + { + ++events; + assert(!stack.empty() && stack.back() == 'a'); + stack.pop_back(); + return true; + } + + bool parse_error(std::size_t /*position*/, const std::string& /*last_token*/, const nlohmann::detail::exception& /*ex*/) override + { + ++errors; + return true; + } + + bool complete() const + { + return stack.empty(); + } + + std::size_t events = 0; + std::size_t errors = 0; + + private: + bool value() + { + ++events; + if (!stack.empty()) + { + // an array element, or the value of a key + assert(stack.back() != 'o'); + if (stack.back() == 'v') + { + stack.back() = 'o'; + } + } + return true; + } + + // 'a' for an array, 'o' for an object that expects a key, 'v' for an + // object that expects the value of a key + std::vector stack {}; // NOLINT(readability-redundant-member-init) +}; + +/// parses @a data with a recovering_checker from memory and from a stream, +/// checks that both see the same, that the events are balanced, and that the +/// number of errors is bounded, and returns the checker (see #3989) +inline recovering_checker check_recovering_parse(const std::uint8_t* data, const std::size_t size, const nlohmann::json::input_format_t format) +{ + recovering_checker checker; + const bool ok = nlohmann::json::sax_parse(data, data + size, &checker, format); + assert(checker.complete()); + assert(checker.errors <= size + 1); + assert(ok == (checker.errors == 0)); + + std::istringstream stream(std::string(reinterpret_cast(data), size)); + recovering_checker stream_checker; + assert(nlohmann::json::sax_parse(stream, &stream_checker, format) == ok); + assert(stream_checker.complete()); + assert(stream_checker.events == checker.events); + assert(stream_checker.errors == checker.errors); + + return checker; +} +} // namespace diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index 193464fca..efae420fe 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -972,8 +972,9 @@ class RecoveringParser : public nlohmann::detail::json_sax_dom_parser return base::end_array(); } - bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& /*unused*/) + bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& ex) { + messages.emplace_back(ex.what()); // a limit, so that a reader that does not stop fails the test // instead of making it hang return ++errors < 100; @@ -986,6 +987,7 @@ class RecoveringParser : public nlohmann::detail::json_sax_dom_parser } std::size_t errors = 0; + std::vector messages {}; // NOLINT(readability-redundant-member-init) std::vector stack {}; // NOLINT(readability-redundant-member-init) bool well_formed = true; @@ -1011,6 +1013,7 @@ struct BinaryParseResult { json value; std::size_t errors; + std::vector messages; bool ok; bool balanced; }; @@ -1020,13 +1023,117 @@ BinaryParseResult parse_binary_recovering(const std::vector& input json j; RecoveringParser sax(j); const bool ok = json::sax_parse(input, &sax, format); - return {j, sax.errors, ok, sax.balanced()}; + return {j, sax.errors, sax.messages, ok, sax.balanced()}; } + +/// the message of the exception that reading @a input into a JSON value +/// throws, or an empty string if reading succeeds +std::string binary_error_message(const std::vector& input, const json::input_format_t format) +{ + try + { + json _; + switch (format) + { + case json::input_format_t::cbor: + _ = json::from_cbor(input); + break; + case json::input_format_t::msgpack: + _ = json::from_msgpack(input); + break; + case json::input_format_t::ubjson: + _ = json::from_ubjson(input); + break; + case json::input_format_t::bjdata: + _ = json::from_bjdata(input); + break; + case json::input_format_t::bson: + _ = json::from_bson(input); + break; + case json::input_format_t::bon8: + _ = json::from_bon8(input); + break; + case json::input_format_t::json: + default: + break; + } + } + catch (const json::exception& e) + { + return e.what(); + } + return ""; +} + +/// a BSON element: its type, its name, and its value +std::vector bson_element(const std::uint8_t type, const std::string& name, const std::vector& value) +{ + std::vector result = {type}; + result.insert(result.end(), name.begin(), name.end()); + result.push_back(0x00); + result.insert(result.end(), value.begin(), value.end()); + return result; +} + +/// a BSON document of the given elements; @a size_offset is added to the +/// size it declares +std::vector bson_document(const std::vector>& elements, const int size_offset = 0) +{ + std::vector body; + for (const auto& element : elements) + { + body.insert(body.end(), element.begin(), element.end()); + } + const auto size = static_cast(static_cast(body.size()) + 5 + size_offset); + std::vector result = {static_cast(size & 0xFFu), static_cast((size >> 8u) & 0xFFu), + static_cast((size >> 16u) & 0xFFu), static_cast((size >> 24u) & 0xFFu) + }; + result.insert(result.end(), body.begin(), body.end()); + result.push_back(0x00); + return result; +} + +/// a BSON int32 value +std::vector bson_int32(const std::int32_t value) +{ + const auto u = static_cast(value); + return {static_cast(u & 0xFFu), static_cast((u >> 8u) & 0xFFu), + static_cast((u >> 16u) & 0xFFu), static_cast((u >> 24u) & 0xFFu)}; +} + +/// a BSON string value, whose length is @a length_offset off +std::vector bson_string(const std::string& value, const std::int32_t length_offset = 0) +{ + auto result = bson_int32(static_cast(value.size() + 1) + length_offset); + result.insert(result.end(), value.begin(), value.end()); + result.push_back(0x00); + return result; +} + +/// @a count bytes of value 0xAB +std::vector bytes(const std::size_t count) +{ + return std::vector(count, 0xAB); +} + +template +std::vector concatenated(const std::vector& first, const Parts& ... rest) +{ + std::vector result = first; + for (const auto& part : std::initializer_list> {rest...}) + { + result.insert(result.end(), part.begin(), part.end()); + } + return result; +} + +/// U+FFFD REPLACEMENT CHARACTER +const std::string replacement_character = "\xEF\xBF\xBD"; } // namespace TEST_CASE("regression test - #3989 SAX parse_error() returning true") { - SECTION("binary formats stop after an error and complete what was read") + SECTION("binary formats complete what was read before the input ends") { const json j = {{"a", {1, -2, {{"b", "c"}}, json::array()}}, {"d", {{"e", nullptr}, {"f", true}}}, {"g", 1.5}, {"h", json::binary({1, 2, 3})}}; @@ -1108,6 +1215,210 @@ TEST_CASE("regression test - #3989 SAX parse_error() returning true") CHECK(result.value == json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2}}})); } + SECTION("binary formats repair items whose end is known") + { + struct Repair + { + json::input_format_t format; + std::vector input; + json expected; + std::size_t errors; + }; + + const std::vector repairs = + { + // CBOR: tags are ignored (here tag 1 and the self-describe tag 55799) + {json::input_format_t::cbor, {0x82, 0xC1, 0x05, 0xD9, 0xD9, 0xF7, 0x06}, {5, 6}, 2}, + // CBOR: undefined and other simple values become null + {json::input_format_t::cbor, {0x84, 0xF7, 0xE0, 0xF8, 0x20, 0x01}, {nullptr, nullptr, nullptr, 1}, 3}, + // CBOR: ill-formed UTF-8 becomes U+FFFD, also in keys + {json::input_format_t::cbor, {0xA1, 0x61, 0xFF, 0x62, 0xC3, 0x28}, {{replacement_character, replacement_character + "("}}, 2}, + // CBOR: members whose key is not a string are skipped, whatever their key and value + {json::input_format_t::cbor, {0xA4, 0x01, 0x02, 0x82, 0x01, 0x02, 0xA1, 0x61, 'x', 0x9F, 0xFF, 0xC1, 0x01, 0x5F, 0x41, 0x00, 0xFF, 0x61, 'a', 0x03}, {{"a", 3}}, 3}, + {json::input_format_t::cbor, {0xBF, 0xF5, 0xBF, 0x61, 'x', 0x7F, 0x61, 'y', 0xFF, 0xFF, 0x61, 'a', 0x03, 0xFF}, {{"a", 3}}, 1}, + // MessagePack: members whose key is not a string are skipped + {json::input_format_t::msgpack, {0x84, 0x01, 0x02, 0x81, 0xA1, 'x', 0x01, 0x92, 0x01, 0x02, 0xD4, 0x01, 0x02, 0xC0, 0xA1, 'a', 0x04}, {{"a", 4}}, 3}, + // MessagePack: ill-formed UTF-8 becomes U+FFFD + {json::input_format_t::msgpack, {0x92, 0xA2, 0xC3, 0x28, 0xA3, 0xE2, 0x82, 'x'}, {replacement_character + "(", replacement_character + "x"}, 2}, + // UBJSON: a char that is not ASCII becomes U+FFFD + {json::input_format_t::ubjson, {'[', 'C', 0x80, 'C', 'A', ']'}, {replacement_character, "A"}, 1}, + // UBJSON: the longest beginning of a high-precision number is kept + {json::input_format_t::ubjson, {'[', 'H', 'i', 5, '1', '2', 'a', 'b', 'c', 'H', 'i', 2, '1', '.', 'H', 'i', 3, 'a', 'b', 'c', 'H', 'i', 3, '4', '.', '5', ']'}, {12, 1, nullptr, 4.5}, 3}, + // BJData, too + {json::input_format_t::bjdata, {'[', 'C', 0xFF, 'H', 'i', 2, '-', '1', 'H', 'i', 2, '-', 'x', ']'}, {replacement_character, -1, nullptr}, 2}, + // BON8: members whose key is not a string are skipped + {json::input_format_t::bon8, {0x89, 0x91, 0x92, 0xC9, 0x40, 0x82, 0x91, 0x92, 0x61, 0x93}, {{"a", 3}}, 2}, + {json::input_format_t::bon8, {0x8B, 0x91, 0x85, 0x91, 0xFE, 0xFA, 0x8B, 'x', 0x91, 0xFE, 0x61, 0x93, 0xFE}, {{"a", 3}}, 2}, + // BSON: elements of types the library does not read become null + { + json::input_format_t::bson, bson_document( + { + bson_element(0x07, "_id", bytes(12)), // ObjectId + bson_element(0x09, "date", bytes(8)), // UTC datetime + bson_element(0x13, "decimal", bytes(16)), // 128-bit decimal + bson_element(0x0B, "regex", {'a', '+', 0, 'i', 0}), // regular expression + bson_element(0x0D, "code", bson_string("f()")), // JavaScript code + bson_element(0x0E, "symbol", bson_string("s")), // symbol + bson_element(0x0C, "pointer", concatenated(bson_string("c"), bytes(12))), // DBPointer + bson_element(0x0F, "scope", concatenated(bson_int32(15), bson_string("g"), bson_document({}))), // code with scope + bson_element(0x06, "undefined", {}), // undefined + bson_element(0xFF, "min", {}), // min key + bson_element(0x7F, "max", {}), // max key + bson_element(0x10, "z", bson_int32(7)), + }), + {{"_id", nullptr}, {"date", nullptr}, {"decimal", nullptr}, {"regex", nullptr}, {"code", nullptr}, {"symbol", nullptr}, {"pointer", nullptr}, {"scope", nullptr}, {"undefined", nullptr}, {"min", nullptr}, {"max", nullptr}, {"z", 7}}, + 11 + }, + // BSON: an element of an unknown type becomes null, and the rest of its document is skipped + { + json::input_format_t::bson, bson_document( + { + bson_element(0x03, "inner", bson_document({bson_element(0x10, "a", bson_int32(1)), bson_element(0x42, "x", bytes(3)), bson_element(0x10, "b", bson_int32(2))})), + bson_element(0x04, "array", bson_document({bson_element(0x10, "0", bson_int32(1)), bson_element(0x42, "1", bytes(3))})), + bson_element(0x10, "after", bson_int32(3)), + }), + {{"inner", {{"a", 1}, {"x", nullptr}}}, {"array", {1, nullptr}}, {"after", 3}}, + 2 + }, + // BSON: so does a string or byte array whose length cannot be right + { + json::input_format_t::bson, bson_document( + { + bson_element(0x03, "inner", bson_document({bson_element(0x02, "s", bson_string("abc", -10)), bson_element(0x10, "b", bson_int32(2))})), + bson_element(0x03, "bin", bson_document({bson_element(0x05, "b", concatenated(bson_int32(-1), bytes(1))), bson_element(0x10, "b", bson_int32(2))})), + bson_element(0x10, "after", bson_int32(3)), + }), + {{"inner", {{"s", nullptr}}}, {"bin", {{"b", nullptr}}}, {"after", 3}}, + 2 + }, + // BSON: a string without its terminator, and a document whose size does not match, are kept + { + json::input_format_t::bson, bson_document( + { + bson_element(0x02, "s", {2, 0, 0, 0, 'a', 'X'}), + bson_element(0x03, "inner", bson_document({bson_element(0x10, "a", bson_int32(1))}, 1)), + }), + {{"s", "a"}, {"inner", {{"a", 1}}}}, + 2 + }, + }; + + for (const auto& repair : repairs) + { + CAPTURE(repair.format); + CAPTURE(repair.input); + const auto result = parse_binary_recovering(repair.input, repair.format); + CHECK(!result.ok); + CHECK(result.balanced); + CHECK(result.errors == repair.errors); + CHECK(result.value == repair.expected); + // the first error is the one reported without recovering + REQUIRE(!result.messages.empty()); + CHECK(result.messages.front() == binary_error_message(repair.input, repair.format)); + } + } + + SECTION("binary formats repair numbers that are out of range") + { + // CBOR: a negative integer below the range of number_integer_t + const auto cbor = parse_binary_recovering({0x3B, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF}, json::input_format_t::cbor); + CHECK(cbor.errors == 1); + CHECK(cbor.value.is_number_float()); + CHECK(cbor.value.get() == -18446744073709551616.0); + + // UBJSON: a high-precision number too large for number_float_t + const auto ubjson = parse_binary_recovering({'H', 'i', 5, '1', 'e', '9', '9', '9'}, json::input_format_t::ubjson); + CHECK(ubjson.errors == 1); + CHECK(ubjson.value.is_number_float()); + CHECK(std::isinf(ubjson.value.get())); + } + + SECTION("binary formats stop where the end of an item is not known") + { + // a byte that begins no item + const auto cbor = parse_binary_recovering({0x82, 0x01, 0x1C, 0x02}, json::input_format_t::cbor); + CHECK(cbor.errors == 1); + CHECK(cbor.value == json({1})); + + // a key that is no item: the unused MessagePack byte, a CBOR break + // in a map of known size, and the end of a BON8 container + const auto msgpack = parse_binary_recovering({0x82, 0xA1, 'a', 0x01, 0xC1, 0x02}, json::input_format_t::msgpack); + CHECK(msgpack.errors == 1); + CHECK(msgpack.value == json({{"a", 1}})); + const auto cbor_break = parse_binary_recovering({0xA2, 0x61, 'a', 0x01, 0xFF, 0x02}, json::input_format_t::cbor); + CHECK(cbor_break.errors == 1); + CHECK(cbor_break.value == json({{"a", 1}})); + const auto bon8 = parse_binary_recovering({0x88, 0x61, 0x91, 0xFE}, json::input_format_t::bon8); + CHECK(bon8.errors == 1); + CHECK(bon8.value == json({{"a", 1}})); + + // a skipped member that the input ends in + const auto truncated = parse_binary_recovering({0xA2, 0x01, 0x82, 0x01}, json::input_format_t::cbor); + CHECK(truncated.errors == 2); + CHECK(truncated.balanced); + CHECK(truncated.value == json::object()); + + // a BSON element of an unknown type in a document whose size cannot be right + const auto bson = parse_binary_recovering(bson_document({bson_element(0x10, "a", bson_int32(1)), bson_element(0x42, "x", bytes(3))}, -10), json::input_format_t::bson); + CHECK(bson.errors == 1); + CHECK(bson.value == json({{"a", 1}, {"x", nullptr}})); + } + + SECTION("changed bytes in binary input") + { + const json j = {{"a", {1, -2, {{"b", "c"}}, json::array()}}, {"d", {{"e", nullptr}, {"f", true}}}, {"g", 1.5}, {"h", json::binary({1, 2, 3})}, {"i", "\xC3\xA4"}}; + + const std::vector>> encodings = + { + {json::input_format_t::cbor, json::to_cbor(j)}, + {json::input_format_t::msgpack, json::to_msgpack(j)}, + {json::input_format_t::ubjson, json::to_ubjson(j)}, + {json::input_format_t::ubjson, json::to_ubjson(j, true, true)}, + {json::input_format_t::bjdata, json::to_bjdata(j)}, + {json::input_format_t::bjdata, json::to_bjdata(j, true, true)}, + {json::input_format_t::bson, json::to_bson(j)}, + {json::input_format_t::bon8, json::to_bon8(j)}, + }; + const std::vector replacements = {0x00, 0x01, 0x7F, 0x80, 0xC1, 0xD9, 0xE0, 0xF7, 0xFE, 0xFF}; + + for (const auto& encoding : encodings) + { + const auto format = encoding.first; + const auto& original = encoding.second; + CAPTURE(format); + + std::vector> inputs; + for (std::size_t position = 0; position < original.size(); ++position) + { + for (const auto replacement : replacements) + { + auto changed = original; + changed[position] = replacement; + inputs.push_back(changed); + } + auto removed = original; + removed.erase(removed.begin() + static_cast(position)); + inputs.push_back(removed); + } + + for (const auto& input : inputs) + { + CAPTURE(input); + const auto result = parse_binary_recovering(input, format); + CHECK(result.balanced); + CHECK(result.errors <= input.size() + 1); + // an error is reported exactly if reading into a JSON value + // fails, and the first one is the same + const auto message = binary_error_message(input, format); + CHECK(result.ok == message.empty()); + if (!result.ok && result.errors < 100) + { + CHECK(result.messages.front() == message); + } + } + } + } + SECTION("JSON text") { // the parser stopped, but reported success