From f56b418c56dcbf7c05fe6cca72e2a4de2d072b99 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 4 Oct 2026 12:13:48 +0200 Subject: [PATCH] Follow each binary format's UTF-8 rule: strict writers (CBOR/UBJSON/BJData/BSON), lenient readers (#5741) Signed-off-by: Niels Lohmann --- CMakeLists.txt | 6 + cmake/ci.cmake | 2 +- docs/mkdocs/docs/api/basic_json/to_bjdata.md | 7 +- docs/mkdocs/docs/api/basic_json/to_bson.md | 6 + docs/mkdocs/docs/api/basic_json/to_cbor.md | 8 ++ docs/mkdocs/docs/api/basic_json/to_ubjson.md | 5 + docs/mkdocs/docs/api/macros/index.md | 2 + .../api/macros/json_strict_binary_utf8.md | 97 +++++++++++++++ .../docs/features/binary_formats/bjdata.md | 16 +++ .../docs/features/binary_formats/bson.md | 16 +-- .../docs/features/binary_formats/cbor.md | 17 +-- .../features/binary_formats/messagepack.md | 15 +-- .../docs/features/binary_formats/ubjson.md | 16 +++ docs/mkdocs/docs/features/macros.md | 14 +++ docs/mkdocs/docs/features/namespace.md | 1 + docs/mkdocs/docs/home/exceptions.md | 13 ++- docs/mkdocs/docs/integration/cmake.md | 5 + docs/mkdocs/mkdocs.yml | 1 + include/nlohmann/detail/abi_macros.hpp | 19 ++- .../nlohmann/detail/input/binary_reader.hpp | 29 ++--- include/nlohmann/detail/macro_unscope.hpp | 1 + .../nlohmann/detail/output/binary_writer.hpp | 70 ++++++++++- include/nlohmann/detail/string_utils.hpp | 44 +------ single_include/nlohmann/json_fwd.hpp | 19 ++- tests/abi/config/default.cpp | 4 + tests/abi/config/noversion.cpp | 4 + tests/src/unit-binary_utf8_strict.cpp | 110 ++++++++++++++++++ tests/src/unit-bjdata.cpp | 37 ++++++ tests/src/unit-bson.cpp | 37 ++++++ tests/src/unit-cbor.cpp | 85 +++++++++++--- tests/src/unit-msgpack.cpp | 36 ++++-- tests/src/unit-ubjson.cpp | 37 ++++++ 32 files changed, 651 insertions(+), 128 deletions(-) create mode 100644 docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md create mode 100644 tests/src/unit-binary_utf8_strict.cpp diff --git a/CMakeLists.txt b/CMakeLists.txt index f0a6771dc..b5092ae49 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -61,6 +61,7 @@ option(JSON_Install "Install CMake targets during install option(JSON_MultipleHeaders "Use non-amalgamated version of the library." ON) option(JSON_SystemInclude "Include as system headers (skip for clang-tidy)." OFF) option(JSON_StrictNulHandling "Build with strict NUL-byte handling enabled." OFF) +option(JSON_StrictBinaryUTF8 "Build with UTF-8 checks in the CBOR, UBJSON, BJData, and BSON writers enabled." OFF) if (JSON_CI) include(ci) @@ -118,6 +119,10 @@ if (JSON_StrictNulHandling) message(STATUS "Strict NUL-byte handling enabled (JSON_STRICT_NUL_HANDLING=1)") endif() +if (JSON_StrictBinaryUTF8) + message(STATUS "Strict UTF-8 checks in binary writers enabled (JSON_STRICT_BINARY_UTF8=1)") +endif() + if (JSON_Diagnostic_Positions) message(STATUS "Diagnostic positions enabled (JSON_DIAGNOSTIC_POSITIONS=1)") endif() @@ -153,6 +158,7 @@ target_compile_definitions( $<$:JSON_DIAGNOSTIC_POSITIONS=1> $<$:JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1> $<$:JSON_STRICT_NUL_HANDLING=1> + $<$:JSON_STRICT_BINARY_UTF8=1> ) target_include_directories( diff --git a/cmake/ci.cmake b/cmake/ci.cmake index e67aba6d3..2aceb5afd 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -701,7 +701,7 @@ ci_get_cmake(4.0.0 CMAKE_4_0_0_BINARY) # the tests require CMake 3.13 or later, so they are excluded for CMake 3.5.0 set(JSON_CMAKE_FLAGS_3_5_0 JSON_Diagnostics JSON_Diagnostic_Positions JSON_GlobalUDLs JSON_ImplicitConversions JSON_DisableEnumSerialization JSON_LegacyDiscardedValueComparison JSON_Install JSON_MultipleHeaders JSON_SystemInclude JSON_Valgrind - JSON_StrictNulHandling) + JSON_StrictNulHandling JSON_StrictBinaryUTF8) set(JSON_CMAKE_FLAGS_3_31_6 JSON_BuildTests ${JSON_CMAKE_FLAGS_3_5_0}) set(JSON_CMAKE_FLAGS_4_0_0 JSON_BuildTests ${JSON_CMAKE_FLAGS_3_5_0}) diff --git a/docs/mkdocs/docs/api/basic_json/to_bjdata.md b/docs/mkdocs/docs/api/basic_json/to_bjdata.md index 44cc399e1..63dc5379e 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/to_bjdata.md @@ -56,6 +56,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false, and `j` contains a non-empty array, object, or binary value. +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not + valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are + written unchanged ## Complexity @@ -104,4 +107,6 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 3.11.0. -- BJData version parameter (for draft3 binary encoding) added in version 3.12.0. \ No newline at end of file +- BJData version parameter (for draft3 binary encoding) added in version 3.12.0. +- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. \ No newline at end of file diff --git a/docs/mkdocs/docs/api/basic_json/to_bson.md b/docs/mkdocs/docs/api/basic_json/to_bson.md index c79f39ffc..e3104b13c 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bson.md +++ b/docs/mkdocs/docs/api/basic_json/to_bson.md @@ -46,6 +46,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value exceeds 255, the maximum of the BSON binary subtype; example: `"subtype 70000 is too large for the BSON binary subtype (max 255)"` +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key is not valid + UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are + written unchanged ## Complexity @@ -98,3 +101,6 @@ pass before anything is written. - Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0. - Linear in the size of `j`, and no longer limited by the call stack for deeply nested values, since version 3.13.0. - `out_of_range.415` is now detected before anything is written, like the other exceptions above, since version 3.13.0. +- Throwing `type_error.316` for a string value or object key that is not valid UTF-8 if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, detected before anything is written, + added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/to_cbor.md b/docs/mkdocs/docs/api/basic_json/to_cbor.md index 3bbd9c7d3..cac8a0917 100644 --- a/docs/mkdocs/docs/api/basic_json/to_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/to_cbor.md @@ -35,6 +35,12 @@ The exact mapping and its limitations are described on a [dedicated page](../../ Strong guarantee: if an exception is thrown, there are no changes in the JSON value. +## Exceptions + +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not + valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are + written unchanged + ## Complexity Linear in the size of the JSON value `j`. @@ -68,3 +74,5 @@ Linear in the size of the JSON value `j`. - Added in version 2.0.9. - Compact representation of floating-point numbers added in version 3.8.0. +- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/to_ubjson.md b/docs/mkdocs/docs/api/basic_json/to_ubjson.md index 1b7f7767e..8f2ea5eb9 100644 --- a/docs/mkdocs/docs/api/basic_json/to_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/to_ubjson.md @@ -49,6 +49,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false, and `j` contains a non-empty array, object, or binary value. +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not + valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are + written unchanged ## Complexity @@ -97,3 +100,5 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 3.1.0. +- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index 3c9b42af2..872d6cb8c 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -18,6 +18,8 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_PRECISE_STREAM_POSITION**](json_precise_stream_position.md) - opt in to leaving an input stream positioned right after a parsed number +- [**JSON_STRICT_BINARY_UTF8**](json_strict_binary_utf8.md) - opt in to checking strings for valid UTF-8 in the CBOR, + UBJSON, BJData, and BSON writers - [**JSON_STRICT_NUL_HANDLING**](json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of treating it as end of input diff --git a/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md new file mode 100644 index 000000000..7fe6d9a58 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md @@ -0,0 +1,97 @@ +# JSON_STRICT_BINARY_UTF8 + +```cpp +#define JSON_STRICT_BINARY_UTF8 /* value */ +``` + +When defined to `1`, the binary writers [`to_cbor`](../basic_json/to_cbor.md), [`to_ubjson`](../basic_json/to_ubjson.md), +[`to_bjdata`](../basic_json/to_bjdata.md), and [`to_bson`](../basic_json/to_bson.md) check every string value and +object key for valid UTF-8 and throw [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for +ill-formed UTF-8, like [`dump`](../basic_json/dump.md) does. Without it, they write the bytes unchanged. + +The macro does not affect: + +- [`to_msgpack`](../basic_json/to_msgpack.md): the MessagePack specification allows a `str` value to contain bytes that + are not valid UTF-8, so it always writes them unchanged. +- [`to_bon8`](../basic_json/to_bon8.md): BON8 always checks, because the UTF-8 lead bytes mark where a string ends. +- The binary readers ([`from_cbor`](../basic_json/from_cbor.md), [`from_msgpack`](../basic_json/from_msgpack.md), + [`from_ubjson`](../basic_json/from_ubjson.md), [`from_bjdata`](../basic_json/from_bjdata.md), + [`from_bson`](../basic_json/from_bson.md)): none of these formats requires a decoder to reject ill-formed UTF-8, so + they always return the bytes unchanged. + +## Default definition + +The default value is `0` (disabled, the behavior of version 3.12.0 and earlier is preserved). + +```cpp +#define JSON_STRICT_BINARY_UTF8 0 +``` + +## Notes + +!!! note "Background" + + CBOR, UBJSON, BJData, and BSON all require strings to be UTF-8. Up to version 3.12.0, the writers did not check + this, so they could produce output that other decoders reject. Checking by default would break code that stores + other encodings (for instance ISO 8859-1) in a string and only ever writes it to a binary format, so this macro + offers the check as an opt-in ahead of version 4.0.0, where it is planned to become the default (see + [#5529](https://github.com/nlohmann/json/issues/5529) and [#5651](https://github.com/nlohmann/json/issues/5651)). + +!!! warning "Opt-in only" + + This macro must be defined **before** including ``. Defining it after the include has no + effect. + +!!! note "ABI compatibility" + + The value of this macro is encoded in the [namespace](../../features/namespace.md) (tag `_sbu8`), resulting in + distinct symbol names. Translation units compiled with and without it can therefore be linked into the same program + without One Definition Rule (ODR) violations, but they cannot exchange instances of library types. + +## Examples + +??? example "Default behavior (macro not defined)" + + Without the macro, the bytes are written unchanged: + + ```cpp + #include + + using json = nlohmann::json; + + int main() + { + auto v = json::to_cbor(json("\xFF")); + // v is {0x61, 0xFF} + } + ``` + +??? example "Opt-in check (macro defined to 1)" + + With the macro, ill-formed UTF-8 is rejected: + + ```cpp + #define JSON_STRICT_BINARY_UTF8 1 + #include + + using json = nlohmann::json; + + int main() + { + auto v = json::to_cbor(json("\xFF")); + // throws type_error.316: invalid UTF-8 byte at index 0: 0xFF + } + ``` + +## See also + +- [**to_cbor**](../basic_json/to_cbor.md) - create a CBOR serialization of a JSON value +- [**to_ubjson**](../basic_json/to_ubjson.md) - create a UBJSON serialization of a JSON value +- [**to_bjdata**](../basic_json/to_bjdata.md) - create a BJData serialization of a JSON value +- [**to_bson**](../basic_json/to_bson.md) - create a BSON serialization of a JSON value +- [**error_handler_t**](../basic_json/error_handler_t.md) - how [`dump`](../basic_json/dump.md) treats ill-formed UTF-8 + +## Version history + +- Added in version 3.13.0. +- Planned to become the default (with the macro removed) in version 4.0.0. diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index 8e3ed41f6..12d1af51b 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -63,6 +63,13 @@ The library uses the following mapping from JSON values types to BJData types ac - strings with more than 18446744073709551615 bytes, i.e., 264-1 bytes (theoretical) +!!! warning "UTF-8 validation of string values and object keys" + + BJData strings must use UTF-8 encoding. By default, `to_bjdata()` writes the bytes of string values and object keys + unchanged, even if they are not valid UTF-8. If + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead. + !!! info "Unused BJData markers" The following markers are not used in the conversion: @@ -208,6 +215,15 @@ The library maps BJData types to JSON value types as follows: The mapping is **complete** in the sense that any BJData value can be converted to a JSON value. +!!! warning "Ill-formed UTF-8 in string values and object keys" + + BJData strings must use UTF-8 encoding, but this is not enforced on read: `from_bjdata()` accepts a string + value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However, + [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error + handler is passed that replaces or ignores the ill-formed bytes. By default, `to_bjdata()` writes such a value + back unchanged (see above). + !!! info "Round trips" A value returned by [`from_bjdata`](../../api/basic_json/from_bjdata.md) can be serialized with diff --git a/docs/mkdocs/docs/features/binary_formats/bson.md b/docs/mkdocs/docs/features/binary_formats/bson.md index cca11451e..245ed3ebf 100644 --- a/docs/mkdocs/docs/features/binary_formats/bson.md +++ b/docs/mkdocs/docs/features/binary_formats/bson.md @@ -109,14 +109,16 @@ The library maps BSON record types to JSON value types as follows: If BSON input must be validated for strict specification compliance, validate it separately before passing it to `from_bson()`. -!!! warning "UTF-8 validation of string values" +!!! warning "Ill-formed UTF-8 in string values" - The BSON specification requires `string` values (type `0x02`) to be valid UTF-8. This library validates the - bytes of every such string at decode time and rejects ill-formed UTF-8 with a - [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with `allow_exceptions` - set to `false`, a discarded value), rather than only failing later when the resulting value is dumped. Element - (key) names and `binary` values (type `0x05`) are unaffected and are never validated, since they are read - byte-by-byte as a C string, or are not required to hold text, respectively. + The BSON specification requires `string` values (type `0x02`) to be valid UTF-8, but this is not required of a + decoder. `from_bson()` accepts a `string` value whose bytes are not valid UTF-8 and hands them back unchanged. + However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error handler is + passed that replaces or ignores the ill-formed bytes. By default, `to_bson()` writes such a string value or element + (key) name unchanged; if [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it + throws the same exception instead. Element (key) names are never validated on read, since they are read byte-by-byte + as a C string. `binary` values (type `0x05`) are unaffected, since they are not required to hold text. ??? example "Example: deserialize a JSON value from BSON" diff --git a/docs/mkdocs/docs/features/binary_formats/cbor.md b/docs/mkdocs/docs/features/binary_formats/cbor.md index eb7bc7f41..66300c5fa 100644 --- a/docs/mkdocs/docs/features/binary_formats/cbor.md +++ b/docs/mkdocs/docs/features/binary_formats/cbor.md @@ -189,15 +189,16 @@ The library maps CBOR types to JSON value types as follows: ([RFC 8392](https://www.rfc-editor.org/rfc/rfc8392.html)), cannot be read with this library and need a general-purpose CBOR library instead. -!!! warning "UTF-8 validation of text strings" +!!! warning "Ill-formed UTF-8 in text strings" - [RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings - (major type 3) to be valid UTF-8. This library validates the bytes of every text string (object keys included) at - decode time and rejects ill-formed UTF-8 with a - [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with - `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value is - dumped. Byte strings (major type 2) are unaffected and are never validated, since they are not required to hold - text. + [RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings (major + type 3) to be valid UTF-8, but leaves it up to the decoder whether to enforce this. This library does not: + `from_cbor()` accepts a text string (object keys included) whose bytes are not valid UTF-8 and hands them back + unchanged. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error handler is + passed that replaces or ignores the ill-formed bytes. By default, `to_cbor()` writes such a value back unchanged; if + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws the same exception + instead. Byte strings (major type 2) are unaffected, since they are not required to hold text. !!! warning "Tagged items" diff --git a/docs/mkdocs/docs/features/binary_formats/messagepack.md b/docs/mkdocs/docs/features/binary_formats/messagepack.md index 2e674252a..047944852 100644 --- a/docs/mkdocs/docs/features/binary_formats/messagepack.md +++ b/docs/mkdocs/docs/features/binary_formats/messagepack.md @@ -153,14 +153,15 @@ The library maps MessagePack types to JSON value types as follows: This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed on. Such input needs a general-purpose MessagePack library instead. -!!! warning "UTF-8 validation of string values" +!!! warning "Ill-formed UTF-8 in string values" - The MessagePack specification requires `str` values (`fixstr`, `str 8`, `str 16`, `str 32`) to be valid UTF-8. - This library validates the bytes of every such string (object keys included) at decode time and rejects - ill-formed UTF-8 with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, - with `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting - value is dumped. `bin`/`ext`/`fixext` values are unaffected and are never validated, since they are not required - to hold text. + The MessagePack specification explicitly allows a `str` value (`fixstr`, `str 8`, `str 16`, `str 32`) to contain + a byte sequence that is not valid UTF-8, and expects a deserializer to hand the original bytes back unchanged. + This library follows that: `from_msgpack()` reads `str` bytes (object keys included) as-is, without validating + them, and `to_msgpack()` writes them back as-is, so such a value round-trips through `from_msgpack(to_msgpack(j))` + byte for byte. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read this way, unless an + error handler is passed that replaces or ignores the ill-formed bytes. ??? example "Example: deserialize a JSON value from MessagePack" diff --git a/docs/mkdocs/docs/features/binary_formats/ubjson.md b/docs/mkdocs/docs/features/binary_formats/ubjson.md index 37aa069e2..dbdff6e6c 100644 --- a/docs/mkdocs/docs/features/binary_formats/ubjson.md +++ b/docs/mkdocs/docs/features/binary_formats/ubjson.md @@ -47,6 +47,13 @@ The library uses the following mapping from JSON values types to UBJSON types ac - strings with more than 9223372036854775807 bytes (theoretical) +!!! warning "UTF-8 validation of string values and object keys" + + UBJSON's required string encoding is UTF-8. By default, `to_ubjson()` writes the bytes of string values and object + keys unchanged, even if they are not valid UTF-8. If + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead. + !!! info "Unused UBJSON markers" The following markers are not used in the conversion: @@ -120,6 +127,15 @@ The library maps UBJSON types to JSON value types as follows: The mapping is **complete** in the sense that any UBJSON value can be converted to a JSON value. +!!! warning "Ill-formed UTF-8 in string values and object keys" + + UBJSON's required string encoding is UTF-8, but this is not enforced on read: `from_ubjson()` accepts a string + value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However, + [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error + handler is passed that replaces or ignores the ill-formed bytes. By default, `to_ubjson()` writes such a value + back unchanged (see above). + ??? example "Example: deserialize a JSON value from UBJSON" ```cpp diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index 90bb352c3..681980315 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -138,6 +138,20 @@ using the library with compilers that do not fully support C++11 and may only wo See [full documentation of `JSON_SKIP_UNSUPPORTED_COMPILER_CHECK`](../api/macros/json_skip_unsupported_compiler_check.md). +## `JSON_STRICT_BINARY_UTF8` + +When defined to `1`, [`to_cbor`](../api/basic_json/to_cbor.md), [`to_ubjson`](../api/basic_json/to_ubjson.md), +[`to_bjdata`](../api/basic_json/to_bjdata.md), and [`to_bson`](../api/basic_json/to_bson.md) throw +[`type_error.316`](../home/exceptions.md#jsonexceptiontype_error316) for a string value or object key that is not +valid UTF-8. The default value is `0`, which writes the bytes unchanged as before version 3.13.0; this is planned to +become the default in version 4.0.0. + +The check can also be enabled with the CMake option +[`JSON_StrictBinaryUTF8`](../integration/cmake.md#json_strictbinaryutf8) (`OFF` by default) which sets +`JSON_STRICT_BINARY_UTF8` accordingly. + +See [full documentation of `JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md). + ## `JSON_STRICT_NUL_HANDLING` When defined to `1`, a `'\0'` (NUL) byte anywhere in the input is rejected with `parse_error.101`, like any other diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index 577f5e221..dbddac13d 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -20,6 +20,7 @@ The complete default namespace name is derived as follows: `_bics`. - [`JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md) defined non-zero appends `_psp`. - [`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md) defined non-zero appends `_snul`. + - [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) defined non-zero appends `_sbu8`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index c78aeaa68..a3797cf04 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -340,8 +340,9 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde ### json.exception.parse_error.113 A string could not be read from a [binary format](../features/binary_formats/index.md): either a value that is not a -string was read where one was required (for instance as a map key), the string's length specification is invalid, or -the string's bytes are not valid UTF-8. +string was read where one was required (for instance as a map key), or the string's length specification is invalid. +The bytes of a string itself are not checked for valid UTF-8 on read; see the ill-formed UTF-8 notes on the +individual [binary format](../features/binary_formats/index.md) pages for how such a string is handled afterward. CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings. Maps with keys of any other type (for instance integers or `null`) are therefore not supported; see the notes on @@ -364,9 +365,6 @@ type (for instance integers or `null`) are therefore not supported; see the note ``` [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData string: string length must not be negative ``` - ``` - [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte - ``` ### json.exception.parse_error.114 @@ -749,6 +747,11 @@ The [`unflatten()`](../api/basic_json/unflatten.md) function only works for an o The [`dump()`](../api/basic_json/dump.md) function only works with UTF-8 encoded strings; that is, if you assign a `std::string` to a JSON value, make sure it is UTF-8 encoded. See the FAQ entry on [serializing untrusted or invalid UTF-8](faq.md#serializing-untrusted-or-invalid-utf-8) for background and the recommended fix. +If [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) is enabled, the binary writers +[`to_cbor()`](../api/basic_json/to_cbor.md), [`to_ubjson()`](../api/basic_json/to_ubjson.md), +[`to_bjdata()`](../api/basic_json/to_bjdata.md), and [`to_bson()`](../api/basic_json/to_bson.md) throw this exception +for a string value or object key that is not valid UTF-8 as well. + !!! failure "Example message" Calling `dump()` on a JSON value containing an ISO 8859-1 encoded string: diff --git a/docs/mkdocs/docs/integration/cmake.md b/docs/mkdocs/docs/integration/cmake.md index 785fc3ed6..9151d12be 100644 --- a/docs/mkdocs/docs/integration/cmake.md +++ b/docs/mkdocs/docs/integration/cmake.md @@ -212,6 +212,11 @@ Use the non-amalgamated version of the library. This option is `ON` by default. Treat the library headers like system headers (i.e., adding `SYSTEM` to the [`target_include_directories`](https://cmake.org/cmake/help/latest/command/target_include_directories.html) call) to check for this library by tools like Clang-Tidy. This option is `OFF` by default. +### `JSON_StrictBinaryUTF8` + +Check string values and object keys for valid UTF-8 in the CBOR, UBJSON, BJData, and BSON writers, by defining the +macro [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md). This option is `OFF` by default. + ### `JSON_StrictNulHandling` Reject a `'\0'` (NUL) byte in the input instead of treating it as end of input, by defining the macro diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 7b5512084..3cb25ba9b 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -307,6 +307,7 @@ nav: - 'JSON_PRECISE_STREAM_POSITION': api/macros/json_precise_stream_position.md - 'JSON_SKIP_LIBRARY_VERSION_CHECK': api/macros/json_skip_library_version_check.md - 'JSON_SKIP_UNSUPPORTED_COMPILER_CHECK': api/macros/json_skip_unsupported_compiler_check.md + - 'JSON_STRICT_BINARY_UTF8': api/macros/json_strict_binary_utf8.md - 'JSON_STRICT_NUL_HANDLING': api/macros/json_strict_nul_handling.md - 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md - 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md diff --git a/include/nlohmann/detail/abi_macros.hpp b/include/nlohmann/detail/abi_macros.hpp index 0bace616a..0153c8706 100644 --- a/include/nlohmann/detail/abi_macros.hpp +++ b/include/nlohmann/detail/abi_macros.hpp @@ -46,6 +46,10 @@ #define JSON_STRICT_NUL_HANDLING 0 #endif +#ifndef JSON_STRICT_BINARY_UTF8 + #define JSON_STRICT_BINARY_UTF8 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -82,14 +86,20 @@ #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING #endif +#if JSON_STRICT_BINARY_UTF8 + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8 +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -98,7 +108,8 @@ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ - NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \ + NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index 86f00bef6..3d79ad41b 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -4044,28 +4044,13 @@ class binary_reader const NumberType len, string_t& result) { - // get_bytes() appends to result, and CBOR indefinite-length strings - // collect all their chunks in the same result; validating only the - // newly read bytes keeps the check linear in the input size - const std::size_t old_size = result.size(); - if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result))) - { - return false; - } - - // RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications - // all require text strings to be valid UTF-8; reject anything else - // right here so malformed input is caught at decode time instead of - // only surfacing later as a type_error.316 when the value is dumped - // (which would defeat allow_exceptions=false / strict discarding). - if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) - { - return sax->parse_error(chars_read, get_token_string(), - parse_error::create(113, chars_read, - exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr)); - } - - return true; + // Strings are taken as is: none of CBOR (RFC 8949 §3.1 leaves the + // choice to the decoder), MessagePack (whose spec explicitly allows + // a str object to contain an invalid byte sequence), UBJSON, BJData, + // or BSON requires a decoder to reject ill-formed UTF-8. The bytes + // are kept unchanged; dump() and the binary writers are the ones + // that check them and report type_error.316 if they are not valid. + return get_bytes(format, len, "string", result); } /*! diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index 8e1d49842..55b2ac99e 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -42,6 +42,7 @@ #undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_PRECISE_STREAM_POSITION #undef JSON_STRICT_NUL_HANDLING + #undef JSON_STRICT_BINARY_UTF8 #endif #include diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 9d844d85e..da5aa0cf4 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -115,6 +115,8 @@ class binary_writer /*! @param[in] j JSON value to serialize + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or an object key is not valid UTF-8 @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) @@ -145,6 +147,8 @@ class binary_writer /*! @param[in] j JSON value to serialize + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or an object key is not valid UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -211,6 +215,8 @@ class binary_writer case value_t::string: { + check_text_utf8(*j.m_data.m_value.string, j); + // step 1: write control byte and the string length write_cbor_head(0x60, j.m_data.m_value.string->size()); @@ -287,6 +293,11 @@ class binary_writer // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { + // el.first is checked here, against the object as + // diagnostics context, because write_cbor(el.first) + // converts it to a temporary basic_json that would be + // used as the context instead + check_text_utf8(el.first, j); write_cbor(el.first); write_cbor(el.second); } @@ -629,6 +640,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or an object key is not valid UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -678,6 +691,8 @@ class binary_writer case value_t::string: { + check_text_utf8(*j.m_data.m_value.string, j); + if (add_prefix) { oa.write_character(to_char_type('S')); @@ -840,6 +855,7 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { + check_text_utf8(el.first, j); write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(el.first.data()), @@ -884,6 +900,10 @@ class binary_writer /*! @return The size of a BSON document entry header, including the id marker and the entry name size (and its null-terminator). + @throw out_of_range.409 if @a name contains U+0000, before anything is + written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a name is + not valid UTF-8, before anything is written */ static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { @@ -893,7 +913,8 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - static_cast(j); + check_text_utf8(name, j); + return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; } @@ -949,9 +970,21 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a value + is not valid UTF-8, before anything is written + + @note The UTF-8 check is skipped if @a value is already too long for the + 32-bit BSON length field (@ref to_bson_length rejects it later, once + the size of the whole document is known); this also keeps the check + from reading past a StringType that reports a size larger than what + it actually holds. */ - static std::size_t calc_bson_string_size(const string_t& value) + static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j) { + if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) + { + check_text_utf8(value, j); + } return sizeof(std::int32_t) + value.size() + 1ul; } @@ -1080,6 +1113,8 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a j is a + string that is not valid UTF-8, before anything is written */ static std::size_t calc_bson_value_size(const BasicJsonType& j) { @@ -1101,7 +1136,7 @@ class binary_writer return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); case value_t::string: - return calc_bson_string_size(*j.m_data.m_value.string); + return calc_bson_string_size(*j.m_data.m_value.string, j); case value_t::null: return 0ul; @@ -1214,6 +1249,8 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or a key is not valid UTF-8, before anything is written */ static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { @@ -2092,7 +2129,7 @@ class binary_writer */ void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context) { - check_bon8_utf8(s, context); + check_utf8(s, context); // a string that follows another string terminates it if (string_open) @@ -2122,7 +2159,7 @@ class binary_writer @throw type_error.316 if @a s is not valid UTF-8; the message names the first byte of the first invalid or incomplete sequence */ - static void check_bon8_utf8(const string_t& s, const BasicJsonType& context) + static void check_utf8(const string_t& s, const BasicJsonType& context) { static_cast(context); // only used when exceptions are enabled const auto* data = reinterpret_cast(s.data()); @@ -2133,6 +2170,29 @@ class binary_writer } } + /*! + @brief check a CBOR, UBJSON, BJData, or BSON text string for valid UTF-8 + + The check only happens if JSON_STRICT_BINARY_UTF8 is enabled. Otherwise, + the bytes are written unchanged, as before version 3.13.0. MessagePack + always writes the bytes as is, and BON8 always checks them (see + @ref check_utf8). + + @param[in] s the string to check + @param[in] context the value that holds @a s (for diagnostics) + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a s is + not valid UTF-8 + */ + static void check_text_utf8(const string_t& s, const BasicJsonType& context) + { +#if JSON_STRICT_BINARY_UTF8 + check_utf8(s, context); +#else + static_cast(s); + static_cast(context); +#endif + } + /*! @brief write an integer in the shortest encoding diff --git a/include/nlohmann/detail/string_utils.hpp b/include/nlohmann/detail/string_utils.hpp index 7c40f7395..2b6864d0d 100644 --- a/include/nlohmann/detail/string_utils.hpp +++ b/include/nlohmann/detail/string_utils.hpp @@ -117,13 +117,14 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally written by Björn Hoehrmann. See http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details. -The library checks UTF-8 well-formedness (RFC 3629, section 4) in four +The library checks UTF-8 well-formedness (RFC 3629, section 4) in three places, which differ in speed, diagnostics, and how they read the input: -- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in - strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR, - MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in - text strings at decode time). +- decode() below: the serializer, to escape and, in strict mode, reject + ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON, + UBJSON and BJData readers do not use it: none of those specs requires a + decoder to reject ill-formed UTF-8 in text strings, so the readers keep + the bytes as is and leave the check to dump() and the binary writers. - the per-lead-byte switch in lexer::scan_string(): JSON text, with a diagnostic for each kind of error. - validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's @@ -178,38 +179,5 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std: return state; } -/*! -@brief check whether a string consists solely of valid UTF-8 - -Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text -strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the -MessagePack/BSON specifications all require text strings to be UTF-8), so -that malformed input is caught immediately instead of only surfacing later -as a type_error.316 when the resulting value is dumped. - -@param[in] s the string to check -@param[in] first index of the first byte to check; the bytes before it are - assumed to have been validated already and to end on a - code point boundary -@return whether @a s (from index @a first on) is valid UTF-8 -*/ -template -inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept -{ - std::uint8_t state = UTF8_ACCEPT; - std::uint32_t codepoint = 0; - - for (std::size_t i = first; i < s.size(); ++i) - { - decode(state, codepoint, static_cast(s[i])); - if (state == UTF8_REJECT) - { - return false; - } - } - - return state == UTF8_ACCEPT; -} - } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/single_include/nlohmann/json_fwd.hpp b/single_include/nlohmann/json_fwd.hpp index 3ee7afa73..7cadc9b1c 100644 --- a/single_include/nlohmann/json_fwd.hpp +++ b/single_include/nlohmann/json_fwd.hpp @@ -63,6 +63,10 @@ #define JSON_STRICT_NUL_HANDLING 0 #endif +#ifndef JSON_STRICT_BINARY_UTF8 + #define JSON_STRICT_BINARY_UTF8 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -99,14 +103,20 @@ #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING #endif +#if JSON_STRICT_BINARY_UTF8 + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8 +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -115,7 +125,8 @@ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ - NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \ + NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index 879322dd0..e4c627060 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -44,6 +44,10 @@ TEST_CASE("default namespace") expected += "_snul"; #endif +#if JSON_STRICT_BINARY_UTF8 + expected += "_sbu8"; +#endif + expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json"; diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index 4b1eb6ee4..858964695 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -45,6 +45,10 @@ TEST_CASE("default namespace without version component") expected += "_snul"; #endif +#if JSON_STRICT_BINARY_UTF8 + expected += "_sbu8"; +#endif + expected += "::basic_json"; // fallback for Clang diff --git a/tests/src/unit-binary_utf8_strict.cpp b/tests/src/unit-binary_utf8_strict.cpp new file mode 100644 index 000000000..c83b5938c --- /dev/null +++ b/tests/src/unit-binary_utf8_strict.cpp @@ -0,0 +1,110 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// The binary writers check strings and object keys for valid UTF-8 only if +// JSON_STRICT_BINARY_UTF8 is enabled (planned to be the default in 4.0.0). +// Without it, they write the bytes unchanged, as before version 3.13.0; the +// tests for that are next to the other tests of each format. +#ifdef JSON_STRICT_BINARY_UTF8 + #undef JSON_STRICT_BINARY_UTF8 +#endif + +#define JSON_STRICT_BINARY_UTF8 1 + +#include +using nlohmann::json; + +#include +#include + +TEST_CASE("JSON_STRICT_BINARY_UTF8 (see #5529, #5651)") +{ + SECTION("CBOR") + { + // a string value with ill-formed UTF-8 is rejected + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_cbor(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + + // binary values are not text and are unaffected + CHECK_NOTHROW(json::to_cbor(json::binary(std::vector({0xFF})))); + + // a value read back from CBOR with ill-formed bytes cannot be written + // back either (the reader is lenient regardless of the macro) + const json j = json::from_cbor(std::vector({0x62, 0xc0, 0xae})); + CHECK_THROWS_WITH_AS(json::to_cbor(j), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + } + + SECTION("UBJSON") + { + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_ubjson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + + SECTION("BJData") + { + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_bjdata(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + + SECTION("BSON") + { + // to_bson() rejects the same kind of ill-formed string value, before + // any bytes reach the output adapter (the BSON document length + // prefix must be known up front, so nothing is written incrementally) + std::vector out{0x42}; // a sentinel byte the writer must not touch + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}, nlohmann::detail::output_adapter(out)), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + CHECK(out == std::vector {0x42}); + + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC3"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xED\xA0\x80"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC0\xAF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected as well; unlike + // the reader (which never validates element names), the writer + // checks both string values and object keys + CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + + SECTION("MessagePack and BON8 are unaffected") + { + // MessagePack allows any bytes in a str, so to_msgpack() writes them as + // is; BON8 always checks, because the lead bytes mark where strings end + CHECK(json::to_msgpack(json("\xFF")) == std::vector({0xa1, 0xff})); + CHECK_THROWS_AS(json::to_bon8(json("\xFF")), json::type_error&); + } +} diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index b6be66c9e..f34df1717 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -3906,6 +3906,43 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") CHECK(json::to_bjdata(j) == v); CHECK(json::from_bjdata(v) == j); } + + SECTION("ill-formed UTF-8 (see #5529, #5651)") + { + // none of the binary format specs requires a decoder to reject + // ill-formed UTF-8 in a text string, so a value whose bytes are + // not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') + // round-trips byte for byte as a string value; to_bjdata() writes + // the bytes unchanged, as before 3.13.0, unless + // JSON_STRICT_BINARY_UTF8 is enabled (see + // unit-binary_utf8_strict.cpp) + const std::vector v = {'S', 'i', 2, 0xc0, 0xae}; + json j; + CHECK_NOTHROW(j = json::from_bjdata(v)); + REQUIRE(j.is_string()); + CHECK(j.get_ref() == std::string("\xc0\xae")); + CHECK_THROWS_AS(j.dump(), json::type_error&); + CHECK(json::from_bjdata(json::to_bjdata(j)) == j); + + // the same bytes as an object key round-trip as well + const std::vector v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'}; + json j_key; + CHECK_NOTHROW(j_key = json::from_bjdata(v_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_bjdata(json::to_bjdata(j_key)) == j_key); + + CHECK(json::from_bjdata(json::to_bjdata(json("\xFF"))) == json("\xFF")); + // a truncated multi-byte sequence + CHECK(json::from_bjdata(json::to_bjdata(json("\xC3"))) == json("\xC3")); + // an encoded surrogate half (U+D800) + CHECK(json::from_bjdata(json::to_bjdata(json("\xED\xA0\x80"))) == json("\xED\xA0\x80")); + // an overlong encoding of '.' + CHECK(json::from_bjdata(json::to_bjdata(json("\xC0\xAF"))) == json("\xC0\xAF")); + + // an object key with ill-formed UTF-8 is kept the same way + CHECK(json::from_bjdata(json::to_bjdata(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); + } } SECTION("Array Type") diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index 2f9a727ab..92a14e6fe 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -154,6 +154,43 @@ TEST_CASE("BSON") #endif } + SECTION("ill-formed UTF-8 (see #5529, #5651)") + { + // a BSON document {"s": "\xC0\xAE"} (0xC0 0xAE is an overlong + // encoding of '.'); the BSON spec does not require a decoder to + // reject ill-formed UTF-8 in a string value, so the reader hands the + // bytes back unchanged + const std::vector v = + { + 0x0F, 0x00, 0x00, 0x00, // document length + 0x02, 's', 0x00, // type 0x02 (string), key "s" + 0x03, 0x00, 0x00, 0x00, // string length (including null) + 0xc0, 0xae, 0x00, // string content and its null terminator + 0x00 // document terminator + }; + json j; + CHECK_NOTHROW(j = json::from_bson(v)); + REQUIRE(j.is_object()); + REQUIRE(j.contains("s")); + CHECK(j["s"].get_ref() == std::string("\xc0\xae")); + // dump() still requires valid UTF-8 and throws for such a value + CHECK_THROWS_AS(j.dump(), json::type_error&); + // to_bson() writes the bytes back unchanged, as before 3.13.0, + // unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp) + CHECK(json::from_bson(json::to_bson(j)) == j); + + CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}})) == json{{"s", "\xFF"}}); + // a truncated multi-byte sequence + CHECK(json::from_bson(json::to_bson(json{{"s", "\xC3"}})) == json{{"s", "\xC3"}}); + // an encoded surrogate half (U+D800) + CHECK(json::from_bson(json::to_bson(json{{"s", "\xED\xA0\x80"}})) == json{{"s", "\xED\xA0\x80"}}); + // an overlong encoding of '.' + CHECK(json::from_bson(json::to_bson(json{{"s", "\xC0\xAF"}})) == json{{"s", "\xC0\xAF"}}); + + // an object key with ill-formed UTF-8 is kept as well + CHECK(json::from_bson(json::to_bson(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); + } + SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON") { // out_of_range.412 is thrown from a single shared helper diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index c3a425a16..633f50fea 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -1801,19 +1801,41 @@ TEST_CASE("CBOR") CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&); } - SECTION("invalid UTF-8 in string (see #5529)") + SECTION("ill-formed UTF-8 in string (see #5529, #5651)") { + // RFC 8949 §3.1 leaves it up to the decoder whether to reject + // ill-formed UTF-8 in a text string; this library does not, and + // hands the original bytes back unchanged, matching the + // MessagePack reader and the behavior before #5185/#5531 (not in + // any release) + // a two-character text string (major type 3) whose bytes are not - // valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be - // rejected at decode time, matching every other kind of - // malformed binary input, rather than only failing later when - // the resulting value is dumped - json _; - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x62, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_cbor(std::vector({0x62, 0xc0, 0xae}), true, false).is_discarded()); + // valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') round-trips + // byte for byte as a string value + const std::vector ill_formed_value = {0x62, 0xc0, 0xae}; + json j_value; + CHECK_NOTHROW(j_value = json::from_cbor(ill_formed_value)); + REQUIRE(j_value.is_string()); + CHECK(j_value.get_ref() == std::string("\xc0\xae")); + // dump() still requires valid UTF-8 and throws for such a value, + // unless an error handler that replaces or ignores the bytes is + // passed + CHECK_THROWS_AS(j_value.dump(), json::type_error&); + // to_cbor() writes the bytes back unchanged, as before 3.13.0, + // unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp) + CHECK(json::from_cbor(json::to_cbor(j_value)) == j_value); + + // the same bytes as an object key round-trip as well + const std::vector ill_formed_key = {0xa1, 0x62, 0xc0, 0xae, 0x01}; + json j_key; + CHECK_NOTHROW(j_key = json::from_cbor(ill_formed_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_cbor(json::to_cbor(j_key)) == j_key); // a CBOR byte string (major type 2) with the very same bytes is // NOT text and must still be accepted as-is + json _; CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x42, 0xc0, 0xae}))); CHECK(_ == json::binary(std::vector({0xc0, 0xae}))); @@ -1822,17 +1844,47 @@ TEST_CASE("CBOR") CHECK(json::from_cbor(json::to_cbor(j)) == j); } - SECTION("invalid UTF-8 in indefinite-length string") + SECTION("to_cbor keeps ill-formed UTF-8 (see #5651)") + { + // to_cbor() writes the bytes unchanged, as before 3.13.0, unless + // JSON_STRICT_BINARY_UTF8 is enabled (see + // unit-binary_utf8_strict.cpp); from_cbor() reads them back as is + CHECK(json::from_cbor(json::to_cbor(json("\xFF"))) == json("\xFF")); + // a truncated multi-byte sequence + CHECK(json::from_cbor(json::to_cbor(json("\xC3"))) == json("\xC3")); + // an encoded surrogate half (U+D800) + CHECK(json::from_cbor(json::to_cbor(json("\xED\xA0\x80"))) == json("\xED\xA0\x80")); + // an overlong encoding of '.' + CHECK(json::from_cbor(json::to_cbor(json("\xC0\xAF"))) == json("\xC0\xAF")); + + // an object key with ill-formed UTF-8 is kept the same way + CHECK(json::from_cbor(json::to_cbor(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); + + // binary values are not text and are unaffected + CHECK_NOTHROW(json::to_cbor(json::binary(std::vector({0xFF})))); + } + + SECTION("ill-formed UTF-8 in indefinite-length string") { json _; - // every chunk must be valid UTF-8 on its own (RFC 8949, Section - // 3.2.3), so a code point split across two chunks is rejected - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}), true, false).is_discarded()); + // the chunks are concatenated as is, without checking that each + // chunk is valid UTF-8 on its own (RFC 8949, Section 3.2.3), so + // a code point split across two chunks yields a valid string + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}))); + CHECK(_ == "\xc3\xa9"); + CHECK(_.dump() == "\"\xc3\xa9\""); - // an ill-formed later chunk is rejected after valid ones - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); + // a truncated code point is kept as is + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0xff}))); + CHECK(_ == "\xc3"); + CHECK_THROWS_AS(_.dump(), json::type_error&); + CHECK(json::from_cbor(json::to_cbor(_)) == _); + + // an ill-formed later chunk is kept after valid ones + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff}))); + CHECK(_ == "\xc3\xa9\xc0\xae"); + CHECK_THROWS_AS(_.dump(), json::type_error&); // valid multi-byte chunks are accepted CHECK(json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc3, 0xb6, 0xff})) == "\xc3\xa9\xc3\xb6"); @@ -1840,9 +1892,6 @@ TEST_CASE("CBOR") SECTION("many chunks in indefinite-length string") { - // only the newly read chunk is validated, not the whole string - // collected so far; validating the latter made this input take - // quadratic time (about ten seconds for 100000 chunks) constexpr std::size_t chunks = 100000; std::vector v{0x7f}; for (std::size_t i = 0; i < chunks; ++i) diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index afabb85c6..a92496144 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -1540,19 +1540,39 @@ TEST_CASE("MessagePack") CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0x81})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing MessagePack string: unexpected end of input", json::parse_error&); } - SECTION("invalid UTF-8 in string (see #5529)") + SECTION("ill-formed UTF-8 in string (see #5529, #5651)") { + // the MessagePack specification explicitly allows a str object to + // contain a byte sequence that is not valid UTF-8 and expects a + // deserializer to hand the original bytes back unchanged; this + // library follows that, unlike CBOR/UBJSON/BJData/BSON, whose + // specifications require text strings to be valid UTF-8 + // a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8 - // (0xC0 0xAE is an overlong encoding of '.') must be rejected at - // decode time, matching every other kind of malformed binary - // input, rather than only failing later when the resulting - // value is dumped - json _; - CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0xa2, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing MessagePack string: invalid string: ill-formed UTF-8 byte", json::parse_error&); - CHECK(json::from_msgpack(std::vector({0xa2, 0xc0, 0xae}), true, false).is_discarded()); + // (0xC0 0xAE is an overlong encoding of '.') round-trips byte for + // byte as a string value + const std::vector ill_formed_value = {0xa2, 0xc0, 0xae}; + json j_value; + CHECK_NOTHROW(j_value = json::from_msgpack(ill_formed_value)); + REQUIRE(j_value.is_string()); + CHECK(j_value.get_ref() == std::string("\xc0\xae")); + CHECK(json::from_msgpack(json::to_msgpack(j_value)) == j_value); + // dump() still requires valid UTF-8 and throws for such a value, + // unless an error handler that replaces or ignores the bytes is + // passed + CHECK_THROWS_AS(j_value.dump(), json::type_error&); + + // the same bytes as an object key round-trip as well + const std::vector ill_formed_key = {0x81, 0xa2, 0xc0, 0xae, 0x01}; + json j_key; + CHECK_NOTHROW(j_key = json::from_msgpack(ill_formed_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_msgpack(json::to_msgpack(j_key)) == j_key); // a MessagePack bin8 blob with the very same bytes is NOT text // and must still be accepted as-is + json _; CHECK_NOTHROW(_ = json::from_msgpack(std::vector({0xc4, 0x02, 0xc0, 0xae}))); CHECK(_ == json::binary(std::vector({0xc0, 0xae}))); diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index 9a9710b48..450882a12 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -2505,6 +2505,43 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") CHECK(json::to_ubjson(j) == v); CHECK(json::from_ubjson(v) == j); } + + SECTION("ill-formed UTF-8 (see #5529, #5651)") + { + // none of the binary format specs requires a decoder to reject + // ill-formed UTF-8 in a text string, so a value whose bytes are + // not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') + // round-trips byte for byte as a string value; to_ubjson() writes + // the bytes unchanged, as before 3.13.0, unless + // JSON_STRICT_BINARY_UTF8 is enabled (see + // unit-binary_utf8_strict.cpp) + const std::vector v = {'S', 'i', 2, 0xc0, 0xae}; + json j; + CHECK_NOTHROW(j = json::from_ubjson(v)); + REQUIRE(j.is_string()); + CHECK(j.get_ref() == std::string("\xc0\xae")); + CHECK_THROWS_AS(j.dump(), json::type_error&); + CHECK(json::from_ubjson(json::to_ubjson(j)) == j); + + // the same bytes as an object key round-trip as well + const std::vector v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'}; + json j_key; + CHECK_NOTHROW(j_key = json::from_ubjson(v_key)); + REQUIRE(j_key.is_object()); + CHECK(j_key.contains(std::string("\xc0\xae"))); + CHECK(json::from_ubjson(json::to_ubjson(j_key)) == j_key); + + CHECK(json::from_ubjson(json::to_ubjson(json("\xFF"))) == json("\xFF")); + // a truncated multi-byte sequence + CHECK(json::from_ubjson(json::to_ubjson(json("\xC3"))) == json("\xC3")); + // an encoded surrogate half (U+D800) + CHECK(json::from_ubjson(json::to_ubjson(json("\xED\xA0\x80"))) == json("\xED\xA0\x80")); + // an overlong encoding of '.' + CHECK(json::from_ubjson(json::to_ubjson(json("\xC0\xAF"))) == json("\xC0\xAF")); + + // an object key with ill-formed UTF-8 is kept the same way + CHECK(json::from_ubjson(json::to_ubjson(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); + } } SECTION("Array Type")