From 3901b223e58671d0e8c460b530de2dfe731a269e Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:28:00 +0200 Subject: [PATCH 01/10] Complete the architecture documentation page (#5570) * Complete the architecture documentation page Replace the placeholder bullets and TODOs with a description of the component pipeline (with a diagram), the source layout, the template parameters, the value storage (now in struct data), the input and output adapters, and the SAX interface. Signed-off-by: Niels Lohmann * Link sources and basic_json, document full input adapter interface Signed-off-by: Niels Lohmann * Align the default column of the template parameter table Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/home/architecture.md | 212 +++++++++++++++++++++----- 1 file changed, 177 insertions(+), 35 deletions(-) diff --git a/docs/mkdocs/docs/home/architecture.md b/docs/mkdocs/docs/home/architecture.md index aba2be580..6c0892f11 100644 --- a/docs/mkdocs/docs/home/architecture.md +++ b/docs/mkdocs/docs/home/architecture.md @@ -1,34 +1,125 @@ # Architecture -!!! info - - This page is still under construction. Its goal is to provide a high-level overview of the library's architecture. - This should help new contributors to get an idea of the used concepts and where to make changes. +This page gives a high-level overview of the library's architecture. It should help new contributors to get an idea of +the used concepts and where to make changes. ## Overview -The main structure is class [nlohmann::basic_json](../api/basic_json/index.md). +The library is built around a single class template, [`nlohmann::basic_json`](../api/basic_json/index.md). A +`basic_json` value is a node in a tree of JSON values. All other components either create such a tree from an input +(parsing), write a tree to an output (serialization), or give access to it (iterators, JSON Pointer, conversions). -- public API -- container interface -- iterators +```mermaid +flowchart LR + input[/"input
(string, stream,
iterator range, file)"/] + ia["input adapter"] + lexer["lexer"] + parser["parser"] + breader["binary_reader"] + sax["SAX interface"] + value[("basic_json
value tree")] + serializer["serializer"] + bwriter["binary_writer"] + oa["output adapter"] + output[/"output
(string, stream,
vector)"/] -## Template specializations + input --> ia + ia --> lexer --> parser --> sax + ia --> breader --> sax + sax --> value + value --> serializer --> oa + value --> bwriter --> oa + oa --> output +``` -- describe template parameters of `basic_json` -- [`json`](../api/json.md) -- [`ordered_json`](../api/ordered_json.md) via [`ordered_map`](../api/ordered_map.md) +- **JSON text** is read by an [input adapter](#input-adapters), tokenized by the lexer, and turned into SAX events by + the parser. +- **Binary formats** (BJData, BSON, CBOR, MessagePack, UBJSON) are read by an input adapter and turned into the same SAX + events by the `binary_reader`. +- A [SAX consumer](#sax-interface) receives the events. The one used by [`parse`](../api/basic_json/parse.md) builds a + `basic_json` value tree. +- The `serializer` (JSON text) or the `binary_writer` (binary formats) writes a value tree to an + [output adapter](#output-adapters). + +## Source layout + +The public headers are in [`include/nlohmann`](https://github.com/nlohmann/json/tree/develop/include/nlohmann): + +- [`json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/json.hpp) defines class [`basic_json`](../api/basic_json/index.md). +- [`json_fwd.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/json_fwd.hpp) contains forward declarations. +- [`adl_serializer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/adl_serializer.hpp), [`byte_container_with_subtype.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/byte_container_with_subtype.hpp), and [`ordered_map.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/ordered_map.hpp) define + [`adl_serializer`](../api/adl_serializer/index.md), + [`byte_container_with_subtype`](../api/byte_container_with_subtype/index.md), and + [`ordered_map`](../api/ordered_map.md). + +Everything else lives in [`detail/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail) and namespace `nlohmann::detail`, which is not part of the public API. Paths +below are relative to `include/nlohmann`. + +| Component | Location | +|-----------|----------| +| Value type enumeration | [`detail/value_t.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/value_t.hpp) | +| Input adapters | [`detail/input/input_adapters.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/input_adapters.hpp) | +| Lexer | [`detail/input/lexer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/lexer.hpp), [`detail/input/number_parse.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/number_parse.hpp), [`detail/input/string_scan.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/string_scan.hpp) | +| Parser | [`detail/input/parser.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/parser.hpp) | +| SAX interface and DOM builders | [`detail/input/json_sax.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/json_sax.hpp) | +| Binary format readers | [`detail/input/binary_reader.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/binary_reader.hpp) | +| JSON serializer | [`detail/output/serializer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/serializer.hpp), [`detail/conversions/to_chars.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/to_chars.hpp) | +| Binary format writers | [`detail/output/binary_writer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/binary_writer.hpp) | +| Output adapters | [`detail/output/output_adapters.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/output_adapters.hpp) | +| Iterators | [`detail/iterators/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail/iterators) | +| Conversions from/to arbitrary types | [`detail/conversions/from_json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/from_json.hpp), [`detail/conversions/to_json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/to_json.hpp) | +| JSON Pointer | [`detail/json_pointer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/json_pointer.hpp) | +| Exceptions | [`detail/exceptions.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/exceptions.hpp) | +| Type traits and C++ feature backports | [`detail/meta/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail/meta) | +| Macros | [`detail/macro_scope.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/macro_scope.hpp), [`detail/macro_unscope.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/macro_unscope.hpp), [`detail/abi_macros.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/abi_macros.hpp) | + +The single-header version [`single_include/nlohmann/json.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json.hpp) +is generated from these files with `make amalgamate` and must not be edited by hand. + +## Template parameters + +[`basic_json`](../api/basic_json/index.md) is parameterized by the types it uses to store values and to convert from and to other types: + +| Template parameter | Default | Used for | +|----------------------|-----------------------------|-------------------------------------------------------------------| +| `ObjectType` | `std::map` | objects, see [`object_t`](../api/basic_json/object_t.md) | +| `ArrayType` | `std::vector` | arrays, see [`array_t`](../api/basic_json/array_t.md) | +| `StringType` | `std::string` | strings and object keys, see [`string_t`](../api/basic_json/string_t.md) | +| `BooleanType` | `bool` | Booleans, see [`boolean_t`](../api/basic_json/boolean_t.md) | +| `NumberIntegerType` | `std::int64_t` | signed integers, see [`number_integer_t`](../api/basic_json/number_integer_t.md) | +| `NumberUnsignedType` | `std::uint64_t` | unsigned integers, see [`number_unsigned_t`](../api/basic_json/number_unsigned_t.md) | +| `NumberFloatType` | `double` | floating-point numbers, see [`number_float_t`](../api/basic_json/number_float_t.md) | +| `AllocatorType` | `std::allocator` | allocating objects, arrays, strings, and binary values | +| `JSONSerializer` | `adl_serializer` | conversions from/to other types, see [`adl_serializer`](../api/adl_serializer/index.md) | +| `BinaryType` | `std::vector` | binary values, see [`binary_t`](../api/basic_json/binary_t.md) | +| `CustomBaseClass` | `void` | an optional base class, see [`json_base_class_t`](../api/basic_json/json_base_class_t.md) | + +The library provides two specializations: + +- [`json`](../api/json.md) uses all default template arguments. +- [`ordered_json`](../api/ordered_json.md) uses [`ordered_map`](../api/ordered_map.md) as `ObjectType` to keep the + insertion order of object keys. + +The requirements on the template arguments are listed in +[Template Parameter Requirements](../features/types/template_parameters.md). ## Value storage -Values are stored as a tagged union of [value_t](../api/basic_json/value_t.md) and json_value. +Each [`basic_json`](../api/basic_json/index.md) value stores its content as a tagged union: an enumeration [`value_t`](../api/basic_json/value_t.md) +names the type of the value, and a union `json_value` holds the value itself. Both are members of the nested struct +`data`, which is the only data member `m_data` of `basic_json`: ```cpp -/// the type of the current element -value_t m_type = value_t::null; +struct data +{ + /// the type of the current element + value_t m_type = value_t::null; -/// the value of the current element -json_value m_value = {}; + /// the value of the current element + json_value m_value = {}; +}; + +data m_data = {}; ``` with @@ -68,42 +159,83 @@ union json_value { }; ``` -## Parsing inputs (deserialization) +Objects, arrays, strings, and binary values are allocated on the heap with `AllocatorType`, and the union only stores a +pointer to them. This keeps a `basic_json` value small: one pointer-sized union and one byte for the type. The class +maintains the invariant that the pointer matching `m_type` is never null; `assert_invariant()` checks it with +[runtime assertions](../features/assertions.md). -Input is read via **input adapters** that abstract a source with a common interface: +## Input adapters + +Input is read via **input adapters** that abstract a source. Every input adapter provides this interface: ```cpp -/// read a single character -std::char_traits::int_type get_character() noexcept; +/// the type of the characters in the input +using char_type = ...; -/// read multiple characters to a destination buffer and -/// returns the number of characters successfully read +/// read a single character; returns std::char_traits::eof() at the end of the input +typename std::char_traits::int_type get_character(); + +/// read up to count * sizeof(T) bytes into dest and return the number of bytes read +/// (used by the binary readers) template std::size_t get_elements(T* dest, std::size_t count = 1); ``` -List examples of input adapters. +The lexer detects two optional extensions at compile time. Only `iterator_input_adapter` provides them, and only for +random-access input of single-byte characters: -## SAX Interface +- `supports_seek`, `get_consumed_count()`, and `copy_consumed_range()` let the lexer reconstruct already consumed input + for error messages instead of copying every character it reads. +- `supports_bulk_scan`, `bulk_data()`, `bulk_remaining()`, and `bulk_skip()` let the lexer scan strings directly in + contiguous memory, several bytes at a time. -TODO +The function `input_adapter` picks the right adapter for the argument passed to `parse`, `accept`, `sax_parse`, or the +`from_*` functions: -## Writing outputs (serialization) +- `iterator_input_adapter` reads from an iterator range, which also covers strings, containers, and pointers. +- `wide_string_input_adapter` reads from ranges of `wchar_t`, `char16_t`, or `char32_t` and converts them to UTF-8. + It cannot be used for binary formats; its `get_elements()` throws. +- `input_stream_adapter` reads from a `std::istream`. +- `file_input_adapter` reads from a `std::FILE*`. + +## SAX interface + +The parser does not build values itself. It reports what it reads as events to a [SAX](../features/parsing/sax_interface.md) +consumer, which implements the interface [`json_sax`](../api/json_sax/index.md): `null`, `boolean`, `number_integer`, +`number_unsigned`, `number_float`, `string`, `binary`, `start_object`, `key`, `end_object`, `start_array`, `end_array`, +and `parse_error`. + +The library comes with two consumers in `detail/input/json_sax.hpp`: + +- `json_sax_dom_parser` builds a [`basic_json`](../api/basic_json/index.md) value tree. [`parse`](../api/basic_json/parse.md) uses it. +- `json_sax_dom_callback_parser` does the same, but calls a [parser callback](../features/parsing/parser_callbacks.md) + for each event, which can skip values. `parse` uses it when a callback is given. + +The `binary_reader` emits the same events for binary formats, so [`sax_parse`](../api/basic_json/sax_parse.md) works +with a user-defined consumer for JSON and for all binary formats alike. + +## Output adapters Output is written via **output adapters**: ```cpp -template void write_character(CharType c); -template void write_characters(const CharType* s, std::size_t length); ``` -List examples of output adapters. +The `serializer` (used by [`dump`](../api/basic_json/dump.md) and [`operator<<`](../api/operator_ltlt.md)) and the +`binary_writer` (used by the `to_*` functions) write to one of these adapters: + +- `output_vector_adapter` appends to a `std::vector`. +- `output_stream_adapter` writes to a `std::ostream`. +- `output_string_adapter` appends to a string. ## Value conversion +Values are converted from and to other types with the `JSONSerializer` template parameter. The default, +[`adl_serializer`](../api/adl_serializer/index.md), calls the free functions + ```cpp template void to_json(basic_json& j, const T& t); @@ -112,13 +244,23 @@ template void from_json(const basic_json& j, T& t); ``` +found by argument-dependent lookup. The library defines them for standard types in `detail/conversions`; users add them +for their own types, see [Arbitrary Type Conversions](../features/arbitrary_types.md). The +[serialization macros](../features/macros.md) generate these functions. + ## Additional features -- JSON Pointers -- Binary formats -- Custom base class -- Conversion macros +- [JSON Pointer](../features/json_pointer.md) (class `json_pointer`) addresses values inside a tree. It is also the + basis of [JSON Patch](../features/json_patch.md). +- [Binary formats](../features/binary_formats/index.md) are read by `binary_reader` and written by `binary_writer`. +- A [custom base class](../api/basic_json/json_base_class_t.md) can add members to every [`basic_json`](../api/basic_json/index.md) value. +- [Serialization macros](../features/macros.md) generate `to_json` and `from_json` functions for user-defined types. ## Details namespace -- C++ feature backports +Namespace `nlohmann::detail` contains all implementation details. It is not part of the public API and may change in any +release. Besides the components above, it contains: + +- type traits to detect the capabilities of user-defined types (`detail/meta/type_traits.hpp`), +- backports of C++14/17 features to C++11 (`detail/meta/cpp_future.hpp`), and +- helpers such as `string_concat` and `string_escape`. From aada27405dca0771eca5f8eaa1d087c4b13c0ff0 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:28:16 +0200 Subject: [PATCH 02/10] Add a roadmap page to the documentation (#5571) Describe what the project will and will not do over the next year, and point to issue #3453 for the open question of a 4.0 release. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/community/index.md | 1 + docs/mkdocs/docs/community/roadmap.md | 43 +++++++++++++++++++++++++++ docs/mkdocs/mkdocs.yml | 1 + 3 files changed, 45 insertions(+) create mode 100644 docs/mkdocs/docs/community/roadmap.md diff --git a/docs/mkdocs/docs/community/index.md b/docs/mkdocs/docs/community/index.md index 50baeab25..c7d0cf079 100644 --- a/docs/mkdocs/docs/community/index.md +++ b/docs/mkdocs/docs/community/index.md @@ -5,4 +5,5 @@ - [Contribution Guidelines](contribution_guidelines.md) - guidelines how to contribute to this project - [Governance](governance.md) - the governance model of this project - [Quality Assurance](quality_assurance.md) - how the quality of this project is assured +- [Roadmap](roadmap.md) - what the project will and will not do - [Security Policy](security_policy.md) - the security policy of the project diff --git a/docs/mkdocs/docs/community/roadmap.md b/docs/mkdocs/docs/community/roadmap.md new file mode 100644 index 000000000..e8c407d3f --- /dev/null +++ b/docs/mkdocs/docs/community/roadmap.md @@ -0,0 +1,43 @@ +# Roadmap + +This page describes what the project intends to do, and what it does not intend to do, over the next year. Concrete +work items are tracked in the [GitHub milestones](https://github.com/nlohmann/json/milestones) and the +[issue tracker](https://github.com/nlohmann/json/issues). + +## What the project will do + +- **Keep the C++11 baseline.** The library will continue to compile with every + [supported C++11 compiler](https://github.com/nlohmann/json/blob/develop/README.md#supported-compilers). Features of + later standards are only used when they are guarded by the `JSON_HAS_CPP_*` macros. +- **Stay conformant to JSON.** The parser and serializer follow [RFC 8259](https://datatracker.ietf.org/doc/html/rfc8259). + Extensions such as [comments](../features/comments.md) or [trailing commas](../features/trailing_commas.md) remain + opt-in. +- **Keep the 3.x public API stable.** Releases follow [semantic versioning](https://semver.org). Changes that would + break existing code are only added behind a feature macro, so users can opt in and test their code before a next + major release. +- **Support a broad range of compilers and platforms.** The [CI](quality_assurance.md) keeps testing old and new + versions of GCC, Clang, MSVC, and other compilers on Linux, macOS, and Windows. +- **Keep the quality assurance up.** Every change keeps the test coverage at 100%, passes the static and dynamic + analysis, and is fuzz-tested by OSS-Fuzz, see [Quality assurance](quality_assurance.md). +- **Harden the library against hostile input.** Handling deeply nested values without exhausting the call stack is + ongoing work. +- **Fix bugs and security issues** reported through the issue tracker and the [security policy](security_policy.md). + +## What the project will not do + +- **Break the public API of version 3.x.** See the + [contribution guidelines](https://github.com/nlohmann/json/blob/develop/.github/CONTRIBUTING.md#break-the-public-api) + for what counts as a breaking change. +- **Require a newer C++ standard than C++11.** +- **Break JSON conformance** or enable non-standard extensions by default. +- **Add dependencies** or require a build step. The library remains header-only, and the single header + `json.hpp` remains a complete distribution. +- **Trade simplicity for speed or memory efficiency.** Performance improvements are welcome, but the library is not + meant to compete with the fastest JSON libraries, see [Design goals](../home/design_goals.md). + +## Version 4.0 + +There is no decision yet on whether or when a version 4.0 with breaking changes will be released. Proposals that need +a major version, for instance stricter type conversions, are collected in issue +[#3453](https://github.com/nlohmann/json/issues/3453). Until then, such changes are only added as opt-in behavior +behind feature macros. diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 8f92a838f..9400222b1 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -317,6 +317,7 @@ nav: - community/contribution_guidelines.md - community/quality_assurance.md - community/governance.md + - community/roadmap.md - community/security_policy.md # Extras From 80bf54a5a2294f35b3858b602af0d9b795889b0a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:28:34 +0200 Subject: [PATCH 03/10] Add a security assurance case to the documentation (#5572) Describe the threat model, the trust boundaries, the secure-design argument, and how common weaknesses are countered, with links to the quality assurance page as evidence. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/community/assurance_case.md | 70 ++++++++++++++++++++ docs/mkdocs/docs/community/index.md | 1 + docs/mkdocs/mkdocs.yml | 1 + 3 files changed, 72 insertions(+) create mode 100644 docs/mkdocs/docs/community/assurance_case.md diff --git a/docs/mkdocs/docs/community/assurance_case.md b/docs/mkdocs/docs/community/assurance_case.md new file mode 100644 index 000000000..87d8d6f14 --- /dev/null +++ b/docs/mkdocs/docs/community/assurance_case.md @@ -0,0 +1,70 @@ +# Assurance case + +This page argues why the library meets its security requirements. It describes the threats the library faces, where the +trust boundaries lie, and how the library's design and the [quality assurance](quality_assurance.md) counter these +threats. To report a vulnerability, see the [security policy](security_policy.md). + +## Threat model + +The library parses, stores, and serializes JSON values in memory. It does not open network connections, does not open +files (it only reads from streams or `std::FILE*` handles that the caller has already opened), does not read environment +variables, and does not implement cryptography or handle credentials. + +The primary threat is therefore **untrusted input**: JSON text or binary data (BJData, BSON, CBOR, MessagePack, UBJSON) +that an attacker controls, passed to [`parse`](../api/basic_json/parse.md), [`accept`](../api/basic_json/accept.md), +[`sax_parse`](../api/basic_json/sax_parse.md), or one of the `from_*` functions such as +[`from_cbor`](../api/basic_json/from_cbor.md). Such input may try to + +- make the library read or write out of bounds (malformed lengths, truncated input, invalid UTF-8), +- trigger undefined behavior (integer overflow in sizes or numbers, invalid casts), +- exhaust memory (huge announced sizes), or +- exhaust the call stack (deeply nested arrays and objects). + +## Trust boundaries + +- **Untrusted:** all serialized input read by the parser, the SAX interface, and the binary readers. The library must + handle every possible input by either producing a value or throwing a [`parse_error`](../home/exceptions.md#parse-errors) + (or returning `false` when exceptions are disabled for the call). +- **Trusted:** the C++ code that calls the library. Calling a function with violated preconditions, for instance + accessing an array with [`operator[]`](../api/basic_json/operator%5B%5D.md) out of range, is a programming error and + not a security boundary. Such preconditions are checked with [runtime assertions](../features/assertions.md) in debug + builds; functions such as [`at`](../api/basic_json/at.md) offer checked access with exceptions. + +## Secure design + +- **Strict parsing.** The parser accepts exactly the JSON grammar of [RFC 8259](https://datatracker.ietf.org/doc/html/rfc8259). + Extensions such as [comments](../features/comments.md) and [trailing commas](../features/trailing_commas.md) must be + enabled explicitly. Invalid UTF-8 is rejected. +- **Errors are reported, not ignored.** Malformed input results in a [`parse_error`](../home/exceptions.md#parse-errors) + with the byte position of the error. Binary readers do not trust announced sizes: strings and binary values grow + only as bytes are actually read, arrays reserve at most a fixed number of elements up front, and sizes that no + container can hold are rejected. +- **Memory is owned by values.** Each `basic_json` value owns its content, and there is no manual memory management in + user code. The destructor does not recurse, so destroying a deeply nested value does not exhaust the stack. +- **Bounded recursion.** The JSON parser and the binary readers keep their state in explicit stacks instead of + recursing per nesting level. Operations that walk a value, such as [`dump`](../api/basic_json/dump.md), copying, + hashing, and [`merge_patch`](../api/basic_json/merge_patch.md), recurse only up to a fixed depth and continue with an + explicit stack below it. Some operations, such as comparison, [`diff`](../api/basic_json/diff.md), + [`flatten`](../api/basic_json/flatten.md), and the binary writers, still recurse once per nesting level; work on them + is in progress. Applications that process untrusted input can limit its nesting depth with a + [parser callback](../features/parsing/parser_callbacks.md). +- **Invariants are checked.** The class invariant (for instance, that the pointer for the stored type is never null) is + checked with runtime assertions throughout the test suite. + +## Common weaknesses + +The following table maps the relevant classes of the [Common Weakness Enumeration](https://cwe.mitre.org) to the +measures that counter them. The measures are described in detail in [Quality assurance](quality_assurance.md). + +| Weakness | Countermeasures | +|---------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------| +| Out-of-bounds read/write ([CWE-125](https://cwe.mitre.org/data/definitions/125.html), [CWE-787](https://cwe.mitre.org/data/definitions/787.html)) | bounds checks on all reads from the input; AddressSanitizer and Valgrind on the test suite; OSS-Fuzz | +| Integer overflow ([CWE-190](https://cwe.mitre.org/data/definitions/190.html)) | UndefinedBehaviorSanitizer with integer overflow detection; Clang-Tidy; Cppcheck | +| Use after free, double free ([CWE-416](https://cwe.mitre.org/data/definitions/416.html), [CWE-415](https://cwe.mitre.org/data/definitions/415.html)) | ownership of all memory by values; AddressSanitizer and Valgrind; Clang Static Analyzer | +| Memory leaks ([CWE-401](https://cwe.mitre.org/data/definitions/401.html)) | Valgrind (Memcheck) on the test suite | +| Uncontrolled recursion ([CWE-674](https://cwe.mitre.org/data/definitions/674.html)) | iterative parser, binary readers, and destructor; bounded recursion in value operations; tests with deeply nested inputs | +| Uncontrolled resource consumption ([CWE-400](https://cwe.mitre.org/data/definitions/400.html)) | allocations based on announced sizes are capped; OSS-Fuzz with memory limits | +| Undefined behavior in general ([CWE-758](https://cwe.mitre.org/data/definitions/758.html)) | UndefinedBehaviorSanitizer; runtime assertions; Clang-Tidy, Cppcheck, Clang Static Analyzer, Infer | + +In addition, every line of the library is covered by the unit tests, and all parsers are fuzz-tested around the clock +by [OSS-Fuzz](https://github.com/google/oss-fuzz/tree/master/projects/json). diff --git a/docs/mkdocs/docs/community/index.md b/docs/mkdocs/docs/community/index.md index c7d0cf079..7b7f5c07a 100644 --- a/docs/mkdocs/docs/community/index.md +++ b/docs/mkdocs/docs/community/index.md @@ -7,3 +7,4 @@ - [Quality Assurance](quality_assurance.md) - how the quality of this project is assured - [Roadmap](roadmap.md) - what the project will and will not do - [Security Policy](security_policy.md) - the security policy of the project +- [Assurance Case](assurance_case.md) - why the library meets its security requirements diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 9400222b1..ffb0fae80 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -319,6 +319,7 @@ nav: - community/governance.md - community/roadmap.md - community/security_policy.md + - community/assurance_case.md # Extras extra: From cc472af13f8a09d4caccc9df755435d06efaa3a9 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:29:02 +0200 Subject: [PATCH 04/10] Check the fuzzers' UBJSON/BJData round-trip invariants in the unit tests (#5569) * Check the fuzzers' UBJSON/BJData round-trip invariants in the unit tests The strongest correctness checks for the UBJSON and BJData writers lived only in the OSS-Fuzz drivers: anything from_ubjson()/from_bjdata() returns must serialize with every option combination, parse back, and re-serialize stably. Those checks only run at OSS-Fuzz, so regressions surfaced days later as external reports - the same BJData assert pair was reported five times over three years, and #5494's harness change was followed by OSS-Fuzz 563659413 within a day. Add "UBJSON round-trip invariants" and "BJData round-trip invariants" test cases that run the drivers' checks on a fixed, deterministic corpus (tests/src/round_trip_corpus.hpp): integer and float boundaries, non-finite numbers, strings, binary values, optimized containers, deep nesting, the JData annotated-array matrix, and seeded random containers. They also check two properties the drivers do not: the first round trip preserves the value, and re-serializing reproduces the exact bytes. For BJData both exclude values containing a binary value, which is read back as an array of integers unless it was written as a Draft 3 optimized binary array; this carve-out is now documented in bjdata.md. Run against the headers before #5542, the BJData test fails, including on the shape from OSS-Fuzz 563659413. Also document how OSS-Fuzz reports are handled (reference them as "OSS-Fuzz: ", turn the reproducer into a unit test, keep drivers and unit tests in sync) in tests/fuzzing.md, and link it from the PR template and the quality assurance page. Signed-off-by: Niels Lohmann * Add the OSS-Fuzz reproducers for 474400817 and 474480402 as unit tests Following the convention added to tests/fuzzing.md, the reproducers of the two BJData fuzzer asserts tracked since January are now unit tests: - 474400817 (assert(false)): an empty object _ArraySize_ was written as the ND-array header length, which from_bjdata() could not read back. Fixed by #5455. - 474480402 (to_bjdata(j2, false, false) == vec2): a one-byte Draft 3 binary array is written in Draft 2 mode as a uint8 array and then re-serialized with the int8 marker. This is the documented exception to byte stability, not a library bug; OSS-Fuzz closed it after #5494 relaxed the harness to value stability. The test pins the exact bytes so the exception stays deliberate. The 563659413 reproducer is already a unit test (#5542). A comment also ties the existing UBJSON excessive-count test to the timeout OSS-Fuzz reported for that shape (testcase 6347769435193344). OSS-Fuzz: 474400817 OSS-Fuzz: 474480402 Signed-off-by: Niels Lohmann * Fix GCC -Weffc++ and -Wuseless-cast warnings in the round-trip corpus Initialize the atoms in the member initialization list, and drop the cast of the generator's result, which already is std::size_t on 64-bit Linux. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/PULL_REQUEST_TEMPLATE.md | 1 + .../docs/community/quality_assurance.md | 3 + .../docs/features/binary_formats/bjdata.md | 10 + tests/fuzzing.md | 23 ++ tests/src/fuzzer-parse_bjdata.cpp | 3 + tests/src/fuzzer-parse_ubjson.cpp | 3 + tests/src/round_trip_corpus.hpp | 213 ++++++++++++++++++ tests/src/unit-bjdata.cpp | 103 +++++++++ tests/src/unit-ubjson.cpp | 50 +++- 9 files changed, 408 insertions(+), 1 deletion(-) create mode 100644 tests/src/round_trip_corpus.hpp diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md index 537095324..16d808485 100644 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -2,6 +2,7 @@ - [ ] The changes are described in detail, both the what and why. - [ ] If applicable, an [existing issue](https://github.com/nlohmann/json/issues) is referenced. +- [ ] If applicable, a fixed [OSS-Fuzz](https://issues.oss-fuzz.com) issue is referenced as `OSS-Fuzz: ` (see [fuzz testing](https://github.com/nlohmann/json/blob/develop/tests/fuzzing.md#handling-oss-fuzz-reports)). - [ ] The [Code coverage](https://coveralls.io/github/nlohmann/json) remained at 100%. A test case for every new line of code. - [ ] If applicable, the [documentation](https://json.nlohmann.me) is updated. - [ ] The source code is amalgamated by running `make amalgamate`. diff --git a/docs/mkdocs/docs/community/quality_assurance.md b/docs/mkdocs/docs/community/quality_assurance.md index 4196f3532..bd35516b8 100644 --- a/docs/mkdocs/docs/community/quality_assurance.md +++ b/docs/mkdocs/docs/community/quality_assurance.md @@ -164,6 +164,9 @@ Note: Some modern features (like C++20 ranges or filesystem support) may be disa - [x] The parser is tested against extensive correctness suites for JSON compliance. - [x] In addition, the library is continuously fuzz-tested at [OSS-Fuzz](https://google.github.io/oss-fuzz/) where the library is checked against billions of inputs. +- [x] Every crash reported by OSS-Fuzz is fixed together with a unit test that reproduces it, and the fix references + the OSS-Fuzz issue. The round-trip checks of the fuzzer drivers are also part of the unit tests. See the + [fuzz testing documentation](https://github.com/nlohmann/json/blob/develop/tests/fuzzing.md#handling-oss-fuzz-reports). ## Static analysis diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index 4cb61053f..a0c84edaf 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -208,6 +208,16 @@ The library maps BJData types to JSON value types as follows: The mapping is **complete** in the sense that any BJData value can be converted to a JSON value. +!!! info "Round trips" + + A value returned by [`from_bjdata`](../../api/basic_json/from_bjdata.md) can be serialized with + [`to_bjdata`](../../api/basic_json/to_bjdata.md) using any combination of options and parsed back into an equal + value, and serializing that value again with the same options produces the same bytes. The exception is binary + values: they are only written as an optimized binary array (`[$B`) if Draft 3 is enabled and both `use_size` and + `use_type` are set. Otherwise, they are written as arrays of integers and parsed back as such (see the notes on + binary values above), and serializing such an array again may choose different, but equally valid, type markers. + The bytes can then differ, but parsing them again yields the same value. + ??? example ```cpp diff --git a/tests/fuzzing.md b/tests/fuzzing.md index cfbf4f249..b3bf90d5d 100644 --- a/tests/fuzzing.md +++ b/tests/fuzzing.md @@ -79,3 +79,26 @@ the same `fuzzers` target as above and also relies on the `FUZZER_ENGINE` variab [build script](https://github.com/google/oss-fuzz/blob/master/projects/json/build.sh) for more information. In case the build at OSS-Fuzz fails, an issue will be created automatically. + +### Handling OSS-Fuzz reports + +OSS-Fuzz files the crashes it finds in its own [issue tracker](https://issues.oss-fuzz.com), not on GitHub. So that +each report can be traced to the change that fixed it, and each fix to the report it answers, fixes follow these +conventions: + +- **Reference the OSS-Fuzz issue in the pull request**, next to any GitHub issue it closes, as `OSS-Fuzz: ` (for + example, `OSS-Fuzz: 563659413`), and in the commit message. The ID alone does not disclose the crash. If the report + was triaged into a GitHub issue, link the OSS-Fuzz issue there too. +- **Turn the reproducer into a unit test.** Download the testcase from the OSS-Fuzz report, reduce it if possible, and + add it as a regression test to the unit test of the affected format (e.g., `tests/src/unit-bjdata.cpp`), with a + comment naming the OSS-Fuzz issue. This way the input is checked by every CI run rather than only by OSS-Fuzz, and + it stays covered even if OSS-Fuzz later closes the report as not reproducible. +- **Keep the fuzzer drivers and the unit tests in sync.** The round-trip checks of the UBJSON and BJData drivers are + also run on a fixed corpus in the unit tests (see `tests/src/round_trip_corpus.hpp` and the "round-trip invariants" + test cases), so a regression shows up in CI first. When a driver's checks change, change the unit tests with them. +- **Record in the report whether the bug shipped.** OSS-Fuzz asks whether a crash was a short-lived regression or + affects a released version; answer it when the fix is merged, as it decides whether the fix needs a release note or + a security advisory (see the [security policy](../.github/SECURITY.md)). + +After the fix is merged, OSS-Fuzz re-runs the reproducer on its next build and marks the report as verified and +closed. If it does not, the fix is incomplete. diff --git a/tests/src/fuzzer-parse_bjdata.cpp b/tests/src/fuzzer-parse_bjdata.cpp index 41c51a311..d3c9e7a33 100644 --- a/tests/src/fuzzer-parse_bjdata.cpp +++ b/tests/src/fuzzer-parse_bjdata.cpp @@ -42,6 +42,9 @@ dump() serializes any non-finite double the same deterministic way (as JSON `null`, since JSON itself cannot represent NaN/Infinity), so comparing dumps is stable under exactly the same values that break operator==. +The unit tests run the same checks on a fixed corpus (see the "BJData round-trip +invariants" test case), so keep both in sync. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ diff --git a/tests/src/fuzzer-parse_ubjson.cpp b/tests/src/fuzzer-parse_ubjson.cpp index 20c20eda7..ebf775b59 100644 --- a/tests/src/fuzzer-parse_ubjson.cpp +++ b/tests/src/fuzzer-parse_ubjson.cpp @@ -21,6 +21,9 @@ array data, it performs the following steps: - j4 = from_ubjson(vec3) - assert(j1 == j4) +The unit tests run the same checks on a fixed corpus (see the "UBJSON round-trip +invariants" test case), so keep both in sync. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ diff --git a/tests/src/round_trip_corpus.hpp b/tests/src/round_trip_corpus.hpp new file mode 100644 index 000000000..41cdac3e6 --- /dev/null +++ b/tests/src/round_trip_corpus.hpp @@ -0,0 +1,213 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // nan +#include // size_t +#include // int32_t, int64_t, uint32_t, uint64_t +#include // numeric_limits +#include // mt19937 +#include // string, to_string +#include // move +#include // vector + +#include + +// Values for the round-trip property tests of the UBJSON and BJData writers. +// +// The fuzzer drivers (tests/src/fuzzer-parse_ubjson.cpp and +// fuzzer-parse_bjdata.cpp) check that anything the library parses can be +// serialized, parsed back, and serialized again without loss. Those checks +// only run at OSS-Fuzz, so a regression used to surface days later as an +// external report. The unit tests run the same checks on this corpus in CI. +// +// The corpus is deterministic: std::mt19937's output sequence is fixed by +// the standard, and it is used directly rather than through a distribution +// (whose results are implementation-defined). +namespace utils +{ + +class round_trip_corpus +{ + public: + using json = nlohmann::json; + + static std::vector values() + { + round_trip_corpus corpus; + return corpus.build(); + } + + // whether a value contains a binary value, which a BJData or UBJSON round + // trip may turn into an array of integers + static bool contains_binary(const json& j) + { + if (j.is_binary()) + { + return true; + } + if (j.is_structured()) + { + for (const auto& element : j) + { + if (contains_binary(element)) + { + return true; + } + } + } + return false; + } + + private: + std::vector atoms; + // a fixed seed is the point: the corpus must be the same in every run + std::mt19937 generator{42}; // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) + + round_trip_corpus() + : atoms + { + nullptr, true, false, + // integers at the boundaries of every UBJSON/BJData integer type + 0, 1, -1, 127, 128, 255, 256, -128, -129, + 32767, 32768, 65535, 65536, -32768, -32769, + (std::numeric_limits::min)(), (std::numeric_limits::max)(), + (std::numeric_limits::max)(), + (std::numeric_limits::min)(), (std::numeric_limits::max)(), + static_cast((std::numeric_limits::max)()) + 1u, + (std::numeric_limits::max)(), + // floating-point numbers, including non-finite ones + 0.0, -0.0, 1.5, -2.25, 3.4e38, (std::numeric_limits::max)(), + std::nan(""), std::numeric_limits::infinity(), -std::numeric_limits::infinity(), + // strings, including a non-ASCII one and one longer than 255 bytes + "", "a", "\xC3\xA4", std::string(300, 'x'), + // binary values with and without subtype + json::binary({}), json::binary({1, 2, 255}), json::binary({0x80, 0x7F}, 42), json::binary({1}, 0) + } + {} + + std::vector build() + { + std::vector result = atoms; + + // each atom inside containers, including homogeneous ones that the + // writers encode as optimized (typed) containers + result.emplace_back(json::array()); + result.emplace_back(json::object()); + for (const auto& atom : atoms) + { + result.push_back(json::array({atom})); + result.push_back(json::array({atom, atom, atom})); + result.push_back(json::array({json::array({atom})})); + result.push_back(json::object({{"key", atom}})); + } + result.push_back(json::array({1, 1.5})); + result.push_back(json::array({-1, 255})); + result.push_back(json::array({"a", "b"})); + + // deep, but well below any recursion or depth limit + json nested_array = 1; + json nested_object = 1; + for (int i = 0; i < 300; ++i) + { + nested_array = json::array({nested_array}); + nested_object = json::object({{"key", nested_object}}); + } + result.push_back(nested_array); + result.push_back(nested_object); + + add_annotated_arrays(result); + add_random_values(result); + return result; + } + + // objects in the JData annotated array format, which the BJData writer + // encodes as ND-arrays when the annotation describes a packed array, and + // as plain objects otherwise (see #5398, #5399, #5403, #5404, and #5542) + static void add_annotated_arrays(std::vector& result) + { + const std::vector types = + { + "uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", + "single", "double", "char", "byte", "bool", "unknown", 5, nullptr + }; + const std::vector sizes = + { + json::array(), {3}, {1, 3}, {3, 1}, {2, 3}, {2, 0}, {0, 2}, {2, 2, 2}, {-1, 2}, {2, 1.5}, + "3", 3, nullptr, json::binary({}) + }; + const std::vector data = + { + nullptr, 5, "s", json::object({{"a", 1}}), json::array(), + {1, 2, 3}, {1, 2, 3, 4, 5, 6}, {1, 2, 3, 4, 5, 6, 7, 8}, + {1.5, 2.5, 3.5, 4.5, 5.5, 6.5}, {300, -300, 70000, -70000, 1, 2}, + {"a", "b", "c", "d", "e", "f"}, {json::array({1, 2, 3}), json::array({4, 5, 6})} + }; + + for (const auto& type : types) + { + for (const auto& size : sizes) + { + for (const auto& d : data) + { + result.push_back({{"_ArrayType_", type}, {"_ArraySize_", size}, {"_ArrayData_", d}}); + } + } + } + + // incomplete annotations and annotations with an extra key + result.push_back({{"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}, {"extra", 1}}); + } + + // random containers of atoms, both homogeneous and mixed + void add_random_values(std::vector& result) + { + for (int i = 0; i < 1000; ++i) + { + result.push_back(random_value(0)); + } + } + + std::size_t random_below(std::size_t bound) + { + return generator() % bound; + } + + json random_value(int depth) + { + const auto kind = random_below(10); + if (depth > 3 || kind < 5) + { + return atoms[random_below(atoms.size())]; + } + + json result = kind < 8 ? json::array() : json::object(); + const auto count = random_below(5); + const bool homogeneous = random_below(2) == 0; + const json fixed = atoms[random_below(atoms.size())]; + for (std::size_t i = 0; i < count; ++i) + { + json element = homogeneous ? fixed : random_value(depth + 1); + if (result.is_array()) + { + result.push_back(std::move(element)); + } + else + { + result[std::to_string(i)] = std::move(element); + } + } + return result; + } +}; + +} // namespace utils diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 9bba141d2..ebbbbfaf6 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -19,6 +19,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "round_trip_corpus.hpp" #include "test_utils.hpp" namespace @@ -2867,6 +2868,21 @@ TEST_CASE("BJData") const auto out_num = json::to_bjdata(j_num); CHECK(out_num.at(0) == '{'); CHECK(json::from_bjdata(out_num) == j_num); + + // OSS-Fuzz issue 474400817: an empty object _ArraySize_ was + // written as the ND-array header length, which from_bjdata() + // could not read back + const std::vector input = + { + '[', '{', 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'D', 'a', 't', 'a', '_', 'Z', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'T', 'y', 'p', 'e', '_', 'S', 'i', 5, 'i', 'n', 't', '1', '6', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'S', 'i', 'z', 'e', '_', '{', '}', '}', ']' + }; + const json j1 = json::from_bjdata(input); + CHECK(j1 == json::parse(R"([{"_ArrayType_":"int16","_ArraySize_":{},"_ArrayData_":null}])")); + json j2; + CHECK_NOTHROW(j2 = json::from_bjdata(json::to_bjdata(j1, false, false))); + CHECK(j2 == j1); } SECTION("ndarray with out-of-range _ArrayData_ elements stays as object") @@ -4273,6 +4289,93 @@ TEST_CASE("BJData use_type requires use_size") } } +TEST_CASE("BJData round-trip invariants") +{ + // This checks what the parse_bjdata_fuzzer driver checks (see + // tests/src/fuzzer-parse_bjdata.cpp), so that a regression shows up in CI + // rather than as an OSS-Fuzz report: every value from_bjdata() returns + // (j1) can be serialized with any combination of options, the result can + // be parsed back (j2), and serializing j2 again with the same options + // yields a value-equal result. + // + // Beyond the driver, this also checks that j2 equals j1 and that + // serializing j2 reproduces the exact bytes, both except for values that + // contain a binary value: a binary value is only written as a binary + // value with Draft 3's optimized binary array, and otherwise read back as + // an array of integers, for which the writer may choose different (but + // equally valid) type markers when it is serialized again (see #5494). + // + // Values are compared with dump() rather than operator==, because a NaN + // never compares equal to itself. + struct options + { + bool use_size; + bool use_type; + json::bjdata_version_t version; + }; + const std::vector all_options = + { + {false, false, json::bjdata_version_t::draft2}, + {true, false, json::bjdata_version_t::draft2}, + {true, true, json::bjdata_version_t::draft2}, + {false, false, json::bjdata_version_t::draft3}, + {true, false, json::bjdata_version_t::draft3}, + {true, true, json::bjdata_version_t::draft3}, + }; + + for (const auto& j0 : utils::round_trip_corpus::values()) + { + // turn the corpus value into a value as from_bjdata() returns it + for (const auto& initial : all_options) + { + const json j1 = json::from_bjdata(json::to_bjdata(j0, initial.use_size, initial.use_type, initial.version)); + const bool has_binary = utils::round_trip_corpus::contains_binary(j1); + + for (const auto& o : all_options) + { + INFO("j1 = " << j1.dump() << ", use_size = " << o.use_size << ", use_type = " << o.use_type + << ", draft3 = " << (o.version == json::bjdata_version_t::draft3)); + + const std::vector vec = json::to_bjdata(j1, o.use_size, o.use_type, o.version); + json j2; + // anything the library writes must be parsable by the library + REQUIRE_NOTHROW(j2 = json::from_bjdata(vec)); + const std::vector vec2 = json::to_bjdata(j2, o.use_size, o.use_type, o.version); + CHECK(json::from_bjdata(vec2).dump() == j2.dump()); + + if (!has_binary) + { + CHECK(j2.dump() == j1.dump()); + CHECK(vec2 == vec); + } + } + } + } +} + +TEST_CASE("BJData round trip of a binary value is value-stable, not byte-stable") +{ + // OSS-Fuzz issue 474480402: a Draft 3 optimized binary array is read as a + // binary value, which to_bjdata() writes in the default Draft 2 mode as a + // plain array of uint8 numbers. That is read back as an array of numbers, + // for which the writer then picks the smallest type marker, int8 ('i'), + // so re-serializing changes the bytes, but not the value. This is the + // exception described in the "Round trips" note of the BJData + // documentation, and why the fuzzer checks value stability (see #5494). + const std::vector input = {'[', '$', 'B', '#', 'U', 1, 0x20}; + const json j1 = json::from_bjdata(input); + CHECK(j1 == json::binary({0x20})); + + const std::vector vec = json::to_bjdata(j1, false, false); + CHECK(vec == std::vector({'[', 'U', 0x20, ']'})); + const json j2 = json::from_bjdata(vec); + CHECK(j2 == json::array({0x20})); + + const std::vector vec2 = json::to_bjdata(j2, false, false); + CHECK(vec2 == std::vector({'[', 'i', 0x20, ']'})); + CHECK(json::from_bjdata(vec2) == j2); +} + TEST_CASE("BJData roundtrips" * doctest::skip()) { SECTION("input from self-generated BJData files") diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index aafbbf5a4..c8458c44d 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -15,6 +15,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "round_trip_corpus.hpp" #include "test_utils.hpp" namespace @@ -2265,7 +2266,9 @@ TEST_CASE("UBJSON optimized arrays of a valueless type are bounded") SECTION("an excessive count is rejected") { - // 'l' is a big-endian int32: 0x7FFFFFFF elements, about 34 GB of value + // 'l' is a big-endian int32: 0x7FFFFFFF elements, about 34 GB of value; + // OSS-Fuzz reported this shape as a parse_ubjson_fuzzer timeout + // (testcase 6347769435193344, no issue filed) for (const auto marker : {'Z', 'T', 'F' }) @@ -2817,6 +2820,51 @@ TEST_CASE("UBJSON use_type requires use_size") } } +TEST_CASE("UBJSON round-trip invariants") +{ + // This checks what the parse_ubjson_fuzzer driver checks (see + // tests/src/fuzzer-parse_ubjson.cpp), so that a regression shows up in CI + // rather than as an OSS-Fuzz report: every value from_ubjson() returns + // (j1) can be serialized with any combination of options, the result can + // be parsed back (j2), and serializing j2 again with the same options + // reproduces the exact bytes. Beyond the driver, this also checks that j2 + // equals j1. Values are compared with dump() rather than operator==, + // because a NaN never compares equal to itself. + struct options + { + bool use_size; + bool use_type; + }; + const std::vector all_options = + { + {false, false}, + {true, false}, + {true, true}, + }; + + for (const auto& j0 : utils::round_trip_corpus::values()) + { + // turn the corpus value into a value as from_ubjson() returns it; this + // has no binary values, as UBJSON writes them as arrays of integers + for (const auto& initial : all_options) + { + const json j1 = json::from_ubjson(json::to_ubjson(j0, initial.use_size, initial.use_type)); + + for (const auto& o : all_options) + { + INFO("j1 = " << j1.dump() << ", use_size = " << o.use_size << ", use_type = " << o.use_type); + + const std::vector vec = json::to_ubjson(j1, o.use_size, o.use_type); + json j2; + // anything the library writes must be parsable by the library + REQUIRE_NOTHROW(j2 = json::from_ubjson(vec)); + CHECK(j2.dump() == j1.dump()); + CHECK(json::to_ubjson(j2, o.use_size, o.use_type) == vec); + } + } + } +} + TEST_CASE("UBJSON roundtrips" * doctest::skip()) { SECTION("input from self-generated UBJSON files") From 01b53c8c15ae94e3b790ec9578a43966f0262c30 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:29:36 +0200 Subject: [PATCH 05/10] Keep JSON_DIAGNOSTICS parent pointers of ordered_json members after erase() and update() (#5552) * Keep JSON_DIAGNOSTICS parent pointers of ordered_json members after erase() and update() ordered_json stores its members in a vector, and two operations moved members without restoring their parent pointers afterwards: - ordered_map::erase() re-constructs every member after the erased one in place. The basic_json move constructor leaves m_parent at nullptr, and none of the object branches of basic_json::erase() (by key, iterator, or iterator range) called set_parents(). This also affected merge_patch() with a null member and patch() with a remove operation. - update() only set the parent pointer of the inserted member. Adding a key can reallocate the vector, which copies all other members and leaves their m_parent at nullptr. The set_parents() call added for #4813 only repaired this for the nested object of a merge, not for the target. The next assert_invariant() on such an object (for instance, when copying it) aborted, and diagnostic messages lost the path prefix above the moved member. std::map-based json was not affected, because its nodes do not move. Erasing from an ordered_map object now calls set_parents(), and update() uses set_parent(), which already refreshes all members for vector-based objects. This makes the #4813 workaround redundant. Signed-off-by: Niels Lohmann * Account for JSON_DIAGNOSTIC_POSITIONS in the ordered_json parent-pointer test The merge_patch() case parses its input, so with JSON_DIAGNOSTIC_POSITIONS the exception message also carries the byte range of the parsed value. Signed-off-by: Niels Lohmann * Silence clang-tidy for the intentional copy in the ordered_json parent-pointer test Signed-off-by: Niels Lohmann * Keep parent pointers when update() merges past its descent bound The iterative path of update() only set the parent pointer of the member it inserted, like the recursive one did before. It now uses set_parent() too, so ordered_json members that move when a nested object grows keep their parents, and the set_parents() calls that patched this up after each nested merge are gone. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 48 ++++++++++----- single_include/nlohmann/json.hpp | 48 ++++++++++----- tests/src/unit-diagnostics.cpp | 100 +++++++++++++++++++++++++++++++ 3 files changed, 166 insertions(+), 30 deletions(-) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index e40a2aa27..09dca4694 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -1243,6 +1243,27 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } + /// @brief restore the parent pointers after erasing from an object + /// ordered_json keeps its members in a vector, and erasing a member + /// re-constructs every member after it in place, which resets their + /// parent pointers + void set_parents_after_object_erase() + { +#if JSON_DIAGNOSTICS +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning(push ) +#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr +#endif + if (detail::is_ordered_map::value) + { + set_parents(); + } +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning( pop ) +#endif +#endif + } + public: ////////////////////////// // JSON parser callback // @@ -2931,6 +2952,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { result.m_it.object_iterator = erase_from_object(pos.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -3003,6 +3025,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { result.m_it.object_iterator = m_data.m_value.object->erase(first.m_it.object_iterator, last.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -3033,7 +3056,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - return m_data.m_value.object->erase(std::forward(key)); + const auto erased = m_data.m_value.object->erase(std::forward(key)); + set_parents_after_object_erase(); + return erased; } template < typename KeyType, detail::enable_if_t < @@ -3050,6 +3075,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); + set_parents_after_object_erase(); return 1; } return 0; @@ -3961,16 +3987,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); -#if JSON_DIAGNOSTICS - it2->second.set_parents(); -#endif continue; } } - m_data.m_value.object->operator[](it.key()) = it.value(); -#if JSON_DIAGNOSTICS - m_data.m_value.object->operator[](it.key()).m_parent = this; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + set_parent(m_data.m_value.object->operator[](it.key()) = it.value()); } } @@ -3999,9 +4021,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // a nested object is merged: continue with its parent -#if JSON_DIAGNOSTICS - target->set_parents(); -#endif target = stack.back().target; first = stack.back().position; last = stack.back().last; @@ -4023,10 +4042,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec continue; } } - target->m_data.m_value.object->operator[](first.key()) = first.value(); -#if JSON_DIAGNOSTICS - target->m_data.m_value.object->operator[](first.key()).m_parent = target; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + target->set_parent(target->m_data.m_value.object->operator[](first.key()) = first.value()); ++first; } } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 3c5821d92..9091c1198 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -25701,6 +25701,27 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } + /// @brief restore the parent pointers after erasing from an object + /// ordered_json keeps its members in a vector, and erasing a member + /// re-constructs every member after it in place, which resets their + /// parent pointers + void set_parents_after_object_erase() + { +#if JSON_DIAGNOSTICS +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning(push ) +#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr +#endif + if (detail::is_ordered_map::value) + { + set_parents(); + } +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning( pop ) +#endif +#endif + } + public: ////////////////////////// // JSON parser callback // @@ -27389,6 +27410,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { result.m_it.object_iterator = erase_from_object(pos.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -27461,6 +27483,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { result.m_it.object_iterator = m_data.m_value.object->erase(first.m_it.object_iterator, last.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -27491,7 +27514,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - return m_data.m_value.object->erase(std::forward(key)); + const auto erased = m_data.m_value.object->erase(std::forward(key)); + set_parents_after_object_erase(); + return erased; } template < typename KeyType, detail::enable_if_t < @@ -27508,6 +27533,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); + set_parents_after_object_erase(); return 1; } return 0; @@ -28419,16 +28445,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); -#if JSON_DIAGNOSTICS - it2->second.set_parents(); -#endif continue; } } - m_data.m_value.object->operator[](it.key()) = it.value(); -#if JSON_DIAGNOSTICS - m_data.m_value.object->operator[](it.key()).m_parent = this; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + set_parent(m_data.m_value.object->operator[](it.key()) = it.value()); } } @@ -28457,9 +28479,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // a nested object is merged: continue with its parent -#if JSON_DIAGNOSTICS - target->set_parents(); -#endif target = stack.back().target; first = stack.back().position; last = stack.back().last; @@ -28481,10 +28500,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec continue; } } - target->m_data.m_value.object->operator[](first.key()) = first.value(); -#if JSON_DIAGNOSTICS - target->m_data.m_value.object->operator[](first.key()).m_parent = target; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + target->set_parent(target->m_data.m_value.object->operator[](first.key()) = first.value()); ++first; } } diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index cdb6185b3..3ae649e5b 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -361,6 +361,106 @@ TEST_CASE("Regression tests for extended diagnostics") CHECK(p == o); } } + + SECTION("Regression test - erase() and update() must keep JSON_DIAGNOSTICS parent pointers of ordered_json members") + { + // ordered_json keeps its members in a vector: erasing a member + // re-constructs all members after it in place, and adding a key may + // reallocate the vector; both reset the parent pointers of the members + // that were moved + using nlohmann::ordered_json; + + const auto check_parents = [](const ordered_json & j) + { + // const access, so operator[] cannot repair the parent pointers + CHECK_THROWS_WITH_AS(j["z"]["x"].at(0), "[json.exception.type_error.304] (/z/x) cannot use at() with number", ordered_json::type_error); + + // must not trigger assert_invariant() in a debug/assert-enabled build + ordered_json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + }; + + // erase(key) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + CHECK(j.erase("a") == 1); + check_parents(j); + } + + // erase(iterator) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + j.erase(j.begin()); + check_parents(j); + } + + // erase(iterator, iterator) + { + ordered_json j = {{"a", 1}, {"b", 2}, {"z", {{"x", 1}}}}; + j.erase(j.begin(), j.find("z")); + check_parents(j); + } + + // patch() removes via erase(iterator) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + j.patch_inplace(ordered_json::parse(R"([{"op": "remove", "path": "/a"}])")); + check_parents(j); + } + + // update(j) + { + ordered_json j = {{"z", {{"x", 1}}}}; + j.update({{"a", 1}, {"b", 2}}); + check_parents(j); + } + + // update(j, true), the outer and the nested vector both grow + { + ordered_json j = {{"z", {{"x", 1}}}}; + j.update({{"z", {{"y", 2}}}, {"a", 1}}, true); + check_parents(j); + } + + // update(j, true) around its descent bound, where the nested vectors + // grow while the objects are merged without recursing + for (const std::size_t depth : + { + nlohmann::detail::recursion_depth_limit() - 1, nlohmann::detail::recursion_depth_limit(), nlohmann::detail::recursion_depth_limit() + 2 + }) + { + ordered_json j = {{"z", {{"x", 1}}}}; + ordered_json patch = {{"a", 1}, {"b", 2}, {"c", {{"d", 3}}}}; + for (std::size_t i = 0; i < depth; ++i) + { + j = ordered_json{{"k", 0}, {"n", std::move(j)}}; + patch = ordered_json{{"n", std::move(patch)}, {"l", 1}, {"m", 2}}; + } + j.update(patch, true); + + // must not trigger assert_invariant() on any level in a + // debug/assert-enabled build + ordered_json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + } + + // merge_patch() inserts "c" and removes "d" at /a/c, then inserts "e" + // at /a, which copies /a/c + { + auto j = ordered_json::parse(R"({"a": {"c": {"d": {}}}})"); + j.merge_patch(ordered_json::parse(R"({"a": {"c": {"c": "s", "d": null}, "e": "s"}})")); + CHECK(j.dump() == R"({"a":{"c":{"c":"s"},"e":"s"}})"); + + auto const& constJ = j; +#if JSON_DIAGNOSTIC_POSITIONS + CHECK_THROWS_WITH_AS(constJ["a"]["c"]["c"].at(0), "[json.exception.type_error.304] (/a/c/c) (bytes 18-21) cannot use at() with string", ordered_json::type_error); +#else + CHECK_THROWS_WITH_AS(constJ["a"]["c"]["c"].at(0), "[json.exception.type_error.304] (/a/c/c) cannot use at() with string", ordered_json::type_error); +#endif + ordered_json const copy = j; + CHECK(copy == j); + } + } } TEST_CASE("Better diagnostics past the descent bound of update() and merge_patch()") From 6c8ea0a6d1b36dfc9a11a83cb052c9436500d529 Mon Sep 17 00:00:00 2001 From: Jeremy Nimmer Date: Thu, 24 Sep 2026 23:46:19 -0700 Subject: [PATCH 06/10] Remove Bazel alwayslink=True (#5376) This should have no effect for header only libraries as mentioned. It was previously removed in e509007d but then accidentally added again in 26cfec34. Signed-off-by: Jeremy Nimmer --- BUILD.bazel | 1 - cmake/scripts/gen_bazel_build_file.cmake | 1 - 2 files changed, 2 deletions(-) diff --git a/BUILD.bazel b/BUILD.bazel index b13e62c22..db59095cd 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -71,7 +71,6 @@ cc_library( ], includes = ["include"], visibility = ["//visibility:public"], - alwayslink = True, ) cc_library( diff --git a/cmake/scripts/gen_bazel_build_file.cmake b/cmake/scripts/gen_bazel_build_file.cmake index 3c7db9493..a781a2e3f 100644 --- a/cmake/scripts/gen_bazel_build_file.cmake +++ b/cmake/scripts/gen_bazel_build_file.cmake @@ -42,7 +42,6 @@ string(APPEND CONTENT [=[ ], includes = ["include"], visibility = ["//visibility:public"], - alwayslink = True, ) cc_library( From 98278dc3f6899244cb5b3ffd07dd8c2379dcd4dd Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 17:54:40 +0200 Subject: [PATCH 07/10] Fix CI: disable MSVC warning C5285 for the vendored doctest (#5577) The windows-11-arm runner now ships MSVC 19.51, which reports doctest's forward declaration of std::tuple as C5285 ("cannot declare a specialization for 'std::tuple'"). With /WX this breaks the msvc-arm64 job on develop and on every open pull request. Disable the warning for the test targets, like the other MSVC warnings already disabled there. Signed-off-by: Niels Lohmann --- tests/CMakeLists.txt | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 7a9bc5471..ea6da107c 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -80,7 +80,10 @@ target_compile_options(test_main PUBLIC # std::allocator_traits<...>::construct for the custom-base-class # test's map type past VS2015's limit. The name is only used for # debug info, so truncation does not affect the build. - $<$:/W4;/wd4566;/wd4996;/wd4702;/wd4503> + # Disable warning C5285: cannot declare a specialization for 'std::tuple'; MSVC 19.51 + # reports the forward declarations of standard library + # templates in the vendored doctest.h + $<$:/W4;/wd4566;/wd4996;/wd4702;/wd4503;/wd5285> # https://github.com/nlohmann/json/issues/1114 $<$:/bigobj> $<$:-Wa,-mbig-obj> From abbe52d6de81323fa0b159da2c36f0589756af16 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 17:56:18 +0200 Subject: [PATCH 08/10] Add JSON_PRECISE_STREAM_POSITION to leave the character that terminates a number in the stream (#5344) * docs: qualify the operator>> stream positioning guarantee operator>>'s notes state that it leaves the stream positioned right after the parsed value, so that concatenated JSON values can be read back to back. That does not hold when the value is a number: a number is only terminated by the character that follows it, and the lexer's unget() is simulated (it rewinds only the lexer's own bookkeeping), so that character stays consumed from the stream. Document the actual behaviour: the guarantee holds for all value types except numbers, which must be followed by whitespace. Also qualify the cross-reference on the JSON Lines page, which repeated the unqualified claim. Documentation only; the behaviour itself is tracked in #5340. Signed-off-by: Niels Lohmann * fix: restore the character that terminates a number (#5340) operator>> is documented to leave the stream positioned right after the parsed value, so that concatenated JSON values can be read back to back. That did not hold for numbers: a number is only terminated by the character following it, and lexer::scan_number() reads that character and calls unget() -- which is simulated and rewinds only the lexer's own bookkeeping. input_stream_adapter consumes via sbumpc() with no matching sungetc(), so the terminating character stayed consumed and the next extraction started one byte too late ('1true' left the stream at 'rue'). Propagating unget() to the adapter directly does not work: next_unget makes the following get() replay the cached character, so the terminator would be delivered twice. Instead, restore the still-pending character once at the end of a non-strict parse, where the input is handed back to the caller: - input_stream_adapter gains unget_character() (sungetc()) and advertises it via supports_unget, detected the same way as supports_seek. - lexer::restore_pending_unget() turns a pending simulated unget of a real (non-EOF) character into a real one and clears next_unget so the character is not also replayed. It is a no-op for adapters that cannot unget, and reports failure when sungetc() fails, in which case the input is left as it was before. - parser calls it on the three non-strict paths, i.e. for operator>> and sax_parse(strict = false). Strict parse()/accept() are unaffected: they require the input to end after the value, so the character is consumed by the end-of-input check anyway. Parse error messages and reported positions are unchanged. Signed-off-by: Niels Lohmann * tests: fix CI failures in the #5340 test helpers Four CI failures, all in the new test code: - GCC (-Werror=useless-cast): drop the `json(...)` wrapper around `json::parse(...)`, which already returns a `json`. - GCC (-Werror=unused-result): assign the discarded `json::parse()` result to a dummy, the idiom used elsewhere in the test suite, and catch `json::parse_error&` for consistency. - clang-tidy (google-default-arguments): remove the default argument from the `pbackfail()` override; `sungetc()` supplies the base declaration's default. - MSVC (bad allocation): `no_putback_streambuf::underflow()` set a one-character get area without advancing `m_pos`, so an implementation whose `istream::get` peeks before it bumps re-read the same character forever. Keep no get area at all: `underflow()` peeks, `uflow()` consumes, and `sungetc()` still always lands in `pbackfail()`, which is what the test needs. Signed-off-by: Niels Lohmann * fix: leave the character that terminates a number in the input Read the character following a number without consuming it, instead of consuming it and putting it back. input_stream_adapter now peeks with sgetc() and only steps over the character when the next one is requested or when the adapter is destroyed, so releasing it cannot fail - no putback position is required from the streambuf. Suggested by gregmarr in #5344. Signed-off-by: Niels Lohmann * docs: match the version history wording to the peek-based fix Signed-off-by: Niels Lohmann * docs: drop the whitespace-separator caveat from the parsing pages The caveat added in #5343 describes the behavior this branch fixes: a number no longer consumes the character that terminates it, so concatenated values need no separator. Signed-off-by: Niels Lohmann * refactor: split the strict and non-strict paths in parser Folding the release_lookahead() call into the existing strict check left the "in strict mode" comment on an else-if branch, and made the strict condition in sax_parse() redundant with the branch it followed. Signed-off-by: Niels Lohmann * Put the stream position fix behind JSON_PRECISE_STREAM_POSITION Leaving the character that terminates a number in the stream is observable: reading "1,2,3" with repeated operator>> works today only because the comma after each number is swallowed, and std::getline after a number skips the line break. Both break with the fix, so make it opt-in for 3.x, as suggested by @gregmarr in the review. - JSON_PRECISE_STREAM_POSITION (default 0) selects the peek-based input_stream_adapter. Without it, the adapter is the consuming one from develop and has no supports_lookahead, so lexer::release_lookahead() and the parser's calls to it compile to nothing. - The macro changes input_stream_adapter's layout and member functions, so it gets the ABI tag _psp, after _bics. The ABI config tests, the natvis generator, and nlohmann_json.natvis (regenerated) know the tag. - The tests for the fix move to unit-precise-stream-position.cpp, which defines the macro itself and runs in every build, and gain the two cases above. unit-deserialization.cpp pins the default behavior instead. - The docs describe the default behavior again and point to the new macro page; version history says "added in 3.13.0, planned default in 4.0.0". Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/sax_parse.md | 6 +- docs/mkdocs/docs/api/macros/index.md | 2 + .../macros/json_precise_stream_position.md | 131 +++ docs/mkdocs/docs/api/operator_gtgt.md | 7 +- docs/mkdocs/docs/features/macros.md | 9 + docs/mkdocs/docs/features/namespace.md | 1 + docs/mkdocs/docs/features/parsing/index.md | 3 +- docs/mkdocs/mkdocs.yml | 1 + include/nlohmann/detail/abi_macros.hpp | 19 +- .../nlohmann/detail/input/input_adapters.hpp | 82 ++ include/nlohmann/detail/input/lexer.hpp | 65 ++ include/nlohmann/detail/input/parser.hpp | 61 +- include/nlohmann/detail/macro_unscope.hpp | 1 + nlohmann_json.natvis | 960 ++++++++++++++++++ single_include/nlohmann/json.hpp | 228 ++++- single_include/nlohmann/json_fwd.hpp | 19 +- tests/abi/config/default.cpp | 4 + tests/abi/config/noversion.cpp | 4 + tests/src/unit-deserialization.cpp | 52 + tests/src/unit-precise-stream-position.cpp | 237 +++++ tools/generate_natvis/generate_natvis.py | 2 +- 21 files changed, 1846 insertions(+), 48 deletions(-) create mode 100644 docs/mkdocs/docs/api/macros/json_precise_stream_position.md create mode 100644 tests/src/unit-precise-stream-position.cpp diff --git a/docs/mkdocs/docs/api/basic_json/sax_parse.md b/docs/mkdocs/docs/api/basic_json/sax_parse.md index fc8ce07b6..bf61ea9eb 100644 --- a/docs/mkdocs/docs/api/basic_json/sax_parse.md +++ b/docs/mkdocs/docs/api/basic_json/sax_parse.md @@ -69,7 +69,9 @@ The SAX event lister must follow the interface of [`json_sax`](../json_sax/index [`input_format_t`](input_format_t.md) for more information `strict` (in) -: whether the input has to be consumed completely (optional, `#!cpp true` by default) +: whether the input has to be consumed completely (optional, `#!cpp true` by default); when `#!cpp false` and the + input is a `#!cpp std::istream`, the character that terminates a number is consumed unless + [`JSON_PRECISE_STREAM_POSITION`](../macros/json_precise_stream_position.md) is defined to `1`; see [`operator>>`](../operator_gtgt.md#notes) `ignore_comments` (in) : whether comments should be ignored and treated like whitespace (`#!cpp true`) or yield a parse error @@ -136,6 +138,8 @@ A UTF-8 byte order mark is silently ignored. - Added `ignore_trailing_commas` in version 3.13.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- `JSON_PRECISE_STREAM_POSITION` added in version 3.13.0 to optionally leave a `#!cpp std::istream` positioned right + after the parsed value when `strict` is `#!cpp false`. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index 70f02a7a6..bf773b5c4 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -16,6 +16,8 @@ header. See also the [macro overview page](../../features/macros.md). ## Parsing +- [**JSON_PRECISE_STREAM_POSITION**](json_precise_stream_position.md) - opt in to leaving an input stream positioned + right after a parsed number - [**JSON_STRICT_NUL_HANDLING**](json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of treating it as end of input diff --git a/docs/mkdocs/docs/api/macros/json_precise_stream_position.md b/docs/mkdocs/docs/api/macros/json_precise_stream_position.md new file mode 100644 index 000000000..5710e9975 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_precise_stream_position.md @@ -0,0 +1,131 @@ +# JSON_PRECISE_STREAM_POSITION + +```cpp +#define JSON_PRECISE_STREAM_POSITION /* value */ +``` + +When defined to `1`, [`operator>>`](../operator_gtgt.md) and [`sax_parse`](../basic_json/sax_parse.md) with +`strict = false` leave a `#!cpp std::istream` positioned right after the parsed value for every value type. By default, +the character that terminates a number is consumed as well. + +The macro only affects reading from a `#!cpp std::istream` when the rest of the stream is not required to be consumed. +[`parse`](../basic_json/parse.md), [`accept`](../basic_json/accept.md), and all other inputs (strings, iterators, +containers, `#!cpp FILE*`) are never affected. + +## Default definition + +The default value is `0` (disabled — existing behavior is preserved). + +```cpp +#define JSON_PRECISE_STREAM_POSITION 0 +``` + +## Notes + +!!! note "Background" + + A number is the only JSON value whose end can be detected solely by reading the character that follows it. By + default, that character is consumed and not put back, so the stream is left one byte too far after a number, and + only after a number: + + ```cpp + std::istringstream input("1true"); + json j; + input >> j; // j == 1, but the stream now starts at "rue" + ``` + + With this macro, the character is only looked at and left in the stream, so the stream starts at `true`. This + does not require the stream buffer to support putting a character back. + + This was not changed unconditionally, because code can depend on the consumed character, even unknowingly (see + [#5340](https://github.com/nlohmann/json/issues/5340)). Both of the following work by default only because the + character after each number is swallowed, and behave differently with this macro: + + ```cpp + std::istringstream input("1,2,3"); + json j1, j2, j3; + input >> j1 >> j2 >> j3; // default: 1, 2, 3 + // with the macro: throws parse_error.101 at the ',' + ``` + + ```cpp + std::istringstream input("42\nfoo"); + json j; + std::string line; + input >> j; + std::getline(input, line); // default: "foo" + // with the macro: "" (like after reading an int with >>) + ``` + + In both cases, the behavior with the macro is what you already get today when the value is not a number: `"a","b"` + fails at the `,`, and `std::getline` after `{}` returns an empty string. This macro offers an opt-in path to + the consistent behavior ahead of version 4.0.0, where it is planned to become the default. + +!!! warning "Opt-in only" + + This macro must be defined **before** including ``. Defining it after the include has no + effect. + +!!! note "ABI compatibility" + + The value of this macro is encoded in the [namespace](../../features/namespace.md) (tag `_psp`), resulting in + distinct symbol names. Translation units compiled with and without it can therefore be linked into the same program + without One Definition Rule (ODR) violations, but they cannot exchange instances of library types. + +!!! tip "Workaround without the macro" + + Separate the values in the stream with whitespace. The character consumed after a number is then the separator, + and whitespace before the next value is skipped anyway. + +## Examples + +??? example "Default behavior (macro not defined)" + + Without the macro, the character after a number is consumed: + + ```cpp + #include + #include + #include + + using json = nlohmann::json; + + int main() + { + std::istringstream input("1true"); + json j1, j2; + input >> j1; // j1 == 1 + input >> j2; // throws parse_error.101: the stream now starts at "rue" + } + ``` + +??? example "Opt-in precise stream position (macro defined to 1)" + + With the macro, the stream is positioned right after the number: + + ```cpp + #define JSON_PRECISE_STREAM_POSITION 1 + #include + #include + #include + + using json = nlohmann::json; + + int main() + { + std::istringstream input("1true"); + json j1, j2; + input >> j1; // j1 == 1 + input >> j2; // j2 == true + } + ``` + +## See also + +- [**operator>>**](../operator_gtgt.md) - deserialize from stream +- [**sax_parse**](../basic_json/sax_parse.md) - generate SAX events + +## Version history + +- Added in version 3.13.0. +- Planned to become the default (with the macro removed) in version 4.0.0. diff --git a/docs/mkdocs/docs/api/operator_gtgt.md b/docs/mkdocs/docs/api/operator_gtgt.md index 3e60d5236..0173b9fb3 100644 --- a/docs/mkdocs/docs/api/operator_gtgt.md +++ b/docs/mkdocs/docs/api/operator_gtgt.md @@ -67,7 +67,9 @@ input >> j2; // parses the next value Only numbers are affected. Values ending in a self-delimiting character do not read past themselves, so `truefalse`, `[1][2]`, `{"a":1}{"b":2}`, and `"a""b"` can be read back to back without a separator. - This is tracked in [#5340](https://github.com/nlohmann/json/issues/5340). + Define [`JSON_PRECISE_STREAM_POSITION`](macros/json_precise_stream_position.md) to `1` to leave the terminating character in the stream + instead, so that the stream is positioned right after the value for every value type and no separator is + needed. This is tracked in [#5340](https://github.com/nlohmann/json/issues/5340). Note that reading concatenated values does **not** work for [JSON Lines](../features/parsing/json_lines.md) (newline-delimited JSON) input -- see that page for why and for the recommended alternative. @@ -107,9 +109,12 @@ being read. - [parse](basic_json/parse.md) - deserialize from a compatible input - [`JSON_STRICT_NUL_HANDLING`](macros/json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of treating it as end of input +- [`JSON_PRECISE_STREAM_POSITION`](macros/json_precise_stream_position.md) - opt in to leaving the stream positioned right after a number ## Version history - Added in version 1.0.0. - `JSON_STRICT_NUL_HANDLING` added in version 3.13.0 to optionally reject a NUL byte in the input instead of treating it as end of input; planned to become the default in version 4.0.0. +- `JSON_PRECISE_STREAM_POSITION` added in version 3.13.0 to optionally leave the character that terminates a number in + the stream; planned to become the default in version 4.0.0. diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index e7baba0ae..7d6d23148 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -98,6 +98,15 @@ rather than descending into a bounded number of levels first, which is slower bu See [full documentation of `JSON_NO_THREAD_LOCAL`](../api/macros/json_no_thread_local.md). +## `JSON_PRECISE_STREAM_POSITION` + +When defined to `1`, [`operator>>`](../api/operator_gtgt.md) and non-strict +[`sax_parse`](../api/basic_json/sax_parse.md) leave an input stream positioned right after the parsed value, instead of +also consuming the character that terminates a number. The default value is `0`, which preserves the existing behavior; +this is planned to become the default in version 4.0.0. + +See [full documentation of `JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md). + ## `JSON_SKIP_LIBRARY_VERSION_CHECK` When defined, the library will not create a compiler warning when a different version of the library was already diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index 09e53f3a2..5eb4a76a9 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -18,6 +18,7 @@ The complete default namespace name is derived as follows: - [`JSON_DIAGNOSTIC_POSITIONS`](../api/macros/json_diagnostic_positions.md) defined non-zero appends `_dp`. - [`JSON_BRACE_INIT_COPY_SEMANTICS`](../api/macros/json_brace_init_copy_semantics.md) defined non-zero appends `_bics`. + - [`JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md) defined non-zero appends `_psp`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/docs/mkdocs/docs/features/parsing/index.md b/docs/mkdocs/docs/features/parsing/index.md index 17624b8a2..476f024fa 100644 --- a/docs/mkdocs/docs/features/parsing/index.md +++ b/docs/mkdocs/docs/features/parsing/index.md @@ -41,7 +41,8 @@ document followed by trailing bytes" is accepted rather than rejected. If you ar reject any input that is not exactly one JSON document, prefer `parse`. When using `operator>>` to read several concatenated values this way, a value that is a number must be followed by -whitespace, because `operator>>` consumes the character that terminates a number — see the +whitespace, because `operator>>` consumes the character that terminates a number, unless +[`JSON_PRECISE_STREAM_POSITION`](../../api/macros/json_precise_stream_position.md) is defined to `1` — see the [`operator>>` notes](../../api/operator_gtgt.md#notes) for details and examples. ## SAX vs. DOM parsing diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index ffb0fae80..a05ce2dff 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -293,6 +293,7 @@ nav: - 'JSON_NOEXCEPTION': api/macros/json_noexception.md - 'JSON_NO_IO': api/macros/json_no_io.md - 'JSON_NO_THREAD_LOCAL': api/macros/json_no_thread_local.md + - 'JSON_PRECISE_STREAM_POSITION': api/macros/json_precise_stream_position.md - 'JSON_SKIP_LIBRARY_VERSION_CHECK': api/macros/json_skip_library_version_check.md - 'JSON_SKIP_UNSUPPORTED_COMPILER_CHECK': api/macros/json_skip_unsupported_compiler_check.md - 'JSON_STRICT_NUL_HANDLING': api/macros/json_strict_nul_handling.md diff --git a/include/nlohmann/detail/abi_macros.hpp b/include/nlohmann/detail/abi_macros.hpp index cca04e8ec..a6666c66e 100644 --- a/include/nlohmann/detail/abi_macros.hpp +++ b/include/nlohmann/detail/abi_macros.hpp @@ -38,6 +38,10 @@ #define JSON_BRACE_INIT_COPY_SEMANTICS 0 #endif +#ifndef JSON_PRECISE_STREAM_POSITION + #define JSON_PRECISE_STREAM_POSITION 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -62,21 +66,28 @@ #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS #endif +#if JSON_PRECISE_STREAM_POSITION + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION _psp +#else + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ - NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ + NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index 775a8398c..317b38806 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -101,6 +101,11 @@ class input_stream_adapter // maintain ifstream flags, except eof if (is != nullptr) { +#if JSON_PRECISE_STREAM_POSITION + // consume the character last returned by get_character() unless it + // was given back with release_lookahead() + commit_lookahead(); +#endif is->clear(is->rdstate() & std::ios::eofbit); } } @@ -114,6 +119,58 @@ class input_stream_adapter input_stream_adapter& operator=(input_stream_adapter&) = delete; input_stream_adapter& operator=(input_stream_adapter&&) = delete; +#if JSON_PRECISE_STREAM_POSITION + input_stream_adapter(input_stream_adapter&& rhs) noexcept + : is(rhs.is), sb(rhs.sb), lookahead(rhs.lookahead) + { + rhs.is = nullptr; + rhs.sb = nullptr; + rhs.lookahead = false; + } + + // Whether the character last returned by get_character() can be given back + // to the input with release_lookahead(). + static constexpr bool supports_lookahead = true; + + // std::istream/std::streambuf use std::char_traits::to_int_type, to + // ensure that std::char_traits::eof() and the character 0xFF do not + // end up as the same value, e.g., 0xFFFFFFFF. + // + // The character is peeked rather than consumed: it is only stepped over + // once the next character is requested, or when the adapter is destroyed. + // Until then, release_lookahead() can leave it in the input. + std::char_traits::int_type get_character() + { + if (lookahead) + { + // step over the character returned by the previous call + sb->sbumpc(); + } + + auto res = sb->sgetc(); + // set eof manually, as we don't use the istream interface. + if (JSON_HEDLEY_UNLIKELY(res == std::char_traits::eof())) + { + // there is nothing to step over next time + lookahead = false; + is->clear(is->rdstate() | std::ios::eofbit); + } + else + { + lookahead = true; + } + return res; + } + + // Leave the character last returned by get_character() in the input, so + // that the next read from the stream - by this adapter or by the caller + // once parsing is done - sees it again. Unlike putting a consumed + // character back, this cannot fail. + void release_lookahead() noexcept + { + lookahead = false; + } +#else input_stream_adapter(input_stream_adapter&& rhs) noexcept : is(rhs.is), sb(rhs.sb) { @@ -124,6 +181,9 @@ class input_stream_adapter // std::istream/std::streambuf use std::char_traits::to_int_type, to // ensure that std::char_traits::eof() and the character 0xFF do not // end up as the same value, e.g., 0xFFFFFFFF. + // + // The character is consumed, so the character that terminates a number + // stays consumed after parsing; see JSON_PRECISE_STREAM_POSITION. std::char_traits::int_type get_character() { auto res = sb->sbumpc(); @@ -134,10 +194,14 @@ class input_stream_adapter } return res; } +#endif template std::size_t get_elements(T* dest, std::size_t count = 1) { +#if JSON_PRECISE_STREAM_POSITION + commit_lookahead(); +#endif auto res = static_cast(sb->sgetn(reinterpret_cast(dest), static_cast(count * sizeof(T)))); if (JSON_HEDLEY_UNLIKELY(res < count * sizeof(T))) { @@ -147,9 +211,27 @@ class input_stream_adapter } private: +#if JSON_PRECISE_STREAM_POSITION + // Step over the character last returned by get_character(). The character + // has already been peeked successfully, so for every streambuf with a get + // area this is a pointer increment that cannot fail. + void commit_lookahead() + { + if (lookahead) + { + lookahead = false; + sb->sbumpc(); + } + } +#endif + /// the associated input stream std::istream* is = nullptr; std::streambuf* sb = nullptr; +#if JSON_PRECISE_STREAM_POSITION + /// whether get_character() peeked a character that is not consumed yet + bool lookahead = false; +#endif }; #endif // JSON_NO_IO diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index fe85cd53d..98c0fd76a 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -127,6 +127,25 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/) return false; } +// Detect whether an input adapter reads with one character of lookahead that +// can be left in the input (see input_stream_adapter::supports_lookahead, +// which is only defined with JSON_PRECISE_STREAM_POSITION), detected like +// supports_seek above. +template +using detect_supports_lookahead = decltype(InputAdapterType::supports_lookahead); + +template +constexpr bool input_adapter_supports_lookahead(std::true_type /*detected*/) +{ + return InputAdapterType::supports_lookahead; +} + +template +constexpr bool input_adapter_supports_lookahead(std::false_type /*detected*/) +{ + return false; +} + // Detect whether an input adapter exposes a contiguous byte block that the // lexer can scan directly (see iterator_input_adapter::supports_bulk_scan). // Adapters without the flag - file, stream, wide-string, user-defined - fall @@ -167,6 +186,12 @@ class lexer : public lexer_base static constexpr bool lazy_token_string = input_adapter_supports_seek(is_detected {}); + /// whether a simulated unget can be passed on to the input adapter, which + /// then leaves the character in the input; see + /// input_adapter_supports_lookahead + static constexpr bool can_release_lookahead = + input_adapter_supports_lookahead(is_detected {}); + /// whether string scanning may bulk-consume runs of ordinary characters /// directly from a contiguous input buffer (SWAR fast path). This requires /// the token to be reconstructible lazily (lazy_token_string), so bypassing @@ -1898,6 +1923,21 @@ scan_number_done: uncapture_char(std::integral_constant {}); } + /// adapter without lookahead: nothing to do (see release_lookahead) + void release_lookahead_impl(std::false_type /*can_release*/) const noexcept {} + + /// adapter with lookahead: leave the character in the input instead + void release_lookahead_impl(std::true_type /*can_release*/) + { + if (next_unget) + { + // the character is read from the input again rather than replayed + // from current, so the adapter must not step over it + next_unget = false; + ia.release_lookahead(); + } + } + /// seekable adapter: nothing was captured, so nothing to undo void uncapture_char(std::true_type /*lazy*/) const noexcept {} @@ -1961,6 +2001,31 @@ scan_number_done: return position; } + /*! + @brief pass a pending simulated unget on to the input + + unget() only rewinds the lexer's own bookkeeping, so the character that + terminated the last token (e.g. the character after a number) would still + be stepped over when the input adapter is done. Callers that hand the + input back to the user afterwards - operator>> and non-strict sax_parse - + call this once when scanning is done, so that the input is positioned + right after the value. + + Adapters without lookahead (see input_adapter_supports_lookahead) are not + handed back to the user, so this is a no-op for them. Without + JSON_PRECISE_STREAM_POSITION, no adapter has lookahead, so this is always a + no-op and the terminating character stays consumed. + + Scanning may continue after this call: @a next_unget is cleared, and the + character is read from the input again instead of being replayed from + @a current. A pending unget of EOF needs no special case, because reaching + EOF leaves no lookahead to release. + */ + void release_lookahead() + { + release_lookahead_impl(std::integral_constant {}); + } + #if JSON_DIAGNOSTIC_POSITIONS /// return the offset of the first character of the last read token; unlike /// the token's parsed value, this accounts for escape sequences diff --git a/include/nlohmann/detail/input/parser.hpp b/include/nlohmann/detail/input/parser.hpp index a45ee4a0a..5fec57a70 100644 --- a/include/nlohmann/detail/input/parser.hpp +++ b/include/nlohmann/detail/input/parser.hpp @@ -100,13 +100,22 @@ class parser json_sax_dom_callback_parser sdp(result, callback, allow_exceptions, &m_lexer); sax_parse_internal(&sdp); - // in strict mode, input must be completely read - if (strict && (get_token() != token_type::end_of_input)) + if (strict) { - sdp.parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), - exception_message(token_type::end_of_input, "value"), nullptr)); + // in strict mode, input must be completely read + if (get_token() != token_type::end_of_input) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), + exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // the caller keeps using the input: position it right after + // the value by leaving the character that terminated it + m_lexer.release_lookahead(); } // in case of an error, return a discarded value @@ -128,12 +137,20 @@ class parser json_sax_dom_parser sdp(result, allow_exceptions, &m_lexer); sax_parse_internal(&sdp); - // in strict mode, input must be completely read - if (strict && (get_token() != token_type::end_of_input)) + if (strict) { - sdp.parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + // in strict mode, input must be completely read + if (get_token() != token_type::end_of_input) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // see above + m_lexer.release_lookahead(); } // in case of an error, return a discarded value @@ -166,12 +183,24 @@ class parser (void)detail::is_sax_static_asserts {}; const bool result = sax_parse_internal(sax); - // strict mode: next byte must be EOF - if (result && strict && (get_token() != token_type::end_of_input)) + if (result) { - return sax->parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + if (strict) + { + // strict mode: next byte must be EOF + if (get_token() != token_type::end_of_input) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // the caller keeps using the input: position it right after + // the value by leaving the character that terminated it + m_lexer.release_lookahead(); + } } return result; diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index afcbfc38b..6ace4cf4a 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -45,6 +45,7 @@ #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON #undef JSON_BRACE_INIT_COPY_SEMANTICS + #undef JSON_PRECISE_STREAM_POSITION #endif #include diff --git a/nlohmann_json.natvis b/nlohmann_json.natvis index 2eccbe17c..8f4eec31a 100644 --- a/nlohmann_json.natvis +++ b/nlohmann_json.natvis @@ -335,6 +335,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -515,6 +575,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -635,6 +755,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -695,6 +875,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -815,6 +1115,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -875,6 +1235,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -935,6 +1415,186 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -995,4 +1655,304 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 9091c1198..e3ca0f0d0 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -95,6 +95,10 @@ #define JSON_BRACE_INIT_COPY_SEMANTICS 0 #endif +#ifndef JSON_PRECISE_STREAM_POSITION + #define JSON_PRECISE_STREAM_POSITION 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -119,21 +123,28 @@ #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS #endif +#if JSON_PRECISE_STREAM_POSITION + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION _psp +#else + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ - NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ + NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ @@ -7419,6 +7430,11 @@ class input_stream_adapter // maintain ifstream flags, except eof if (is != nullptr) { +#if JSON_PRECISE_STREAM_POSITION + // consume the character last returned by get_character() unless it + // was given back with release_lookahead() + commit_lookahead(); +#endif is->clear(is->rdstate() & std::ios::eofbit); } } @@ -7432,6 +7448,58 @@ class input_stream_adapter input_stream_adapter& operator=(input_stream_adapter&) = delete; input_stream_adapter& operator=(input_stream_adapter&&) = delete; +#if JSON_PRECISE_STREAM_POSITION + input_stream_adapter(input_stream_adapter&& rhs) noexcept + : is(rhs.is), sb(rhs.sb), lookahead(rhs.lookahead) + { + rhs.is = nullptr; + rhs.sb = nullptr; + rhs.lookahead = false; + } + + // Whether the character last returned by get_character() can be given back + // to the input with release_lookahead(). + static constexpr bool supports_lookahead = true; + + // std::istream/std::streambuf use std::char_traits::to_int_type, to + // ensure that std::char_traits::eof() and the character 0xFF do not + // end up as the same value, e.g., 0xFFFFFFFF. + // + // The character is peeked rather than consumed: it is only stepped over + // once the next character is requested, or when the adapter is destroyed. + // Until then, release_lookahead() can leave it in the input. + std::char_traits::int_type get_character() + { + if (lookahead) + { + // step over the character returned by the previous call + sb->sbumpc(); + } + + auto res = sb->sgetc(); + // set eof manually, as we don't use the istream interface. + if (JSON_HEDLEY_UNLIKELY(res == std::char_traits::eof())) + { + // there is nothing to step over next time + lookahead = false; + is->clear(is->rdstate() | std::ios::eofbit); + } + else + { + lookahead = true; + } + return res; + } + + // Leave the character last returned by get_character() in the input, so + // that the next read from the stream - by this adapter or by the caller + // once parsing is done - sees it again. Unlike putting a consumed + // character back, this cannot fail. + void release_lookahead() noexcept + { + lookahead = false; + } +#else input_stream_adapter(input_stream_adapter&& rhs) noexcept : is(rhs.is), sb(rhs.sb) { @@ -7442,6 +7510,9 @@ class input_stream_adapter // std::istream/std::streambuf use std::char_traits::to_int_type, to // ensure that std::char_traits::eof() and the character 0xFF do not // end up as the same value, e.g., 0xFFFFFFFF. + // + // The character is consumed, so the character that terminates a number + // stays consumed after parsing; see JSON_PRECISE_STREAM_POSITION. std::char_traits::int_type get_character() { auto res = sb->sbumpc(); @@ -7452,10 +7523,14 @@ class input_stream_adapter } return res; } +#endif template std::size_t get_elements(T* dest, std::size_t count = 1) { +#if JSON_PRECISE_STREAM_POSITION + commit_lookahead(); +#endif auto res = static_cast(sb->sgetn(reinterpret_cast(dest), static_cast(count * sizeof(T)))); if (JSON_HEDLEY_UNLIKELY(res < count * sizeof(T))) { @@ -7465,9 +7540,27 @@ class input_stream_adapter } private: +#if JSON_PRECISE_STREAM_POSITION + // Step over the character last returned by get_character(). The character + // has already been peeked successfully, so for every streambuf with a get + // area this is a pointer increment that cannot fail. + void commit_lookahead() + { + if (lookahead) + { + lookahead = false; + sb->sbumpc(); + } + } +#endif + /// the associated input stream std::istream* is = nullptr; std::streambuf* sb = nullptr; +#if JSON_PRECISE_STREAM_POSITION + /// whether get_character() peeked a character that is not consumed yet + bool lookahead = false; +#endif }; #endif // JSON_NO_IO @@ -8879,6 +8972,25 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/) return false; } +// Detect whether an input adapter reads with one character of lookahead that +// can be left in the input (see input_stream_adapter::supports_lookahead, +// which is only defined with JSON_PRECISE_STREAM_POSITION), detected like +// supports_seek above. +template +using detect_supports_lookahead = decltype(InputAdapterType::supports_lookahead); + +template +constexpr bool input_adapter_supports_lookahead(std::true_type /*detected*/) +{ + return InputAdapterType::supports_lookahead; +} + +template +constexpr bool input_adapter_supports_lookahead(std::false_type /*detected*/) +{ + return false; +} + // Detect whether an input adapter exposes a contiguous byte block that the // lexer can scan directly (see iterator_input_adapter::supports_bulk_scan). // Adapters without the flag - file, stream, wide-string, user-defined - fall @@ -8919,6 +9031,12 @@ class lexer : public lexer_base static constexpr bool lazy_token_string = input_adapter_supports_seek(is_detected {}); + /// whether a simulated unget can be passed on to the input adapter, which + /// then leaves the character in the input; see + /// input_adapter_supports_lookahead + static constexpr bool can_release_lookahead = + input_adapter_supports_lookahead(is_detected {}); + /// whether string scanning may bulk-consume runs of ordinary characters /// directly from a contiguous input buffer (SWAR fast path). This requires /// the token to be reconstructible lazily (lazy_token_string), so bypassing @@ -10650,6 +10768,21 @@ scan_number_done: uncapture_char(std::integral_constant {}); } + /// adapter without lookahead: nothing to do (see release_lookahead) + void release_lookahead_impl(std::false_type /*can_release*/) const noexcept {} + + /// adapter with lookahead: leave the character in the input instead + void release_lookahead_impl(std::true_type /*can_release*/) + { + if (next_unget) + { + // the character is read from the input again rather than replayed + // from current, so the adapter must not step over it + next_unget = false; + ia.release_lookahead(); + } + } + /// seekable adapter: nothing was captured, so nothing to undo void uncapture_char(std::true_type /*lazy*/) const noexcept {} @@ -10713,6 +10846,31 @@ scan_number_done: return position; } + /*! + @brief pass a pending simulated unget on to the input + + unget() only rewinds the lexer's own bookkeeping, so the character that + terminated the last token (e.g. the character after a number) would still + be stepped over when the input adapter is done. Callers that hand the + input back to the user afterwards - operator>> and non-strict sax_parse - + call this once when scanning is done, so that the input is positioned + right after the value. + + Adapters without lookahead (see input_adapter_supports_lookahead) are not + handed back to the user, so this is a no-op for them. Without + JSON_PRECISE_STREAM_POSITION, no adapter has lookahead, so this is always a + no-op and the terminating character stays consumed. + + Scanning may continue after this call: @a next_unget is cleared, and the + character is read from the input again instead of being replayed from + @a current. A pending unget of EOF needs no special case, because reaching + EOF leaves no lookahead to release. + */ + void release_lookahead() + { + release_lookahead_impl(std::integral_constant {}); + } + #if JSON_DIAGNOSTIC_POSITIONS /// return the offset of the first character of the last read token; unlike /// the token's parsed value, this accounts for escape sequences @@ -15960,13 +16118,22 @@ class parser json_sax_dom_callback_parser sdp(result, callback, allow_exceptions, &m_lexer); sax_parse_internal(&sdp); - // in strict mode, input must be completely read - if (strict && (get_token() != token_type::end_of_input)) + if (strict) { - sdp.parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), - exception_message(token_type::end_of_input, "value"), nullptr)); + // in strict mode, input must be completely read + if (get_token() != token_type::end_of_input) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), + exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // the caller keeps using the input: position it right after + // the value by leaving the character that terminated it + m_lexer.release_lookahead(); } // in case of an error, return a discarded value @@ -15988,12 +16155,20 @@ class parser json_sax_dom_parser sdp(result, allow_exceptions, &m_lexer); sax_parse_internal(&sdp); - // in strict mode, input must be completely read - if (strict && (get_token() != token_type::end_of_input)) + if (strict) { - sdp.parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + // in strict mode, input must be completely read + if (get_token() != token_type::end_of_input) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // see above + m_lexer.release_lookahead(); } // in case of an error, return a discarded value @@ -16026,12 +16201,24 @@ class parser (void)detail::is_sax_static_asserts {}; const bool result = sax_parse_internal(sax); - // strict mode: next byte must be EOF - if (result && strict && (get_token() != token_type::end_of_input)) + if (result) { - return sax->parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + if (strict) + { + // strict mode: next byte must be EOF + if (get_token() != token_type::end_of_input) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // the caller keeps using the input: position it right after + // the value by leaving the character that terminated it + m_lexer.release_lookahead(); + } } return result; @@ -30765,6 +30952,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON #undef JSON_BRACE_INIT_COPY_SEMANTICS + #undef JSON_PRECISE_STREAM_POSITION #endif // #include diff --git a/single_include/nlohmann/json_fwd.hpp b/single_include/nlohmann/json_fwd.hpp index 281c05efa..af776d652 100644 --- a/single_include/nlohmann/json_fwd.hpp +++ b/single_include/nlohmann/json_fwd.hpp @@ -56,6 +56,10 @@ #define JSON_BRACE_INIT_COPY_SEMANTICS 0 #endif +#ifndef JSON_PRECISE_STREAM_POSITION + #define JSON_PRECISE_STREAM_POSITION 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -80,21 +84,28 @@ #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS #endif +#if JSON_PRECISE_STREAM_POSITION + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION _psp +#else + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ - NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ + NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index d0b4ba54b..8f66dfbf1 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -36,6 +36,10 @@ TEST_CASE("default namespace") expected += "_bics"; #endif +#if JSON_PRECISE_STREAM_POSITION + expected += "_psp"; +#endif + expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json"; diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index 789107181..12b7603b9 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -37,6 +37,10 @@ TEST_CASE("default namespace without version component") expected += "_bics"; #endif +#if JSON_PRECISE_STREAM_POSITION + expected += "_psp"; +#endif + expected += "::basic_json"; // fallback for Clang diff --git a/tests/src/unit-deserialization.cpp b/tests/src/unit-deserialization.cpp index 0dbfdd15c..28359804b 100644 --- a/tests/src/unit-deserialization.cpp +++ b/tests/src/unit-deserialization.cpp @@ -25,6 +25,7 @@ using nlohmann::json; #include #include #include +#include #include #if defined(_WIN32) @@ -1233,6 +1234,57 @@ TEST_CASE("deserialization") } } + SECTION("stream position after extraction without JSON_PRECISE_STREAM_POSITION (#5340)") + { + // By default, the character that terminates a number is consumed, so + // the stream is left one byte too far after a number (and only after a + // number). JSON_PRECISE_STREAM_POSITION changes this; see + // unit-precise-stream-position.cpp. These checks pin the default. + const auto remaining = [](std::istream & is) + { + return std::string(std::istreambuf_iterator(is), std::istreambuf_iterator()); + }; + + SECTION("the character after a number is consumed") + { + std::istringstream ss("1true"); + json j; + ss >> j; + CHECK(j == 1); + CHECK(remaining(ss) == "rue"); + } + + SECTION("the character after other values is not consumed") + { + std::istringstream ss("[1]true"); + json j; + ss >> j; + CHECK(j == json::parse("[1]")); + CHECK(remaining(ss) == "true"); + } + + SECTION("comma-separated numbers can be read one by one") + { + std::istringstream ss("1,2,3"); + json j1, j2, j3; + ss >> j1 >> j2 >> j3; + CHECK(j1 == 1); + CHECK(j2 == 2); + CHECK(j3 == 3); + } + + SECTION("std::getline after a number skips the line break") + { + std::istringstream ss("42\nfoo"); + json j; + std::string line; + ss >> j; + std::getline(ss, line); + CHECK(j == 42); + CHECK(line == "foo"); + } + } + // build with C++20 // JSON_HAS_CPP_20 #if defined(__cpp_char8_t) diff --git a/tests/src/unit-precise-stream-position.cpp b/tests/src/unit-precise-stream-position.cpp new file mode 100644 index 000000000..5b6bff682 --- /dev/null +++ b/tests/src/unit-precise-stream-position.cpp @@ -0,0 +1,237 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// This file tests the opt-in JSON_PRECISE_STREAM_POSITION, so it defines the +// macro itself rather than relying on a -D flag, and runs in every build. The +// default behavior is pinned in unit-deserialization.cpp. +#ifdef JSON_PRECISE_STREAM_POSITION + #undef JSON_PRECISE_STREAM_POSITION +#endif + +#define JSON_PRECISE_STREAM_POSITION 1 + +#include +using nlohmann::json; + +#include +#include +#include +#include +#include +#include + +#define STRINGIZE_EX(x) #x +#define STRINGIZE(x) STRINGIZE_EX(x) + +namespace +{ +// A streambuf that keeps no get area at all and refuses every putback: with an +// empty get area, sungetc() always ends up in pbackfail(). Used to check that +// the character terminating a number is left in the input without relying on +// the streambuf being able to put a consumed character back. +class no_putback_streambuf : public std::streambuf +{ + public: + explicit no_putback_streambuf(std::string s) : m_data(std::move(s)) {} + + protected: + // peek at the next character without consuming it + int_type underflow() override + { + if (m_pos >= m_data.size()) + { + return traits_type::eof(); + } + return traits_type::to_int_type(m_data[m_pos]); + } + + // consume the next character + int_type uflow() override + { + if (m_pos >= m_data.size()) + { + return traits_type::eof(); + } + return traits_type::to_int_type(m_data[m_pos++]); + } + + int_type pbackfail(int_type /*c*/) override + { + return traits_type::eof(); + } + + private: + std::string m_data; + std::size_t m_pos = 0; +}; + +// read the characters that are left in a stream +std::string remaining(std::istream& is) +{ + std::string result; + char c = 0; + while (is.get(c)) + { + result += c; + } + return result; +} +} // namespace + +TEST_CASE("JSON_PRECISE_STREAM_POSITION") +{ + SECTION("the macro is part of the ABI tag") + { + const std::string ns = STRINGIZE(NLOHMANN_JSON_NAMESPACE); + // other tags may come before it, e.g. json_abi_diag_psp + CHECK(ns.find("_psp") != std::string::npos); + } + + SECTION("a number does not consume the character that terminates it") + { + // a number is only terminated by the character following it; that + // character must be given back so the stream is positioned right + // after the value + const std::vector> tests = + { + {"1true", "true"}, + {"1[2]", "[2]"}, + {"1{}", "{}"}, + {R"(1"a")", R"("a")"}, + {"1 true", " true"}, + {"12,", ","}, + {"-0.5e3x", "x"}, + {"1null", "null"} + }; + + for (const auto& test : tests) + { + CAPTURE(test.first); + std::istringstream ss(test.first); + json j; + ss >> j; + CHECK(j == json::parse(test.first.substr(0, test.first.size() - test.second.size()))); + CHECK(remaining(ss) == test.second); + } + } + + SECTION("values that are self-delimiting are unaffected") + { + const std::vector> tests = + { + {"truefalse", "false"}, + {"[1][2]", "[2]"}, + {R"({"a":1}{"b":2})", R"({"b":2})"}, + {R"("a""b")", R"("b")"}, + {"null null", " null"} + }; + + for (const auto& test : tests) + { + CAPTURE(test.first); + std::istringstream ss(test.first); + json j; + ss >> j; + CHECK(remaining(ss) == test.second); + } + } + + SECTION("a number at the end of the input leaves nothing behind") + { + for (const std::string s : + {"1", "12", "-3.5e2", " 7 " + }) + { + CAPTURE(s); + std::istringstream ss(s); + json j; + ss >> j; + CHECK(remaining(ss).find_first_not_of(" \t\n\r") == std::string::npos); + } + } + + SECTION("repeated extraction of concatenated values") + { + std::istringstream ss(R"(1true[2]3"x"{"a":4}5)"); + const std::vector expected = + { + json(1), json(true), json::parse("[2]"), json(3), + json("x"), json::parse(R"({"a":4})"), json(5) + }; + + for (const auto& e : expected) + { + json j; + ss >> j; + CHECK(j == e); + } + } + + SECTION("differences to the default behavior") + { + // both of these work by accident without the macro, because the + // character after a number is swallowed; see unit-deserialization.cpp + + SECTION("a separator after a number is not skipped") + { + std::istringstream ss("1,2"); + json j; + ss >> j; + CHECK(j == 1); + CHECK_THROWS_AS(ss >> j, json::parse_error&); + } + + SECTION("std::getline after a number sees the line break") + { + std::istringstream ss("42\nfoo"); + json j; + std::string line; + ss >> j; + std::getline(ss, line); + CHECK(j == 42); + CHECK(line.empty()); + std::getline(ss, line); + CHECK(line == "foo"); + } + } + + SECTION("sax_parse with strict == false") + { + std::istringstream ss("1true"); + json j; + nlohmann::detail::json_sax_dom_parser sdp(j, true); + CHECK(json::sax_parse(ss, &sdp, nlohmann::detail::input_format_t::json, false)); + CHECK(j == 1); + CHECK(remaining(ss) == "true"); + } + + SECTION("strict parsing still rejects trailing data") + { + std::istringstream ss("1true"); + json _; + CHECK_THROWS_WITH_AS(_ = json::parse(ss), + "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - unexpected true literal; expected end of input", json::parse_error&); + + std::istringstream ss2("1true"); + CHECK_FALSE(json::accept(ss2)); + } + + SECTION("a streambuf that cannot put back is not needed") + { + // the terminating character is never consumed, so no putback + // position is required + no_putback_streambuf buf("1true"); + std::istream is(&buf); + json j; + is >> j; + CHECK(j == json(1)); + CHECK(remaining(is) == "true"); + } +} diff --git a/tools/generate_natvis/generate_natvis.py b/tools/generate_natvis/generate_natvis.py index 968690abe..fb1210db1 100755 --- a/tools/generate_natvis/generate_natvis.py +++ b/tools/generate_natvis/generate_natvis.py @@ -20,7 +20,7 @@ if __name__ == '__main__': namespaces = ['nlohmann'] abi_prefix = 'json_abi' - abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics'] + abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics', '_psp'] version = '_v' + args.version.replace('.', '_') inline_namespaces = [] From 02dd3e67f2ec25539d9763a7137a1950db7e94d4 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 17:58:59 +0200 Subject: [PATCH 09/10] Fix stack overflow and exponential runtime when comparing nested values (#5390) * Compare values without recursing, and without comparing them twice Comparing two values compared their containers, which compare their elements, which brought the comparison back once per nesting level. Two values nested deeply enough exhausted the call stack and terminated the process with a segmentation fault - the same bug as #5387, in the last operation that still had it. Worse, an ordered comparison took exponentially long in the nesting depth before C++20. std::vector's operator< is a lexicographical comparison, which asks whether an element is less than its counterpart and then whether the counterpart is less than it - two full comparisons of everything below that element, at every level. Comparing two equal values nested 30 levels deep, which is nothing unusual, took 3.8 seconds; 40 levels would have taken an hour, and nothing about the value has to be pathological to get there. C++20 is unaffected: std::lexicographical_compare_three_way asks once. Compare a value that is nested too deeply to descend into on an explicit stack instead, in a single pass that yields less, equal, greater or unordered at once. Equality and the three-way comparison descend as they always did for the first 128 levels, which nothing measurable costs them; an ordered comparison no longer descends at all, which is what takes the exponent out of it. Objects and arrays that are not nested deeply are otherwise compared exactly as before. The results are unchanged for every pair of values: 68121 comparisons of a corpus that covers NaN, discarded values, mixed number types, binary values, empty containers and both object types are identical to develop, in C++11, C++17 and C++20, with and without thread_local storage and legacy discarded comparison. Reproducing that meant reproducing two subtleties: a lexicographic comparison steps over a pair it cannot order, where a three-way comparison stops at it, and an object compares its keys with < where its entries are ordered but with == where they are only checked for equality - not with the object's own comparator, which for nlohmann::ordered_map tells equality. Equality needs no ordering, so it no longer asks for any: a key or string type that can only be compared for equality still works. Measured (medians of 7 interleaved runs, clang -O3, C++11): comparing two equal values nested 30 levels deep 3778 ms -> 0.002 ms; ordering flat objects -33.6%; ordering flat arrays of numbers +27.3%, the one shape that pays for the single pass; equality unchanged throughout. Signed-off-by: Niels Lohmann * Describe comparison in the no-thread-local docs and CI target Comparing two values now bounds its descent with a thread_local counter just as copying does, so the JSON_NO_THREAD_LOCAL page, the macro overview and the ci_test_no_thread_local target cover both rather than copying alone. Also record what switching the macro on costs a comparison: on the benchmark documents, comparing two equal values takes 10% to 90% longer. Signed-off-by: Niels Lohmann * Take the descent flag as an argument rather than testing it MSVC reports the test of a constant as C4127 ("conditional expression is constant"), which the Windows builds treat as an error: may_descend is false for operator<, so the operand short-circuits the whole condition. Passing it to compare_descent_exhausted() puts the test where the value is an ordinary parameter, and leaves the call sites with no condition of their own. Signed-off-by: Niels Lohmann * Note the comparison fallback in the no-thread-local documentation The macro page describes what the library defines JSON_NO_THREAD_LOCAL for by itself in terms of copying alone; comparing falls back the same way. Signed-off-by: Niels Lohmann * Parenthesise the reserve() computation in the comparison test clang-tidy reports the mixed * and + as readability-math-missing- parentheses, as it does for the identical line in the copy test. Signed-off-by: Niels Lohmann * Use the shared descent bookkeeping rather than a second set Comparing kept a thread_local count, a limit and a guard of its own beside the ones copying already had, all three the same thing under a different name. They are gone; the shared count, limit and guard do the work. The guard grows a second constructor here, because the comparison operators are written as a macro and a macro cannot use the preprocessor: it cannot look the count up behind an #ifdef the way copy_structured does, so the guard looks it up for it. nesting_depth_exhausted() arrives for the same reason - whether an operator descends at all is a constant at every call site, and testing it there is what MSVC reports as C4127. Also say in compare_leaves what happens to a pair that is an array on one side and an object on the other, since the answer is not obvious from the code: an operator only descends into two values of the same type, so such a pair is told apart by its types alone - unequal, and ordered the way the types are - exactly as it is above the bound. And record what the explicit stack costs: the comparison operators are noexcept and the container comparison this replaces allocated nothing, so running out of memory here ends the process instead of throwing. It takes a value nested past the bound and an exhausted heap to reach, and the same comparison used to exhaust the call stack, but it is a new way to fail. Signed-off-by: Niels Lohmann * Amalgamate Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- cmake/ci.cmake | 8 +- .../docs/api/macros/json_no_thread_local.md | 19 +- docs/mkdocs/docs/features/macros.md | 5 +- include/nlohmann/json.hpp | 304 +++++++++++++++++- single_include/nlohmann/json.hpp | 304 +++++++++++++++++- tests/src/unit-large_json.cpp | 48 +++ 6 files changed, 659 insertions(+), 29 deletions(-) diff --git a/cmake/ci.cmake b/cmake/ci.cmake index a99788633..752bdc6f8 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -300,10 +300,10 @@ add_custom_target(ci_test_skiplibraryversioncheck # Disable thread-local storage. ############################################################################### -# Without thread-local storage, the copy constructor cannot bound its descent -# and copies every object and array without the call stack. That path is -# otherwise only reached by values nested deeper than the bound, so this target -# is what runs the whole test suite through it. +# Without thread-local storage, copying and comparing cannot bound their +# descent and handle every object and array without the call stack. Those paths +# are otherwise only reached by values nested deeper than the bound, so this +# target is what runs the whole test suite through them. add_custom_target(ci_test_no_thread_local COMMAND ${CMAKE_COMMAND} -DCMAKE_BUILD_TYPE=Debug -GNinja diff --git a/docs/mkdocs/docs/api/macros/json_no_thread_local.md b/docs/mkdocs/docs/api/macros/json_no_thread_local.md index 126116ec3..0a001ac36 100644 --- a/docs/mkdocs/docs/api/macros/json_no_thread_local.md +++ b/docs/mkdocs/docs/api/macros/json_no_thread_local.md @@ -7,16 +7,16 @@ When defined, the library does not use `#!cpp thread_local` storage. This is relevant for the few environments whose toolchain does not support it. -The copy constructor copies the first levels of a value by copying the containers, which copy their elements, and -completes whatever is nested deeper than that without the call stack, so that copying a value cannot exhaust the stack -however deeply it is nested. It counts the levels it has descended into in a `#!cpp thread_local` variable, as a counter -shared between threads would be raced. +Copying a value and comparing two values both descend into the first levels by letting the containers copy or compare +themselves, and finish whatever is nested deeper than that without the call stack, so that neither can exhaust the stack +however deeply the values are nested. Each counts the levels it has descended into in a `#!cpp thread_local` variable, as +a counter shared between threads would be raced. -Without that counter, no descent can be bounded safely, so objects and arrays are copied without the call stack right -away. Copying keeps working exactly as it does otherwise - the same values come out, and deeply nested values are copied -just as safely - but copying is slower, because the containers no longer copy themselves. Copying the benchmark -documents takes 9% (`canada.json`) to 34% (`twitter.json`) longer; values built mostly from objects are affected the -most. +Without those counters, no descent can be bounded safely, so objects and arrays are copied and compared without the call +stack right away. Both keep working exactly as they do otherwise - the same values come out, the same comparisons hold, +and deeply nested values are handled just as safely - but both are slower, because the containers no longer copy or +compare themselves. Copying the benchmark documents takes 9% (`canada.json`) to 34% (`twitter.json`) longer, and +comparing two equal ones 10% (`citm_catalog.json`) to 90% (`canada.json`) longer. ## Default definition @@ -28,6 +28,7 @@ By default, `#!cpp JSON_NO_THREAD_LOCAL` is not defined. The library defines it by itself for Clang targeting MinGW, which does not survive the `#!cpp thread_local` storage: copying a value segfaults there, with both old and current Clang versions, while GCC targeting MinGW is unaffected. +Copying and comparing fall back to working without the call stack there, as they do whenever the macro is defined. ## Examples diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index 7d6d23148..2bc8a4a5b 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -93,8 +93,9 @@ See [full documentation of `JSON_NO_IO`](../api/macros/json_no_io.md). ## `JSON_NO_THREAD_LOCAL` -When defined, the library does not use `#!cpp thread_local` storage. Copying a value then always avoids the call stack -rather than descending into a bounded number of levels first, which is slower but yields the same values. +When defined, the library does not use `#!cpp thread_local` storage. Copying a value and comparing two values then +always avoid the call stack rather than descending into a bounded number of levels first, which is slower but yields the +same values and the same comparisons. See [full documentation of `JSON_NO_THREAD_LOCAL`](../api/macros/json_no_thread_local.md). diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 09dca4694..0123bad29 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -924,6 +924,31 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } #endif + /*! + @brief whether a descent must stop here and finish without the call stack + + @a may_descend says whether the operator descends at all; it is a constant + at every call site, and is passed rather than tested by the caller so that + the test does not become a constant condition there, which MSVC reports as + C4127. + + The comparison operators use this rather than @ref nesting_depth_guard::okay, + because they are written as a macro and a macro cannot use the preprocessor + the way the guard's constructor does; @ref copy_structured, which can, asks + the guard instead and never calls this. + */ + static bool nesting_depth_exhausted(bool may_descend = true) noexcept + { +#ifdef JSON_NO_THREAD_LOCAL + // without a count of its own per thread, a descent cannot be bounded + // without racing another one, so none is made + static_cast(may_descend); + return true; +#else + return !may_descend || nesting_depth() >= nesting_depth_limit(); +#endif + } + /*! @brief counts one level of a bounded descent for as long as it runs, and reports whether the descent was still within the limit when it began @@ -1243,6 +1268,253 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } + /// the result of comparing two values, including values that cannot be + /// ordered at all, such as a discarded value or a NaN + enum class compare_result { less, equal, greater, unordered }; + +#if JSON_HAS_THREE_WAY_COMPARISON + /// @brief the ordering that @a result stands for + static std::partial_ordering to_partial_ordering(compare_result result) noexcept // *NOPAD* + { + switch (result) + { + case compare_result::less: + return std::partial_ordering::less; + case compare_result::greater: + return std::partial_ordering::greater; + case compare_result::equal: + return std::partial_ordering::equivalent; + case compare_result::unordered: + default: + return std::partial_ordering::unordered; + } + } +#endif + + /*! + @brief compare two values that are not both an array or both an object + + Such a pair is compared by the operators themselves, which cannot descend + into it and therefore cannot recurse. + + That holds for a pair whose types differ as much as for a pair of leaves: an + array and an object are told apart by their types alone, because an operator + only ever descends into two values of the same type. So `==` reports them as + unequal without looking inside either, and an ordering falls back to the + order of the types - an object sorts before an array - exactly as it does + for a value that is not nested deeply enough to get here. + */ + template + static compare_result compare_leaves(const_reference lhs, const_reference rhs) noexcept + { + if (lhs == rhs) + { + return compare_result::equal; + } + + return order_leaves(lhs, rhs, std::integral_constant {}); + } + + /*! + @brief compare two object keys + + An object compares its entries as pairs of a key and a value, so its keys + are compared exactly as std::pair compares them: with < where the objects + are being ordered, and with == where they are only checked for equality. + Note that this is not the object's own comparator, which for a vector-backed + object type such as nlohmann::ordered_map tells equality rather than order. + */ + static compare_result compare_keys(const typename object_t::key_type& lhs, + const typename object_t::key_type& rhs, + std::true_type /*ordered*/) + { + if (lhs < rhs) + { + return compare_result::less; + } + + if (rhs < lhs) + { + return compare_result::greater; + } + + return compare_result::equal; + } + + /// @brief check two object keys for equality + static compare_result compare_keys(const typename object_t::key_type& lhs, + const typename object_t::key_type& rhs, + std::false_type /*ordered*/) + { + return lhs == rhs ? compare_result::equal : compare_result::unordered; + } + + /// @brief tell apart two values that are not equal + /// @note only instantiated where the values are being ordered, as a key or + /// string type is not required to be ordered to be compared for equality + static compare_result order_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept + { + if (lhs < rhs) + { + return compare_result::less; + } + + if (rhs < lhs) + { + return compare_result::greater; + } + + return compare_result::unordered; + } + + /// @brief report two values as not equal without ordering them + static compare_result order_leaves(const_reference /*lhs*/, const_reference /*rhs*/, std::false_type /*ordered*/) noexcept + { + return compare_result::unordered; + } + + /*! + @brief compare @a lhs and @a rhs without descending into them + + Reached once a comparison has descended @ref nesting_depth_limit levels, so + that comparing values cannot exhaust the call stack however deeply they are + nested. The two values are walked in lockstep on an explicit stack and + compared lexicographically, element by element in the order the containers + enumerate them - which is how the container types this library ships compare + themselves: a std::map enumerates its entries in key order, and + nlohmann::ordered_map in insertion order. An object type that enumerates its + entries in an unspecified order, such as std::unordered_map, compares them + pairwise instead; the difference could only ever show below the bound. + + Note that the stack this walks with is allocated, while the comparison + operators are noexcept and the container comparison this replaces allocated + nothing. Failing that allocation therefore ends the process rather than + throwing. It only arises for values nested past the bound, and only when + memory has run out - where the same comparison used to exhaust the call + stack instead - but it is a way to fail that the operators did not have. + */ + template + static compare_result compare_iteratively(const_reference lhs, const_reference rhs, + const bool unordered_compares_equal) noexcept + { + /// a pair of containers being compared in lockstep + struct frame + { + const basic_json* lhs_value{nullptr}; + const basic_json* rhs_value{nullptr}; + typename array_t::const_iterator lhs_array_it{}; + typename array_t::const_iterator rhs_array_it{}; + typename object_t::const_iterator lhs_object_it{}; + typename object_t::const_iterator rhs_object_it{}; + }; + + std::vector stack; + const basic_json* left = &lhs; + const basic_json* right = &rhs; + + for (;;) + { + const auto type = left->m_data.m_type; + + if (type == right->m_data.m_type && (type == value_t::array || type == value_t::object)) + { + // descend: the elements decide, and are compared further down + stack.emplace_back(); + frame& pushed = stack.back(); + pushed.lhs_value = left; + pushed.rhs_value = right; + + if (type == value_t::array) + { + pushed.lhs_array_it = left->m_data.m_value.array->cbegin(); + pushed.rhs_array_it = right->m_data.m_value.array->cbegin(); + } + else + { + pushed.lhs_object_it = left->m_data.m_value.object->cbegin(); + pushed.rhs_object_it = right->m_data.m_value.object->cbegin(); + } + } + else + { + const compare_result result = compare_leaves(*left, *right); + + // Values that cannot be ordered - a NaN, say - end an ordered + // comparison for std::lexicographical_compare_three_way, but + // std::lexicographical_compare treats them as equivalent and + // carries on with the next element. Both are reproduced here, + // so that a value nested too deeply to descend into compares + // exactly as one that is not. + if (result != compare_result::equal && + !(unordered_compares_equal && result == compare_result::unordered)) + { + return result; + } + } + + // walk back up past the containers that are exhausted, then take the + // next pair of elements from the innermost one that is not + for (;;) + { + if (stack.empty()) + { + return compare_result::equal; + } + + frame& current = stack.back(); + const bool is_object = current.lhs_value->m_data.m_type == value_t::object; + + const bool lhs_done = is_object + ? current.lhs_object_it == current.lhs_value->m_data.m_value.object->cend() + : current.lhs_array_it == current.lhs_value->m_data.m_value.array->cend(); + const bool rhs_done = is_object + ? current.rhs_object_it == current.rhs_value->m_data.m_value.object->cend() + : current.rhs_array_it == current.rhs_value->m_data.m_value.array->cend(); + + if (lhs_done || rhs_done) + { + // whichever ran out first holds the smaller container; if + // both did, they are equal and the container above decides + if (lhs_done != rhs_done) + { + return lhs_done ? compare_result::less : compare_result::greater; + } + + stack.pop_back(); + continue; + } + + if (is_object) + { + // an entry is a key and a value, and the key decides first + const compare_result key_result = + compare_keys(current.lhs_object_it->first, current.rhs_object_it->first, + std::integral_constant {}); + + if (key_result != compare_result::equal) + { + return key_result; + } + + left = &(current.lhs_object_it->second); + right = &(current.rhs_object_it->second); + ++current.lhs_object_it; + ++current.rhs_object_it; + } + else + { + left = &(*current.lhs_array_it); + right = &(*current.rhs_array_it); + ++current.lhs_array_it; + ++current.rhs_array_it; + } + + break; + } + } + } + + /// @brief restore the parent pointers after erasing from an object /// ordered_json keeps its members in a vector, and erasing a member /// re-constructs every member after it in place, which resets their @@ -4183,7 +4455,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // because any negative signed value is smaller than any unsigned value. // Otherwise, the non-negative signed value is cast to unsigned before the // comparison to avoid wraparound. -#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result) \ +#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result, deep_result, may_descend) \ const auto lhs_type = lhs.type(); \ const auto rhs_type = rhs.type(); \ \ @@ -4192,11 +4464,25 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec switch (lhs_type) \ { \ case value_t::array: \ + { \ + if (JSON_HEDLEY_UNLIKELY(nesting_depth_exhausted(may_descend))) \ + { \ + return (deep_result); \ + } \ + const nesting_depth_guard guard; \ return (*lhs.m_data.m_value.array) op (*rhs.m_data.m_value.array); \ - \ + } \ + \ case value_t::object: \ + { \ + if (JSON_HEDLEY_UNLIKELY(nesting_depth_exhausted(may_descend))) \ + { \ + return (deep_result); \ + } \ + const nesting_depth_guard guard; \ return (*lhs.m_data.m_value.object) op (*rhs.m_data.m_value.object); \ - \ + } \ + \ case value_t::null: \ return (null_result); \ \ @@ -4296,7 +4582,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif const_reference lhs = *this; - JSON_IMPLEMENT_OPERATOR( ==, true, false, false) + JSON_IMPLEMENT_OPERATOR( ==, true, false, false, + compare_iteratively(lhs, rhs, false) == compare_result::equal, true) #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP #endif @@ -4321,7 +4608,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_IMPLEMENT_OPERATOR(<=>, // *NOPAD* std::partial_ordering::equivalent, std::partial_ordering::unordered, - lhs_type <=> rhs_type) // *NOPAD* + lhs_type <=> rhs_type, // *NOPAD* + to_partial_ordering(compare_iteratively(lhs, rhs, false)), true) } /// @brief comparison: 3-way @@ -4388,7 +4676,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_DIAGNOSTIC_PUSH JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif - JSON_IMPLEMENT_OPERATOR( ==, true, false, false) + JSON_IMPLEMENT_OPERATOR( ==, true, false, false, + compare_iteratively(lhs, rhs, false) == compare_result::equal, true) #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP #endif @@ -4444,7 +4733,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // default_result is used if we cannot compare values. In that case, // we compare types. Note we have to call the operator explicitly, // because MSVC has problems otherwise. - JSON_IMPLEMENT_OPERATOR( <, false, false, operator<(lhs_type, rhs_type)) + JSON_IMPLEMENT_OPERATOR( <, false, false, operator<(lhs_type, rhs_type), + compare_iteratively(lhs, rhs, true) == compare_result::less, false) } /// @brief comparison: less than diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index e3ca0f0d0..4567a2b32 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -25569,6 +25569,31 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } #endif + /*! + @brief whether a descent must stop here and finish without the call stack + + @a may_descend says whether the operator descends at all; it is a constant + at every call site, and is passed rather than tested by the caller so that + the test does not become a constant condition there, which MSVC reports as + C4127. + + The comparison operators use this rather than @ref nesting_depth_guard::okay, + because they are written as a macro and a macro cannot use the preprocessor + the way the guard's constructor does; @ref copy_structured, which can, asks + the guard instead and never calls this. + */ + static bool nesting_depth_exhausted(bool may_descend = true) noexcept + { +#ifdef JSON_NO_THREAD_LOCAL + // without a count of its own per thread, a descent cannot be bounded + // without racing another one, so none is made + static_cast(may_descend); + return true; +#else + return !may_descend || nesting_depth() >= nesting_depth_limit(); +#endif + } + /*! @brief counts one level of a bounded descent for as long as it runs, and reports whether the descent was still within the limit when it began @@ -25888,6 +25913,253 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } + /// the result of comparing two values, including values that cannot be + /// ordered at all, such as a discarded value or a NaN + enum class compare_result { less, equal, greater, unordered }; + +#if JSON_HAS_THREE_WAY_COMPARISON + /// @brief the ordering that @a result stands for + static std::partial_ordering to_partial_ordering(compare_result result) noexcept // *NOPAD* + { + switch (result) + { + case compare_result::less: + return std::partial_ordering::less; + case compare_result::greater: + return std::partial_ordering::greater; + case compare_result::equal: + return std::partial_ordering::equivalent; + case compare_result::unordered: + default: + return std::partial_ordering::unordered; + } + } +#endif + + /*! + @brief compare two values that are not both an array or both an object + + Such a pair is compared by the operators themselves, which cannot descend + into it and therefore cannot recurse. + + That holds for a pair whose types differ as much as for a pair of leaves: an + array and an object are told apart by their types alone, because an operator + only ever descends into two values of the same type. So `==` reports them as + unequal without looking inside either, and an ordering falls back to the + order of the types - an object sorts before an array - exactly as it does + for a value that is not nested deeply enough to get here. + */ + template + static compare_result compare_leaves(const_reference lhs, const_reference rhs) noexcept + { + if (lhs == rhs) + { + return compare_result::equal; + } + + return order_leaves(lhs, rhs, std::integral_constant {}); + } + + /*! + @brief compare two object keys + + An object compares its entries as pairs of a key and a value, so its keys + are compared exactly as std::pair compares them: with < where the objects + are being ordered, and with == where they are only checked for equality. + Note that this is not the object's own comparator, which for a vector-backed + object type such as nlohmann::ordered_map tells equality rather than order. + */ + static compare_result compare_keys(const typename object_t::key_type& lhs, + const typename object_t::key_type& rhs, + std::true_type /*ordered*/) + { + if (lhs < rhs) + { + return compare_result::less; + } + + if (rhs < lhs) + { + return compare_result::greater; + } + + return compare_result::equal; + } + + /// @brief check two object keys for equality + static compare_result compare_keys(const typename object_t::key_type& lhs, + const typename object_t::key_type& rhs, + std::false_type /*ordered*/) + { + return lhs == rhs ? compare_result::equal : compare_result::unordered; + } + + /// @brief tell apart two values that are not equal + /// @note only instantiated where the values are being ordered, as a key or + /// string type is not required to be ordered to be compared for equality + static compare_result order_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept + { + if (lhs < rhs) + { + return compare_result::less; + } + + if (rhs < lhs) + { + return compare_result::greater; + } + + return compare_result::unordered; + } + + /// @brief report two values as not equal without ordering them + static compare_result order_leaves(const_reference /*lhs*/, const_reference /*rhs*/, std::false_type /*ordered*/) noexcept + { + return compare_result::unordered; + } + + /*! + @brief compare @a lhs and @a rhs without descending into them + + Reached once a comparison has descended @ref nesting_depth_limit levels, so + that comparing values cannot exhaust the call stack however deeply they are + nested. The two values are walked in lockstep on an explicit stack and + compared lexicographically, element by element in the order the containers + enumerate them - which is how the container types this library ships compare + themselves: a std::map enumerates its entries in key order, and + nlohmann::ordered_map in insertion order. An object type that enumerates its + entries in an unspecified order, such as std::unordered_map, compares them + pairwise instead; the difference could only ever show below the bound. + + Note that the stack this walks with is allocated, while the comparison + operators are noexcept and the container comparison this replaces allocated + nothing. Failing that allocation therefore ends the process rather than + throwing. It only arises for values nested past the bound, and only when + memory has run out - where the same comparison used to exhaust the call + stack instead - but it is a way to fail that the operators did not have. + */ + template + static compare_result compare_iteratively(const_reference lhs, const_reference rhs, + const bool unordered_compares_equal) noexcept + { + /// a pair of containers being compared in lockstep + struct frame + { + const basic_json* lhs_value{nullptr}; + const basic_json* rhs_value{nullptr}; + typename array_t::const_iterator lhs_array_it{}; + typename array_t::const_iterator rhs_array_it{}; + typename object_t::const_iterator lhs_object_it{}; + typename object_t::const_iterator rhs_object_it{}; + }; + + std::vector stack; + const basic_json* left = &lhs; + const basic_json* right = &rhs; + + for (;;) + { + const auto type = left->m_data.m_type; + + if (type == right->m_data.m_type && (type == value_t::array || type == value_t::object)) + { + // descend: the elements decide, and are compared further down + stack.emplace_back(); + frame& pushed = stack.back(); + pushed.lhs_value = left; + pushed.rhs_value = right; + + if (type == value_t::array) + { + pushed.lhs_array_it = left->m_data.m_value.array->cbegin(); + pushed.rhs_array_it = right->m_data.m_value.array->cbegin(); + } + else + { + pushed.lhs_object_it = left->m_data.m_value.object->cbegin(); + pushed.rhs_object_it = right->m_data.m_value.object->cbegin(); + } + } + else + { + const compare_result result = compare_leaves(*left, *right); + + // Values that cannot be ordered - a NaN, say - end an ordered + // comparison for std::lexicographical_compare_three_way, but + // std::lexicographical_compare treats them as equivalent and + // carries on with the next element. Both are reproduced here, + // so that a value nested too deeply to descend into compares + // exactly as one that is not. + if (result != compare_result::equal && + !(unordered_compares_equal && result == compare_result::unordered)) + { + return result; + } + } + + // walk back up past the containers that are exhausted, then take the + // next pair of elements from the innermost one that is not + for (;;) + { + if (stack.empty()) + { + return compare_result::equal; + } + + frame& current = stack.back(); + const bool is_object = current.lhs_value->m_data.m_type == value_t::object; + + const bool lhs_done = is_object + ? current.lhs_object_it == current.lhs_value->m_data.m_value.object->cend() + : current.lhs_array_it == current.lhs_value->m_data.m_value.array->cend(); + const bool rhs_done = is_object + ? current.rhs_object_it == current.rhs_value->m_data.m_value.object->cend() + : current.rhs_array_it == current.rhs_value->m_data.m_value.array->cend(); + + if (lhs_done || rhs_done) + { + // whichever ran out first holds the smaller container; if + // both did, they are equal and the container above decides + if (lhs_done != rhs_done) + { + return lhs_done ? compare_result::less : compare_result::greater; + } + + stack.pop_back(); + continue; + } + + if (is_object) + { + // an entry is a key and a value, and the key decides first + const compare_result key_result = + compare_keys(current.lhs_object_it->first, current.rhs_object_it->first, + std::integral_constant {}); + + if (key_result != compare_result::equal) + { + return key_result; + } + + left = &(current.lhs_object_it->second); + right = &(current.rhs_object_it->second); + ++current.lhs_object_it; + ++current.rhs_object_it; + } + else + { + left = &(*current.lhs_array_it); + right = &(*current.rhs_array_it); + ++current.lhs_array_it; + ++current.rhs_array_it; + } + + break; + } + } + } + + /// @brief restore the parent pointers after erasing from an object /// ordered_json keeps its members in a vector, and erasing a member /// re-constructs every member after it in place, which resets their @@ -28828,7 +29100,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // because any negative signed value is smaller than any unsigned value. // Otherwise, the non-negative signed value is cast to unsigned before the // comparison to avoid wraparound. -#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result) \ +#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result, deep_result, may_descend) \ const auto lhs_type = lhs.type(); \ const auto rhs_type = rhs.type(); \ \ @@ -28837,11 +29109,25 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec switch (lhs_type) \ { \ case value_t::array: \ + { \ + if (JSON_HEDLEY_UNLIKELY(nesting_depth_exhausted(may_descend))) \ + { \ + return (deep_result); \ + } \ + const nesting_depth_guard guard; \ return (*lhs.m_data.m_value.array) op (*rhs.m_data.m_value.array); \ - \ + } \ + \ case value_t::object: \ + { \ + if (JSON_HEDLEY_UNLIKELY(nesting_depth_exhausted(may_descend))) \ + { \ + return (deep_result); \ + } \ + const nesting_depth_guard guard; \ return (*lhs.m_data.m_value.object) op (*rhs.m_data.m_value.object); \ - \ + } \ + \ case value_t::null: \ return (null_result); \ \ @@ -28941,7 +29227,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif const_reference lhs = *this; - JSON_IMPLEMENT_OPERATOR( ==, true, false, false) + JSON_IMPLEMENT_OPERATOR( ==, true, false, false, + compare_iteratively(lhs, rhs, false) == compare_result::equal, true) #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP #endif @@ -28966,7 +29253,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_IMPLEMENT_OPERATOR(<=>, // *NOPAD* std::partial_ordering::equivalent, std::partial_ordering::unordered, - lhs_type <=> rhs_type) // *NOPAD* + lhs_type <=> rhs_type, // *NOPAD* + to_partial_ordering(compare_iteratively(lhs, rhs, false)), true) } /// @brief comparison: 3-way @@ -29033,7 +29321,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_DIAGNOSTIC_PUSH JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif - JSON_IMPLEMENT_OPERATOR( ==, true, false, false) + JSON_IMPLEMENT_OPERATOR( ==, true, false, false, + compare_iteratively(lhs, rhs, false) == compare_result::equal, true) #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP #endif @@ -29089,7 +29378,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // default_result is used if we cannot compare values. In that case, // we compare types. Note we have to call the operator explicitly, // because MSVC has problems otherwise. - JSON_IMPLEMENT_OPERATOR( <, false, false, operator<(lhs_type, rhs_type)) + JSON_IMPLEMENT_OPERATOR( <, false, false, operator<(lhs_type, rhs_type), + compare_iteratively(lhs, rhs, true) == compare_result::less, false) } /// @brief comparison: less than diff --git a/tests/src/unit-large_json.cpp b/tests/src/unit-large_json.cpp index f10c8be41..3734966d5 100644 --- a/tests/src/unit-large_json.cpp +++ b/tests/src/unit-large_json.cpp @@ -157,6 +157,54 @@ TEST_CASE("tests on deeply nested JSONs") CHECK(deep_depth == depth); } + SECTION("comparing") + { + // Comparing used to descend once per level, and an ordered + // comparison used to compare every pair of elements twice, once in + // each direction, which took exponentially long in the nesting + // depth. Both are gone: these finish in milliseconds, where the + // second used to take longer than anyone would wait even for a + // value nested only a few dozen levels deep. + const std::string text = std::string(depth, '[') + '0' + std::string(depth, ']'); + const json j = json::parse(text); + const json same = json::parse(text); + const json larger = json::parse(std::string(depth, '[') + '1' + std::string(depth, ']')); + + CHECK(j == same); + CHECK_FALSE(j == larger); + CHECK(j != larger); + + CHECK(j < larger); + CHECK_FALSE(larger < j); + CHECK(larger > j); + CHECK(j <= same); + CHECK(j >= same); + + // a value that ends earlier is the smaller one + const json shorter = json::parse(std::string(depth - 1, '[') + '0' + std::string(depth - 1, ']')); + CHECK_FALSE(j == shorter); + } + + SECTION("comparing objects") + { + std::string text; + text.reserve((6 * depth) + 1); + for (std::size_t i = 0; i < depth; ++i) + { + text += "{\"a\":"; + } + text += '1'; + text.append(depth, '}'); + + const json j = json::parse(text); + const json same = json::parse(text); + + CHECK(j == same); + CHECK_FALSE(j != same); + CHECK(j <= same); + CHECK(j >= same); + } + SECTION("the copy is independent of the original") { const json j = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']')); From 465407f3ce046558b712fe8f9d9db74686cf9d55 Mon Sep 17 00:00:00 2001 From: Alexander Lanin Date: Fri, 25 Sep 2026 18:02:38 +0200 Subject: [PATCH 10/10] Improve error message for const fields (#2818) * Improve error message for const fields * Reject const arguments to get_to() with a clear message Reword the static_assert, add it to the C array overload of get_to() as well, and document that v must not be const. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann Co-authored-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/get_to.md | 5 +++++ include/nlohmann/json.hpp | 2 ++ single_include/nlohmann/json.hpp | 2 ++ 3 files changed, 9 insertions(+) diff --git a/docs/mkdocs/docs/api/basic_json/get_to.md b/docs/mkdocs/docs/api/basic_json/get_to.md index 8334ba5cf..c50c077b0 100644 --- a/docs/mkdocs/docs/api/basic_json/get_to.md +++ b/docs/mkdocs/docs/api/basic_json/get_to.md @@ -21,6 +21,10 @@ This overload is chosen if: - `ValueType` is not `basic_json`, - `json_serializer` has a `from_json()` method of the form `void from_json(const basic_json&, ValueType&)` +`v` must not be `const`. Passing a `const` object is a compile-time error. For types such as arithmetic types, enums, +and C arrays, the error is a `static_assert` that names the problem. For other types, the overload is not viable, and +the compiler reports that no matching `get_to` was found. + ## Template parameters `ValueType` @@ -67,3 +71,4 @@ Depends on the `json_serializer::from_json()` implementation. ## Version history - Since version 3.3.0. +- Added a `static_assert` with a clear message for `const` arguments in version 3.13.0. diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 0123bad29..1090ed105 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -2553,6 +2553,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec ValueType & get_to(ValueType& v) const noexcept(noexcept( JSONSerializer::from_json(std::declval(), v))) { + static_assert(!std::is_const::value, "get_to() cannot deserialize into a const value"); JSONSerializer::from_json(*this, v); return v; } @@ -2578,6 +2579,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec noexcept(noexcept(JSONSerializer::from_json( std::declval(), v))) { + static_assert(!std::is_const::value, "get_to() cannot deserialize into a const value"); JSONSerializer::from_json(*this, v); return v; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4567a2b32..8c3a1319b 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -27198,6 +27198,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec ValueType & get_to(ValueType& v) const noexcept(noexcept( JSONSerializer::from_json(std::declval(), v))) { + static_assert(!std::is_const::value, "get_to() cannot deserialize into a const value"); JSONSerializer::from_json(*this, v); return v; } @@ -27223,6 +27224,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec noexcept(noexcept(JSONSerializer::from_json( std::declval(), v))) { + static_assert(!std::is_const::value, "get_to() cannot deserialize into a const value"); JSONSerializer::from_json(*this, v); return v; }