From 9e1a09eec0242380339b43691fa6f9174caf8ada Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Mon, 28 Sep 2026 17:51:07 +0200 Subject: [PATCH 01/31] Name the key type when rejecting non-string CBOR/MessagePack map keys (#5594) * Name the key type when rejecting non-string CBOR/MessagePack map keys CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings, so such maps are rejected. The error so far was the one for a malformed string (e.g. "expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0xC0" for a nil key), which does not tell the user what went wrong. Report the type of the key instead: syntax error while parsing MessagePack object key: only string keys are supported, but found nil; last byte: 0xC0 The exception id (parse_error.113) and type are unchanged. Malformed string keys and a missing key keep their previous messages. Document the restriction on the CBOR and MessagePack pages. Refs #2766, #3381 Signed-off-by: Niels Lohmann * Point the MessagePack key note to the spec's profile section The note linked to "Serialization: type to format conversion", which says nothing about key types. Restricting map keys to strings is only mentioned in the "Profile" section (under "Future discussion") as an example of a JSON-compatible profile, so link there and describe it as such instead of as a permission. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/from_cbor.md | 4 +- .../docs/api/basic_json/from_msgpack.md | 4 +- .../docs/features/binary_formats/cbor.md | 15 +- .../features/binary_formats/messagepack.md | 15 ++ docs/mkdocs/docs/home/exceptions.md | 11 +- .../nlohmann/detail/input/binary_reader.hpp | 170 +++++++++++++++++- single_include/nlohmann/json.hpp | 170 +++++++++++++++++- tests/src/unit-cbor.cpp | 45 ++++- tests/src/unit-msgpack.cpp | 61 ++++++- tests/src/unit-regression1.cpp | 6 +- 10 files changed, 484 insertions(+), 17 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/from_cbor.md b/docs/mkdocs/docs/api/basic_json/from_cbor.md index b72f55280..8c1062da8 100644 --- a/docs/mkdocs/docs/api/basic_json/from_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/from_cbor.md @@ -80,8 +80,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from CBOR were used in the given input or if the input is not valid CBOR -- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string was expected as a map key, - but not found +- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other + types are not supported, as JSON object keys are always strings) or a string is malformed ## Complexity diff --git a/docs/mkdocs/docs/api/basic_json/from_msgpack.md b/docs/mkdocs/docs/api/basic_json/from_msgpack.md index 2f4b7bb3b..e41edfe7e 100644 --- a/docs/mkdocs/docs/api/basic_json/from_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/from_msgpack.md @@ -73,8 +73,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va the end of the file was not reached when `strict` was set to true - Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from MessagePack were used in the given input or if the input is not valid MessagePack -- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string was expected as a map key, - but not found +- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other + types are not supported, as JSON object keys are always strings) or a string is malformed ## Complexity diff --git a/docs/mkdocs/docs/features/binary_formats/cbor.md b/docs/mkdocs/docs/features/binary_formats/cbor.md index 8e6acf0fb..a488466d4 100644 --- a/docs/mkdocs/docs/features/binary_formats/cbor.md +++ b/docs/mkdocs/docs/features/binary_formats/cbor.md @@ -174,7 +174,20 @@ The library maps CBOR types to JSON value types as follows: !!! warning "Object keys" - CBOR allows map keys of any type, whereas JSON only allows strings as keys in object values. Therefore, CBOR maps with keys other than UTF-8 strings are rejected. + CBOR allows map keys of any type, whereas JSON only allows strings as keys in object values. Therefore, CBOR maps + with keys other than text strings (major type 3) are rejected with a + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with `allow_exceptions` set + to `false`, a discarded value) naming the type of the key that was found, for instance: + + ``` + [json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found an unsigned integer; last byte: 0x01 + ``` + + This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed + on. This is a deliberate restriction of the library's JSON value model, not an oversight: formats built on CBOR + maps with integer keys, such as COSE ([RFC 9052](https://www.rfc-editor.org/rfc/rfc9052.html)) or CWT + ([RFC 8392](https://www.rfc-editor.org/rfc/rfc8392.html)), cannot be read with this library and need a + general-purpose CBOR library instead. !!! warning "UTF-8 validation of text strings" diff --git a/docs/mkdocs/docs/features/binary_formats/messagepack.md b/docs/mkdocs/docs/features/binary_formats/messagepack.md index 0ca82c145..3ce5f7620 100644 --- a/docs/mkdocs/docs/features/binary_formats/messagepack.md +++ b/docs/mkdocs/docs/features/binary_formats/messagepack.md @@ -138,6 +138,21 @@ The library maps MessagePack types to JSON value types as follows: Any MessagePack output created by `to_msgpack` can be successfully parsed by `from_msgpack`. +!!! warning "Object keys" + + MessagePack allows map keys of any type, whereas JSON only allows strings as keys in object values. Like the + JSON-compatible [profile](https://github.com/msgpack/msgpack/blob/master/spec.md#profile) sketched in the + MessagePack specification, this library restricts map keys to `str` values. Maps with keys of any other type are + rejected with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with + `allow_exceptions` set to `false`, a discarded value) naming the type of the key that was found, for instance: + + ``` + [json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack object key: only string keys are supported, but found nil; last byte: 0xC0 + ``` + + This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed + on. Such input needs a general-purpose MessagePack library instead. + !!! warning "UTF-8 validation of string values" The MessagePack specification requires `str` values (`fixstr`, `str 8`, `str 16`, `str 32`) to be valid UTF-8. diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index bf18baab1..407f3c3f1 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -343,13 +343,20 @@ A string could not be read from a [binary format](../features/binary_formats/ind string was read where one was required (for instance as a map key), the string's length specification is invalid, or the string's bytes are not valid UTF-8. +CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings. Maps with keys of any other +type (for instance integers or `null`) are therefore not supported; see the notes on +[CBOR](../features/binary_formats/cbor.md) and [MessagePack](../features/binary_formats/messagepack.md). + !!! failure "Example messages" ``` - [json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xFF + [json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found an unsigned integer; last byte: 0x01 ``` ``` - [json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack string: expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0xFF + [json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack object key: only string keys are supported, but found nil; last byte: 0xC0 + ``` + ``` + [json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C ``` ``` [json.exception.parse_error.113] parse error at byte 2: syntax error while parsing UBJSON char: byte after 'C' must be in range 0x00..0x7F; last byte: 0x82 diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index b0675c626..b9e6b304b 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -1324,6 +1324,80 @@ class binary_reader } } + /*! + @brief reads a CBOR object key + + RFC 8949 allows any data item as a map key, but only strings have a + counterpart in JSON. A key of any other type is rejected with a message + naming that type, rather than the one @ref get_cbor_string gives for a + malformed string. + + @param[out] result created key + + @return whether key creation completed + */ + bool get_cbor_object_key(string_t& result) + { + // EOF and major type 3 (text string) are left to get_cbor_string + if (current == char_traits::eof() || (static_cast(current) & 0xE0u) == 0x60u) + { + return get_cbor_string(result); + } + + const char* found = nullptr; + switch (static_cast(current) >> 5u) + { + case 0: + found = "an unsigned integer"; + break; + case 1: + found = "a negative integer"; + break; + case 2: + found = "a byte string"; + break; + case 4: + found = "an array"; + break; + case 5: + found = "a map"; + break; + case 6: + found = "a tag"; + break; + default: // major type 7 + switch (current) + { + case 0xF4: + case 0xF5: + found = "a boolean"; + break; + case 0xF6: + found = "null"; + break; + case 0xF7: + found = "undefined"; + break; + case 0xF9: + case 0xFA: + case 0xFB: + found = "a floating-point number"; + break; + case 0xFF: + found = "a break stop code"; + break; + default: + found = "a simple value"; + break; + } + break; + } + + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format_t::cbor, concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr)); + } + /*! @brief reads a definite-length CBOR byte array @@ -1568,7 +1642,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_cbor_object_key(key) || !sax->key(key))) { return false; } @@ -2069,6 +2143,98 @@ class binary_reader } } + /*! + @brief reads a MessagePack object key + + The MessagePack specification allows any type as a map key, but only + strings have a counterpart in JSON. A key of any other type is rejected + with a message naming that type, rather than the one @ref + get_msgpack_string gives for a malformed string. + + @param[out] result created key + + @return whether key creation completed + */ + bool get_msgpack_object_key(string_t& result) + { + const char* found = nullptr; + switch (current) + { + case 0xC0: + found = "nil"; + break; + case 0xC2: + case 0xC3: + found = "a boolean"; + break; + case 0xCA: + case 0xCB: + found = "a float"; + break; + case 0xC4: + case 0xC5: + case 0xC6: + found = "a bin"; + break; + case 0xC7: + case 0xC8: + case 0xC9: + case 0xD4: + case 0xD5: + case 0xD6: + case 0xD7: + case 0xD8: + found = "an ext"; + break; + case 0xCC: + case 0xCD: + case 0xCE: + case 0xCF: + case 0xD0: + case 0xD1: + case 0xD2: + case 0xD3: + found = "an integer"; + break; + case 0xDC: + case 0xDD: + found = "an array"; + break; + case 0xDE: + case 0xDF: + found = "a map"; + break; + default: + // fixint, fixmap, and fixarray; strings, EOF, and the unused + // byte 0xC1 are left to get_msgpack_string + if (current == char_traits::eof()) + { + return get_msgpack_string(result); + } + if (current <= 0x7F || current >= 0xE0) + { + found = "an integer"; + } + else if (current <= 0x8F) + { + found = "a map"; + } + else if (current <= 0x9F) + { + found = "an array"; + } + else + { + return get_msgpack_string(result); + } + break; + } + + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format_t::msgpack, concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr)); + } + /*! @brief reads a MessagePack byte array @@ -2231,7 +2397,7 @@ class binary_reader { get(); key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_msgpack_object_key(key) || !sax->key(key))) { return false; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 0e3cae486..591a00b74 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -14059,6 +14059,80 @@ class binary_reader } } + /*! + @brief reads a CBOR object key + + RFC 8949 allows any data item as a map key, but only strings have a + counterpart in JSON. A key of any other type is rejected with a message + naming that type, rather than the one @ref get_cbor_string gives for a + malformed string. + + @param[out] result created key + + @return whether key creation completed + */ + bool get_cbor_object_key(string_t& result) + { + // EOF and major type 3 (text string) are left to get_cbor_string + if (current == char_traits::eof() || (static_cast(current) & 0xE0u) == 0x60u) + { + return get_cbor_string(result); + } + + const char* found = nullptr; + switch (static_cast(current) >> 5u) + { + case 0: + found = "an unsigned integer"; + break; + case 1: + found = "a negative integer"; + break; + case 2: + found = "a byte string"; + break; + case 4: + found = "an array"; + break; + case 5: + found = "a map"; + break; + case 6: + found = "a tag"; + break; + default: // major type 7 + switch (current) + { + case 0xF4: + case 0xF5: + found = "a boolean"; + break; + case 0xF6: + found = "null"; + break; + case 0xF7: + found = "undefined"; + break; + case 0xF9: + case 0xFA: + case 0xFB: + found = "a floating-point number"; + break; + case 0xFF: + found = "a break stop code"; + break; + default: + found = "a simple value"; + break; + } + break; + } + + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format_t::cbor, concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr)); + } + /*! @brief reads a definite-length CBOR byte array @@ -14303,7 +14377,7 @@ class binary_reader if (top.is_object) { key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_cbor_object_key(key) || !sax->key(key))) { return false; } @@ -14804,6 +14878,98 @@ class binary_reader } } + /*! + @brief reads a MessagePack object key + + The MessagePack specification allows any type as a map key, but only + strings have a counterpart in JSON. A key of any other type is rejected + with a message naming that type, rather than the one @ref + get_msgpack_string gives for a malformed string. + + @param[out] result created key + + @return whether key creation completed + */ + bool get_msgpack_object_key(string_t& result) + { + const char* found = nullptr; + switch (current) + { + case 0xC0: + found = "nil"; + break; + case 0xC2: + case 0xC3: + found = "a boolean"; + break; + case 0xCA: + case 0xCB: + found = "a float"; + break; + case 0xC4: + case 0xC5: + case 0xC6: + found = "a bin"; + break; + case 0xC7: + case 0xC8: + case 0xC9: + case 0xD4: + case 0xD5: + case 0xD6: + case 0xD7: + case 0xD8: + found = "an ext"; + break; + case 0xCC: + case 0xCD: + case 0xCE: + case 0xCF: + case 0xD0: + case 0xD1: + case 0xD2: + case 0xD3: + found = "an integer"; + break; + case 0xDC: + case 0xDD: + found = "an array"; + break; + case 0xDE: + case 0xDF: + found = "a map"; + break; + default: + // fixint, fixmap, and fixarray; strings, EOF, and the unused + // byte 0xC1 are left to get_msgpack_string + if (current == char_traits::eof()) + { + return get_msgpack_string(result); + } + if (current <= 0x7F || current >= 0xE0) + { + found = "an integer"; + } + else if (current <= 0x8F) + { + found = "a map"; + } + else if (current <= 0x9F) + { + found = "an array"; + } + else + { + return get_msgpack_string(result); + } + break; + } + + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format_t::msgpack, concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr)); + } + /*! @brief reads a MessagePack byte array @@ -14966,7 +15132,7 @@ class binary_reader { get(); key.clear(); - if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key))) + if (JSON_HEDLEY_UNLIKELY(!get_msgpack_object_key(key) || !sax->key(key))) { return false; } diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index 6bd792f8a..fe0fb2644 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -1830,10 +1830,51 @@ TEST_CASE("CBOR") SECTION("invalid string in map") { json _; - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0xa1, 0xff, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xFF", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0xa1, 0xff, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found a break stop code; last byte: 0xFF", json::parse_error&); CHECK(json::from_cbor(std::vector({0xa1, 0xff, 0x01}), true, false).is_discarded()); } + SECTION("non-string key (see #2766 and #3381)") + { + // only text strings map to JSON object keys; any other key is + // rejected with a message naming its type + const std::vector, std::string>> cases = + { + {{0xA1, 0x01, 0x01}, "an unsigned integer; last byte: 0x01"}, + {{0xA1, 0x20, 0x01}, "a negative integer; last byte: 0x20"}, + {{0xA1, 0x41, 0x61, 0x01}, "a byte string; last byte: 0x41"}, + {{0xA1, 0x80, 0x01}, "an array; last byte: 0x80"}, + {{0xA1, 0xA0, 0x01}, "a map; last byte: 0xA0"}, + {{0xA1, 0xC0, 0x61, 0x61, 0x01}, "a tag; last byte: 0xC0"}, + {{0xA1, 0xF4, 0x01}, "a boolean; last byte: 0xF4"}, + {{0xA1, 0xF5, 0x01}, "a boolean; last byte: 0xF5"}, + {{0xA1, 0xF6, 0x01}, "null; last byte: 0xF6"}, + {{0xA1, 0xF7, 0x01}, "undefined; last byte: 0xF7"}, + {{0xA1, 0xF9, 0x3C, 0x00, 0x01}, "a floating-point number; last byte: 0xF9"}, + {{0xA1, 0xFA, 0x3F, 0x80, 0x00, 0x00, 0x01}, "a floating-point number; last byte: 0xFA"}, + {{0xA1, 0xFB, 0x3F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01}, "a floating-point number; last byte: 0xFB"}, + {{0xA1, 0xE0, 0x01}, "a simple value; last byte: 0xE0"}, + {{0xA1, 0xF8, 0x20, 0x01}, "a simple value; last byte: 0xF8"}, + // indefinite-length map + {{0xBF, 0x01, 0x01, 0xFF}, "an unsigned integer; last byte: 0x01"}, + }; + + for (const auto& c : cases) + { + CAPTURE(c.first) + const std::string expected = "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found " + c.second; + json _; + CHECK_THROWS_WITH_AS(_ = json::from_cbor(c.first), expected.c_str(), json::parse_error&); + CHECK(json::from_cbor(c.first, true, false).is_discarded()); + } + + // a key of major type 3 with a reserved length is still reported as + // a malformed string, and a missing key as the end of input + json _; + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0xA1})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing CBOR string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&); + } + SECTION("invalid UTF-8 in string (see #5529)") { // a two-character text string (major type 3) whose bytes are not @@ -2284,7 +2325,7 @@ TEST_CASE("CBOR indefinite-length strings do not recurse per chunk") SECTION("a break marker outside an indefinite-length string is not a string") { // 0xFF only closes a string that was opened; on its own it is not one - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0xA1, 0xFF, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xFF", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0xA1, 0xFF, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found a break stop code; last byte: 0xFF", json::parse_error&); } } diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index de4255b4a..498dec859 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -1551,10 +1551,69 @@ TEST_CASE("MessagePack") SECTION("invalid string in map") { json _; - CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0x81, 0xff, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack string: expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0xFF", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0x81, 0xff, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack object key: only string keys are supported, but found an integer; last byte: 0xFF", json::parse_error&); CHECK(json::from_msgpack(std::vector({0x81, 0xff, 0x01}), true, false).is_discarded()); } + SECTION("non-string key (see #3381)") + { + // only strings map to JSON object keys; any other key is rejected + // with a message naming its type + const std::vector, std::string>> cases = + { + {{0x81, 0xC0, 0x01}, "nil; last byte: 0xC0"}, + {{0x81, 0xC2, 0x01}, "a boolean; last byte: 0xC2"}, + {{0x81, 0xC3, 0x01}, "a boolean; last byte: 0xC3"}, + {{0x81, 0xCA, 0x3F, 0x80, 0x00, 0x00, 0x01}, "a float; last byte: 0xCA"}, + {{0x81, 0xCB, 0x3F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01}, "a float; last byte: 0xCB"}, + {{0x81, 0xC4, 0x00, 0x01}, "a bin; last byte: 0xC4"}, + {{0x81, 0xC5, 0x00, 0x00, 0x01}, "a bin; last byte: 0xC5"}, + {{0x81, 0xC6, 0x00, 0x00, 0x00, 0x00, 0x01}, "a bin; last byte: 0xC6"}, + {{0x81, 0xC7, 0x00, 0x01, 0x01}, "an ext; last byte: 0xC7"}, + {{0x81, 0xC8, 0x00, 0x00, 0x01, 0x01}, "an ext; last byte: 0xC8"}, + {{0x81, 0xC9, 0x00, 0x00, 0x00, 0x00, 0x01, 0x01}, "an ext; last byte: 0xC9"}, + {{0x81, 0xD4, 0x01, 0x00, 0x01}, "an ext; last byte: 0xD4"}, + {{0x81, 0xD5, 0x01, 0x00, 0x00, 0x01}, "an ext; last byte: 0xD5"}, + {{0x81, 0xD6, 0x01, 0x00, 0x00, 0x00, 0x00, 0x01}, "an ext; last byte: 0xD6"}, + {{0x81, 0xD7, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01}, "an ext; last byte: 0xD7"}, + {{0x81, 0xD8, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01}, "an ext; last byte: 0xD8"}, + {{0x81, 0xCC, 0x01, 0x01}, "an integer; last byte: 0xCC"}, + {{0x81, 0xCD, 0x00, 0x01, 0x01}, "an integer; last byte: 0xCD"}, + {{0x81, 0xCE, 0x00, 0x00, 0x00, 0x01, 0x01}, "an integer; last byte: 0xCE"}, + {{0x81, 0xCF, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x01}, "an integer; last byte: 0xCF"}, + {{0x81, 0xD0, 0x01, 0x01}, "an integer; last byte: 0xD0"}, + {{0x81, 0xD1, 0x00, 0x01, 0x01}, "an integer; last byte: 0xD1"}, + {{0x81, 0xD2, 0x00, 0x00, 0x00, 0x01, 0x01}, "an integer; last byte: 0xD2"}, + {{0x81, 0xD3, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x01}, "an integer; last byte: 0xD3"}, + {{0x81, 0x00, 0x01}, "an integer; last byte: 0x00"}, + {{0x81, 0x7F, 0x01}, "an integer; last byte: 0x7F"}, + {{0x81, 0xE0, 0x01}, "an integer; last byte: 0xE0"}, + {{0x81, 0x80, 0x01}, "a map; last byte: 0x80"}, + {{0x81, 0x8F, 0x01}, "a map; last byte: 0x8F"}, + {{0x81, 0xDE, 0x00, 0x00, 0x01}, "a map; last byte: 0xDE"}, + {{0x81, 0xDF, 0x00, 0x00, 0x00, 0x00, 0x01}, "a map; last byte: 0xDF"}, + {{0x81, 0x90, 0x01}, "an array; last byte: 0x90"}, + {{0x81, 0x9F, 0x01}, "an array; last byte: 0x9F"}, + {{0x81, 0xDC, 0x00, 0x00, 0x01}, "an array; last byte: 0xDC"}, + {{0x81, 0xDD, 0x00, 0x00, 0x00, 0x00, 0x01}, "an array; last byte: 0xDD"}, + }; + + for (const auto& c : cases) + { + CAPTURE(c.first) + const std::string expected = "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack object key: only string keys are supported, but found " + c.second; + json _; + CHECK_THROWS_WITH_AS(_ = json::from_msgpack(c.first), expected.c_str(), json::parse_error&); + CHECK(json::from_msgpack(c.first, true, false).is_discarded()); + } + + json _; + // the unused byte 0xC1 is still reported as a malformed string + CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0x81, 0xC1, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack string: expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0xC1", json::parse_error&); + // a missing key is still reported as the end of input + CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0x81})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing MessagePack string: unexpected end of input", json::parse_error&); + } + SECTION("invalid UTF-8 in string (see #5529)") { // a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8 diff --git a/tests/src/unit-regression1.cpp b/tests/src/unit-regression1.cpp index 0529f83dd..43cd18438 100644 --- a/tests/src/unit-regression1.cpp +++ b/tests/src/unit-regression1.cpp @@ -1018,7 +1018,7 @@ TEST_CASE("regression tests 1") }; json _; - CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x98", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found an array; last byte: 0x98", json::parse_error&); // related test case: nonempty UTF-8 string (indefinite length) std::vector const vec1 {0x7f, 0x61, 0x61}; @@ -1065,7 +1065,7 @@ TEST_CASE("regression tests 1") }; json _; - CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec1), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xB4", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec1), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing CBOR object key: only string keys are supported, but found a map; last byte: 0xB4", json::parse_error&); // related test case: double-precision std::vector const vec2 @@ -1077,7 +1077,7 @@ TEST_CASE("regression tests 1") 0x96, 0x96, 0xb4, 0xb4, 0xfa, 0x94, 0x94, 0x61, 0x61, 0x61, 0x61, 0x61, 0x61, 0x61, 0x61, 0xfb }; - CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec2), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xB4", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec2), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing CBOR object key: only string keys are supported, but found a map; last byte: 0xB4", json::parse_error&); } SECTION("issue #452 - Heap-buffer-overflow (OSS-Fuzz issue 585)") From fc03b9912eab296efbfc31f0b7da5568b7c9bb53 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Mon, 28 Sep 2026 17:56:11 +0200 Subject: [PATCH 02/31] Look up the locale decimal point at conversion time, not lexer construction (#5597) * Look up the locale decimal point at conversion time, not lexer construction The lexer read localeconv()->decimal_point once in its constructor and wrote that character into token_buffer in place of '.'. The strtod fallback then used the locale current at conversion time, so an LC_NUMERIC change in between (parser callback, SAX handler, another thread) truncated the value in release builds and fired the endptr assertion in debug builds. token_buffer now always holds '.'. Only the strtof/strtod/strtold fallback depends on the locale: it looks up the decimal point right before the call, restores '.' afterwards, and repeats the conversion if the locale changed in between. As a side effect, std::from_chars and Clinger's fast path now also apply under locales whose decimal point is not '.'. Fixes #5198 Signed-off-by: Niels Lohmann * Stop the strtod retry loop when the decimal point is unchanged convert_float_locale_aware() repeated the conversion until strtod consumed the whole token, assuming an early stop can only mean a locale change. Under a locale whose decimal point is not a single character (e.g. the two-byte U+066B of ar_EG.UTF-8, ar_SA.UTF-8, or fa_IR.UTF-8, all available on macOS), the in-place substitution can never succeed, so parsing any float that reaches the strtod fallback (for example 3.14159265358979323846 at C++11) hung forever. Before this branch, the same input was truncated. Retry only if the decimal point changed since the previous attempt; otherwise keep the value strtod parsed so far, as before. Add a test that parses such numbers under a multi-byte decimal point locale; it hangs without this change. Signed-off-by: Niels Lohmann * Fix -Weffc++ errors in the #5198 locale test GCC's -Weffc++ (an error in ci_test_gcc and ci_test_standards_gcc) rejected LocaleSwitchingSax: it has a pointer data member but does not declare its copy operations, and its vectors are not initialized in the member initializer list. Store the locale name as a std::string and give the vectors brace initializers, like SaxEventLogger in unit-deserialization.cpp. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/detail/input/lexer.hpp | 115 ++++++---- .../nlohmann/detail/input/number_parse.hpp | 21 +- single_include/nlohmann/json.hpp | 136 +++++++----- tests/src/unit-class_lexer.cpp | 2 +- tests/src/unit-locale-cpp.cpp | 210 ++++++++++++++++++ 5 files changed, 379 insertions(+), 105 deletions(-) diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index 98c0fd76a..00a964a17 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -206,7 +206,6 @@ class lexer : public lexer_base explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept : ia(std::move(adapter)) , ignore_comments(ignore_comments_) - , decimal_point_char(static_cast(get_decimal_point())) , discard_number_values(discard_number_values_) {} @@ -222,8 +221,7 @@ class lexer : public lexer_base // locales ///////////////////// - /// return the locale-dependent decimal point - JSON_HEDLEY_PURE + /// return the decimal point of the current locale static char get_decimal_point() noexcept { const auto* loc = localeconv(); @@ -1092,9 +1090,10 @@ class lexer : public lexer_base token_type::value_float if number could be successfully scanned, token_type::parse_error otherwise - @note The scanner is independent of the current locale. Internally, the - locale's decimal point is used instead of `.` to work with the - locale-dependent converters. + @note The scanner is independent of the current locale: token_buffer + always holds `.`. Only the std::strtod fallback of convert_number() + depends on the locale, and it looks up the decimal point right + before converting (see convert_float_locale_aware()). */ token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated. { @@ -1183,7 +1182,7 @@ scan_number_zero: { case '.': { - add(decimal_point_char); + add(current); decimal_point_position = token_buffer.size() - 1; goto scan_number_decimal1; } @@ -1220,7 +1219,7 @@ scan_number_any1: case '.': { - add(decimal_point_char); + add(current); decimal_point_position = token_buffer.size() - 1; goto scan_number_decimal1; } @@ -1462,9 +1461,9 @@ scan_number_done: // Only a number below 1 can carry further insignificant zeros, and only // while the count stays at the limit does removing them change the - // answer - so this loop is skipped for all but a few tokens. Note - // token_buffer holds the locale's decimal point, so the fraction is - // located through decimal_point_position rather than by searching '.'. + // answer - so this loop is skipped for all but a few tokens. The + // fraction is located through decimal_point_position rather than by + // searching '.'. if (lead_zero != 0) { JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit @@ -1482,8 +1481,8 @@ scan_number_done: @brief convert the number text in token_buffer to its value and token type The digit sequence in token_buffer has already been validated (by the - scan_number() state machine or by the contiguous fast path) and holds the - locale decimal point in place of '.'. Integers are parsed first and fall + scan_number() state machine or by the contiguous fast path) and holds '.' + as decimal point, independent of the locale. Integers are parsed first and fall back to floating point on overflow. This is shared so both scanners produce identical results. @@ -1563,7 +1562,7 @@ scan_number_done: // integer conversion above overflowed. Prefer std::from_chars // (Eisel-Lemire, locale-independent, correctly rounded) when available; // otherwise the exact Clinger fast path (double only); otherwise the - // locale-aware strtof/strtod. + // locale-aware strtof/strtod/strtold. if (parse_float_from_chars(num_begin, num_end, value_float)) { return token_type::value_float; @@ -1572,26 +1571,75 @@ scan_number_done: // extra pass over the token's bytes, which otherwise shows up on // high-precision inputs such as canada.json if (mantissa_fits_clinger(mantissa_end) - && parse_float_fast(num_begin, num_end, decimal_point_char, value_float)) + && parse_float_fast(num_begin, num_end, value_float)) { return token_type::value_float; } - char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) - strtof(value_float, token_buffer.data(), &endptr); - - // we checked the number format before - JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); - + convert_float_locale_aware(); return token_type::value_float; } + /*! + @brief convert the float in token_buffer with strtof/strtod/strtold + + These functions expect the decimal point of the *current* locale, so it is + looked up right before the conversion instead of once when the lexer is + constructed: a locale change in between (by a parser callback, a SAX + handler, or another thread) must not truncate the value (#5198). The + token has been validated before, so if the conversion stops early and the + decimal point changed in the meantime, the locale changed between the + lookup and the call, and the conversion is repeated with the new decimal + point. If the decimal point did not change, a retry cannot succeed: the + locale's decimal point is not a single character (e.g., the two-byte + U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place. + The value strtod parsed up to that point is kept, as before this change. + + Note that changing the locale in another thread *while* strtod runs is + undefined behavior of the C library, which this function cannot prevent. + */ + void convert_float_locale_aware() + { + const bool has_dot = decimal_point_position != std::string::npos; + char decimal_point = get_decimal_point(); + for (;;) + { + const bool substitute = has_dot && decimal_point != '.'; + if (substitute) + { + token_buffer[decimal_point_position] = static_cast(decimal_point); + } + + char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) + strtof(value_float, token_buffer.data(), &endptr); + + if (substitute) + { + // get_string() hands the token to the SAX interface with '.' + token_buffer[decimal_point_position] = '.'; + } + + if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size())) + { + return; + } + + // retry only if the locale changed; otherwise, this would loop forever + const char current_decimal_point = get_decimal_point(); + if (current_decimal_point == decimal_point) + { + return; + } + decimal_point = current_decimal_point; + } + } + /*! @brief contiguous fast path for scanning a number Parses the whole number token straight from the input buffer, avoiding the per-character get()/add() of scan_number(). On success it fills token_buffer - (with the locale decimal point substituted, as scan_number() does) and + (as scan_number() does) and returns the token type. On anything it does not fully recognize as a well-formed number it makes no state change and returns token_type::uninitialized, so the caller falls back to scan_number(), which @@ -1707,16 +1755,11 @@ scan_number_done: } #endif - // materialize the token exactly as scan_number() would, substituting the - // locale decimal point so convert_number()'s strtof fallback stays valid. - // reset() already cleared token_buffer, so append() fills it (assign() is - // avoided because custom string_t types need not provide it) + // materialize the token exactly as scan_number() would. reset() already + // cleared token_buffer, so append() fills it (assign() is avoided + // because custom string_t types need not provide it) token_buffer.append(reinterpret_cast(data), len); - if (dot_index != std::string::npos) - { - token_buffer[dot_index] = static_cast(decimal_point_char); - decimal_point_position = dot_index; - } + decimal_point_position = dot_index; ia.bulk_skip(len - 1); position.chars_read_total += (len - 1); @@ -1983,11 +2026,7 @@ scan_number_done: /// return current string value (implicitly resets the token; useful only once) string_t& get_string() { - // translate decimal points from locale back to '.' (#4084) - if (decimal_point_char != '.' && decimal_point_position != std::string::npos) - { - token_buffer[decimal_point_position] = '.'; - } + // a number token holds '.' regardless of the locale (#4084) return token_buffer; } @@ -2283,9 +2322,7 @@ scan_number_done: number_unsigned_t value_unsigned = 0; number_float_t value_float = 0; - /// the decimal point - const char_int_type decimal_point_char = '.'; - /// the position of the decimal point in the input + /// the position of the decimal point in token_buffer std::size_t decimal_point_position = std::string::npos; /// whether the caller (e.g. accept()/json_sax_acceptor) only needs the diff --git a/include/nlohmann/detail/input/number_parse.hpp b/include/nlohmann/detail/input/number_parse.hpp index e50c3f67f..25f6cac91 100644 --- a/include/nlohmann/detail/input/number_parse.hpp +++ b/include/nlohmann/detail/input/number_parse.hpp @@ -118,14 +118,12 @@ std::strtod. The parser only activates for number_float_t == double; float and long double keep the std::strtof/std::strtold paths (see the templated overload below). -@param[in] first pointer to the first character of the number -@param[in] last pointer past the last character -@param[in] decimal_point the (locale-dependent) decimal point character -@param[out] out the parsed value on success +@param[in] first pointer to the first character of the number +@param[in] last pointer past the last character +@param[out] out the parsed value on success @return true if the value was parsed exactly; false to fall back to strtod */ -template -bool parse_float_fast(const char* first, const char* last, DecimalPointType decimal_point, double& out) noexcept +inline bool parse_float_fast(const char* first, const char* last, double& out) noexcept { #if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0 // Clinger's fast path is only exact when double operations are evaluated in @@ -136,7 +134,6 @@ bool parse_float_fast(const char* first, const char* last, DecimalPointType deci // std::from_chars / std::strtod path. static_cast(first); static_cast(last); - static_cast(decimal_point); static_cast(out); return false; #else @@ -175,7 +172,7 @@ bool parse_float_fast(const char* first, const char* last, DecimalPointType deci ++num_digits; fractional_digits += static_cast(seen_dot); } - else if (static_cast(c) == decimal_point) + else if (c == '.') { if (JSON_HEDLEY_UNLIKELY(seen_dot)) { @@ -260,8 +257,8 @@ bool parse_float_fast(const char* first, const char* last, DecimalPointType deci } /// fast float path is only exact for `double`; decline for float/long double -template -bool parse_float_fast(const char* /*first*/, const char* /*last*/, DecimalPointType /*decimal_point*/, FloatType& /*out*/) noexcept +template +bool parse_float_fast(const char* /*first*/, const char* /*last*/, FloatType& /*out*/) noexcept { return false; } @@ -273,9 +270,7 @@ std::from_chars is locale-independent, correctly rounded, and - via the Eisel-Lemire algorithm in modern standard libraries - much faster than strtod over the whole value range (not just the Clinger subset). It is used only when __cpp_lib_to_chars indicates full floating-point support and only when it -consumes the entire token ([first, last)); a partial parse means the buffer -uses a non-'.' locale decimal point, in which case the caller falls back to the -locale-aware path. An under-/overflow (result_out_of_range) also declines, so +consumes the entire token ([first, last)). An under-/overflow (result_out_of_range) also declines, so the caller's strtod fallback supplies the well-defined ±inf/0 result the parser expects (side-stepping the P4168 divergence between implementations). diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 591a00b74..576498738 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -8605,14 +8605,12 @@ std::strtod. The parser only activates for number_float_t == double; float and long double keep the std::strtof/std::strtold paths (see the templated overload below). -@param[in] first pointer to the first character of the number -@param[in] last pointer past the last character -@param[in] decimal_point the (locale-dependent) decimal point character -@param[out] out the parsed value on success +@param[in] first pointer to the first character of the number +@param[in] last pointer past the last character +@param[out] out the parsed value on success @return true if the value was parsed exactly; false to fall back to strtod */ -template -bool parse_float_fast(const char* first, const char* last, DecimalPointType decimal_point, double& out) noexcept +inline bool parse_float_fast(const char* first, const char* last, double& out) noexcept { #if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0 // Clinger's fast path is only exact when double operations are evaluated in @@ -8623,7 +8621,6 @@ bool parse_float_fast(const char* first, const char* last, DecimalPointType deci // std::from_chars / std::strtod path. static_cast(first); static_cast(last); - static_cast(decimal_point); static_cast(out); return false; #else @@ -8662,7 +8659,7 @@ bool parse_float_fast(const char* first, const char* last, DecimalPointType deci ++num_digits; fractional_digits += static_cast(seen_dot); } - else if (static_cast(c) == decimal_point) + else if (c == '.') { if (JSON_HEDLEY_UNLIKELY(seen_dot)) { @@ -8747,8 +8744,8 @@ bool parse_float_fast(const char* first, const char* last, DecimalPointType deci } /// fast float path is only exact for `double`; decline for float/long double -template -bool parse_float_fast(const char* /*first*/, const char* /*last*/, DecimalPointType /*decimal_point*/, FloatType& /*out*/) noexcept +template +bool parse_float_fast(const char* /*first*/, const char* /*last*/, FloatType& /*out*/) noexcept { return false; } @@ -8760,9 +8757,7 @@ std::from_chars is locale-independent, correctly rounded, and - via the Eisel-Lemire algorithm in modern standard libraries - much faster than strtod over the whole value range (not just the Clinger subset). It is used only when __cpp_lib_to_chars indicates full floating-point support and only when it -consumes the entire token ([first, last)); a partial parse means the buffer -uses a non-'.' locale decimal point, in which case the caller falls back to the -locale-aware path. An under-/overflow (result_out_of_range) also declines, so +consumes the entire token ([first, last)). An under-/overflow (result_out_of_range) also declines, so the caller's strtod fallback supplies the well-defined ±inf/0 result the parser expects (side-stepping the P4168 divergence between implementations). @@ -9303,7 +9298,6 @@ class lexer : public lexer_base explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept : ia(std::move(adapter)) , ignore_comments(ignore_comments_) - , decimal_point_char(static_cast(get_decimal_point())) , discard_number_values(discard_number_values_) {} @@ -9319,8 +9313,7 @@ class lexer : public lexer_base // locales ///////////////////// - /// return the locale-dependent decimal point - JSON_HEDLEY_PURE + /// return the decimal point of the current locale static char get_decimal_point() noexcept { const auto* loc = localeconv(); @@ -10189,9 +10182,10 @@ class lexer : public lexer_base token_type::value_float if number could be successfully scanned, token_type::parse_error otherwise - @note The scanner is independent of the current locale. Internally, the - locale's decimal point is used instead of `.` to work with the - locale-dependent converters. + @note The scanner is independent of the current locale: token_buffer + always holds `.`. Only the std::strtod fallback of convert_number() + depends on the locale, and it looks up the decimal point right + before converting (see convert_float_locale_aware()). */ token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated. { @@ -10280,7 +10274,7 @@ scan_number_zero: { case '.': { - add(decimal_point_char); + add(current); decimal_point_position = token_buffer.size() - 1; goto scan_number_decimal1; } @@ -10317,7 +10311,7 @@ scan_number_any1: case '.': { - add(decimal_point_char); + add(current); decimal_point_position = token_buffer.size() - 1; goto scan_number_decimal1; } @@ -10559,9 +10553,9 @@ scan_number_done: // Only a number below 1 can carry further insignificant zeros, and only // while the count stays at the limit does removing them change the - // answer - so this loop is skipped for all but a few tokens. Note - // token_buffer holds the locale's decimal point, so the fraction is - // located through decimal_point_position rather than by searching '.'. + // answer - so this loop is skipped for all but a few tokens. The + // fraction is located through decimal_point_position rather than by + // searching '.'. if (lead_zero != 0) { JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit @@ -10579,8 +10573,8 @@ scan_number_done: @brief convert the number text in token_buffer to its value and token type The digit sequence in token_buffer has already been validated (by the - scan_number() state machine or by the contiguous fast path) and holds the - locale decimal point in place of '.'. Integers are parsed first and fall + scan_number() state machine or by the contiguous fast path) and holds '.' + as decimal point, independent of the locale. Integers are parsed first and fall back to floating point on overflow. This is shared so both scanners produce identical results. @@ -10660,7 +10654,7 @@ scan_number_done: // integer conversion above overflowed. Prefer std::from_chars // (Eisel-Lemire, locale-independent, correctly rounded) when available; // otherwise the exact Clinger fast path (double only); otherwise the - // locale-aware strtof/strtod. + // locale-aware strtof/strtod/strtold. if (parse_float_from_chars(num_begin, num_end, value_float)) { return token_type::value_float; @@ -10669,26 +10663,75 @@ scan_number_done: // extra pass over the token's bytes, which otherwise shows up on // high-precision inputs such as canada.json if (mantissa_fits_clinger(mantissa_end) - && parse_float_fast(num_begin, num_end, decimal_point_char, value_float)) + && parse_float_fast(num_begin, num_end, value_float)) { return token_type::value_float; } - char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) - strtof(value_float, token_buffer.data(), &endptr); - - // we checked the number format before - JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); - + convert_float_locale_aware(); return token_type::value_float; } + /*! + @brief convert the float in token_buffer with strtof/strtod/strtold + + These functions expect the decimal point of the *current* locale, so it is + looked up right before the conversion instead of once when the lexer is + constructed: a locale change in between (by a parser callback, a SAX + handler, or another thread) must not truncate the value (#5198). The + token has been validated before, so if the conversion stops early and the + decimal point changed in the meantime, the locale changed between the + lookup and the call, and the conversion is repeated with the new decimal + point. If the decimal point did not change, a retry cannot succeed: the + locale's decimal point is not a single character (e.g., the two-byte + U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place. + The value strtod parsed up to that point is kept, as before this change. + + Note that changing the locale in another thread *while* strtod runs is + undefined behavior of the C library, which this function cannot prevent. + */ + void convert_float_locale_aware() + { + const bool has_dot = decimal_point_position != std::string::npos; + char decimal_point = get_decimal_point(); + for (;;) + { + const bool substitute = has_dot && decimal_point != '.'; + if (substitute) + { + token_buffer[decimal_point_position] = static_cast(decimal_point); + } + + char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) + strtof(value_float, token_buffer.data(), &endptr); + + if (substitute) + { + // get_string() hands the token to the SAX interface with '.' + token_buffer[decimal_point_position] = '.'; + } + + if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size())) + { + return; + } + + // retry only if the locale changed; otherwise, this would loop forever + const char current_decimal_point = get_decimal_point(); + if (current_decimal_point == decimal_point) + { + return; + } + decimal_point = current_decimal_point; + } + } + /*! @brief contiguous fast path for scanning a number Parses the whole number token straight from the input buffer, avoiding the per-character get()/add() of scan_number(). On success it fills token_buffer - (with the locale decimal point substituted, as scan_number() does) and + (as scan_number() does) and returns the token type. On anything it does not fully recognize as a well-formed number it makes no state change and returns token_type::uninitialized, so the caller falls back to scan_number(), which @@ -10804,16 +10847,11 @@ scan_number_done: } #endif - // materialize the token exactly as scan_number() would, substituting the - // locale decimal point so convert_number()'s strtof fallback stays valid. - // reset() already cleared token_buffer, so append() fills it (assign() is - // avoided because custom string_t types need not provide it) + // materialize the token exactly as scan_number() would. reset() already + // cleared token_buffer, so append() fills it (assign() is avoided + // because custom string_t types need not provide it) token_buffer.append(reinterpret_cast(data), len); - if (dot_index != std::string::npos) - { - token_buffer[dot_index] = static_cast(decimal_point_char); - decimal_point_position = dot_index; - } + decimal_point_position = dot_index; ia.bulk_skip(len - 1); position.chars_read_total += (len - 1); @@ -11080,11 +11118,7 @@ scan_number_done: /// return current string value (implicitly resets the token; useful only once) string_t& get_string() { - // translate decimal points from locale back to '.' (#4084) - if (decimal_point_char != '.' && decimal_point_position != std::string::npos) - { - token_buffer[decimal_point_position] = '.'; - } + // a number token holds '.' regardless of the locale (#4084) return token_buffer; } @@ -11380,9 +11414,7 @@ scan_number_done: number_unsigned_t value_unsigned = 0; number_float_t value_float = 0; - /// the decimal point - const char_int_type decimal_point_char = '.'; - /// the position of the decimal point in the input + /// the position of the decimal point in token_buffer std::size_t decimal_point_position = std::string::npos; /// whether the caller (e.g. accept()/json_sax_acceptor) only needs the diff --git a/tests/src/unit-class_lexer.cpp b/tests/src/unit-class_lexer.cpp index cb8ceed6b..5d52179d7 100644 --- a/tests/src/unit-class_lexer.cpp +++ b/tests/src/unit-class_lexer.cpp @@ -666,7 +666,7 @@ TEST_CASE("parse_float_fast declines what it cannot convert exactly") // always safe: the caller then falls back to a slower, exact conversion. const auto fast = [](const std::string & s, double & out) { - return nlohmann::detail::parse_float_fast(s.data(), s.data() + s.size(), '.', out); + return nlohmann::detail::parse_float_fast(s.data(), s.data() + s.size(), out); }; double out = 0; diff --git a/tests/src/unit-locale-cpp.cpp b/tests/src/unit-locale-cpp.cpp index c2a113d06..9f62fc0ca 100644 --- a/tests/src/unit-locale-cpp.cpp +++ b/tests/src/unit-locale-cpp.cpp @@ -12,7 +12,12 @@ #include using nlohmann::json; +#include #include +#include +#include +#include +#include struct ParserImpl final: public nlohmann::json_sax { @@ -175,3 +180,208 @@ TEST_CASE("locale-dependent test (LC_NUMERIC=de_DE)") MESSAGE("locale de_DE is not usable"); } } + +namespace +{ +// records the numbers of a flat array and switches LC_NUMERIC to the given +// locale once the array opens - after the lexer was constructed, but before +// any number in the array is lexed +struct LocaleSwitchingSax final: public nlohmann::json_sax +{ + explicit LocaleSwitchingSax(const char* switch_to) + : locale_after_open(switch_to) + {} + + bool null() override + { + return true; + } + bool boolean(bool /*val*/) override + { + return true; + } + bool number_integer(json::number_integer_t /*val*/) override + { + return true; + } + bool number_unsigned(json::number_unsigned_t /*val*/) override + { + return true; + } + bool number_float(json::number_float_t val, const json::string_t& s) override + { + values.push_back(val); + strings.push_back(s); + return true; + } + bool string(json::string_t& /*val*/) override + { + return true; + } + bool binary(json::binary_t& /*val*/) override + { + return true; + } + bool start_object(std::size_t /*val*/) override + { + return true; + } + bool key(json::string_t& /*val*/) override + { + return true; + } + bool end_object() override + { + return true; + } + bool start_array(std::size_t /*val*/) override + { + switched = std::setlocale(LC_NUMERIC, locale_after_open.c_str()) != nullptr; + return true; + } + bool end_array() override + { + return true; + } + bool parse_error(std::size_t /*val*/, const std::string& /*val*/, const nlohmann::detail::exception& /*val*/) override + { + return false; + } + + std::string locale_after_open; + bool switched = false; + std::vector values {}; // NOLINT(readability-redundant-member-init) + std::vector strings {}; // NOLINT(readability-redundant-member-init) +}; +} // namespace + +TEST_CASE("locale changes between lexer construction and number conversion (#5198)") +{ + // The numbers are chosen so that the conversion also takes the strtod + // fallback, which honors the locale that is current at conversion time: + // too many significant digits for Clinger's fast path, an underflow that + // std::from_chars rejects, and a plain value. + const std::vector numbers = {"3.14159265358979323846", "1.5e-400", "12.34", "-0.000123456789012345678"}; + std::string text = "["; + for (const auto& n : numbers) + { + text += (text.size() == 1 ? "" : ",") + n; + } + text += "]"; + + using long_double_json = nlohmann::basic_json; + + // reference values, parsed without a locale switch + REQUIRE(std::setlocale(LC_NUMERIC, "C") != nullptr); + const json expected = json::parse(text); + const long_double_json expected_ld = long_double_json::parse(text); + + const std::array, 2> transitions = + { + { + {"C", "de_DE"}, + {"de_DE", "C"} + } + }; + + for (const auto& transition : transitions) + { + CAPTURE(transition.first); + CAPTURE(transition.second); + + if (std::setlocale(LC_NUMERIC, transition.first) == nullptr) + { + MESSAGE("locale is not usable"); + continue; + } + + // SAX parsing + { + LocaleSwitchingSax sax(transition.second); + CHECK(json::sax_parse(text, &sax)); + if (sax.switched) + { + CHECK(sax.values == expected.get>()); + CHECK(sax.strings == numbers); + } + } + + // DOM parsing with a callback + { + bool switched = false; + const auto cb = [&](int /*depth*/, json::parse_event_t event, json& /*parsed*/) + { + if (event == json::parse_event_t::array_start) + { + switched = std::setlocale(LC_NUMERIC, transition.second) != nullptr; + } + return true; + }; + const json j = json::parse(text, cb); + if (switched) + { + CHECK(j == expected); + } + } + + // a long double goes through std::strtold unless std::from_chars supports it + { + bool switched = false; + const auto cb = [&](int /*depth*/, long_double_json::parse_event_t event, long_double_json& /*parsed*/) + { + if (event == long_double_json::parse_event_t::array_start) + { + switched = std::setlocale(LC_NUMERIC, transition.second) != nullptr; + } + return true; + }; + const long_double_json j = long_double_json::parse(text, cb); + if (switched) + { + CHECK(j == expected_ld); + } + } + } + + std::setlocale(LC_NUMERIC, "C"); +} + +TEST_CASE("locale with a multi-byte decimal point") +{ + // Some locales use a decimal point that is not a single character, e.g. + // U+066B ARABIC DECIMAL SEPARATOR (two bytes in UTF-8). It cannot be + // substituted in place for '.', so the strtod fallback stops early. The + // conversion must still terminate rather than retry forever. + const std::array names = {{"ar_EG.UTF-8", "ar_SA.UTF-8", "fa_IR.UTF-8", "ps_AF.UTF-8", "ar_EG", "fa_IR"}}; + bool tested = false; + for (const char* name : names) + { + if (std::setlocale(LC_NUMERIC, name) == nullptr) + { + continue; + } + const std::string decimal_point = std::localeconv()->decimal_point; + if (decimal_point.size() < 2) + { + continue; + } + CAPTURE(name); + tested = true; + + // too many significant digits for Clinger's fast path, and an underflow + // that std::from_chars rejects: both reach the strtod fallback + json j; + CHECK_NOTHROW(j = json::parse("[3.14159265358979323846, 1.5e-400, -0.000123456789012345678]")); + CHECK(j.is_array()); + CHECK(json::accept("3.14159265358979323846")); + + // a value the locale-independent paths convert is not affected + CHECK(json::parse("12.5") == 12.5); + } + if (!tested) + { + MESSAGE("no locale with a multi-byte decimal point is usable"); + } + + std::setlocale(LC_NUMERIC, "C"); +} From 633de8e44bbbc29383d72f8c920343ea5d1eaa21 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Mon, 28 Sep 2026 22:20:43 +0200 Subject: [PATCH 03/31] Fix CI: clang-tidy and GCC -Wnoexcept in the locale test (#5613) #5597 was merged before all of its CI jobs had run, and two of them fail on develop now, and so on every pull request: - ci_clang_tidy: cert-err33-c for the two std::setlocale(LC_NUMERIC, "C") calls whose result was discarded. Check the result, like the other resets in the file. - ci_test_standards_gcc (20) with GCC 16: -Wnoexcept for the two parser callbacks, which cannot throw but were not declared noexcept. Signed-off-by: Niels Lohmann --- tests/src/unit-locale-cpp.cpp | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/src/unit-locale-cpp.cpp b/tests/src/unit-locale-cpp.cpp index 9f62fc0ca..14f743a66 100644 --- a/tests/src/unit-locale-cpp.cpp +++ b/tests/src/unit-locale-cpp.cpp @@ -309,7 +309,7 @@ TEST_CASE("locale changes between lexer construction and number conversion (#519 // DOM parsing with a callback { bool switched = false; - const auto cb = [&](int /*depth*/, json::parse_event_t event, json& /*parsed*/) + const auto cb = [&](int /*depth*/, json::parse_event_t event, json& /*parsed*/) noexcept { if (event == json::parse_event_t::array_start) { @@ -327,7 +327,7 @@ TEST_CASE("locale changes between lexer construction and number conversion (#519 // a long double goes through std::strtold unless std::from_chars supports it { bool switched = false; - const auto cb = [&](int /*depth*/, long_double_json::parse_event_t event, long_double_json& /*parsed*/) + const auto cb = [&](int /*depth*/, long_double_json::parse_event_t event, long_double_json& /*parsed*/) noexcept { if (event == long_double_json::parse_event_t::array_start) { @@ -343,7 +343,7 @@ TEST_CASE("locale changes between lexer construction and number conversion (#519 } } - std::setlocale(LC_NUMERIC, "C"); + CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr); } TEST_CASE("locale with a multi-byte decimal point") @@ -383,5 +383,5 @@ TEST_CASE("locale with a multi-byte decimal point") MESSAGE("no locale with a multi-byte decimal point is usable"); } - std::setlocale(LC_NUMERIC, "C"); + CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr); } From 04f6ddd227026faaa12dd8c7962645937b50e60d Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:05:51 +0200 Subject: [PATCH 04/31] Keep external headers as #include when amalgamating (#5615) A header that builds on json.hpp (such as the planned json_view.hpp) must not inline json.hpp: its single-header version would contain a second copy of the library, and that copy would change with every library change. The optional config key "external" lists include paths that are kept as #include directives. Only the first directive per path is kept; repeated ones are commented out, as the tool already does for inlined headers. config_json_view.json uses it for json_view.hpp; the existing configs do not set it, and json.hpp and json_fwd.hpp regenerate byte-identically. Signed-off-by: Niels Lohmann --- tools/amalgamate/CHANGES.md | 3 +++ tools/amalgamate/README.md | 5 +++++ tools/amalgamate/amalgamate.py | 28 +++++++++++++++++++++----- tools/amalgamate/config_json_view.json | 9 +++++++++ 4 files changed, 40 insertions(+), 5 deletions(-) create mode 100644 tools/amalgamate/config_json_view.json diff --git a/tools/amalgamate/CHANGES.md b/tools/amalgamate/CHANGES.md index 728b93319..684a62148 100644 --- a/tools/amalgamate/CHANGES.md +++ b/tools/amalgamate/CHANGES.md @@ -8,3 +8,6 @@ The following changes have been made to the code with respect to Date: Wed, 30 Sep 2026 20:06:25 +0200 Subject: [PATCH 05/31] Move the float conversion chain out of the lexer (#5616) lexer::convert_number() converted float tokens with std::from_chars (when available), Clinger's fast path, and the locale-aware strtod fallback, all as lexer members. They are now free functions in number_parse.hpp: - convert_float_fast(): std::from_chars, then Clinger's fast path, skipped when the mantissa has too many significant digits - convert_float_locale_aware(): strtof/strtod/strtold with the decimal point of the current locale, retried when the locale changed (#5198) so that other code converting JSON number tokens gets the same values. No change in behavior; the lexer no longer includes and . Signed-off-by: Niels Lohmann --- include/nlohmann/detail/input/lexer.hpp | 155 +------- .../nlohmann/detail/input/number_parse.hpp | 186 +++++++++- single_include/nlohmann/json.hpp | 341 ++++++++++-------- 3 files changed, 376 insertions(+), 306 deletions(-) diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index 00a964a17..47de76c22 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -9,10 +9,8 @@ #pragma once #include // array -#include // localeconv #include // size_t #include // snprintf -#include // strtof, strtod, strtold, strtoll, strtoull #include // initializer_list #include // char_traits, string #include // move @@ -217,18 +215,6 @@ class lexer : public lexer_base ~lexer() = default; private: - ///////////////////// - // locales - ///////////////////// - - /// return the decimal point of the current locale - static char get_decimal_point() noexcept - { - const auto* loc = localeconv(); - JSON_ASSERT(loc != nullptr); - return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point); - } - ///////////////////// // scan functions ///////////////////// @@ -1036,24 +1022,6 @@ class lexer : public lexer_base } } - JSON_HEDLEY_NON_NULL(2) - static void strtof(float& f, const char* str, char** endptr) noexcept - { - f = std::strtof(str, endptr); - } - - JSON_HEDLEY_NON_NULL(2) - static void strtof(double& f, const char* str, char** endptr) noexcept - { - f = std::strtod(str, endptr); - } - - JSON_HEDLEY_NON_NULL(2) - static void strtof(long double& f, const char* str, char** endptr) noexcept - { - f = std::strtold(str, endptr); - } - /*! @brief scan a number literal @@ -1093,7 +1061,7 @@ class lexer : public lexer_base @note The scanner is independent of the current locale: token_buffer always holds `.`. Only the std::strtod fallback of convert_number() depends on the locale, and it looks up the decimal point right - before converting (see convert_float_locale_aware()). + before converting (see detail::convert_float_locale_aware()). */ token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated. { @@ -1424,59 +1392,6 @@ scan_number_done: return token_type::uninitialized; } - /*! - @brief check whether Clinger's fast path can still succeed for this token - - parse_float_fast() needs a significand below 2^53. A mantissa with 17 or - more significant digits is at least 10^16 and therefore always exceeds it, - so calling the fast path would walk the token one extra time only to - decline before strtod has to run anyway. - - Significant digits are the mantissa's digits from the first nonzero one on; - the sign, the decimal point, leading zeros, and the exponent do not count. - The answer is derived from indices - the digits are not scanned again - so - this stays off the hot path of the number scanners. - - @param[in] mantissa_end offset just past the last mantissa byte in - token_buffer - @return false if parse_float_fast() is guaranteed to decline - */ - bool mantissa_fits_clinger(std::size_t mantissa_end) const - { - // 10^16 already exceeds 2^53, so 17 digits can never fit - constexpr std::size_t limit = 17; - - const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u; - const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u; - // the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so - // a leading zero can only be a lone "0", which is not significant - const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u; - JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero); - std::size_t digits = mantissa_end - neg - has_dot - lead_zero; - - if (JSON_HEDLEY_LIKELY(digits < limit)) - { - return true; - } - - // Only a number below 1 can carry further insignificant zeros, and only - // while the count stays at the limit does removing them change the - // answer - so this loop is skipped for all but a few tokens. The - // fraction is located through decimal_point_position rather than by - // searching '.'. - if (lead_zero != 0) - { - JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit - for (std::size_t i = decimal_point_position + 1; - digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i) - { - --digits; - } - } - - return digits < limit; - } - /*! @brief convert the number text in token_buffer to its value and token type @@ -1490,7 +1405,7 @@ scan_number_done: token_buffer (the index of 'e'/'E', or token_buffer.size() when there is no exponent); used to skip Clinger's fast path when it cannot - possibly succeed - see mantissa_fits_clinger() + possibly succeed - see detail::mantissa_fits_clinger() */ token_type convert_number(token_type number_type, std::size_t mantissa_end) { @@ -1563,77 +1478,15 @@ scan_number_done: // (Eisel-Lemire, locale-independent, correctly rounded) when available; // otherwise the exact Clinger fast path (double only); otherwise the // locale-aware strtof/strtod/strtold. - if (parse_float_from_chars(num_begin, num_end, value_float)) - { - return token_type::value_float; - } - // Skipping a fast path that cannot succeed is lossless and saves a full - // extra pass over the token's bytes, which otherwise shows up on - // high-precision inputs such as canada.json - if (mantissa_fits_clinger(mantissa_end) - && parse_float_fast(num_begin, num_end, value_float)) + if (convert_float_fast(num_begin, num_end, decimal_point_position, mantissa_end, value_float)) { return token_type::value_float; } - convert_float_locale_aware(); + convert_float_locale_aware(token_buffer, decimal_point_position, value_float); return token_type::value_float; } - /*! - @brief convert the float in token_buffer with strtof/strtod/strtold - - These functions expect the decimal point of the *current* locale, so it is - looked up right before the conversion instead of once when the lexer is - constructed: a locale change in between (by a parser callback, a SAX - handler, or another thread) must not truncate the value (#5198). The - token has been validated before, so if the conversion stops early and the - decimal point changed in the meantime, the locale changed between the - lookup and the call, and the conversion is repeated with the new decimal - point. If the decimal point did not change, a retry cannot succeed: the - locale's decimal point is not a single character (e.g., the two-byte - U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place. - The value strtod parsed up to that point is kept, as before this change. - - Note that changing the locale in another thread *while* strtod runs is - undefined behavior of the C library, which this function cannot prevent. - */ - void convert_float_locale_aware() - { - const bool has_dot = decimal_point_position != std::string::npos; - char decimal_point = get_decimal_point(); - for (;;) - { - const bool substitute = has_dot && decimal_point != '.'; - if (substitute) - { - token_buffer[decimal_point_position] = static_cast(decimal_point); - } - - char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) - strtof(value_float, token_buffer.data(), &endptr); - - if (substitute) - { - // get_string() hands the token to the SAX interface with '.' - token_buffer[decimal_point_position] = '.'; - } - - if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size())) - { - return; - } - - // retry only if the locale changed; otherwise, this would loop forever - const char current_decimal_point = get_decimal_point(); - if (current_decimal_point == decimal_point) - { - return; - } - decimal_point = current_decimal_point; - } - } - /*! @brief contiguous fast path for scanning a number diff --git a/include/nlohmann/detail/input/number_parse.hpp b/include/nlohmann/detail/input/number_parse.hpp index 25f6cac91..c3b6cd28c 100644 --- a/include/nlohmann/detail/input/number_parse.hpp +++ b/include/nlohmann/detail/input/number_parse.hpp @@ -10,9 +10,12 @@ #include // array #include // FLT_EVAL_METHOD +#include // localeconv #include // size_t #include // int64_t, uint64_t +#include // strtof, strtod, strtold #include // numeric_limits +#include // string #include @@ -29,8 +32,9 @@ // This file contains the value-conversion helpers used by the lexer to turn an // already-validated number token into a value, without the locale/errno -// overhead of std::strtoull/std::strtod. They are free functions so the lexer -// stays focused on scanning; see lexer::convert_number(). +// overhead of std::strtoull/std::strtod where possible. They are free functions +// so the lexer stays focused on scanning (see lexer::convert_number()) and so +// that other parsers of JSON text can convert tokens exactly like it does. NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -293,5 +297,183 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out) #endif } +/*! +@brief check whether Clinger's fast path can still succeed for a float token + +parse_float_fast() needs a significand below 2^53. A mantissa with 17 or +more significant digits is at least 10^16 and therefore always exceeds it, +so calling the fast path would walk the token one extra time only to +decline before strtod has to run anyway. + +Significant digits are the mantissa's digits from the first nonzero one on; +the sign, the decimal point, leading zeros, and the exponent do not count. +The answer is derived from indices - the digits are not scanned again - so +this stays off the hot path of the number scanners. + +@param[in] token the validated number token ('.' as decimal point) +@param[in] decimal_point_position index of the '.' in @a token, or + std::string::npos if there is none +@param[in] mantissa_end offset just past the last mantissa byte +@return false if parse_float_fast() is guaranteed to decline +*/ +inline bool mantissa_fits_clinger(const char* token, std::size_t decimal_point_position, std::size_t mantissa_end) noexcept +{ + // 10^16 already exceeds 2^53, so 17 digits can never fit + constexpr std::size_t limit = 17; + + const std::size_t neg = (token[0] == '-') ? 1u : 0u; + const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u; + // the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so + // a leading zero can only be a lone "0", which is not significant + const std::size_t lead_zero = (token[neg] == '0') ? 1u : 0u; + JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero); + std::size_t digits = mantissa_end - neg - has_dot - lead_zero; + + if (JSON_HEDLEY_LIKELY(digits < limit)) + { + return true; + } + + // Only a number below 1 can carry further insignificant zeros, and only + // while the count stays at the limit does removing them change the + // answer - so this loop is skipped for all but a few tokens. The + // fraction is located through decimal_point_position rather than by + // searching '.'. + if (lead_zero != 0) + { + JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit + for (std::size_t i = decimal_point_position + 1; + digits >= limit && i < mantissa_end && token[i] == '0'; ++i) + { + --digits; + } + } + + return digits < limit; +} + +/*! +@brief convert a validated float token without the C library, if possible + +Tries std::from_chars (when available) and then Clinger's exact fast path +(double only), skipping the latter when it cannot succeed. + +@param[in] first pointer to the first character of the token +@param[in] last pointer past the last character +@param[in] decimal_point_position index of the '.' in the token, or + std::string::npos if there is none +@param[in] mantissa_end offset just past the last mantissa byte (the + index of 'e'/'E', or the token length) +@param[out] value the converted value on success +@return true if the value was converted; false if convert_float_locale_aware() + must convert it +*/ +template +bool convert_float_fast(const char* first, const char* last, std::size_t decimal_point_position, + std::size_t mantissa_end, FloatType& value) noexcept +{ + if (parse_float_from_chars(first, last, value)) + { + return true; + } + // Skipping a fast path that cannot succeed is lossless and saves a full + // extra pass over the token's bytes, which otherwise shows up on + // high-precision inputs such as canada.json + return mantissa_fits_clinger(first, decimal_point_position, mantissa_end) + && parse_float_fast(first, last, value); +} + +/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f +JSON_HEDLEY_NON_NULL(2) +inline void strtof_by_type(float& f, const char* str, char** endptr) noexcept +{ + f = std::strtof(str, endptr); +} + +/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f +JSON_HEDLEY_NON_NULL(2) +inline void strtof_by_type(double& f, const char* str, char** endptr) noexcept +{ + f = std::strtod(str, endptr); +} + +/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f +JSON_HEDLEY_NON_NULL(2) +inline void strtof_by_type(long double& f, const char* str, char** endptr) noexcept +{ + f = std::strtold(str, endptr); +} + +/// return the decimal point of the current locale +inline char get_decimal_point() noexcept +{ + const auto* loc = localeconv(); + JSON_ASSERT(loc != nullptr); + return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point); +} + +/*! +@brief convert a validated float token with strtof/strtod/strtold + +These functions expect the decimal point of the *current* locale, so it is +looked up right before the conversion instead of once when the lexer is +constructed: a locale change in between (by a parser callback, a SAX +handler, or another thread) must not truncate the value (#5198). The +token has been validated before, so if the conversion stops early and the +decimal point changed in the meantime, the locale changed between the +lookup and the call, and the conversion is repeated with the new decimal +point. If the decimal point did not change, a retry cannot succeed: the +locale's decimal point is not a single character (e.g., the two-byte +U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place. +The value strtod parsed up to that point is kept, as before this change. + +Note that changing the locale in another thread *while* strtod runs is +undefined behavior of the C library, which this function cannot prevent. + +@param[in,out] token the token with '.' as decimal point; its + decimal point is replaced during the + conversion and restored afterwards + (data() must be NUL-terminated) +@param[in] decimal_point_position index of the '.' in @a token, or + std::string::npos if there is none +@param[out] value the converted value +*/ +template +void convert_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& value) +{ + const bool has_dot = decimal_point_position != std::string::npos; + char decimal_point = get_decimal_point(); + for (;;) + { + const bool substitute = has_dot && decimal_point != '.'; + if (substitute) + { + token[decimal_point_position] = static_cast(decimal_point); + } + + char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) + strtof_by_type(value, token.data(), &endptr); + + if (substitute) + { + // the caller hands the token on (e.g. to the SAX interface) with '.' + token[decimal_point_position] = '.'; + } + + if (JSON_HEDLEY_LIKELY(endptr == token.data() + token.size())) + { + return; + } + + // retry only if the locale changed; otherwise, this would loop forever + const char current_decimal_point = get_decimal_point(); + if (current_decimal_point == decimal_point) + { + return; + } + decimal_point = current_decimal_point; + } +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 576498738..fbb30e7f7 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -8472,10 +8472,8 @@ NLOHMANN_JSON_NAMESPACE_END #include // array -#include // localeconv #include // size_t #include // snprintf -#include // strtof, strtod, strtold, strtoll, strtoull #include // initializer_list #include // char_traits, string #include // move @@ -8496,9 +8494,12 @@ NLOHMANN_JSON_NAMESPACE_END #include // array #include // FLT_EVAL_METHOD +#include // localeconv #include // size_t #include // int64_t, uint64_t +#include // strtof, strtod, strtold #include // numeric_limits +#include // string // #include @@ -8516,8 +8517,9 @@ NLOHMANN_JSON_NAMESPACE_END // This file contains the value-conversion helpers used by the lexer to turn an // already-validated number token into a value, without the locale/errno -// overhead of std::strtoull/std::strtod. They are free functions so the lexer -// stays focused on scanning; see lexer::convert_number(). +// overhead of std::strtoull/std::strtod where possible. They are free functions +// so the lexer stays focused on scanning (see lexer::convert_number()) and so +// that other parsers of JSON text can convert tokens exactly like it does. NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -8780,6 +8782,184 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out) #endif } +/*! +@brief check whether Clinger's fast path can still succeed for a float token + +parse_float_fast() needs a significand below 2^53. A mantissa with 17 or +more significant digits is at least 10^16 and therefore always exceeds it, +so calling the fast path would walk the token one extra time only to +decline before strtod has to run anyway. + +Significant digits are the mantissa's digits from the first nonzero one on; +the sign, the decimal point, leading zeros, and the exponent do not count. +The answer is derived from indices - the digits are not scanned again - so +this stays off the hot path of the number scanners. + +@param[in] token the validated number token ('.' as decimal point) +@param[in] decimal_point_position index of the '.' in @a token, or + std::string::npos if there is none +@param[in] mantissa_end offset just past the last mantissa byte +@return false if parse_float_fast() is guaranteed to decline +*/ +inline bool mantissa_fits_clinger(const char* token, std::size_t decimal_point_position, std::size_t mantissa_end) noexcept +{ + // 10^16 already exceeds 2^53, so 17 digits can never fit + constexpr std::size_t limit = 17; + + const std::size_t neg = (token[0] == '-') ? 1u : 0u; + const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u; + // the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so + // a leading zero can only be a lone "0", which is not significant + const std::size_t lead_zero = (token[neg] == '0') ? 1u : 0u; + JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero); + std::size_t digits = mantissa_end - neg - has_dot - lead_zero; + + if (JSON_HEDLEY_LIKELY(digits < limit)) + { + return true; + } + + // Only a number below 1 can carry further insignificant zeros, and only + // while the count stays at the limit does removing them change the + // answer - so this loop is skipped for all but a few tokens. The + // fraction is located through decimal_point_position rather than by + // searching '.'. + if (lead_zero != 0) + { + JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit + for (std::size_t i = decimal_point_position + 1; + digits >= limit && i < mantissa_end && token[i] == '0'; ++i) + { + --digits; + } + } + + return digits < limit; +} + +/*! +@brief convert a validated float token without the C library, if possible + +Tries std::from_chars (when available) and then Clinger's exact fast path +(double only), skipping the latter when it cannot succeed. + +@param[in] first pointer to the first character of the token +@param[in] last pointer past the last character +@param[in] decimal_point_position index of the '.' in the token, or + std::string::npos if there is none +@param[in] mantissa_end offset just past the last mantissa byte (the + index of 'e'/'E', or the token length) +@param[out] value the converted value on success +@return true if the value was converted; false if convert_float_locale_aware() + must convert it +*/ +template +bool convert_float_fast(const char* first, const char* last, std::size_t decimal_point_position, + std::size_t mantissa_end, FloatType& value) noexcept +{ + if (parse_float_from_chars(first, last, value)) + { + return true; + } + // Skipping a fast path that cannot succeed is lossless and saves a full + // extra pass over the token's bytes, which otherwise shows up on + // high-precision inputs such as canada.json + return mantissa_fits_clinger(first, decimal_point_position, mantissa_end) + && parse_float_fast(first, last, value); +} + +/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f +JSON_HEDLEY_NON_NULL(2) +inline void strtof_by_type(float& f, const char* str, char** endptr) noexcept +{ + f = std::strtof(str, endptr); +} + +/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f +JSON_HEDLEY_NON_NULL(2) +inline void strtof_by_type(double& f, const char* str, char** endptr) noexcept +{ + f = std::strtod(str, endptr); +} + +/// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f +JSON_HEDLEY_NON_NULL(2) +inline void strtof_by_type(long double& f, const char* str, char** endptr) noexcept +{ + f = std::strtold(str, endptr); +} + +/// return the decimal point of the current locale +inline char get_decimal_point() noexcept +{ + const auto* loc = localeconv(); + JSON_ASSERT(loc != nullptr); + return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point); +} + +/*! +@brief convert a validated float token with strtof/strtod/strtold + +These functions expect the decimal point of the *current* locale, so it is +looked up right before the conversion instead of once when the lexer is +constructed: a locale change in between (by a parser callback, a SAX +handler, or another thread) must not truncate the value (#5198). The +token has been validated before, so if the conversion stops early and the +decimal point changed in the meantime, the locale changed between the +lookup and the call, and the conversion is repeated with the new decimal +point. If the decimal point did not change, a retry cannot succeed: the +locale's decimal point is not a single character (e.g., the two-byte +U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place. +The value strtod parsed up to that point is kept, as before this change. + +Note that changing the locale in another thread *while* strtod runs is +undefined behavior of the C library, which this function cannot prevent. + +@param[in,out] token the token with '.' as decimal point; its + decimal point is replaced during the + conversion and restored afterwards + (data() must be NUL-terminated) +@param[in] decimal_point_position index of the '.' in @a token, or + std::string::npos if there is none +@param[out] value the converted value +*/ +template +void convert_float_locale_aware(StringType& token, std::size_t decimal_point_position, FloatType& value) +{ + const bool has_dot = decimal_point_position != std::string::npos; + char decimal_point = get_decimal_point(); + for (;;) + { + const bool substitute = has_dot && decimal_point != '.'; + if (substitute) + { + token[decimal_point_position] = static_cast(decimal_point); + } + + char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) + strtof_by_type(value, token.data(), &endptr); + + if (substitute) + { + // the caller hands the token on (e.g. to the SAX interface) with '.' + token[decimal_point_position] = '.'; + } + + if (JSON_HEDLEY_LIKELY(endptr == token.data() + token.size())) + { + return; + } + + // retry only if the locale changed; otherwise, this would loop forever + const char current_decimal_point = get_decimal_point(); + if (current_decimal_point == decimal_point) + { + return; + } + decimal_point = current_decimal_point; + } +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -9309,18 +9489,6 @@ class lexer : public lexer_base ~lexer() = default; private: - ///////////////////// - // locales - ///////////////////// - - /// return the decimal point of the current locale - static char get_decimal_point() noexcept - { - const auto* loc = localeconv(); - JSON_ASSERT(loc != nullptr); - return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point); - } - ///////////////////// // scan functions ///////////////////// @@ -10128,24 +10296,6 @@ class lexer : public lexer_base } } - JSON_HEDLEY_NON_NULL(2) - static void strtof(float& f, const char* str, char** endptr) noexcept - { - f = std::strtof(str, endptr); - } - - JSON_HEDLEY_NON_NULL(2) - static void strtof(double& f, const char* str, char** endptr) noexcept - { - f = std::strtod(str, endptr); - } - - JSON_HEDLEY_NON_NULL(2) - static void strtof(long double& f, const char* str, char** endptr) noexcept - { - f = std::strtold(str, endptr); - } - /*! @brief scan a number literal @@ -10185,7 +10335,7 @@ class lexer : public lexer_base @note The scanner is independent of the current locale: token_buffer always holds `.`. Only the std::strtod fallback of convert_number() depends on the locale, and it looks up the decimal point right - before converting (see convert_float_locale_aware()). + before converting (see detail::convert_float_locale_aware()). */ token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated. { @@ -10516,59 +10666,6 @@ scan_number_done: return token_type::uninitialized; } - /*! - @brief check whether Clinger's fast path can still succeed for this token - - parse_float_fast() needs a significand below 2^53. A mantissa with 17 or - more significant digits is at least 10^16 and therefore always exceeds it, - so calling the fast path would walk the token one extra time only to - decline before strtod has to run anyway. - - Significant digits are the mantissa's digits from the first nonzero one on; - the sign, the decimal point, leading zeros, and the exponent do not count. - The answer is derived from indices - the digits are not scanned again - so - this stays off the hot path of the number scanners. - - @param[in] mantissa_end offset just past the last mantissa byte in - token_buffer - @return false if parse_float_fast() is guaranteed to decline - */ - bool mantissa_fits_clinger(std::size_t mantissa_end) const - { - // 10^16 already exceeds 2^53, so 17 digits can never fit - constexpr std::size_t limit = 17; - - const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u; - const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u; - // the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so - // a leading zero can only be a lone "0", which is not significant - const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u; - JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero); - std::size_t digits = mantissa_end - neg - has_dot - lead_zero; - - if (JSON_HEDLEY_LIKELY(digits < limit)) - { - return true; - } - - // Only a number below 1 can carry further insignificant zeros, and only - // while the count stays at the limit does removing them change the - // answer - so this loop is skipped for all but a few tokens. The - // fraction is located through decimal_point_position rather than by - // searching '.'. - if (lead_zero != 0) - { - JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit - for (std::size_t i = decimal_point_position + 1; - digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i) - { - --digits; - } - } - - return digits < limit; - } - /*! @brief convert the number text in token_buffer to its value and token type @@ -10582,7 +10679,7 @@ scan_number_done: token_buffer (the index of 'e'/'E', or token_buffer.size() when there is no exponent); used to skip Clinger's fast path when it cannot - possibly succeed - see mantissa_fits_clinger() + possibly succeed - see detail::mantissa_fits_clinger() */ token_type convert_number(token_type number_type, std::size_t mantissa_end) { @@ -10655,77 +10752,15 @@ scan_number_done: // (Eisel-Lemire, locale-independent, correctly rounded) when available; // otherwise the exact Clinger fast path (double only); otherwise the // locale-aware strtof/strtod/strtold. - if (parse_float_from_chars(num_begin, num_end, value_float)) - { - return token_type::value_float; - } - // Skipping a fast path that cannot succeed is lossless and saves a full - // extra pass over the token's bytes, which otherwise shows up on - // high-precision inputs such as canada.json - if (mantissa_fits_clinger(mantissa_end) - && parse_float_fast(num_begin, num_end, value_float)) + if (convert_float_fast(num_begin, num_end, decimal_point_position, mantissa_end, value_float)) { return token_type::value_float; } - convert_float_locale_aware(); + convert_float_locale_aware(token_buffer, decimal_point_position, value_float); return token_type::value_float; } - /*! - @brief convert the float in token_buffer with strtof/strtod/strtold - - These functions expect the decimal point of the *current* locale, so it is - looked up right before the conversion instead of once when the lexer is - constructed: a locale change in between (by a parser callback, a SAX - handler, or another thread) must not truncate the value (#5198). The - token has been validated before, so if the conversion stops early and the - decimal point changed in the meantime, the locale changed between the - lookup and the call, and the conversion is repeated with the new decimal - point. If the decimal point did not change, a retry cannot succeed: the - locale's decimal point is not a single character (e.g., the two-byte - U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place. - The value strtod parsed up to that point is kept, as before this change. - - Note that changing the locale in another thread *while* strtod runs is - undefined behavior of the C library, which this function cannot prevent. - */ - void convert_float_locale_aware() - { - const bool has_dot = decimal_point_position != std::string::npos; - char decimal_point = get_decimal_point(); - for (;;) - { - const bool substitute = has_dot && decimal_point != '.'; - if (substitute) - { - token_buffer[decimal_point_position] = static_cast(decimal_point); - } - - char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) - strtof(value_float, token_buffer.data(), &endptr); - - if (substitute) - { - // get_string() hands the token to the SAX interface with '.' - token_buffer[decimal_point_position] = '.'; - } - - if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size())) - { - return; - } - - // retry only if the locale changed; otherwise, this would loop forever - const char current_decimal_point = get_decimal_point(); - if (current_decimal_point == decimal_point) - { - return; - } - decimal_point = current_decimal_point; - } - } - /*! @brief contiguous fast path for scanning a number From d268eaa693367c3f8a31a6cd2088e6bb731e30bb Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:06:26 +0200 Subject: [PATCH 06/31] Convert floats with Eisel-Lemire when std::from_chars is unavailable (#5617) * Convert floats with Eisel-Lemire when std::from_chars is unavailable Float tokens that Clinger's fast path cannot convert (e.g. the 17-digit coordinates of canada.json) went to strtod unless std::from_chars was available. It is not used in C++11/14, and not with libc++, which does not define __cpp_lib_to_chars. The Eisel-Lemire algorithm (after fast_float's compute_float) now converts them with integer arithmetic, correctly rounded for any token with at most 19 significant digits. Longer tokens are truncated; the result is used if w and w + 1 round alike, else strtod decides as before. Overflow still yields infinity (out_of_range.406). The table of powers of five (fast_float's) lives in pow5_table.hpp; a unit test recomputes every entry with big-integer arithmetic. Further tests: known values generated with Python (whose float() is correctly rounded), 200,000 round trips through to_chars, and the 128-bit multiplication and leading-zero count against big-integer references (both with and without a 128-bit type). Checked against strtod on 6.5 million tokens, among them 60,000 exact halfway cases: no difference. json::parse on canada.json: -8.6% (C++11), -7.6% (C++17, Apple clang); other files unchanged. Compile time of a TU including json.hpp: +0.7%. Signed-off-by: Niels Lohmann * Fix CI: unused parse result and find() == npos in the Eisel-Lemire tests GCC (-Werror=unused-result) rejected CHECK_THROWS_WITH_AS(json::parse(...)) because parse() is [[nodiscard]]; assign the result to a dummy json as the other tests do. clang-tidy flagged longer.find('.') == npos with abseil-string-find-str-contains; store the position in a variable first. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- BUILD.bazel | 2 + README.md | 1 + docs/mkdocs/docs/home/license.md | 2 + include/nlohmann/detail/bit_ops.hpp | 81 ++ .../nlohmann/detail/input/number_parse.hpp | 244 +++++- include/nlohmann/detail/input/pow5_table.hpp | 371 ++++++++++ single_include/nlohmann/json.hpp | 700 +++++++++++++++++- tests/src/unit-class_lexer.cpp | 615 +++++++++++++++ 8 files changed, 2008 insertions(+), 8 deletions(-) create mode 100644 include/nlohmann/detail/bit_ops.hpp create mode 100644 include/nlohmann/detail/input/pow5_table.hpp diff --git a/BUILD.bazel b/BUILD.bazel index db59095cd..899f9196d 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -21,6 +21,7 @@ cc_library( "include/nlohmann/adl_serializer.hpp", "include/nlohmann/byte_container_with_subtype.hpp", "include/nlohmann/detail/abi_macros.hpp", + "include/nlohmann/detail/bit_ops.hpp", "include/nlohmann/detail/conversions/from_json.hpp", "include/nlohmann/detail/conversions/to_chars.hpp", "include/nlohmann/detail/conversions/to_json.hpp", @@ -33,6 +34,7 @@ cc_library( "include/nlohmann/detail/input/number_parse.hpp", "include/nlohmann/detail/input/parser.hpp", "include/nlohmann/detail/input/position_t.hpp", + "include/nlohmann/detail/input/pow5_table.hpp", "include/nlohmann/detail/input/string_scan.hpp", "include/nlohmann/detail/iterators/internal_iterator.hpp", "include/nlohmann/detail/iterators/iter_impl.hpp", diff --git a/README.md b/README.md index cc3546686..9fd3f71fc 100644 --- a/README.md +++ b/README.md @@ -1395,6 +1395,7 @@ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR I - The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2009 [Florian Loitsch](https://florian.loitsch.com/) - The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/). - The class contains parts of [Google Abseil](https://github.com/abseil/abseil-cpp) which is licensed under the [Apache 2.0 License](https://opensource.org/licenses/Apache-2.0). +- The class contains an adapted version of the Eisel-Lemire algorithm and its table of powers of five from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors REUSE Software diff --git a/docs/mkdocs/docs/home/license.md b/docs/mkdocs/docs/home/license.md index 597c69496..863b7c1d6 100644 --- a/docs/mkdocs/docs/home/license.md +++ b/docs/mkdocs/docs/home/license.md @@ -19,3 +19,5 @@ The class contains the UTF-8 Decoder from Bjoern Hoehrmann which is licensed und The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2009 [Florian Loitsch](https://florian.loitsch.com/) The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/). + +The class contains an adapted version of the Eisel-Lemire algorithm and its table of powers of five from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors diff --git a/include/nlohmann/detail/bit_ops.hpp b/include/nlohmann/detail/bit_ops.hpp new file mode 100644 index 000000000..9655ff18b --- /dev/null +++ b/include/nlohmann/detail/bit_ops.hpp @@ -0,0 +1,81 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // uint64_t + +#include + +// Portable bit-level helpers for the number and string scanners. They use +// compiler builtins where available and plain C++ otherwise, so they need no +// platform headers and work regardless of byte order. + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// number of leading zero bits of x (x != 0) +inline int count_leading_zeros(std::uint64_t x) noexcept +{ +#if defined(__GNUC__) || defined(__clang__) + return __builtin_clzll(x); +#else + int n = 0; + for (int shift = 32; shift != 0; shift >>= 1) + { + if ((x >> (64 - shift)) == 0) + { + n += shift; + x <<= shift; + } + } + return n; +#endif +} + +/// the 128-bit product of two 64-bit numbers +struct uint128_parts +{ + std::uint64_t low; + std::uint64_t high; +}; + +inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexcept +{ +#if defined(__SIZEOF_INT128__) + __extension__ using uint128 = unsigned __int128; + const uint128 r = static_cast(a) * b; + return {static_cast(r), static_cast(r >> 64u)}; +#else + const std::uint64_t a_lo = a & 0xFFFFFFFFu; + const std::uint64_t a_hi = a >> 32u; + const std::uint64_t b_lo = b & 0xFFFFFFFFu; + const std::uint64_t b_hi = b >> 32u; + const std::uint64_t lo_lo = a_lo * b_lo; + const std::uint64_t hi_lo = a_hi * b_lo; + const std::uint64_t lo_hi = a_lo * b_hi; + const std::uint64_t hi_hi = a_hi * b_hi; + const std::uint64_t cross = (lo_lo >> 32u) + (hi_lo & 0xFFFFFFFFu) + lo_hi; + return {(cross << 32u) | (lo_lo & 0xFFFFFFFFu), (hi_lo >> 32u) + (cross >> 32u) + hi_hi}; +#endif +} + +/// eight bytes as a little-endian word (compilers fold this into one load on +/// little-endian targets) +inline std::uint64_t read_eight_bytes(const char* p) noexcept +{ + const auto* b = reinterpret_cast(p); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return static_cast(b[0]) | (static_cast(b[1]) << 8u) + | (static_cast(b[2]) << 16u) | (static_cast(b[3]) << 24u) + | (static_cast(b[4]) << 32u) | (static_cast(b[5]) << 40u) + | (static_cast(b[6]) << 48u) | (static_cast(b[7]) << 56u); +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/input/number_parse.hpp b/include/nlohmann/detail/input/number_parse.hpp index c3b6cd28c..b85e758cb 100644 --- a/include/nlohmann/detail/input/number_parse.hpp +++ b/include/nlohmann/detail/input/number_parse.hpp @@ -3,6 +3,7 @@ // | | |__ | | | | | | version 3.12.0 // |_____|_____|_____|_|___| https://github.com/nlohmann/json // +// SPDX-FileCopyrightText: 2021 The fast_float authors // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT @@ -14,9 +15,12 @@ #include // size_t #include // int64_t, uint64_t #include // strtof, strtod, strtold +#include // memcpy #include // numeric_limits #include // string +#include +#include #include // std::from_chars lives in , but being in C++17 mode does not @@ -297,6 +301,233 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out) #endif } +/// whether the eight bytes of @a v (see read_eight_bytes()) are ASCII digits +/// (after fast_float's is_made_of_eight_digits_fast) +inline bool is_eight_digits(std::uint64_t v) noexcept +{ + return ((v & 0xF0F0F0F0F0F0F0F0u) | (((v + 0x0606060606060606u) & 0xF0F0F0F0F0F0F0F0u) >> 4u)) == 0x3333333333333333u; +} + +/// the value of the eight ASCII digits in @a v (see read_eight_bytes()), three +/// multiplications instead of eight (after simdjson and fast_float) +inline std::uint32_t parse_eight_digits(std::uint64_t v) noexcept +{ + v = ((v & 0x0F0F0F0F0F0F0F0Fu) * 2561u) >> 8u; + v = ((v & 0x00FF00FF00FF00FFu) * 6553601u) >> 16u; + return static_cast(((v & 0x0000FFFF0000FFFFu) * 42949672960001u) >> 32u); +} + +/*! +@brief the double nearest to w * 10^q (Eisel-Lemire) + +The algorithm of Daniel Lemire, "Number Parsing at a Gigabyte per Second" +(Software: Practice and Experience, 2021), after fast_float's compute_float +(used under the MIT license). With a 128-bit approximation of 5^q, the product +is always sufficient to round correctly for w with at most 19 digits (Noble +Mushtak and Daniel Lemire, "Fast number parsing without fallback", Software: +Practice and Experience, 2023). Only integer arithmetic is used, so the result +does not depend on the floating-point environment. + +@param[in] q decimal exponent +@param[in] w significand, w != 0 +@return the IEEE-754 bits of the positive result (0 for underflow, infinity + for overflow) +*/ +inline std::uint64_t eisel_lemire(std::int64_t q, std::uint64_t w) noexcept +{ + constexpr int mantissa_bits = 52; + constexpr std::uint64_t infinity = std::uint64_t{0x7FF} << mantissa_bits; + if (q < pow5_128_smallest_power) + { + return 0; + } + if (q > pow5_128_largest_power) + { + return infinity; + } + + const int lz = count_leading_zeros(w); + w <<= static_cast(lz); + const auto index = static_cast(2 * (q - pow5_128_smallest_power)); + uint128_parts product = full_multiplication(w, pow5_128()[index]); + constexpr std::uint64_t precision_mask = 0xFFFFFFFFFFFFFFFFu >> (mantissa_bits + 3); + if ((product.high & precision_mask) == precision_mask) + { + // the lower bits may carry into the result: use the next 64 bits of 5^q + const uint128_parts second = full_multiplication(w, pow5_128()[index + 1]); + product.low += second.high; + if (second.high > product.low) + { + ++product.high; + } + } + + const auto upperbit = static_cast(product.high >> 63u); + const int shift = upperbit + 64 - mantissa_bits - 3; + std::uint64_t mantissa = product.high >> static_cast(shift); + // floor(log2(10^q)) + 63 + 1023, with log2(10) ~ 217706 / 2^16 + std::int64_t power2 = (((152170 + 65536) * q) >> 16) + 63 + upperbit - lz + 1023; + + if (power2 <= 0) // subnormal + { + if (-power2 + 1 >= 64) + { + return 0; + } + mantissa >>= static_cast(-power2 + 1); + mantissa += (mantissa & 1u); + mantissa >>= 1u; + // rounding up may produce the smallest normal number + power2 = (mantissa < (std::uint64_t{1} << mantissa_bits)) ? 0 : 1; + return mantissa | (static_cast(power2) << mantissa_bits); + } + + // a value exactly between two doubles rounds to even; this can only + // happen for small |q|, where 5^q is exact + if (product.low <= 1 && q >= -4 && q <= 23 && (mantissa & 3u) == 1 + && (mantissa << static_cast(shift)) == product.high) + { + mantissa &= ~std::uint64_t{1}; + } + mantissa += (mantissa & 1u); + mantissa >>= 1u; + if (mantissa >= (std::uint64_t{2} << mantissa_bits)) + { + mantissa = std::uint64_t{1} << mantissa_bits; + ++power2; + } + mantissa &= ~(std::uint64_t{1} << mantissa_bits); + if (power2 >= 0x7FF) + { + return infinity; + } + return mantissa | (static_cast(power2) << mantissa_bits); +} + +/*! +@brief parse a validated float token with the Eisel-Lemire algorithm + +The significand is accumulated eight digits at a time where possible. A token +with more than 19 significant digits is truncated to w; the value then lies +in [w, w + 1) * 10^q, and it is only returned if both ends round to the same +double, which covers all but a few such tokens. + +@param[in] first pointer to the first character of the token +@param[in] last pointer past the last character +@param[out] out the correctly rounded value on success (±infinity if it + overflows, like strtod) +@return true on success; false if strtod must decide +*/ +inline bool parse_float_eisel_lemire(const char* first, const char* last, double& out) noexcept +{ + const char* p = first; + const bool negative = (p != last && *p == '-'); + if (negative) + { + ++p; + } + + std::uint64_t w = 0; + int digits = 0; // significant digits in w + std::int64_t exponent = 0; + bool truncated = false; + bool in_fraction = false; + for (;;) + { + // eight digits at a time, as long as they fit into w + while (w != 0 && digits <= 19 - 8 && last - p >= 8) + { + const std::uint64_t v = read_eight_bytes(p); + if (!is_eight_digits(v)) + { + break; + } + w = (w * 100000000u) + parse_eight_digits(v); + digits += 8; + exponent -= in_fraction ? 8 : 0; + p += 8; + } + if (p == last) + { + break; + } + const char c = *p; + if (c >= '0' && c <= '9') + { + if (w == 0 && c == '0') + { + // leading zeros are not significant, but scale a fraction + exponent -= in_fraction ? 1 : 0; + } + else if (digits < 19) + { + w = (w * 10u) + static_cast(c - '0'); + ++digits; + exponent -= in_fraction ? 1 : 0; + } + else + { + // dropped: the value lies between w and w + 1 (in units of + // the last kept digit) unless all dropped digits are zero + truncated = truncated || c != '0'; + exponent += in_fraction ? 0 : 1; + } + ++p; + } + else if (c == '.') + { + in_fraction = true; + ++p; + } + else + { + break; // 'e' or 'E' + } + } + + if (p != last) + { + ++p; // 'e' or 'E' + bool exp_negative = false; + if (p != last && (*p == '-' || *p == '+')) + { + exp_negative = (*p == '-'); + ++p; + } + std::int64_t exp_value = 0; + for (; p != last; ++p) + { + // saturate: any exponent beyond this under- or overflows anyway + if (exp_value < 100000) + { + exp_value = (exp_value * 10) + (*p - '0'); + } + } + exponent += exp_negative ? -exp_value : exp_value; + } + + std::uint64_t bits = 0; + if (w != 0) + { + bits = eisel_lemire(exponent, w); + if (truncated && (w + 1 == 0 || eisel_lemire(exponent, w + 1) != bits)) + { + return false; + } + } + bits |= negative ? (std::uint64_t{1} << 63u) : 0u; + static_assert(sizeof(double) == sizeof(std::uint64_t), "double must have 64 bits"); + std::memcpy(&out, &bits, sizeof(out)); + return true; +} + +/// Eisel-Lemire is only implemented for `double` +template +bool parse_float_eisel_lemire(const char* /*first*/, const char* /*last*/, FloatType& /*out*/) noexcept +{ + return false; +} + /*! @brief check whether Clinger's fast path can still succeed for a float token @@ -355,8 +586,9 @@ inline bool mantissa_fits_clinger(const char* token, std::size_t decimal_point_p /*! @brief convert a validated float token without the C library, if possible -Tries std::from_chars (when available) and then Clinger's exact fast path -(double only), skipping the latter when it cannot succeed. +Tries std::from_chars (when available), Clinger's exact fast path (double +only, skipped when it cannot succeed), and the Eisel-Lemire algorithm (double +only). @param[in] first pointer to the first character of the token @param[in] last pointer past the last character @@ -379,8 +611,12 @@ bool convert_float_fast(const char* first, const char* last, std::size_t decimal // Skipping a fast path that cannot succeed is lossless and saves a full // extra pass over the token's bytes, which otherwise shows up on // high-precision inputs such as canada.json - return mantissa_fits_clinger(first, decimal_point_position, mantissa_end) - && parse_float_fast(first, last, value); + if (mantissa_fits_clinger(first, decimal_point_position, mantissa_end) + && parse_float_fast(first, last, value)) + { + return true; + } + return parse_float_eisel_lemire(first, last, value); } /// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f diff --git a/include/nlohmann/detail/input/pow5_table.hpp b/include/nlohmann/detail/input/pow5_table.hpp new file mode 100644 index 000000000..496fdb2f7 --- /dev/null +++ b/include/nlohmann/detail/input/pow5_table.hpp @@ -0,0 +1,371 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2021 The fast_float authors +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // array +#include // int64_t, uint64_t + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// the range of decimal exponents covered by pow5_128() +constexpr std::int64_t pow5_128_smallest_power = -342; +constexpr std::int64_t pow5_128_largest_power = 308; + +/*! +@brief 128-bit approximations of 5^q for q in [-342, 308] + +Entry q (at index 2 * (q + 342)) holds the most significant 128 bits of 5^q, +normalized so that the highest bit is set: for q >= 0 the truncated value, for +q < 0 the value rounded up. This is the table of fast_float (Daniel Lemire and +contributors, used under the MIT license), generated like its +script/table_generation.py; unit-class_lexer.cpp recomputes every entry. +*/ +inline const std::array& pow5_128() noexcept +{ + static const std::array table = + { + { + 0xeef453d6923bd65au, 0x113faa2906a13b3fu, 0x9558b4661b6565f8u, 0x4ac7ca59a424c507u, + 0xbaaee17fa23ebf76u, 0x5d79bcf00d2df649u, 0xe95a99df8ace6f53u, 0xf4d82c2c107973dcu, + 0x91d8a02bb6c10594u, 0x79071b9b8a4be869u, 0xb64ec836a47146f9u, 0x9748e2826cdee284u, + 0xe3e27a444d8d98b7u, 0xfd1b1b2308169b25u, 0x8e6d8c6ab0787f72u, 0xfe30f0f5e50e20f7u, + 0xb208ef855c969f4fu, 0xbdbd2d335e51a935u, 0xde8b2b66b3bc4723u, 0xad2c788035e61382u, + 0x8b16fb203055ac76u, 0x4c3bcb5021afcc31u, 0xaddcb9e83c6b1793u, 0xdf4abe242a1bbf3du, + 0xd953e8624b85dd78u, 0xd71d6dad34a2af0du, 0x87d4713d6f33aa6bu, 0x8672648c40e5ad68u, + 0xa9c98d8ccb009506u, 0x680efdaf511f18c2u, 0xd43bf0effdc0ba48u, 0x0212bd1b2566def2u, + 0x84a57695fe98746du, 0x014bb630f7604b57u, 0xa5ced43b7e3e9188u, 0x419ea3bd35385e2du, + 0xcf42894a5dce35eau, 0x52064cac828675b9u, 0x818995ce7aa0e1b2u, 0x7343efebd1940993u, + 0xa1ebfb4219491a1fu, 0x1014ebe6c5f90bf8u, 0xca66fa129f9b60a6u, 0xd41a26e077774ef6u, + 0xfd00b897478238d0u, 0x8920b098955522b4u, 0x9e20735e8cb16382u, 0x55b46e5f5d5535b0u, + 0xc5a890362fddbc62u, 0xeb2189f734aa831du, 0xf712b443bbd52b7bu, 0xa5e9ec7501d523e4u, + 0x9a6bb0aa55653b2du, 0x47b233c92125366eu, 0xc1069cd4eabe89f8u, 0x999ec0bb696e840au, + 0xf148440a256e2c76u, 0xc00670ea43ca250du, 0x96cd2a865764dbcau, 0x380406926a5e5728u, + 0xbc807527ed3e12bcu, 0xc605083704f5ecf2u, 0xeba09271e88d976bu, 0xf7864a44c633682eu, + 0x93445b8731587ea3u, 0x7ab3ee6afbe0211du, 0xb8157268fdae9e4cu, 0x5960ea05bad82964u, + 0xe61acf033d1a45dfu, 0x6fb92487298e33bdu, 0x8fd0c16206306babu, 0xa5d3b6d479f8e056u, + 0xb3c4f1ba87bc8696u, 0x8f48a4899877186cu, 0xe0b62e2929aba83cu, 0x331acdabfe94de87u, + 0x8c71dcd9ba0b4925u, 0x9ff0c08b7f1d0b14u, 0xaf8e5410288e1b6fu, 0x07ecf0ae5ee44dd9u, + 0xdb71e91432b1a24au, 0xc9e82cd9f69d6150u, 0x892731ac9faf056eu, 0xbe311c083a225cd2u, + 0xab70fe17c79ac6cau, 0x6dbd630a48aaf406u, 0xd64d3d9db981787du, 0x092cbbccdad5b108u, + 0x85f0468293f0eb4eu, 0x25bbf56008c58ea5u, 0xa76c582338ed2621u, 0xaf2af2b80af6f24eu, + 0xd1476e2c07286faau, 0x1af5af660db4aee1u, 0x82cca4db847945cau, 0x50d98d9fc890ed4du, + 0xa37fce126597973cu, 0xe50ff107bab528a0u, 0xcc5fc196fefd7d0cu, 0x1e53ed49a96272c8u, + 0xff77b1fcbebcdc4fu, 0x25e8e89c13bb0f7au, 0x9faacf3df73609b1u, 0x77b191618c54e9acu, + 0xc795830d75038c1du, 0xd59df5b9ef6a2417u, 0xf97ae3d0d2446f25u, 0x4b0573286b44ad1du, + 0x9becce62836ac577u, 0x4ee367f9430aec32u, 0xc2e801fb244576d5u, 0x229c41f793cda73fu, + 0xf3a20279ed56d48au, 0x6b43527578c1110fu, 0x9845418c345644d6u, 0x830a13896b78aaa9u, + 0xbe5691ef416bd60cu, 0x23cc986bc656d553u, 0xedec366b11c6cb8fu, 0x2cbfbe86b7ec8aa8u, + 0x94b3a202eb1c3f39u, 0x7bf7d71432f3d6a9u, 0xb9e08a83a5e34f07u, 0xdaf5ccd93fb0cc53u, + 0xe858ad248f5c22c9u, 0xd1b3400f8f9cff68u, 0x91376c36d99995beu, 0x23100809b9c21fa1u, + 0xb58547448ffffb2du, 0xabd40a0c2832a78au, 0xe2e69915b3fff9f9u, 0x16c90c8f323f516cu, + 0x8dd01fad907ffc3bu, 0xae3da7d97f6792e3u, 0xb1442798f49ffb4au, 0x99cd11cfdf41779cu, + 0xdd95317f31c7fa1du, 0x40405643d711d583u, 0x8a7d3eef7f1cfc52u, 0x482835ea666b2572u, + 0xad1c8eab5ee43b66u, 0xda3243650005eecfu, 0xd863b256369d4a40u, 0x90bed43e40076a82u, + 0x873e4f75e2224e68u, 0x5a7744a6e804a291u, 0xa90de3535aaae202u, 0x711515d0a205cb36u, + 0xd3515c2831559a83u, 0x0d5a5b44ca873e03u, 0x8412d9991ed58091u, 0xe858790afe9486c2u, + 0xa5178fff668ae0b6u, 0x626e974dbe39a872u, 0xce5d73ff402d98e3u, 0xfb0a3d212dc8128fu, + 0x80fa687f881c7f8eu, 0x7ce66634bc9d0b99u, 0xa139029f6a239f72u, 0x1c1fffc1ebc44e80u, + 0xc987434744ac874eu, 0xa327ffb266b56220u, 0xfbe9141915d7a922u, 0x4bf1ff9f0062baa8u, + 0x9d71ac8fada6c9b5u, 0x6f773fc3603db4a9u, 0xc4ce17b399107c22u, 0xcb550fb4384d21d3u, + 0xf6019da07f549b2bu, 0x7e2a53a146606a48u, 0x99c102844f94e0fbu, 0x2eda7444cbfc426du, + 0xc0314325637a1939u, 0xfa911155fefb5308u, 0xf03d93eebc589f88u, 0x793555ab7eba27cau, + 0x96267c7535b763b5u, 0x4bc1558b2f3458deu, 0xbbb01b9283253ca2u, 0x9eb1aaedfb016f16u, + 0xea9c227723ee8bcbu, 0x465e15a979c1cadcu, 0x92a1958a7675175fu, 0x0bfacd89ec191ec9u, + 0xb749faed14125d36u, 0xcef980ec671f667bu, 0xe51c79a85916f484u, 0x82b7e12780e7401au, + 0x8f31cc0937ae58d2u, 0xd1b2ecb8b0908810u, 0xb2fe3f0b8599ef07u, 0x861fa7e6dcb4aa15u, + 0xdfbdcece67006ac9u, 0x67a791e093e1d49au, 0x8bd6a141006042bdu, 0xe0c8bb2c5c6d24e0u, + 0xaecc49914078536du, 0x58fae9f773886e18u, 0xda7f5bf590966848u, 0xaf39a475506a899eu, + 0x888f99797a5e012du, 0x6d8406c952429603u, 0xaab37fd7d8f58178u, 0xc8e5087ba6d33b83u, + 0xd5605fcdcf32e1d6u, 0xfb1e4a9a90880a64u, 0x855c3be0a17fcd26u, 0x5cf2eea09a55067fu, + 0xa6b34ad8c9dfc06fu, 0xf42faa48c0ea481eu, 0xd0601d8efc57b08bu, 0xf13b94daf124da26u, + 0x823c12795db6ce57u, 0x76c53d08d6b70858u, 0xa2cb1717b52481edu, 0x54768c4b0c64ca6eu, + 0xcb7ddcdda26da268u, 0xa9942f5dcf7dfd09u, 0xfe5d54150b090b02u, 0xd3f93b35435d7c4cu, + 0x9efa548d26e5a6e1u, 0xc47bc5014a1a6dafu, 0xc6b8e9b0709f109au, 0x359ab6419ca1091bu, + 0xf867241c8cc6d4c0u, 0xc30163d203c94b62u, 0x9b407691d7fc44f8u, 0x79e0de63425dcf1du, + 0xc21094364dfb5636u, 0x985915fc12f542e4u, 0xf294b943e17a2bc4u, 0x3e6f5b7b17b2939du, + 0x979cf3ca6cec5b5au, 0xa705992ceecf9c42u, 0xbd8430bd08277231u, 0x50c6ff782a838353u, + 0xece53cec4a314ebdu, 0xa4f8bf5635246428u, 0x940f4613ae5ed136u, 0x871b7795e136be99u, + 0xb913179899f68584u, 0x28e2557b59846e3fu, 0xe757dd7ec07426e5u, 0x331aeada2fe589cfu, + 0x9096ea6f3848984fu, 0x3ff0d2c85def7621u, 0xb4bca50b065abe63u, 0x0fed077a756b53a9u, + 0xe1ebce4dc7f16dfbu, 0xd3e8495912c62894u, 0x8d3360f09cf6e4bdu, 0x64712dd7abbbd95cu, + 0xb080392cc4349decu, 0xbd8d794d96aacfb3u, 0xdca04777f541c567u, 0xecf0d7a0fc5583a0u, + 0x89e42caaf9491b60u, 0xf41686c49db57244u, 0xac5d37d5b79b6239u, 0x311c2875c522ced5u, + 0xd77485cb25823ac7u, 0x7d633293366b828bu, 0x86a8d39ef77164bcu, 0xae5dff9c02033197u, + 0xa8530886b54dbdebu, 0xd9f57f830283fdfcu, 0xd267caa862a12d66u, 0xd072df63c324fd7bu, + 0x8380dea93da4bc60u, 0x4247cb9e59f71e6du, 0xa46116538d0deb78u, 0x52d9be85f074e608u, + 0xcd795be870516656u, 0x67902e276c921f8bu, 0x806bd9714632dff6u, 0x00ba1cd8a3db53b6u, + 0xa086cfcd97bf97f3u, 0x80e8a40eccd228a4u, 0xc8a883c0fdaf7df0u, 0x6122cd128006b2cdu, + 0xfad2a4b13d1b5d6cu, 0x796b805720085f81u, 0x9cc3a6eec6311a63u, 0xcbe3303674053bb0u, + 0xc3f490aa77bd60fcu, 0xbedbfc4411068a9cu, 0xf4f1b4d515acb93bu, 0xee92fb5515482d44u, + 0x991711052d8bf3c5u, 0x751bdd152d4d1c4au, 0xbf5cd54678eef0b6u, 0xd262d45a78a0635du, + 0xef340a98172aace4u, 0x86fb897116c87c34u, 0x9580869f0e7aac0eu, 0xd45d35e6ae3d4da0u, + 0xbae0a846d2195712u, 0x8974836059cca109u, 0xe998d258869facd7u, 0x2bd1a438703fc94bu, + 0x91ff83775423cc06u, 0x7b6306a34627ddcfu, 0xb67f6455292cbf08u, 0x1a3bc84c17b1d542u, + 0xe41f3d6a7377eecau, 0x20caba5f1d9e4a93u, 0x8e938662882af53eu, 0x547eb47b7282ee9cu, + 0xb23867fb2a35b28du, 0xe99e619a4f23aa43u, 0xdec681f9f4c31f31u, 0x6405fa00e2ec94d4u, + 0x8b3c113c38f9f37eu, 0xde83bc408dd3dd04u, 0xae0b158b4738705eu, 0x9624ab50b148d445u, + 0xd98ddaee19068c76u, 0x3badd624dd9b0957u, 0x87f8a8d4cfa417c9u, 0xe54ca5d70a80e5d6u, + 0xa9f6d30a038d1dbcu, 0x5e9fcf4ccd211f4cu, 0xd47487cc8470652bu, 0x7647c3200069671fu, + 0x84c8d4dfd2c63f3bu, 0x29ecd9f40041e073u, 0xa5fb0a17c777cf09u, 0xf468107100525890u, + 0xcf79cc9db955c2ccu, 0x7182148d4066eeb4u, 0x81ac1fe293d599bfu, 0xc6f14cd848405530u, + 0xa21727db38cb002fu, 0xb8ada00e5a506a7cu, 0xca9cf1d206fdc03bu, 0xa6d90811f0e4851cu, + 0xfd442e4688bd304au, 0x908f4a166d1da663u, 0x9e4a9cec15763e2eu, 0x9a598e4e043287feu, + 0xc5dd44271ad3cdbau, 0x40eff1e1853f29fdu, 0xf7549530e188c128u, 0xd12bee59e68ef47cu, + 0x9a94dd3e8cf578b9u, 0x82bb74f8301958ceu, 0xc13a148e3032d6e7u, 0xe36a52363c1faf01u, + 0xf18899b1bc3f8ca1u, 0xdc44e6c3cb279ac1u, 0x96f5600f15a7b7e5u, 0x29ab103a5ef8c0b9u, + 0xbcb2b812db11a5deu, 0x7415d448f6b6f0e7u, 0xebdf661791d60f56u, 0x111b495b3464ad21u, + 0x936b9fcebb25c995u, 0xcab10dd900beec34u, 0xb84687c269ef3bfbu, 0x3d5d514f40eea742u, + 0xe65829b3046b0afau, 0x0cb4a5a3112a5112u, 0x8ff71a0fe2c2e6dcu, 0x47f0e785eaba72abu, + 0xb3f4e093db73a093u, 0x59ed216765690f56u, 0xe0f218b8d25088b8u, 0x306869c13ec3532cu, + 0x8c974f7383725573u, 0x1e414218c73a13fbu, 0xafbd2350644eeacfu, 0xe5d1929ef90898fau, + 0xdbac6c247d62a583u, 0xdf45f746b74abf39u, 0x894bc396ce5da772u, 0x6b8bba8c328eb783u, + 0xab9eb47c81f5114fu, 0x066ea92f3f326564u, 0xd686619ba27255a2u, 0xc80a537b0efefebdu, + 0x8613fd0145877585u, 0xbd06742ce95f5f36u, 0xa798fc4196e952e7u, 0x2c48113823b73704u, + 0xd17f3b51fca3a7a0u, 0xf75a15862ca504c5u, 0x82ef85133de648c4u, 0x9a984d73dbe722fbu, + 0xa3ab66580d5fdaf5u, 0xc13e60d0d2e0ebbau, 0xcc963fee10b7d1b3u, 0x318df905079926a8u, + 0xffbbcfe994e5c61fu, 0xfdf17746497f7052u, 0x9fd561f1fd0f9bd3u, 0xfeb6ea8bedefa633u, + 0xc7caba6e7c5382c8u, 0xfe64a52ee96b8fc0u, 0xf9bd690a1b68637bu, 0x3dfdce7aa3c673b0u, + 0x9c1661a651213e2du, 0x06bea10ca65c084eu, 0xc31bfa0fe5698db8u, 0x486e494fcff30a62u, + 0xf3e2f893dec3f126u, 0x5a89dba3c3efccfau, 0x986ddb5c6b3a76b7u, 0xf89629465a75e01cu, + 0xbe89523386091465u, 0xf6bbb397f1135823u, 0xee2ba6c0678b597fu, 0x746aa07ded582e2cu, + 0x94db483840b717efu, 0xa8c2a44eb4571cdcu, 0xba121a4650e4ddebu, 0x92f34d62616ce413u, + 0xe896a0d7e51e1566u, 0x77b020baf9c81d17u, 0x915e2486ef32cd60u, 0x0ace1474dc1d122eu, + 0xb5b5ada8aaff80b8u, 0x0d819992132456bau, 0xe3231912d5bf60e6u, 0x10e1fff697ed6c69u, + 0x8df5efabc5979c8fu, 0xca8d3ffa1ef463c1u, 0xb1736b96b6fd83b3u, 0xbd308ff8a6b17cb2u, + 0xddd0467c64bce4a0u, 0xac7cb3f6d05ddbdeu, 0x8aa22c0dbef60ee4u, 0x6bcdf07a423aa96bu, + 0xad4ab7112eb3929du, 0x86c16c98d2c953c6u, 0xd89d64d57a607744u, 0xe871c7bf077ba8b7u, + 0x87625f056c7c4a8bu, 0x11471cd764ad4972u, 0xa93af6c6c79b5d2du, 0xd598e40d3dd89bcfu, + 0xd389b47879823479u, 0x4aff1d108d4ec2c3u, 0x843610cb4bf160cbu, 0xcedf722a585139bau, + 0xa54394fe1eedb8feu, 0xc2974eb4ee658828u, 0xce947a3da6a9273eu, 0x733d226229feea32u, + 0x811ccc668829b887u, 0x0806357d5a3f525fu, 0xa163ff802a3426a8u, 0xca07c2dcb0cf26f7u, + 0xc9bcff6034c13052u, 0xfc89b393dd02f0b5u, 0xfc2c3f3841f17c67u, 0xbbac2078d443ace2u, + 0x9d9ba7832936edc0u, 0xd54b944b84aa4c0du, 0xc5029163f384a931u, 0x0a9e795e65d4df11u, + 0xf64335bcf065d37du, 0x4d4617b5ff4a16d5u, 0x99ea0196163fa42eu, 0x504bced1bf8e4e45u, + 0xc06481fb9bcf8d39u, 0xe45ec2862f71e1d6u, 0xf07da27a82c37088u, 0x5d767327bb4e5a4cu, + 0x964e858c91ba2655u, 0x3a6a07f8d510f86fu, 0xbbe226efb628afeau, 0x890489f70a55368bu, + 0xeadab0aba3b2dbe5u, 0x2b45ac74ccea842eu, 0x92c8ae6b464fc96fu, 0x3b0b8bc90012929du, + 0xb77ada0617e3bbcbu, 0x09ce6ebb40173744u, 0xe55990879ddcaabdu, 0xcc420a6a101d0515u, + 0x8f57fa54c2a9eab6u, 0x9fa946824a12232du, 0xb32df8e9f3546564u, 0x47939822dc96abf9u, + 0xdff9772470297ebdu, 0x59787e2b93bc56f7u, 0x8bfbea76c619ef36u, 0x57eb4edb3c55b65au, + 0xaefae51477a06b03u, 0xede622920b6b23f1u, 0xdab99e59958885c4u, 0xe95fab368e45ecedu, + 0x88b402f7fd75539bu, 0x11dbcb0218ebb414u, 0xaae103b5fcd2a881u, 0xd652bdc29f26a119u, + 0xd59944a37c0752a2u, 0x4be76d3346f0495fu, 0x857fcae62d8493a5u, 0x6f70a4400c562ddbu, + 0xa6dfbd9fb8e5b88eu, 0xcb4ccd500f6bb952u, 0xd097ad07a71f26b2u, 0x7e2000a41346a7a7u, + 0x825ecc24c873782fu, 0x8ed400668c0c28c8u, 0xa2f67f2dfa90563bu, 0x728900802f0f32fau, + 0xcbb41ef979346bcau, 0x4f2b40a03ad2ffb9u, 0xfea126b7d78186bcu, 0xe2f610c84987bfa8u, + 0x9f24b832e6b0f436u, 0x0dd9ca7d2df4d7c9u, 0xc6ede63fa05d3143u, 0x91503d1c79720dbbu, + 0xf8a95fcf88747d94u, 0x75a44c6397ce912au, 0x9b69dbe1b548ce7cu, 0xc986afbe3ee11abau, + 0xc24452da229b021bu, 0xfbe85badce996168u, 0xf2d56790ab41c2a2u, 0xfae27299423fb9c3u, + 0x97c560ba6b0919a5u, 0xdccd879fc967d41au, 0xbdb6b8e905cb600fu, 0x5400e987bbc1c920u, + 0xed246723473e3813u, 0x290123e9aab23b68u, 0x9436c0760c86e30bu, 0xf9a0b6720aaf6521u, + 0xb94470938fa89bceu, 0xf808e40e8d5b3e69u, 0xe7958cb87392c2c2u, 0xb60b1d1230b20e04u, + 0x90bd77f3483bb9b9u, 0xb1c6f22b5e6f48c2u, 0xb4ecd5f01a4aa828u, 0x1e38aeb6360b1af3u, + 0xe2280b6c20dd5232u, 0x25c6da63c38de1b0u, 0x8d590723948a535fu, 0x579c487e5a38ad0eu, + 0xb0af48ec79ace837u, 0x2d835a9df0c6d851u, 0xdcdb1b2798182244u, 0xf8e431456cf88e65u, + 0x8a08f0f8bf0f156bu, 0x1b8e9ecb641b58ffu, 0xac8b2d36eed2dac5u, 0xe272467e3d222f3fu, + 0xd7adf884aa879177u, 0x5b0ed81dcc6abb0fu, 0x86ccbb52ea94baeau, 0x98e947129fc2b4e9u, + 0xa87fea27a539e9a5u, 0x3f2398d747b36224u, 0xd29fe4b18e88640eu, 0x8eec7f0d19a03aadu, + 0x83a3eeeef9153e89u, 0x1953cf68300424acu, 0xa48ceaaab75a8e2bu, 0x5fa8c3423c052dd7u, + 0xcdb02555653131b6u, 0x3792f412cb06794du, 0x808e17555f3ebf11u, 0xe2bbd88bbee40bd0u, + 0xa0b19d2ab70e6ed6u, 0x5b6aceaeae9d0ec4u, 0xc8de047564d20a8bu, 0xf245825a5a445275u, + 0xfb158592be068d2eu, 0xeed6e2f0f0d56712u, 0x9ced737bb6c4183du, 0x55464dd69685606bu, + 0xc428d05aa4751e4cu, 0xaa97e14c3c26b886u, 0xf53304714d9265dfu, 0xd53dd99f4b3066a8u, + 0x993fe2c6d07b7fabu, 0xe546a8038efe4029u, 0xbf8fdb78849a5f96u, 0xde98520472bdd033u, + 0xef73d256a5c0f77cu, 0x963e66858f6d4440u, 0x95a8637627989aadu, 0xdde7001379a44aa8u, + 0xbb127c53b17ec159u, 0x5560c018580d5d52u, 0xe9d71b689dde71afu, 0xaab8f01e6e10b4a6u, + 0x9226712162ab070du, 0xcab3961304ca70e8u, 0xb6b00d69bb55c8d1u, 0x3d607b97c5fd0d22u, + 0xe45c10c42a2b3b05u, 0x8cb89a7db77c506au, 0x8eb98a7a9a5b04e3u, 0x77f3608e92adb242u, + 0xb267ed1940f1c61cu, 0x55f038b237591ed3u, 0xdf01e85f912e37a3u, 0x6b6c46dec52f6688u, + 0x8b61313bbabce2c6u, 0x2323ac4b3b3da015u, 0xae397d8aa96c1b77u, 0xabec975e0a0d081au, + 0xd9c7dced53c72255u, 0x96e7bd358c904a21u, 0x881cea14545c7575u, 0x7e50d64177da2e54u, + 0xaa242499697392d2u, 0xdde50bd1d5d0b9e9u, 0xd4ad2dbfc3d07787u, 0x955e4ec64b44e864u, + 0x84ec3c97da624ab4u, 0xbd5af13bef0b113eu, 0xa6274bbdd0fadd61u, 0xecb1ad8aeacdd58eu, + 0xcfb11ead453994bau, 0x67de18eda5814af2u, 0x81ceb32c4b43fcf4u, 0x80eacf948770ced7u, + 0xa2425ff75e14fc31u, 0xa1258379a94d028du, 0xcad2f7f5359a3b3eu, 0x096ee45813a04330u, + 0xfd87b5f28300ca0du, 0x8bca9d6e188853fcu, 0x9e74d1b791e07e48u, 0x775ea264cf55347eu, + 0xc612062576589ddau, 0x95364afe032a819eu, 0xf79687aed3eec551u, 0x3a83ddbd83f52205u, + 0x9abe14cd44753b52u, 0xc4926a9672793543u, 0xc16d9a0095928a27u, 0x75b7053c0f178294u, + 0xf1c90080baf72cb1u, 0x5324c68b12dd6339u, 0x971da05074da7beeu, 0xd3f6fc16ebca5e04u, + 0xbce5086492111aeau, 0x88f4bb1ca6bcf585u, 0xec1e4a7db69561a5u, 0x2b31e9e3d06c32e6u, + 0x9392ee8e921d5d07u, 0x3aff322e62439fd0u, 0xb877aa3236a4b449u, 0x09befeb9fad487c3u, + 0xe69594bec44de15bu, 0x4c2ebe687989a9b4u, 0x901d7cf73ab0acd9u, 0x0f9d37014bf60a11u, + 0xb424dc35095cd80fu, 0x538484c19ef38c95u, 0xe12e13424bb40e13u, 0x2865a5f206b06fbau, + 0x8cbccc096f5088cbu, 0xf93f87b7442e45d4u, 0xafebff0bcb24aafeu, 0xf78f69a51539d749u, + 0xdbe6fecebdedd5beu, 0xb573440e5a884d1cu, 0x89705f4136b4a597u, 0x31680a88f8953031u, + 0xabcc77118461cefcu, 0xfdc20d2b36ba7c3eu, 0xd6bf94d5e57a42bcu, 0x3d32907604691b4du, + 0x8637bd05af6c69b5u, 0xa63f9a49c2c1b110u, 0xa7c5ac471b478423u, 0x0fcf80dc33721d54u, + 0xd1b71758e219652bu, 0xd3c36113404ea4a9u, 0x83126e978d4fdf3bu, 0x645a1cac083126eau, + 0xa3d70a3d70a3d70au, 0x3d70a3d70a3d70a4u, 0xccccccccccccccccu, 0xcccccccccccccccdu, + 0x8000000000000000u, 0x0000000000000000u, 0xa000000000000000u, 0x0000000000000000u, + 0xc800000000000000u, 0x0000000000000000u, 0xfa00000000000000u, 0x0000000000000000u, + 0x9c40000000000000u, 0x0000000000000000u, 0xc350000000000000u, 0x0000000000000000u, + 0xf424000000000000u, 0x0000000000000000u, 0x9896800000000000u, 0x0000000000000000u, + 0xbebc200000000000u, 0x0000000000000000u, 0xee6b280000000000u, 0x0000000000000000u, + 0x9502f90000000000u, 0x0000000000000000u, 0xba43b74000000000u, 0x0000000000000000u, + 0xe8d4a51000000000u, 0x0000000000000000u, 0x9184e72a00000000u, 0x0000000000000000u, + 0xb5e620f480000000u, 0x0000000000000000u, 0xe35fa931a0000000u, 0x0000000000000000u, + 0x8e1bc9bf04000000u, 0x0000000000000000u, 0xb1a2bc2ec5000000u, 0x0000000000000000u, + 0xde0b6b3a76400000u, 0x0000000000000000u, 0x8ac7230489e80000u, 0x0000000000000000u, + 0xad78ebc5ac620000u, 0x0000000000000000u, 0xd8d726b7177a8000u, 0x0000000000000000u, + 0x878678326eac9000u, 0x0000000000000000u, 0xa968163f0a57b400u, 0x0000000000000000u, + 0xd3c21bcecceda100u, 0x0000000000000000u, 0x84595161401484a0u, 0x0000000000000000u, + 0xa56fa5b99019a5c8u, 0x0000000000000000u, 0xcecb8f27f4200f3au, 0x0000000000000000u, + 0x813f3978f8940984u, 0x4000000000000000u, 0xa18f07d736b90be5u, 0x5000000000000000u, + 0xc9f2c9cd04674edeu, 0xa400000000000000u, 0xfc6f7c4045812296u, 0x4d00000000000000u, + 0x9dc5ada82b70b59du, 0xf020000000000000u, 0xc5371912364ce305u, 0x6c28000000000000u, + 0xf684df56c3e01bc6u, 0xc732000000000000u, 0x9a130b963a6c115cu, 0x3c7f400000000000u, + 0xc097ce7bc90715b3u, 0x4b9f100000000000u, 0xf0bdc21abb48db20u, 0x1e86d40000000000u, + 0x96769950b50d88f4u, 0x1314448000000000u, 0xbc143fa4e250eb31u, 0x17d955a000000000u, + 0xeb194f8e1ae525fdu, 0x5dcfab0800000000u, 0x92efd1b8d0cf37beu, 0x5aa1cae500000000u, + 0xb7abc627050305adu, 0xf14a3d9e40000000u, 0xe596b7b0c643c719u, 0x6d9ccd05d0000000u, + 0x8f7e32ce7bea5c6fu, 0xe4820023a2000000u, 0xb35dbf821ae4f38bu, 0xdda2802c8a800000u, + 0xe0352f62a19e306eu, 0xd50b2037ad200000u, 0x8c213d9da502de45u, 0x4526f422cc340000u, + 0xaf298d050e4395d6u, 0x9670b12b7f410000u, 0xdaf3f04651d47b4cu, 0x3c0cdd765f114000u, + 0x88d8762bf324cd0fu, 0xa5880a69fb6ac800u, 0xab0e93b6efee0053u, 0x8eea0d047a457a00u, + 0xd5d238a4abe98068u, 0x72a4904598d6d880u, 0x85a36366eb71f041u, 0x47a6da2b7f864750u, + 0xa70c3c40a64e6c51u, 0x999090b65f67d924u, 0xd0cf4b50cfe20765u, 0xfff4b4e3f741cf6du, + 0x82818f1281ed449fu, 0xbff8f10e7a8921a4u, 0xa321f2d7226895c7u, 0xaff72d52192b6a0du, + 0xcbea6f8ceb02bb39u, 0x9bf4f8a69f764490u, 0xfee50b7025c36a08u, 0x02f236d04753d5b4u, + 0x9f4f2726179a2245u, 0x01d762422c946590u, 0xc722f0ef9d80aad6u, 0x424d3ad2b7b97ef5u, + 0xf8ebad2b84e0d58bu, 0xd2e0898765a7deb2u, 0x9b934c3b330c8577u, 0x63cc55f49f88eb2fu, + 0xc2781f49ffcfa6d5u, 0x3cbf6b71c76b25fbu, 0xf316271c7fc3908au, 0x8bef464e3945ef7au, + 0x97edd871cfda3a56u, 0x97758bf0e3cbb5acu, 0xbde94e8e43d0c8ecu, 0x3d52eeed1cbea317u, + 0xed63a231d4c4fb27u, 0x4ca7aaa863ee4bddu, 0x945e455f24fb1cf8u, 0x8fe8caa93e74ef6au, + 0xb975d6b6ee39e436u, 0xb3e2fd538e122b44u, 0xe7d34c64a9c85d44u, 0x60dbbca87196b616u, + 0x90e40fbeea1d3a4au, 0xbc8955e946fe31cdu, 0xb51d13aea4a488ddu, 0x6babab6398bdbe41u, + 0xe264589a4dcdab14u, 0xc696963c7eed2dd1u, 0x8d7eb76070a08aecu, 0xfc1e1de5cf543ca2u, + 0xb0de65388cc8ada8u, 0x3b25a55f43294bcbu, 0xdd15fe86affad912u, 0x49ef0eb713f39ebeu, + 0x8a2dbf142dfcc7abu, 0x6e3569326c784337u, 0xacb92ed9397bf996u, 0x49c2c37f07965404u, + 0xd7e77a8f87daf7fbu, 0xdc33745ec97be906u, 0x86f0ac99b4e8dafdu, 0x69a028bb3ded71a3u, + 0xa8acd7c0222311bcu, 0xc40832ea0d68ce0cu, 0xd2d80db02aabd62bu, 0xf50a3fa490c30190u, + 0x83c7088e1aab65dbu, 0x792667c6da79e0fau, 0xa4b8cab1a1563f52u, 0x577001b891185938u, + 0xcde6fd5e09abcf26u, 0xed4c0226b55e6f86u, 0x80b05e5ac60b6178u, 0x544f8158315b05b4u, + 0xa0dc75f1778e39d6u, 0x696361ae3db1c721u, 0xc913936dd571c84cu, 0x03bc3a19cd1e38e9u, + 0xfb5878494ace3a5fu, 0x04ab48a04065c723u, 0x9d174b2dcec0e47bu, 0x62eb0d64283f9c76u, + 0xc45d1df942711d9au, 0x3ba5d0bd324f8394u, 0xf5746577930d6500u, 0xca8f44ec7ee36479u, + 0x9968bf6abbe85f20u, 0x7e998b13cf4e1ecbu, 0xbfc2ef456ae276e8u, 0x9e3fedd8c321a67eu, + 0xefb3ab16c59b14a2u, 0xc5cfe94ef3ea101eu, 0x95d04aee3b80ece5u, 0xbba1f1d158724a12u, + 0xbb445da9ca61281fu, 0x2a8a6e45ae8edc97u, 0xea1575143cf97226u, 0xf52d09d71a3293bdu, + 0x924d692ca61be758u, 0x593c2626705f9c56u, 0xb6e0c377cfa2e12eu, 0x6f8b2fb00c77836cu, + 0xe498f455c38b997au, 0x0b6dfb9c0f956447u, 0x8edf98b59a373fecu, 0x4724bd4189bd5eacu, + 0xb2977ee300c50fe7u, 0x58edec91ec2cb657u, 0xdf3d5e9bc0f653e1u, 0x2f2967b66737e3edu, + 0x8b865b215899f46cu, 0xbd79e0d20082ee74u, 0xae67f1e9aec07187u, 0xecd8590680a3aa11u, + 0xda01ee641a708de9u, 0xe80e6f4820cc9495u, 0x884134fe908658b2u, 0x3109058d147fdcddu, + 0xaa51823e34a7eedeu, 0xbd4b46f0599fd415u, 0xd4e5e2cdc1d1ea96u, 0x6c9e18ac7007c91au, + 0x850fadc09923329eu, 0x03e2cf6bc604ddb0u, 0xa6539930bf6bff45u, 0x84db8346b786151cu, + 0xcfe87f7cef46ff16u, 0xe612641865679a63u, 0x81f14fae158c5f6eu, 0x4fcb7e8f3f60c07eu, + 0xa26da3999aef7749u, 0xe3be5e330f38f09du, 0xcb090c8001ab551cu, 0x5cadf5bfd3072cc5u, + 0xfdcb4fa002162a63u, 0x73d9732fc7c8f7f6u, 0x9e9f11c4014dda7eu, 0x2867e7fddcdd9afau, + 0xc646d63501a1511du, 0xb281e1fd541501b8u, 0xf7d88bc24209a565u, 0x1f225a7ca91a4226u, + 0x9ae757596946075fu, 0x3375788de9b06958u, 0xc1a12d2fc3978937u, 0x0052d6b1641c83aeu, + 0xf209787bb47d6b84u, 0xc0678c5dbd23a49au, 0x9745eb4d50ce6332u, 0xf840b7ba963646e0u, + 0xbd176620a501fbffu, 0xb650e5a93bc3d898u, 0xec5d3fa8ce427affu, 0xa3e51f138ab4cebeu, + 0x93ba47c980e98cdfu, 0xc66f336c36b10137u, 0xb8a8d9bbe123f017u, 0xb80b0047445d4184u, + 0xe6d3102ad96cec1du, 0xa60dc059157491e5u, 0x9043ea1ac7e41392u, 0x87c89837ad68db2fu, + 0xb454e4a179dd1877u, 0x29babe4598c311fbu, 0xe16a1dc9d8545e94u, 0xf4296dd6fef3d67au, + 0x8ce2529e2734bb1du, 0x1899e4a65f58660cu, 0xb01ae745b101e9e4u, 0x5ec05dcff72e7f8fu, + 0xdc21a1171d42645du, 0x76707543f4fa1f73u, 0x899504ae72497ebau, 0x6a06494a791c53a8u, + 0xabfa45da0edbde69u, 0x0487db9d17636892u, 0xd6f8d7509292d603u, 0x45a9d2845d3c42b6u, + 0x865b86925b9bc5c2u, 0x0b8a2392ba45a9b2u, 0xa7f26836f282b732u, 0x8e6cac7768d7141eu, + 0xd1ef0244af2364ffu, 0x3207d795430cd926u, 0x8335616aed761f1fu, 0x7f44e6bd49e807b8u, + 0xa402b9c5a8d3a6e7u, 0x5f16206c9c6209a6u, 0xcd036837130890a1u, 0x36dba887c37a8c0fu, + 0x802221226be55a64u, 0xc2494954da2c9789u, 0xa02aa96b06deb0fdu, 0xf2db9baa10b7bd6cu, + 0xc83553c5c8965d3du, 0x6f92829494e5acc7u, 0xfa42a8b73abbf48cu, 0xcb772339ba1f17f9u, + 0x9c69a97284b578d7u, 0xff2a760414536efbu, 0xc38413cf25e2d70du, 0xfef5138519684abau, + 0xf46518c2ef5b8cd1u, 0x7eb258665fc25d69u, 0x98bf2f79d5993802u, 0xef2f773ffbd97a61u, + 0xbeeefb584aff8603u, 0xaafb550ffacfd8fau, 0xeeaaba2e5dbf6784u, 0x95ba2a53f983cf38u, + 0x952ab45cfa97a0b2u, 0xdd945a747bf26183u, 0xba756174393d88dfu, 0x94f971119aeef9e4u, + 0xe912b9d1478ceb17u, 0x7a37cd5601aab85du, 0x91abb422ccb812eeu, 0xac62e055c10ab33au, + 0xb616a12b7fe617aau, 0x577b986b314d6009u, 0xe39c49765fdf9d94u, 0xed5a7e85fda0b80bu, + 0x8e41ade9fbebc27du, 0x14588f13be847307u, 0xb1d219647ae6b31cu, 0x596eb2d8ae258fc8u, + 0xde469fbd99a05fe3u, 0x6fca5f8ed9aef3bbu, 0x8aec23d680043beeu, 0x25de7bb9480d5854u, + 0xada72ccc20054ae9u, 0xaf561aa79a10ae6au, 0xd910f7ff28069da4u, 0x1b2ba1518094da04u, + 0x87aa9aff79042286u, 0x90fb44d2f05d0842u, 0xa99541bf57452b28u, 0x353a1607ac744a53u, + 0xd3fa922f2d1675f2u, 0x42889b8997915ce8u, 0x847c9b5d7c2e09b7u, 0x69956135febada11u, + 0xa59bc234db398c25u, 0x43fab9837e699095u, 0xcf02b2c21207ef2eu, 0x94f967e45e03f4bbu, + 0x8161afb94b44f57du, 0x1d1be0eebac278f5u, 0xa1ba1ba79e1632dcu, 0x6462d92a69731732u, + 0xca28a291859bbf93u, 0x7d7b8f7503cfdcfeu, 0xfcb2cb35e702af78u, 0x5cda735244c3d43eu, + 0x9defbf01b061adabu, 0x3a0888136afa64a7u, 0xc56baec21c7a1916u, 0x088aaa1845b8fdd0u, + 0xf6c69a72a3989f5bu, 0x8aad549e57273d45u, 0x9a3c2087a63f6399u, 0x36ac54e2f678864bu, + 0xc0cb28a98fcf3c7fu, 0x84576a1bb416a7ddu, 0xf0fdf2d3f3c30b9fu, 0x656d44a2a11c51d5u, + 0x969eb7c47859e743u, 0x9f644ae5a4b1b325u, 0xbc4665b596706114u, 0x873d5d9f0dde1feeu, + 0xeb57ff22fc0c7959u, 0xa90cb506d155a7eau, 0x9316ff75dd87cbd8u, 0x09a7f12442d588f2u, + 0xb7dcbf5354e9beceu, 0x0c11ed6d538aeb2fu, 0xe5d3ef282a242e81u, 0x8f1668c8a86da5fau, + 0x8fa475791a569d10u, 0xf96e017d694487bcu, 0xb38d92d760ec4455u, 0x37c981dcc395a9acu, + 0xe070f78d3927556au, 0x85bbe253f47b1417u, 0x8c469ab843b89562u, 0x93956d7478ccec8eu, + 0xaf58416654a6babbu, 0x387ac8d1970027b2u, 0xdb2e51bfe9d0696au, 0x06997b05fcc0319eu, + 0x88fcf317f22241e2u, 0x441fece3bdf81f03u, 0xab3c2fddeeaad25au, 0xd527e81cad7626c3u, + 0xd60b3bd56a5586f1u, 0x8a71e223d8d3b074u, 0x85c7056562757456u, 0xf6872d5667844e49u, + 0xa738c6bebb12d16cu, 0xb428f8ac016561dbu, 0xd106f86e69d785c7u, 0xe13336d701beba52u, + 0x82a45b450226b39cu, 0xecc0024661173473u, 0xa34d721642b06084u, 0x27f002d7f95d0190u, + 0xcc20ce9bd35c78a5u, 0x31ec038df7b441f4u, 0xff290242c83396ceu, 0x7e67047175a15271u, + 0x9f79a169bd203e41u, 0x0f0062c6e984d386u, 0xc75809c42c684dd1u, 0x52c07b78a3e60868u, + 0xf92e0c3537826145u, 0xa7709a56ccdf8a82u, 0x9bbcc7a142b17ccbu, 0x88a66076400bb691u, + 0xc2abf989935ddbfeu, 0x6acff893d00ea435u, 0xf356f7ebf83552feu, 0x0583f6b8c4124d43u, + 0x98165af37b2153deu, 0xc3727a337a8b704au, 0xbe1bf1b059e9a8d6u, 0x744f18c0592e4c5cu, + 0xeda2ee1c7064130cu, 0x1162def06f79df73u, 0x9485d4d1c63e8be7u, 0x8addcb5645ac2ba8u, + 0xb9a74a0637ce2ee1u, 0x6d953e2bd7173692u, 0xe8111c87c5c1ba99u, 0xc8fa8db6ccdd0437u, + 0x910ab1d4db9914a0u, 0x1d9c9892400a22a2u, 0xb54d5e4a127f59c8u, 0x2503beb6d00cab4bu, + 0xe2a0b5dc971f303au, 0x2e44ae64840fd61du, 0x8da471a9de737e24u, 0x5ceaecfed289e5d2u, + 0xb10d8e1456105dadu, 0x7425a83e872c5f47u, 0xdd50f1996b947518u, 0xd12f124e28f77719u, + 0x8a5296ffe33cc92fu, 0x82bd6b70d99aaa6fu, 0xace73cbfdc0bfb7bu, 0x636cc64d1001550bu, + 0xd8210befd30efa5au, 0x3c47f7e05401aa4eu, 0x8714a775e3e95c78u, 0x65acfaec34810a71u, + 0xa8d9d1535ce3b396u, 0x7f1839a741a14d0du, 0xd31045a8341ca07cu, 0x1ede48111209a050u, + 0x83ea2b892091e44du, 0x934aed0aab460432u, 0xa4e4b66b68b65d60u, 0xf81da84d5617853fu, + 0xce1de40642e3f4b9u, 0x36251260ab9d668eu, 0x80d2ae83e9ce78f3u, 0xc1d72b7c6b426019u, + 0xa1075a24e4421730u, 0xb24cf65b8612f81fu, 0xc94930ae1d529cfcu, 0xdee033f26797b627u, + 0xfb9b7cd9a4a7443cu, 0x169840ef017da3b1u, 0x9d412e0806e88aa5u, 0x8e1f289560ee864eu, + 0xc491798a08a2ad4eu, 0xf1a6f2bab92a27e2u, 0xf5b5d7ec8acb58a2u, 0xae10af696774b1dbu, + 0x9991a6f3d6bf1765u, 0xacca6da1e0a8ef29u, 0xbff610b0cc6edd3fu, 0x17fd090a58d32af3u, + 0xeff394dcff8a948eu, 0xddfc4b4cef07f5b0u, 0x95f83d0a1fb69cd9u, 0x4abdaf101564f98eu, + 0xbb764c4ca7a4440fu, 0x9d6d1ad41abe37f1u, 0xea53df5fd18d5513u, 0x84c86189216dc5edu, + 0x92746b9be2f8552cu, 0x32fd3cf5b4e49bb4u, 0xb7118682dbb66a77u, 0x3fbc8c33221dc2a1u, + 0xe4d5e82392a40515u, 0x0fabaf3feaa5334au, 0x8f05b1163ba6832du, 0x29cb4d87f2a7400eu, + 0xb2c71d5bca9023f8u, 0x743e20e9ef511012u, 0xdf78e4b2bd342cf6u, 0x914da9246b255416u, + 0x8bab8eefb6409c1au, 0x1ad089b6c2f7548eu, 0xae9672aba3d0c320u, 0xa184ac2473b529b1u, + 0xda3c0f568cc4f3e8u, 0xc9e5d72d90a2741eu, 0x8865899617fb1871u, 0x7e2fa67c7a658892u, + 0xaa7eebfb9df9de8du, 0xddbb901b98feeab7u, 0xd51ea6fa85785631u, 0x552a74227f3ea565u, + 0x8533285c936b35deu, 0xd53a88958f87275fu, 0xa67ff273b8460356u, 0x8a892abaf368f137u, + 0xd01fef10a657842cu, 0x2d2b7569b0432d85u, 0x8213f56a67f6b29bu, 0x9c3b29620e29fc73u, + 0xa298f2c501f45f42u, 0x8349f3ba91b47b8fu, 0xcb3f2f7642717713u, 0x241c70a936219a73u, + 0xfe0efb53d30dd4d7u, 0xed238cd383aa0110u, 0x9ec95d1463e8a506u, 0xf4363804324a40aau, + 0xc67bb4597ce2ce48u, 0xb143c6053edcd0d5u, 0xf81aa16fdc1b81dau, 0xdd94b7868e94050au, + 0x9b10a4e5e9913128u, 0xca7cf2b4191c8326u, 0xc1d4ce1f63f57d72u, 0xfd1c2f611f63a3f0u, + 0xf24a01a73cf2dccfu, 0xbc633b39673c8cecu, 0x976e41088617ca01u, 0xd5be0503e085d813u, + 0xbd49d14aa79dbc82u, 0x4b2d8644d8a74e18u, 0xec9c459d51852ba2u, 0xddf8e7d60ed1219eu, + 0x93e1ab8252f33b45u, 0xcabb90e5c942b503u, 0xb8da1662e7b00a17u, 0x3d6a751f3b936243u, + 0xe7109bfba19c0c9du, 0x0cc512670a783ad4u, 0x906a617d450187e2u, 0x27fb2b80668b24c5u, + 0xb484f9dc9641e9dau, 0xb1f9f660802dedf6u, 0xe1a63853bbd26451u, 0x5e7873f8a0396973u, + 0x8d07e33455637eb2u, 0xdb0b487b6423e1e8u, 0xb049dc016abc5e5fu, 0x91ce1a9a3d2cda62u, + 0xdc5c5301c56b75f7u, 0x7641a140cc7810fbu, 0x89b9b3e11b6329bau, 0xa9e904c87fcb0a9du, + 0xac2820d9623bf429u, 0x546345fa9fbdcd44u, 0xd732290fbacaf133u, 0xa97c177947ad4095u, + 0x867f59a9d4bed6c0u, 0x49ed8eabcccc485du, 0xa81f301449ee8c70u, 0x5c68f256bfff5a74u, + 0xd226fc195c6a2f8cu, 0x73832eec6fff3111u, 0x83585d8fd9c25db7u, 0xc831fd53c5ff7eabu, + 0xa42e74f3d032f525u, 0xba3e7ca8b77f5e55u, 0xcd3a1230c43fb26fu, 0x28ce1bd2e55f35ebu, + 0x80444b5e7aa7cf85u, 0x7980d163cf5b81b3u, 0xa0555e361951c366u, 0xd7e105bcc332621fu, + 0xc86ab5c39fa63440u, 0x8dd9472bf3fefaa7u, 0xfa856334878fc150u, 0xb14f98f6f0feb951u, + 0x9c935e00d4b9d8d2u, 0x6ed1bf9a569f33d3u, 0xc3b8358109e84f07u, 0x0a862f80ec4700c8u, + 0xf4a642e14c6262c8u, 0xcd27bb612758c0fau, 0x98e7e9cccfbd7dbdu, 0x8038d51cb897789cu, + 0xbf21e44003acdd2cu, 0xe0470a63e6bd56c3u, 0xeeea5d5004981478u, 0x1858ccfce06cac74u, + 0x95527a5202df0ccbu, 0x0f37801e0c43ebc8u, 0xbaa718e68396cffdu, 0xd30560258f54e6bau, + 0xe950df20247c83fdu, 0x47c6b82ef32a2069u, 0x91d28b7416cdd27eu, 0x4cdc331d57fa5441u, + 0xb6472e511c81471du, 0xe0133fe4adf8e952u, 0xe3d8f9e563a198e5u, 0x58180fddd97723a6u, + 0x8e679c2f5e44ff8fu, 0x570f09eaa7ea7648u, + } + }; + return table; +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index fbb30e7f7..1232e184c 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -8487,6 +8487,7 @@ NLOHMANN_JSON_NAMESPACE_END // | | |__ | | | | | | version 3.12.0 // |_____|_____|_____|_|___| https://github.com/nlohmann/json // +// SPDX-FileCopyrightText: 2021 The fast_float authors // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT @@ -8498,9 +8499,468 @@ NLOHMANN_JSON_NAMESPACE_END #include // size_t #include // int64_t, uint64_t #include // strtof, strtod, strtold +#include // memcpy #include // numeric_limits #include // string +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // uint64_t + +// #include + + +// Portable bit-level helpers for the number and string scanners. They use +// compiler builtins where available and plain C++ otherwise, so they need no +// platform headers and work regardless of byte order. + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// number of leading zero bits of x (x != 0) +inline int count_leading_zeros(std::uint64_t x) noexcept +{ +#if defined(__GNUC__) || defined(__clang__) + return __builtin_clzll(x); +#else + int n = 0; + for (int shift = 32; shift != 0; shift >>= 1) + { + if ((x >> (64 - shift)) == 0) + { + n += shift; + x <<= shift; + } + } + return n; +#endif +} + +/// the 128-bit product of two 64-bit numbers +struct uint128_parts +{ + std::uint64_t low; + std::uint64_t high; +}; + +inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexcept +{ +#if defined(__SIZEOF_INT128__) + __extension__ using uint128 = unsigned __int128; + const uint128 r = static_cast(a) * b; + return {static_cast(r), static_cast(r >> 64u)}; +#else + const std::uint64_t a_lo = a & 0xFFFFFFFFu; + const std::uint64_t a_hi = a >> 32u; + const std::uint64_t b_lo = b & 0xFFFFFFFFu; + const std::uint64_t b_hi = b >> 32u; + const std::uint64_t lo_lo = a_lo * b_lo; + const std::uint64_t hi_lo = a_hi * b_lo; + const std::uint64_t lo_hi = a_lo * b_hi; + const std::uint64_t hi_hi = a_hi * b_hi; + const std::uint64_t cross = (lo_lo >> 32u) + (hi_lo & 0xFFFFFFFFu) + lo_hi; + return {(cross << 32u) | (lo_lo & 0xFFFFFFFFu), (hi_lo >> 32u) + (cross >> 32u) + hi_hi}; +#endif +} + +/// eight bytes as a little-endian word (compilers fold this into one load on +/// little-endian targets) +inline std::uint64_t read_eight_bytes(const char* p) noexcept +{ + const auto* b = reinterpret_cast(p); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return static_cast(b[0]) | (static_cast(b[1]) << 8u) + | (static_cast(b[2]) << 16u) | (static_cast(b[3]) << 24u) + | (static_cast(b[4]) << 32u) | (static_cast(b[5]) << 40u) + | (static_cast(b[6]) << 48u) | (static_cast(b[7]) << 56u); +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2021 The fast_float authors +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // array +#include // int64_t, uint64_t + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// the range of decimal exponents covered by pow5_128() +constexpr std::int64_t pow5_128_smallest_power = -342; +constexpr std::int64_t pow5_128_largest_power = 308; + +/*! +@brief 128-bit approximations of 5^q for q in [-342, 308] + +Entry q (at index 2 * (q + 342)) holds the most significant 128 bits of 5^q, +normalized so that the highest bit is set: for q >= 0 the truncated value, for +q < 0 the value rounded up. This is the table of fast_float (Daniel Lemire and +contributors, used under the MIT license), generated like its +script/table_generation.py; unit-class_lexer.cpp recomputes every entry. +*/ +inline const std::array& pow5_128() noexcept +{ + static const std::array table = + { + { + 0xeef453d6923bd65au, 0x113faa2906a13b3fu, 0x9558b4661b6565f8u, 0x4ac7ca59a424c507u, + 0xbaaee17fa23ebf76u, 0x5d79bcf00d2df649u, 0xe95a99df8ace6f53u, 0xf4d82c2c107973dcu, + 0x91d8a02bb6c10594u, 0x79071b9b8a4be869u, 0xb64ec836a47146f9u, 0x9748e2826cdee284u, + 0xe3e27a444d8d98b7u, 0xfd1b1b2308169b25u, 0x8e6d8c6ab0787f72u, 0xfe30f0f5e50e20f7u, + 0xb208ef855c969f4fu, 0xbdbd2d335e51a935u, 0xde8b2b66b3bc4723u, 0xad2c788035e61382u, + 0x8b16fb203055ac76u, 0x4c3bcb5021afcc31u, 0xaddcb9e83c6b1793u, 0xdf4abe242a1bbf3du, + 0xd953e8624b85dd78u, 0xd71d6dad34a2af0du, 0x87d4713d6f33aa6bu, 0x8672648c40e5ad68u, + 0xa9c98d8ccb009506u, 0x680efdaf511f18c2u, 0xd43bf0effdc0ba48u, 0x0212bd1b2566def2u, + 0x84a57695fe98746du, 0x014bb630f7604b57u, 0xa5ced43b7e3e9188u, 0x419ea3bd35385e2du, + 0xcf42894a5dce35eau, 0x52064cac828675b9u, 0x818995ce7aa0e1b2u, 0x7343efebd1940993u, + 0xa1ebfb4219491a1fu, 0x1014ebe6c5f90bf8u, 0xca66fa129f9b60a6u, 0xd41a26e077774ef6u, + 0xfd00b897478238d0u, 0x8920b098955522b4u, 0x9e20735e8cb16382u, 0x55b46e5f5d5535b0u, + 0xc5a890362fddbc62u, 0xeb2189f734aa831du, 0xf712b443bbd52b7bu, 0xa5e9ec7501d523e4u, + 0x9a6bb0aa55653b2du, 0x47b233c92125366eu, 0xc1069cd4eabe89f8u, 0x999ec0bb696e840au, + 0xf148440a256e2c76u, 0xc00670ea43ca250du, 0x96cd2a865764dbcau, 0x380406926a5e5728u, + 0xbc807527ed3e12bcu, 0xc605083704f5ecf2u, 0xeba09271e88d976bu, 0xf7864a44c633682eu, + 0x93445b8731587ea3u, 0x7ab3ee6afbe0211du, 0xb8157268fdae9e4cu, 0x5960ea05bad82964u, + 0xe61acf033d1a45dfu, 0x6fb92487298e33bdu, 0x8fd0c16206306babu, 0xa5d3b6d479f8e056u, + 0xb3c4f1ba87bc8696u, 0x8f48a4899877186cu, 0xe0b62e2929aba83cu, 0x331acdabfe94de87u, + 0x8c71dcd9ba0b4925u, 0x9ff0c08b7f1d0b14u, 0xaf8e5410288e1b6fu, 0x07ecf0ae5ee44dd9u, + 0xdb71e91432b1a24au, 0xc9e82cd9f69d6150u, 0x892731ac9faf056eu, 0xbe311c083a225cd2u, + 0xab70fe17c79ac6cau, 0x6dbd630a48aaf406u, 0xd64d3d9db981787du, 0x092cbbccdad5b108u, + 0x85f0468293f0eb4eu, 0x25bbf56008c58ea5u, 0xa76c582338ed2621u, 0xaf2af2b80af6f24eu, + 0xd1476e2c07286faau, 0x1af5af660db4aee1u, 0x82cca4db847945cau, 0x50d98d9fc890ed4du, + 0xa37fce126597973cu, 0xe50ff107bab528a0u, 0xcc5fc196fefd7d0cu, 0x1e53ed49a96272c8u, + 0xff77b1fcbebcdc4fu, 0x25e8e89c13bb0f7au, 0x9faacf3df73609b1u, 0x77b191618c54e9acu, + 0xc795830d75038c1du, 0xd59df5b9ef6a2417u, 0xf97ae3d0d2446f25u, 0x4b0573286b44ad1du, + 0x9becce62836ac577u, 0x4ee367f9430aec32u, 0xc2e801fb244576d5u, 0x229c41f793cda73fu, + 0xf3a20279ed56d48au, 0x6b43527578c1110fu, 0x9845418c345644d6u, 0x830a13896b78aaa9u, + 0xbe5691ef416bd60cu, 0x23cc986bc656d553u, 0xedec366b11c6cb8fu, 0x2cbfbe86b7ec8aa8u, + 0x94b3a202eb1c3f39u, 0x7bf7d71432f3d6a9u, 0xb9e08a83a5e34f07u, 0xdaf5ccd93fb0cc53u, + 0xe858ad248f5c22c9u, 0xd1b3400f8f9cff68u, 0x91376c36d99995beu, 0x23100809b9c21fa1u, + 0xb58547448ffffb2du, 0xabd40a0c2832a78au, 0xe2e69915b3fff9f9u, 0x16c90c8f323f516cu, + 0x8dd01fad907ffc3bu, 0xae3da7d97f6792e3u, 0xb1442798f49ffb4au, 0x99cd11cfdf41779cu, + 0xdd95317f31c7fa1du, 0x40405643d711d583u, 0x8a7d3eef7f1cfc52u, 0x482835ea666b2572u, + 0xad1c8eab5ee43b66u, 0xda3243650005eecfu, 0xd863b256369d4a40u, 0x90bed43e40076a82u, + 0x873e4f75e2224e68u, 0x5a7744a6e804a291u, 0xa90de3535aaae202u, 0x711515d0a205cb36u, + 0xd3515c2831559a83u, 0x0d5a5b44ca873e03u, 0x8412d9991ed58091u, 0xe858790afe9486c2u, + 0xa5178fff668ae0b6u, 0x626e974dbe39a872u, 0xce5d73ff402d98e3u, 0xfb0a3d212dc8128fu, + 0x80fa687f881c7f8eu, 0x7ce66634bc9d0b99u, 0xa139029f6a239f72u, 0x1c1fffc1ebc44e80u, + 0xc987434744ac874eu, 0xa327ffb266b56220u, 0xfbe9141915d7a922u, 0x4bf1ff9f0062baa8u, + 0x9d71ac8fada6c9b5u, 0x6f773fc3603db4a9u, 0xc4ce17b399107c22u, 0xcb550fb4384d21d3u, + 0xf6019da07f549b2bu, 0x7e2a53a146606a48u, 0x99c102844f94e0fbu, 0x2eda7444cbfc426du, + 0xc0314325637a1939u, 0xfa911155fefb5308u, 0xf03d93eebc589f88u, 0x793555ab7eba27cau, + 0x96267c7535b763b5u, 0x4bc1558b2f3458deu, 0xbbb01b9283253ca2u, 0x9eb1aaedfb016f16u, + 0xea9c227723ee8bcbu, 0x465e15a979c1cadcu, 0x92a1958a7675175fu, 0x0bfacd89ec191ec9u, + 0xb749faed14125d36u, 0xcef980ec671f667bu, 0xe51c79a85916f484u, 0x82b7e12780e7401au, + 0x8f31cc0937ae58d2u, 0xd1b2ecb8b0908810u, 0xb2fe3f0b8599ef07u, 0x861fa7e6dcb4aa15u, + 0xdfbdcece67006ac9u, 0x67a791e093e1d49au, 0x8bd6a141006042bdu, 0xe0c8bb2c5c6d24e0u, + 0xaecc49914078536du, 0x58fae9f773886e18u, 0xda7f5bf590966848u, 0xaf39a475506a899eu, + 0x888f99797a5e012du, 0x6d8406c952429603u, 0xaab37fd7d8f58178u, 0xc8e5087ba6d33b83u, + 0xd5605fcdcf32e1d6u, 0xfb1e4a9a90880a64u, 0x855c3be0a17fcd26u, 0x5cf2eea09a55067fu, + 0xa6b34ad8c9dfc06fu, 0xf42faa48c0ea481eu, 0xd0601d8efc57b08bu, 0xf13b94daf124da26u, + 0x823c12795db6ce57u, 0x76c53d08d6b70858u, 0xa2cb1717b52481edu, 0x54768c4b0c64ca6eu, + 0xcb7ddcdda26da268u, 0xa9942f5dcf7dfd09u, 0xfe5d54150b090b02u, 0xd3f93b35435d7c4cu, + 0x9efa548d26e5a6e1u, 0xc47bc5014a1a6dafu, 0xc6b8e9b0709f109au, 0x359ab6419ca1091bu, + 0xf867241c8cc6d4c0u, 0xc30163d203c94b62u, 0x9b407691d7fc44f8u, 0x79e0de63425dcf1du, + 0xc21094364dfb5636u, 0x985915fc12f542e4u, 0xf294b943e17a2bc4u, 0x3e6f5b7b17b2939du, + 0x979cf3ca6cec5b5au, 0xa705992ceecf9c42u, 0xbd8430bd08277231u, 0x50c6ff782a838353u, + 0xece53cec4a314ebdu, 0xa4f8bf5635246428u, 0x940f4613ae5ed136u, 0x871b7795e136be99u, + 0xb913179899f68584u, 0x28e2557b59846e3fu, 0xe757dd7ec07426e5u, 0x331aeada2fe589cfu, + 0x9096ea6f3848984fu, 0x3ff0d2c85def7621u, 0xb4bca50b065abe63u, 0x0fed077a756b53a9u, + 0xe1ebce4dc7f16dfbu, 0xd3e8495912c62894u, 0x8d3360f09cf6e4bdu, 0x64712dd7abbbd95cu, + 0xb080392cc4349decu, 0xbd8d794d96aacfb3u, 0xdca04777f541c567u, 0xecf0d7a0fc5583a0u, + 0x89e42caaf9491b60u, 0xf41686c49db57244u, 0xac5d37d5b79b6239u, 0x311c2875c522ced5u, + 0xd77485cb25823ac7u, 0x7d633293366b828bu, 0x86a8d39ef77164bcu, 0xae5dff9c02033197u, + 0xa8530886b54dbdebu, 0xd9f57f830283fdfcu, 0xd267caa862a12d66u, 0xd072df63c324fd7bu, + 0x8380dea93da4bc60u, 0x4247cb9e59f71e6du, 0xa46116538d0deb78u, 0x52d9be85f074e608u, + 0xcd795be870516656u, 0x67902e276c921f8bu, 0x806bd9714632dff6u, 0x00ba1cd8a3db53b6u, + 0xa086cfcd97bf97f3u, 0x80e8a40eccd228a4u, 0xc8a883c0fdaf7df0u, 0x6122cd128006b2cdu, + 0xfad2a4b13d1b5d6cu, 0x796b805720085f81u, 0x9cc3a6eec6311a63u, 0xcbe3303674053bb0u, + 0xc3f490aa77bd60fcu, 0xbedbfc4411068a9cu, 0xf4f1b4d515acb93bu, 0xee92fb5515482d44u, + 0x991711052d8bf3c5u, 0x751bdd152d4d1c4au, 0xbf5cd54678eef0b6u, 0xd262d45a78a0635du, + 0xef340a98172aace4u, 0x86fb897116c87c34u, 0x9580869f0e7aac0eu, 0xd45d35e6ae3d4da0u, + 0xbae0a846d2195712u, 0x8974836059cca109u, 0xe998d258869facd7u, 0x2bd1a438703fc94bu, + 0x91ff83775423cc06u, 0x7b6306a34627ddcfu, 0xb67f6455292cbf08u, 0x1a3bc84c17b1d542u, + 0xe41f3d6a7377eecau, 0x20caba5f1d9e4a93u, 0x8e938662882af53eu, 0x547eb47b7282ee9cu, + 0xb23867fb2a35b28du, 0xe99e619a4f23aa43u, 0xdec681f9f4c31f31u, 0x6405fa00e2ec94d4u, + 0x8b3c113c38f9f37eu, 0xde83bc408dd3dd04u, 0xae0b158b4738705eu, 0x9624ab50b148d445u, + 0xd98ddaee19068c76u, 0x3badd624dd9b0957u, 0x87f8a8d4cfa417c9u, 0xe54ca5d70a80e5d6u, + 0xa9f6d30a038d1dbcu, 0x5e9fcf4ccd211f4cu, 0xd47487cc8470652bu, 0x7647c3200069671fu, + 0x84c8d4dfd2c63f3bu, 0x29ecd9f40041e073u, 0xa5fb0a17c777cf09u, 0xf468107100525890u, + 0xcf79cc9db955c2ccu, 0x7182148d4066eeb4u, 0x81ac1fe293d599bfu, 0xc6f14cd848405530u, + 0xa21727db38cb002fu, 0xb8ada00e5a506a7cu, 0xca9cf1d206fdc03bu, 0xa6d90811f0e4851cu, + 0xfd442e4688bd304au, 0x908f4a166d1da663u, 0x9e4a9cec15763e2eu, 0x9a598e4e043287feu, + 0xc5dd44271ad3cdbau, 0x40eff1e1853f29fdu, 0xf7549530e188c128u, 0xd12bee59e68ef47cu, + 0x9a94dd3e8cf578b9u, 0x82bb74f8301958ceu, 0xc13a148e3032d6e7u, 0xe36a52363c1faf01u, + 0xf18899b1bc3f8ca1u, 0xdc44e6c3cb279ac1u, 0x96f5600f15a7b7e5u, 0x29ab103a5ef8c0b9u, + 0xbcb2b812db11a5deu, 0x7415d448f6b6f0e7u, 0xebdf661791d60f56u, 0x111b495b3464ad21u, + 0x936b9fcebb25c995u, 0xcab10dd900beec34u, 0xb84687c269ef3bfbu, 0x3d5d514f40eea742u, + 0xe65829b3046b0afau, 0x0cb4a5a3112a5112u, 0x8ff71a0fe2c2e6dcu, 0x47f0e785eaba72abu, + 0xb3f4e093db73a093u, 0x59ed216765690f56u, 0xe0f218b8d25088b8u, 0x306869c13ec3532cu, + 0x8c974f7383725573u, 0x1e414218c73a13fbu, 0xafbd2350644eeacfu, 0xe5d1929ef90898fau, + 0xdbac6c247d62a583u, 0xdf45f746b74abf39u, 0x894bc396ce5da772u, 0x6b8bba8c328eb783u, + 0xab9eb47c81f5114fu, 0x066ea92f3f326564u, 0xd686619ba27255a2u, 0xc80a537b0efefebdu, + 0x8613fd0145877585u, 0xbd06742ce95f5f36u, 0xa798fc4196e952e7u, 0x2c48113823b73704u, + 0xd17f3b51fca3a7a0u, 0xf75a15862ca504c5u, 0x82ef85133de648c4u, 0x9a984d73dbe722fbu, + 0xa3ab66580d5fdaf5u, 0xc13e60d0d2e0ebbau, 0xcc963fee10b7d1b3u, 0x318df905079926a8u, + 0xffbbcfe994e5c61fu, 0xfdf17746497f7052u, 0x9fd561f1fd0f9bd3u, 0xfeb6ea8bedefa633u, + 0xc7caba6e7c5382c8u, 0xfe64a52ee96b8fc0u, 0xf9bd690a1b68637bu, 0x3dfdce7aa3c673b0u, + 0x9c1661a651213e2du, 0x06bea10ca65c084eu, 0xc31bfa0fe5698db8u, 0x486e494fcff30a62u, + 0xf3e2f893dec3f126u, 0x5a89dba3c3efccfau, 0x986ddb5c6b3a76b7u, 0xf89629465a75e01cu, + 0xbe89523386091465u, 0xf6bbb397f1135823u, 0xee2ba6c0678b597fu, 0x746aa07ded582e2cu, + 0x94db483840b717efu, 0xa8c2a44eb4571cdcu, 0xba121a4650e4ddebu, 0x92f34d62616ce413u, + 0xe896a0d7e51e1566u, 0x77b020baf9c81d17u, 0x915e2486ef32cd60u, 0x0ace1474dc1d122eu, + 0xb5b5ada8aaff80b8u, 0x0d819992132456bau, 0xe3231912d5bf60e6u, 0x10e1fff697ed6c69u, + 0x8df5efabc5979c8fu, 0xca8d3ffa1ef463c1u, 0xb1736b96b6fd83b3u, 0xbd308ff8a6b17cb2u, + 0xddd0467c64bce4a0u, 0xac7cb3f6d05ddbdeu, 0x8aa22c0dbef60ee4u, 0x6bcdf07a423aa96bu, + 0xad4ab7112eb3929du, 0x86c16c98d2c953c6u, 0xd89d64d57a607744u, 0xe871c7bf077ba8b7u, + 0x87625f056c7c4a8bu, 0x11471cd764ad4972u, 0xa93af6c6c79b5d2du, 0xd598e40d3dd89bcfu, + 0xd389b47879823479u, 0x4aff1d108d4ec2c3u, 0x843610cb4bf160cbu, 0xcedf722a585139bau, + 0xa54394fe1eedb8feu, 0xc2974eb4ee658828u, 0xce947a3da6a9273eu, 0x733d226229feea32u, + 0x811ccc668829b887u, 0x0806357d5a3f525fu, 0xa163ff802a3426a8u, 0xca07c2dcb0cf26f7u, + 0xc9bcff6034c13052u, 0xfc89b393dd02f0b5u, 0xfc2c3f3841f17c67u, 0xbbac2078d443ace2u, + 0x9d9ba7832936edc0u, 0xd54b944b84aa4c0du, 0xc5029163f384a931u, 0x0a9e795e65d4df11u, + 0xf64335bcf065d37du, 0x4d4617b5ff4a16d5u, 0x99ea0196163fa42eu, 0x504bced1bf8e4e45u, + 0xc06481fb9bcf8d39u, 0xe45ec2862f71e1d6u, 0xf07da27a82c37088u, 0x5d767327bb4e5a4cu, + 0x964e858c91ba2655u, 0x3a6a07f8d510f86fu, 0xbbe226efb628afeau, 0x890489f70a55368bu, + 0xeadab0aba3b2dbe5u, 0x2b45ac74ccea842eu, 0x92c8ae6b464fc96fu, 0x3b0b8bc90012929du, + 0xb77ada0617e3bbcbu, 0x09ce6ebb40173744u, 0xe55990879ddcaabdu, 0xcc420a6a101d0515u, + 0x8f57fa54c2a9eab6u, 0x9fa946824a12232du, 0xb32df8e9f3546564u, 0x47939822dc96abf9u, + 0xdff9772470297ebdu, 0x59787e2b93bc56f7u, 0x8bfbea76c619ef36u, 0x57eb4edb3c55b65au, + 0xaefae51477a06b03u, 0xede622920b6b23f1u, 0xdab99e59958885c4u, 0xe95fab368e45ecedu, + 0x88b402f7fd75539bu, 0x11dbcb0218ebb414u, 0xaae103b5fcd2a881u, 0xd652bdc29f26a119u, + 0xd59944a37c0752a2u, 0x4be76d3346f0495fu, 0x857fcae62d8493a5u, 0x6f70a4400c562ddbu, + 0xa6dfbd9fb8e5b88eu, 0xcb4ccd500f6bb952u, 0xd097ad07a71f26b2u, 0x7e2000a41346a7a7u, + 0x825ecc24c873782fu, 0x8ed400668c0c28c8u, 0xa2f67f2dfa90563bu, 0x728900802f0f32fau, + 0xcbb41ef979346bcau, 0x4f2b40a03ad2ffb9u, 0xfea126b7d78186bcu, 0xe2f610c84987bfa8u, + 0x9f24b832e6b0f436u, 0x0dd9ca7d2df4d7c9u, 0xc6ede63fa05d3143u, 0x91503d1c79720dbbu, + 0xf8a95fcf88747d94u, 0x75a44c6397ce912au, 0x9b69dbe1b548ce7cu, 0xc986afbe3ee11abau, + 0xc24452da229b021bu, 0xfbe85badce996168u, 0xf2d56790ab41c2a2u, 0xfae27299423fb9c3u, + 0x97c560ba6b0919a5u, 0xdccd879fc967d41au, 0xbdb6b8e905cb600fu, 0x5400e987bbc1c920u, + 0xed246723473e3813u, 0x290123e9aab23b68u, 0x9436c0760c86e30bu, 0xf9a0b6720aaf6521u, + 0xb94470938fa89bceu, 0xf808e40e8d5b3e69u, 0xe7958cb87392c2c2u, 0xb60b1d1230b20e04u, + 0x90bd77f3483bb9b9u, 0xb1c6f22b5e6f48c2u, 0xb4ecd5f01a4aa828u, 0x1e38aeb6360b1af3u, + 0xe2280b6c20dd5232u, 0x25c6da63c38de1b0u, 0x8d590723948a535fu, 0x579c487e5a38ad0eu, + 0xb0af48ec79ace837u, 0x2d835a9df0c6d851u, 0xdcdb1b2798182244u, 0xf8e431456cf88e65u, + 0x8a08f0f8bf0f156bu, 0x1b8e9ecb641b58ffu, 0xac8b2d36eed2dac5u, 0xe272467e3d222f3fu, + 0xd7adf884aa879177u, 0x5b0ed81dcc6abb0fu, 0x86ccbb52ea94baeau, 0x98e947129fc2b4e9u, + 0xa87fea27a539e9a5u, 0x3f2398d747b36224u, 0xd29fe4b18e88640eu, 0x8eec7f0d19a03aadu, + 0x83a3eeeef9153e89u, 0x1953cf68300424acu, 0xa48ceaaab75a8e2bu, 0x5fa8c3423c052dd7u, + 0xcdb02555653131b6u, 0x3792f412cb06794du, 0x808e17555f3ebf11u, 0xe2bbd88bbee40bd0u, + 0xa0b19d2ab70e6ed6u, 0x5b6aceaeae9d0ec4u, 0xc8de047564d20a8bu, 0xf245825a5a445275u, + 0xfb158592be068d2eu, 0xeed6e2f0f0d56712u, 0x9ced737bb6c4183du, 0x55464dd69685606bu, + 0xc428d05aa4751e4cu, 0xaa97e14c3c26b886u, 0xf53304714d9265dfu, 0xd53dd99f4b3066a8u, + 0x993fe2c6d07b7fabu, 0xe546a8038efe4029u, 0xbf8fdb78849a5f96u, 0xde98520472bdd033u, + 0xef73d256a5c0f77cu, 0x963e66858f6d4440u, 0x95a8637627989aadu, 0xdde7001379a44aa8u, + 0xbb127c53b17ec159u, 0x5560c018580d5d52u, 0xe9d71b689dde71afu, 0xaab8f01e6e10b4a6u, + 0x9226712162ab070du, 0xcab3961304ca70e8u, 0xb6b00d69bb55c8d1u, 0x3d607b97c5fd0d22u, + 0xe45c10c42a2b3b05u, 0x8cb89a7db77c506au, 0x8eb98a7a9a5b04e3u, 0x77f3608e92adb242u, + 0xb267ed1940f1c61cu, 0x55f038b237591ed3u, 0xdf01e85f912e37a3u, 0x6b6c46dec52f6688u, + 0x8b61313bbabce2c6u, 0x2323ac4b3b3da015u, 0xae397d8aa96c1b77u, 0xabec975e0a0d081au, + 0xd9c7dced53c72255u, 0x96e7bd358c904a21u, 0x881cea14545c7575u, 0x7e50d64177da2e54u, + 0xaa242499697392d2u, 0xdde50bd1d5d0b9e9u, 0xd4ad2dbfc3d07787u, 0x955e4ec64b44e864u, + 0x84ec3c97da624ab4u, 0xbd5af13bef0b113eu, 0xa6274bbdd0fadd61u, 0xecb1ad8aeacdd58eu, + 0xcfb11ead453994bau, 0x67de18eda5814af2u, 0x81ceb32c4b43fcf4u, 0x80eacf948770ced7u, + 0xa2425ff75e14fc31u, 0xa1258379a94d028du, 0xcad2f7f5359a3b3eu, 0x096ee45813a04330u, + 0xfd87b5f28300ca0du, 0x8bca9d6e188853fcu, 0x9e74d1b791e07e48u, 0x775ea264cf55347eu, + 0xc612062576589ddau, 0x95364afe032a819eu, 0xf79687aed3eec551u, 0x3a83ddbd83f52205u, + 0x9abe14cd44753b52u, 0xc4926a9672793543u, 0xc16d9a0095928a27u, 0x75b7053c0f178294u, + 0xf1c90080baf72cb1u, 0x5324c68b12dd6339u, 0x971da05074da7beeu, 0xd3f6fc16ebca5e04u, + 0xbce5086492111aeau, 0x88f4bb1ca6bcf585u, 0xec1e4a7db69561a5u, 0x2b31e9e3d06c32e6u, + 0x9392ee8e921d5d07u, 0x3aff322e62439fd0u, 0xb877aa3236a4b449u, 0x09befeb9fad487c3u, + 0xe69594bec44de15bu, 0x4c2ebe687989a9b4u, 0x901d7cf73ab0acd9u, 0x0f9d37014bf60a11u, + 0xb424dc35095cd80fu, 0x538484c19ef38c95u, 0xe12e13424bb40e13u, 0x2865a5f206b06fbau, + 0x8cbccc096f5088cbu, 0xf93f87b7442e45d4u, 0xafebff0bcb24aafeu, 0xf78f69a51539d749u, + 0xdbe6fecebdedd5beu, 0xb573440e5a884d1cu, 0x89705f4136b4a597u, 0x31680a88f8953031u, + 0xabcc77118461cefcu, 0xfdc20d2b36ba7c3eu, 0xd6bf94d5e57a42bcu, 0x3d32907604691b4du, + 0x8637bd05af6c69b5u, 0xa63f9a49c2c1b110u, 0xa7c5ac471b478423u, 0x0fcf80dc33721d54u, + 0xd1b71758e219652bu, 0xd3c36113404ea4a9u, 0x83126e978d4fdf3bu, 0x645a1cac083126eau, + 0xa3d70a3d70a3d70au, 0x3d70a3d70a3d70a4u, 0xccccccccccccccccu, 0xcccccccccccccccdu, + 0x8000000000000000u, 0x0000000000000000u, 0xa000000000000000u, 0x0000000000000000u, + 0xc800000000000000u, 0x0000000000000000u, 0xfa00000000000000u, 0x0000000000000000u, + 0x9c40000000000000u, 0x0000000000000000u, 0xc350000000000000u, 0x0000000000000000u, + 0xf424000000000000u, 0x0000000000000000u, 0x9896800000000000u, 0x0000000000000000u, + 0xbebc200000000000u, 0x0000000000000000u, 0xee6b280000000000u, 0x0000000000000000u, + 0x9502f90000000000u, 0x0000000000000000u, 0xba43b74000000000u, 0x0000000000000000u, + 0xe8d4a51000000000u, 0x0000000000000000u, 0x9184e72a00000000u, 0x0000000000000000u, + 0xb5e620f480000000u, 0x0000000000000000u, 0xe35fa931a0000000u, 0x0000000000000000u, + 0x8e1bc9bf04000000u, 0x0000000000000000u, 0xb1a2bc2ec5000000u, 0x0000000000000000u, + 0xde0b6b3a76400000u, 0x0000000000000000u, 0x8ac7230489e80000u, 0x0000000000000000u, + 0xad78ebc5ac620000u, 0x0000000000000000u, 0xd8d726b7177a8000u, 0x0000000000000000u, + 0x878678326eac9000u, 0x0000000000000000u, 0xa968163f0a57b400u, 0x0000000000000000u, + 0xd3c21bcecceda100u, 0x0000000000000000u, 0x84595161401484a0u, 0x0000000000000000u, + 0xa56fa5b99019a5c8u, 0x0000000000000000u, 0xcecb8f27f4200f3au, 0x0000000000000000u, + 0x813f3978f8940984u, 0x4000000000000000u, 0xa18f07d736b90be5u, 0x5000000000000000u, + 0xc9f2c9cd04674edeu, 0xa400000000000000u, 0xfc6f7c4045812296u, 0x4d00000000000000u, + 0x9dc5ada82b70b59du, 0xf020000000000000u, 0xc5371912364ce305u, 0x6c28000000000000u, + 0xf684df56c3e01bc6u, 0xc732000000000000u, 0x9a130b963a6c115cu, 0x3c7f400000000000u, + 0xc097ce7bc90715b3u, 0x4b9f100000000000u, 0xf0bdc21abb48db20u, 0x1e86d40000000000u, + 0x96769950b50d88f4u, 0x1314448000000000u, 0xbc143fa4e250eb31u, 0x17d955a000000000u, + 0xeb194f8e1ae525fdu, 0x5dcfab0800000000u, 0x92efd1b8d0cf37beu, 0x5aa1cae500000000u, + 0xb7abc627050305adu, 0xf14a3d9e40000000u, 0xe596b7b0c643c719u, 0x6d9ccd05d0000000u, + 0x8f7e32ce7bea5c6fu, 0xe4820023a2000000u, 0xb35dbf821ae4f38bu, 0xdda2802c8a800000u, + 0xe0352f62a19e306eu, 0xd50b2037ad200000u, 0x8c213d9da502de45u, 0x4526f422cc340000u, + 0xaf298d050e4395d6u, 0x9670b12b7f410000u, 0xdaf3f04651d47b4cu, 0x3c0cdd765f114000u, + 0x88d8762bf324cd0fu, 0xa5880a69fb6ac800u, 0xab0e93b6efee0053u, 0x8eea0d047a457a00u, + 0xd5d238a4abe98068u, 0x72a4904598d6d880u, 0x85a36366eb71f041u, 0x47a6da2b7f864750u, + 0xa70c3c40a64e6c51u, 0x999090b65f67d924u, 0xd0cf4b50cfe20765u, 0xfff4b4e3f741cf6du, + 0x82818f1281ed449fu, 0xbff8f10e7a8921a4u, 0xa321f2d7226895c7u, 0xaff72d52192b6a0du, + 0xcbea6f8ceb02bb39u, 0x9bf4f8a69f764490u, 0xfee50b7025c36a08u, 0x02f236d04753d5b4u, + 0x9f4f2726179a2245u, 0x01d762422c946590u, 0xc722f0ef9d80aad6u, 0x424d3ad2b7b97ef5u, + 0xf8ebad2b84e0d58bu, 0xd2e0898765a7deb2u, 0x9b934c3b330c8577u, 0x63cc55f49f88eb2fu, + 0xc2781f49ffcfa6d5u, 0x3cbf6b71c76b25fbu, 0xf316271c7fc3908au, 0x8bef464e3945ef7au, + 0x97edd871cfda3a56u, 0x97758bf0e3cbb5acu, 0xbde94e8e43d0c8ecu, 0x3d52eeed1cbea317u, + 0xed63a231d4c4fb27u, 0x4ca7aaa863ee4bddu, 0x945e455f24fb1cf8u, 0x8fe8caa93e74ef6au, + 0xb975d6b6ee39e436u, 0xb3e2fd538e122b44u, 0xe7d34c64a9c85d44u, 0x60dbbca87196b616u, + 0x90e40fbeea1d3a4au, 0xbc8955e946fe31cdu, 0xb51d13aea4a488ddu, 0x6babab6398bdbe41u, + 0xe264589a4dcdab14u, 0xc696963c7eed2dd1u, 0x8d7eb76070a08aecu, 0xfc1e1de5cf543ca2u, + 0xb0de65388cc8ada8u, 0x3b25a55f43294bcbu, 0xdd15fe86affad912u, 0x49ef0eb713f39ebeu, + 0x8a2dbf142dfcc7abu, 0x6e3569326c784337u, 0xacb92ed9397bf996u, 0x49c2c37f07965404u, + 0xd7e77a8f87daf7fbu, 0xdc33745ec97be906u, 0x86f0ac99b4e8dafdu, 0x69a028bb3ded71a3u, + 0xa8acd7c0222311bcu, 0xc40832ea0d68ce0cu, 0xd2d80db02aabd62bu, 0xf50a3fa490c30190u, + 0x83c7088e1aab65dbu, 0x792667c6da79e0fau, 0xa4b8cab1a1563f52u, 0x577001b891185938u, + 0xcde6fd5e09abcf26u, 0xed4c0226b55e6f86u, 0x80b05e5ac60b6178u, 0x544f8158315b05b4u, + 0xa0dc75f1778e39d6u, 0x696361ae3db1c721u, 0xc913936dd571c84cu, 0x03bc3a19cd1e38e9u, + 0xfb5878494ace3a5fu, 0x04ab48a04065c723u, 0x9d174b2dcec0e47bu, 0x62eb0d64283f9c76u, + 0xc45d1df942711d9au, 0x3ba5d0bd324f8394u, 0xf5746577930d6500u, 0xca8f44ec7ee36479u, + 0x9968bf6abbe85f20u, 0x7e998b13cf4e1ecbu, 0xbfc2ef456ae276e8u, 0x9e3fedd8c321a67eu, + 0xefb3ab16c59b14a2u, 0xc5cfe94ef3ea101eu, 0x95d04aee3b80ece5u, 0xbba1f1d158724a12u, + 0xbb445da9ca61281fu, 0x2a8a6e45ae8edc97u, 0xea1575143cf97226u, 0xf52d09d71a3293bdu, + 0x924d692ca61be758u, 0x593c2626705f9c56u, 0xb6e0c377cfa2e12eu, 0x6f8b2fb00c77836cu, + 0xe498f455c38b997au, 0x0b6dfb9c0f956447u, 0x8edf98b59a373fecu, 0x4724bd4189bd5eacu, + 0xb2977ee300c50fe7u, 0x58edec91ec2cb657u, 0xdf3d5e9bc0f653e1u, 0x2f2967b66737e3edu, + 0x8b865b215899f46cu, 0xbd79e0d20082ee74u, 0xae67f1e9aec07187u, 0xecd8590680a3aa11u, + 0xda01ee641a708de9u, 0xe80e6f4820cc9495u, 0x884134fe908658b2u, 0x3109058d147fdcddu, + 0xaa51823e34a7eedeu, 0xbd4b46f0599fd415u, 0xd4e5e2cdc1d1ea96u, 0x6c9e18ac7007c91au, + 0x850fadc09923329eu, 0x03e2cf6bc604ddb0u, 0xa6539930bf6bff45u, 0x84db8346b786151cu, + 0xcfe87f7cef46ff16u, 0xe612641865679a63u, 0x81f14fae158c5f6eu, 0x4fcb7e8f3f60c07eu, + 0xa26da3999aef7749u, 0xe3be5e330f38f09du, 0xcb090c8001ab551cu, 0x5cadf5bfd3072cc5u, + 0xfdcb4fa002162a63u, 0x73d9732fc7c8f7f6u, 0x9e9f11c4014dda7eu, 0x2867e7fddcdd9afau, + 0xc646d63501a1511du, 0xb281e1fd541501b8u, 0xf7d88bc24209a565u, 0x1f225a7ca91a4226u, + 0x9ae757596946075fu, 0x3375788de9b06958u, 0xc1a12d2fc3978937u, 0x0052d6b1641c83aeu, + 0xf209787bb47d6b84u, 0xc0678c5dbd23a49au, 0x9745eb4d50ce6332u, 0xf840b7ba963646e0u, + 0xbd176620a501fbffu, 0xb650e5a93bc3d898u, 0xec5d3fa8ce427affu, 0xa3e51f138ab4cebeu, + 0x93ba47c980e98cdfu, 0xc66f336c36b10137u, 0xb8a8d9bbe123f017u, 0xb80b0047445d4184u, + 0xe6d3102ad96cec1du, 0xa60dc059157491e5u, 0x9043ea1ac7e41392u, 0x87c89837ad68db2fu, + 0xb454e4a179dd1877u, 0x29babe4598c311fbu, 0xe16a1dc9d8545e94u, 0xf4296dd6fef3d67au, + 0x8ce2529e2734bb1du, 0x1899e4a65f58660cu, 0xb01ae745b101e9e4u, 0x5ec05dcff72e7f8fu, + 0xdc21a1171d42645du, 0x76707543f4fa1f73u, 0x899504ae72497ebau, 0x6a06494a791c53a8u, + 0xabfa45da0edbde69u, 0x0487db9d17636892u, 0xd6f8d7509292d603u, 0x45a9d2845d3c42b6u, + 0x865b86925b9bc5c2u, 0x0b8a2392ba45a9b2u, 0xa7f26836f282b732u, 0x8e6cac7768d7141eu, + 0xd1ef0244af2364ffu, 0x3207d795430cd926u, 0x8335616aed761f1fu, 0x7f44e6bd49e807b8u, + 0xa402b9c5a8d3a6e7u, 0x5f16206c9c6209a6u, 0xcd036837130890a1u, 0x36dba887c37a8c0fu, + 0x802221226be55a64u, 0xc2494954da2c9789u, 0xa02aa96b06deb0fdu, 0xf2db9baa10b7bd6cu, + 0xc83553c5c8965d3du, 0x6f92829494e5acc7u, 0xfa42a8b73abbf48cu, 0xcb772339ba1f17f9u, + 0x9c69a97284b578d7u, 0xff2a760414536efbu, 0xc38413cf25e2d70du, 0xfef5138519684abau, + 0xf46518c2ef5b8cd1u, 0x7eb258665fc25d69u, 0x98bf2f79d5993802u, 0xef2f773ffbd97a61u, + 0xbeeefb584aff8603u, 0xaafb550ffacfd8fau, 0xeeaaba2e5dbf6784u, 0x95ba2a53f983cf38u, + 0x952ab45cfa97a0b2u, 0xdd945a747bf26183u, 0xba756174393d88dfu, 0x94f971119aeef9e4u, + 0xe912b9d1478ceb17u, 0x7a37cd5601aab85du, 0x91abb422ccb812eeu, 0xac62e055c10ab33au, + 0xb616a12b7fe617aau, 0x577b986b314d6009u, 0xe39c49765fdf9d94u, 0xed5a7e85fda0b80bu, + 0x8e41ade9fbebc27du, 0x14588f13be847307u, 0xb1d219647ae6b31cu, 0x596eb2d8ae258fc8u, + 0xde469fbd99a05fe3u, 0x6fca5f8ed9aef3bbu, 0x8aec23d680043beeu, 0x25de7bb9480d5854u, + 0xada72ccc20054ae9u, 0xaf561aa79a10ae6au, 0xd910f7ff28069da4u, 0x1b2ba1518094da04u, + 0x87aa9aff79042286u, 0x90fb44d2f05d0842u, 0xa99541bf57452b28u, 0x353a1607ac744a53u, + 0xd3fa922f2d1675f2u, 0x42889b8997915ce8u, 0x847c9b5d7c2e09b7u, 0x69956135febada11u, + 0xa59bc234db398c25u, 0x43fab9837e699095u, 0xcf02b2c21207ef2eu, 0x94f967e45e03f4bbu, + 0x8161afb94b44f57du, 0x1d1be0eebac278f5u, 0xa1ba1ba79e1632dcu, 0x6462d92a69731732u, + 0xca28a291859bbf93u, 0x7d7b8f7503cfdcfeu, 0xfcb2cb35e702af78u, 0x5cda735244c3d43eu, + 0x9defbf01b061adabu, 0x3a0888136afa64a7u, 0xc56baec21c7a1916u, 0x088aaa1845b8fdd0u, + 0xf6c69a72a3989f5bu, 0x8aad549e57273d45u, 0x9a3c2087a63f6399u, 0x36ac54e2f678864bu, + 0xc0cb28a98fcf3c7fu, 0x84576a1bb416a7ddu, 0xf0fdf2d3f3c30b9fu, 0x656d44a2a11c51d5u, + 0x969eb7c47859e743u, 0x9f644ae5a4b1b325u, 0xbc4665b596706114u, 0x873d5d9f0dde1feeu, + 0xeb57ff22fc0c7959u, 0xa90cb506d155a7eau, 0x9316ff75dd87cbd8u, 0x09a7f12442d588f2u, + 0xb7dcbf5354e9beceu, 0x0c11ed6d538aeb2fu, 0xe5d3ef282a242e81u, 0x8f1668c8a86da5fau, + 0x8fa475791a569d10u, 0xf96e017d694487bcu, 0xb38d92d760ec4455u, 0x37c981dcc395a9acu, + 0xe070f78d3927556au, 0x85bbe253f47b1417u, 0x8c469ab843b89562u, 0x93956d7478ccec8eu, + 0xaf58416654a6babbu, 0x387ac8d1970027b2u, 0xdb2e51bfe9d0696au, 0x06997b05fcc0319eu, + 0x88fcf317f22241e2u, 0x441fece3bdf81f03u, 0xab3c2fddeeaad25au, 0xd527e81cad7626c3u, + 0xd60b3bd56a5586f1u, 0x8a71e223d8d3b074u, 0x85c7056562757456u, 0xf6872d5667844e49u, + 0xa738c6bebb12d16cu, 0xb428f8ac016561dbu, 0xd106f86e69d785c7u, 0xe13336d701beba52u, + 0x82a45b450226b39cu, 0xecc0024661173473u, 0xa34d721642b06084u, 0x27f002d7f95d0190u, + 0xcc20ce9bd35c78a5u, 0x31ec038df7b441f4u, 0xff290242c83396ceu, 0x7e67047175a15271u, + 0x9f79a169bd203e41u, 0x0f0062c6e984d386u, 0xc75809c42c684dd1u, 0x52c07b78a3e60868u, + 0xf92e0c3537826145u, 0xa7709a56ccdf8a82u, 0x9bbcc7a142b17ccbu, 0x88a66076400bb691u, + 0xc2abf989935ddbfeu, 0x6acff893d00ea435u, 0xf356f7ebf83552feu, 0x0583f6b8c4124d43u, + 0x98165af37b2153deu, 0xc3727a337a8b704au, 0xbe1bf1b059e9a8d6u, 0x744f18c0592e4c5cu, + 0xeda2ee1c7064130cu, 0x1162def06f79df73u, 0x9485d4d1c63e8be7u, 0x8addcb5645ac2ba8u, + 0xb9a74a0637ce2ee1u, 0x6d953e2bd7173692u, 0xe8111c87c5c1ba99u, 0xc8fa8db6ccdd0437u, + 0x910ab1d4db9914a0u, 0x1d9c9892400a22a2u, 0xb54d5e4a127f59c8u, 0x2503beb6d00cab4bu, + 0xe2a0b5dc971f303au, 0x2e44ae64840fd61du, 0x8da471a9de737e24u, 0x5ceaecfed289e5d2u, + 0xb10d8e1456105dadu, 0x7425a83e872c5f47u, 0xdd50f1996b947518u, 0xd12f124e28f77719u, + 0x8a5296ffe33cc92fu, 0x82bd6b70d99aaa6fu, 0xace73cbfdc0bfb7bu, 0x636cc64d1001550bu, + 0xd8210befd30efa5au, 0x3c47f7e05401aa4eu, 0x8714a775e3e95c78u, 0x65acfaec34810a71u, + 0xa8d9d1535ce3b396u, 0x7f1839a741a14d0du, 0xd31045a8341ca07cu, 0x1ede48111209a050u, + 0x83ea2b892091e44du, 0x934aed0aab460432u, 0xa4e4b66b68b65d60u, 0xf81da84d5617853fu, + 0xce1de40642e3f4b9u, 0x36251260ab9d668eu, 0x80d2ae83e9ce78f3u, 0xc1d72b7c6b426019u, + 0xa1075a24e4421730u, 0xb24cf65b8612f81fu, 0xc94930ae1d529cfcu, 0xdee033f26797b627u, + 0xfb9b7cd9a4a7443cu, 0x169840ef017da3b1u, 0x9d412e0806e88aa5u, 0x8e1f289560ee864eu, + 0xc491798a08a2ad4eu, 0xf1a6f2bab92a27e2u, 0xf5b5d7ec8acb58a2u, 0xae10af696774b1dbu, + 0x9991a6f3d6bf1765u, 0xacca6da1e0a8ef29u, 0xbff610b0cc6edd3fu, 0x17fd090a58d32af3u, + 0xeff394dcff8a948eu, 0xddfc4b4cef07f5b0u, 0x95f83d0a1fb69cd9u, 0x4abdaf101564f98eu, + 0xbb764c4ca7a4440fu, 0x9d6d1ad41abe37f1u, 0xea53df5fd18d5513u, 0x84c86189216dc5edu, + 0x92746b9be2f8552cu, 0x32fd3cf5b4e49bb4u, 0xb7118682dbb66a77u, 0x3fbc8c33221dc2a1u, + 0xe4d5e82392a40515u, 0x0fabaf3feaa5334au, 0x8f05b1163ba6832du, 0x29cb4d87f2a7400eu, + 0xb2c71d5bca9023f8u, 0x743e20e9ef511012u, 0xdf78e4b2bd342cf6u, 0x914da9246b255416u, + 0x8bab8eefb6409c1au, 0x1ad089b6c2f7548eu, 0xae9672aba3d0c320u, 0xa184ac2473b529b1u, + 0xda3c0f568cc4f3e8u, 0xc9e5d72d90a2741eu, 0x8865899617fb1871u, 0x7e2fa67c7a658892u, + 0xaa7eebfb9df9de8du, 0xddbb901b98feeab7u, 0xd51ea6fa85785631u, 0x552a74227f3ea565u, + 0x8533285c936b35deu, 0xd53a88958f87275fu, 0xa67ff273b8460356u, 0x8a892abaf368f137u, + 0xd01fef10a657842cu, 0x2d2b7569b0432d85u, 0x8213f56a67f6b29bu, 0x9c3b29620e29fc73u, + 0xa298f2c501f45f42u, 0x8349f3ba91b47b8fu, 0xcb3f2f7642717713u, 0x241c70a936219a73u, + 0xfe0efb53d30dd4d7u, 0xed238cd383aa0110u, 0x9ec95d1463e8a506u, 0xf4363804324a40aau, + 0xc67bb4597ce2ce48u, 0xb143c6053edcd0d5u, 0xf81aa16fdc1b81dau, 0xdd94b7868e94050au, + 0x9b10a4e5e9913128u, 0xca7cf2b4191c8326u, 0xc1d4ce1f63f57d72u, 0xfd1c2f611f63a3f0u, + 0xf24a01a73cf2dccfu, 0xbc633b39673c8cecu, 0x976e41088617ca01u, 0xd5be0503e085d813u, + 0xbd49d14aa79dbc82u, 0x4b2d8644d8a74e18u, 0xec9c459d51852ba2u, 0xddf8e7d60ed1219eu, + 0x93e1ab8252f33b45u, 0xcabb90e5c942b503u, 0xb8da1662e7b00a17u, 0x3d6a751f3b936243u, + 0xe7109bfba19c0c9du, 0x0cc512670a783ad4u, 0x906a617d450187e2u, 0x27fb2b80668b24c5u, + 0xb484f9dc9641e9dau, 0xb1f9f660802dedf6u, 0xe1a63853bbd26451u, 0x5e7873f8a0396973u, + 0x8d07e33455637eb2u, 0xdb0b487b6423e1e8u, 0xb049dc016abc5e5fu, 0x91ce1a9a3d2cda62u, + 0xdc5c5301c56b75f7u, 0x7641a140cc7810fbu, 0x89b9b3e11b6329bau, 0xa9e904c87fcb0a9du, + 0xac2820d9623bf429u, 0x546345fa9fbdcd44u, 0xd732290fbacaf133u, 0xa97c177947ad4095u, + 0x867f59a9d4bed6c0u, 0x49ed8eabcccc485du, 0xa81f301449ee8c70u, 0x5c68f256bfff5a74u, + 0xd226fc195c6a2f8cu, 0x73832eec6fff3111u, 0x83585d8fd9c25db7u, 0xc831fd53c5ff7eabu, + 0xa42e74f3d032f525u, 0xba3e7ca8b77f5e55u, 0xcd3a1230c43fb26fu, 0x28ce1bd2e55f35ebu, + 0x80444b5e7aa7cf85u, 0x7980d163cf5b81b3u, 0xa0555e361951c366u, 0xd7e105bcc332621fu, + 0xc86ab5c39fa63440u, 0x8dd9472bf3fefaa7u, 0xfa856334878fc150u, 0xb14f98f6f0feb951u, + 0x9c935e00d4b9d8d2u, 0x6ed1bf9a569f33d3u, 0xc3b8358109e84f07u, 0x0a862f80ec4700c8u, + 0xf4a642e14c6262c8u, 0xcd27bb612758c0fau, 0x98e7e9cccfbd7dbdu, 0x8038d51cb897789cu, + 0xbf21e44003acdd2cu, 0xe0470a63e6bd56c3u, 0xeeea5d5004981478u, 0x1858ccfce06cac74u, + 0x95527a5202df0ccbu, 0x0f37801e0c43ebc8u, 0xbaa718e68396cffdu, 0xd30560258f54e6bau, + 0xe950df20247c83fdu, 0x47c6b82ef32a2069u, 0x91d28b7416cdd27eu, 0x4cdc331d57fa5441u, + 0xb6472e511c81471du, 0xe0133fe4adf8e952u, 0xe3d8f9e563a198e5u, 0x58180fddd97723a6u, + 0x8e679c2f5e44ff8fu, 0x570f09eaa7ea7648u, + } + }; + return table; +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // #include @@ -8782,6 +9242,233 @@ bool parse_float_from_chars(const char* first, const char* last, FloatType& out) #endif } +/// whether the eight bytes of @a v (see read_eight_bytes()) are ASCII digits +/// (after fast_float's is_made_of_eight_digits_fast) +inline bool is_eight_digits(std::uint64_t v) noexcept +{ + return ((v & 0xF0F0F0F0F0F0F0F0u) | (((v + 0x0606060606060606u) & 0xF0F0F0F0F0F0F0F0u) >> 4u)) == 0x3333333333333333u; +} + +/// the value of the eight ASCII digits in @a v (see read_eight_bytes()), three +/// multiplications instead of eight (after simdjson and fast_float) +inline std::uint32_t parse_eight_digits(std::uint64_t v) noexcept +{ + v = ((v & 0x0F0F0F0F0F0F0F0Fu) * 2561u) >> 8u; + v = ((v & 0x00FF00FF00FF00FFu) * 6553601u) >> 16u; + return static_cast(((v & 0x0000FFFF0000FFFFu) * 42949672960001u) >> 32u); +} + +/*! +@brief the double nearest to w * 10^q (Eisel-Lemire) + +The algorithm of Daniel Lemire, "Number Parsing at a Gigabyte per Second" +(Software: Practice and Experience, 2021), after fast_float's compute_float +(used under the MIT license). With a 128-bit approximation of 5^q, the product +is always sufficient to round correctly for w with at most 19 digits (Noble +Mushtak and Daniel Lemire, "Fast number parsing without fallback", Software: +Practice and Experience, 2023). Only integer arithmetic is used, so the result +does not depend on the floating-point environment. + +@param[in] q decimal exponent +@param[in] w significand, w != 0 +@return the IEEE-754 bits of the positive result (0 for underflow, infinity + for overflow) +*/ +inline std::uint64_t eisel_lemire(std::int64_t q, std::uint64_t w) noexcept +{ + constexpr int mantissa_bits = 52; + constexpr std::uint64_t infinity = std::uint64_t{0x7FF} << mantissa_bits; + if (q < pow5_128_smallest_power) + { + return 0; + } + if (q > pow5_128_largest_power) + { + return infinity; + } + + const int lz = count_leading_zeros(w); + w <<= static_cast(lz); + const auto index = static_cast(2 * (q - pow5_128_smallest_power)); + uint128_parts product = full_multiplication(w, pow5_128()[index]); + constexpr std::uint64_t precision_mask = 0xFFFFFFFFFFFFFFFFu >> (mantissa_bits + 3); + if ((product.high & precision_mask) == precision_mask) + { + // the lower bits may carry into the result: use the next 64 bits of 5^q + const uint128_parts second = full_multiplication(w, pow5_128()[index + 1]); + product.low += second.high; + if (second.high > product.low) + { + ++product.high; + } + } + + const auto upperbit = static_cast(product.high >> 63u); + const int shift = upperbit + 64 - mantissa_bits - 3; + std::uint64_t mantissa = product.high >> static_cast(shift); + // floor(log2(10^q)) + 63 + 1023, with log2(10) ~ 217706 / 2^16 + std::int64_t power2 = (((152170 + 65536) * q) >> 16) + 63 + upperbit - lz + 1023; + + if (power2 <= 0) // subnormal + { + if (-power2 + 1 >= 64) + { + return 0; + } + mantissa >>= static_cast(-power2 + 1); + mantissa += (mantissa & 1u); + mantissa >>= 1u; + // rounding up may produce the smallest normal number + power2 = (mantissa < (std::uint64_t{1} << mantissa_bits)) ? 0 : 1; + return mantissa | (static_cast(power2) << mantissa_bits); + } + + // a value exactly between two doubles rounds to even; this can only + // happen for small |q|, where 5^q is exact + if (product.low <= 1 && q >= -4 && q <= 23 && (mantissa & 3u) == 1 + && (mantissa << static_cast(shift)) == product.high) + { + mantissa &= ~std::uint64_t{1}; + } + mantissa += (mantissa & 1u); + mantissa >>= 1u; + if (mantissa >= (std::uint64_t{2} << mantissa_bits)) + { + mantissa = std::uint64_t{1} << mantissa_bits; + ++power2; + } + mantissa &= ~(std::uint64_t{1} << mantissa_bits); + if (power2 >= 0x7FF) + { + return infinity; + } + return mantissa | (static_cast(power2) << mantissa_bits); +} + +/*! +@brief parse a validated float token with the Eisel-Lemire algorithm + +The significand is accumulated eight digits at a time where possible. A token +with more than 19 significant digits is truncated to w; the value then lies +in [w, w + 1) * 10^q, and it is only returned if both ends round to the same +double, which covers all but a few such tokens. + +@param[in] first pointer to the first character of the token +@param[in] last pointer past the last character +@param[out] out the correctly rounded value on success (±infinity if it + overflows, like strtod) +@return true on success; false if strtod must decide +*/ +inline bool parse_float_eisel_lemire(const char* first, const char* last, double& out) noexcept +{ + const char* p = first; + const bool negative = (p != last && *p == '-'); + if (negative) + { + ++p; + } + + std::uint64_t w = 0; + int digits = 0; // significant digits in w + std::int64_t exponent = 0; + bool truncated = false; + bool in_fraction = false; + for (;;) + { + // eight digits at a time, as long as they fit into w + while (w != 0 && digits <= 19 - 8 && last - p >= 8) + { + const std::uint64_t v = read_eight_bytes(p); + if (!is_eight_digits(v)) + { + break; + } + w = (w * 100000000u) + parse_eight_digits(v); + digits += 8; + exponent -= in_fraction ? 8 : 0; + p += 8; + } + if (p == last) + { + break; + } + const char c = *p; + if (c >= '0' && c <= '9') + { + if (w == 0 && c == '0') + { + // leading zeros are not significant, but scale a fraction + exponent -= in_fraction ? 1 : 0; + } + else if (digits < 19) + { + w = (w * 10u) + static_cast(c - '0'); + ++digits; + exponent -= in_fraction ? 1 : 0; + } + else + { + // dropped: the value lies between w and w + 1 (in units of + // the last kept digit) unless all dropped digits are zero + truncated = truncated || c != '0'; + exponent += in_fraction ? 0 : 1; + } + ++p; + } + else if (c == '.') + { + in_fraction = true; + ++p; + } + else + { + break; // 'e' or 'E' + } + } + + if (p != last) + { + ++p; // 'e' or 'E' + bool exp_negative = false; + if (p != last && (*p == '-' || *p == '+')) + { + exp_negative = (*p == '-'); + ++p; + } + std::int64_t exp_value = 0; + for (; p != last; ++p) + { + // saturate: any exponent beyond this under- or overflows anyway + if (exp_value < 100000) + { + exp_value = (exp_value * 10) + (*p - '0'); + } + } + exponent += exp_negative ? -exp_value : exp_value; + } + + std::uint64_t bits = 0; + if (w != 0) + { + bits = eisel_lemire(exponent, w); + if (truncated && (w + 1 == 0 || eisel_lemire(exponent, w + 1) != bits)) + { + return false; + } + } + bits |= negative ? (std::uint64_t{1} << 63u) : 0u; + static_assert(sizeof(double) == sizeof(std::uint64_t), "double must have 64 bits"); + std::memcpy(&out, &bits, sizeof(out)); + return true; +} + +/// Eisel-Lemire is only implemented for `double` +template +bool parse_float_eisel_lemire(const char* /*first*/, const char* /*last*/, FloatType& /*out*/) noexcept +{ + return false; +} + /*! @brief check whether Clinger's fast path can still succeed for a float token @@ -8840,8 +9527,9 @@ inline bool mantissa_fits_clinger(const char* token, std::size_t decimal_point_p /*! @brief convert a validated float token without the C library, if possible -Tries std::from_chars (when available) and then Clinger's exact fast path -(double only), skipping the latter when it cannot succeed. +Tries std::from_chars (when available), Clinger's exact fast path (double +only, skipped when it cannot succeed), and the Eisel-Lemire algorithm (double +only). @param[in] first pointer to the first character of the token @param[in] last pointer past the last character @@ -8864,8 +9552,12 @@ bool convert_float_fast(const char* first, const char* last, std::size_t decimal // Skipping a fast path that cannot succeed is lossless and saves a full // extra pass over the token's bytes, which otherwise shows up on // high-precision inputs such as canada.json - return mantissa_fits_clinger(first, decimal_point_position, mantissa_end) - && parse_float_fast(first, last, value); + if (mantissa_fits_clinger(first, decimal_point_position, mantissa_end) + && parse_float_fast(first, last, value)) + { + return true; + } + return parse_float_eisel_lemire(first, last, value); } /// std::strtof, std::strtod, or std::strtold, chosen by the type of @a f diff --git a/tests/src/unit-class_lexer.cpp b/tests/src/unit-class_lexer.cpp index 5d52179d7..b56e4bd0b 100644 --- a/tests/src/unit-class_lexer.cpp +++ b/tests/src/unit-class_lexer.cpp @@ -12,10 +12,14 @@ #include using nlohmann::json; +#include // array #include // FLT_EVAL_METHOD +#include // uint32_t, uint64_t #include // strtod +#include // memcpy #include // stringstream #include // string +#include // pair #include // vector namespace @@ -700,3 +704,614 @@ TEST_CASE("parse_float_fast declines what it cannot convert exactly") CHECK_FALSE(fast("1e23", out)); CHECK_FALSE(fast("1e-23", out)); } + +namespace +{ +// arbitrary-precision unsigned integers, just enough to recompute the table of +// powers of five (little-endian 32-bit limbs) +using big_uint = std::vector; + +void big_trim(big_uint& a) +{ + while (!a.empty() && a.back() == 0) + { + a.pop_back(); + } +} + +big_uint big_from(std::uint64_t high, std::uint64_t low) +{ + big_uint a = {static_cast(low), static_cast(low >> 32u), + static_cast(high), static_cast(high >> 32u) + }; + big_trim(a); + return a; +} + +big_uint big_mul(const big_uint& a, const big_uint& b) +{ + big_uint r(a.size() + b.size(), 0); + for (std::size_t i = 0; i < a.size(); ++i) + { + std::uint64_t carry = 0; + for (std::size_t j = 0; j < b.size(); ++j) + { + const std::uint64_t t = (static_cast(a[i]) * b[j]) + r[i + j] + carry; + r[i + j] = static_cast(t); + carry = t >> 32u; + } + r[i + b.size()] = static_cast(carry); + } + big_trim(r); + return r; +} + +big_uint big_shl(const big_uint& a, std::size_t s) +{ + big_uint r(s / 32, 0); + std::uint32_t carry = 0; + for (const std::uint32_t x : a) + { + const std::uint64_t t = static_cast(x) << (s % 32); + r.push_back(static_cast(t) | carry); + carry = static_cast(t >> 32u); + } + r.push_back(carry); + big_trim(r); + return r; +} + +// a + 1 (add) or a - 1 (!add, a > 0) +big_uint big_step(big_uint a, bool add) +{ + for (auto& x : a) + { + const std::uint32_t old = x; + x = add ? x + 1 : x - 1; + if ((add && x > old) || (!add && x < old)) + { + break; + } + } + if (add && (a.empty() || a.back() == 0)) + { + a.push_back(1); + } + big_trim(a); + return a; +} + +bool big_less_equal(const big_uint& a, const big_uint& b) +{ + if (a.size() != b.size()) + { + return a.size() < b.size(); + } + for (std::size_t i = a.size(); i-- > 0;) + { + if (a[i] != b[i]) + { + return a[i] < b[i]; + } + } + return true; +} + +std::size_t big_bit_length(const big_uint& a) +{ + std::size_t n = a.size() * 32; + for (std::uint32_t top = a.back(); (top & 0x80000000u) == 0; top <<= 1u) + { + --n; + } + return n; +} + +std::uint64_t bits_of(double d) +{ + std::uint64_t b = 0; + std::memcpy(&b, &d, sizeof(b)); + return b; +} + +bool eisel_lemire(const std::string& s, double& out) +{ + return nlohmann::detail::parse_float_eisel_lemire(s.data(), s.data() + s.size(), out); +} + +// significant digits of a token, without trailing zeros +std::size_t significant_digits(const std::string& s) +{ + std::string digits; + for (const char c : s) + { + if (c == 'e' || c == 'E') + { + break; + } + if (c >= '0' && c <= '9' && !(digits.empty() && c == '0')) + { + digits += c; + } + } + while (!digits.empty() && digits.back() == '0') + { + digits.pop_back(); + } + return digits.size(); +} +} // namespace + +TEST_CASE("Eisel-Lemire float conversion") +{ + SECTION("the table of powers of five") + { + // Recompute every entry the way fast_float's table_generation.py + // defines it, using only multiplications and comparisons: for q >= 0 + // the most significant 128 bits of 5^q; for q < 0 floor(2^b / 5^-q) + 1 + // for b = z + 127 (q >= -27), or that value for b = 2z + 128 cut to + // its most significant 128 bits (q < -27), where z is the bit length + // of 5^-q. + const auto& table = nlohmann::detail::pow5_128(); + big_uint power5 = {1}; + for (std::int64_t q = 0; q <= nlohmann::detail::pow5_128_largest_power; ++q) + { + const auto index = static_cast(2 * (q - nlohmann::detail::pow5_128_smallest_power)); + const big_uint entry = big_from(table[index], table[index + 1]); + const std::size_t bits = big_bit_length(power5); + if (bits <= 128) + { + CHECK(entry == big_shl(power5, 128 - bits)); + } + else + { + // floor(5^q / 2^(bits - 128)) + CHECK(big_less_equal(big_shl(entry, bits - 128), power5)); + CHECK_FALSE(big_less_equal(big_shl(big_step(entry, true), bits - 128), power5)); + } + power5 = big_mul(power5, {5}); + } + + power5 = {5}; + for (std::int64_t q = -1; q >= nlohmann::detail::pow5_128_smallest_power; --q) + { + const auto index = static_cast(2 * (q - nlohmann::detail::pow5_128_smallest_power)); + const big_uint entry = big_from(table[index], table[index + 1]); + CHECK(big_bit_length(entry) == 128); + const std::size_t z = big_bit_length(power5); + const big_uint two_b = big_shl({1}, q >= -27 ? z + 127 : (2 * z) + 128); + // c = floor(2^b / p) + 1, stored as floor(c / 2^s): + // (entry * 2^s - 1) * p <= 2^b < ((entry + 1) * 2^s - 1) * p + const std::size_t s = q >= -27 ? 0 : z + 1; + CHECK(big_less_equal(big_mul(big_step(big_shl(entry, s), false), power5), two_b)); + CHECK_FALSE(big_less_equal(big_mul(big_step(big_shl(big_step(entry, true), s), false), power5), two_b)); + power5 = big_mul(power5, {5}); + } + } + + SECTION("128-bit products and leading zeros") + { + // whichever implementation the compiler gets (with or without a + // 128-bit integer type or a builtin) + std::uint64_t state = 42; + for (int i = 0; i < 10000; ++i) + { + state ^= state << 13u; + state ^= state >> 7u; + state ^= state << 17u; + const std::uint64_t a = state; + const std::uint64_t b = (state * 0x9E3779B97F4A7C15u) >> (i % 64); + const auto product = nlohmann::detail::full_multiplication(a, b); + CHECK(big_from(product.high, product.low) == big_mul(big_from(0, a), big_from(0, b))); + + const int k = i % 64; + const std::uint64_t x = (std::uint64_t{1} << k) | (a & ((std::uint64_t{1} << k) - 1)); + CHECK(nlohmann::detail::count_leading_zeros(x) == 63 - k); + } + } + + SECTION("known values") + { + // Generated with Python, whose float() is correctly rounded: + // cases = [, 2**53 + 2k + 1, 2**54 + 4k + 2, and exact midpoints + // between neighbouring doubles, also 1e-60 above and below them] + // print('{"%s", 0x%016xu},' % (s, struct.unpack('> known = + { + {"0", 0x0000000000000000u}, + {"-0", 0x8000000000000000u}, + {"0.0", 0x0000000000000000u}, + {"-0.0", 0x8000000000000000u}, + {"0e5", 0x0000000000000000u}, + {"0.000e-9", 0x0000000000000000u}, + {"1", 0x3ff0000000000000u}, + {"-1", 0xbff0000000000000u}, + {"0.1", 0x3fb999999999999au}, + {"0.3", 0x3fd3333333333333u}, + {"1.5", 0x3ff8000000000000u}, + {"-2.5e-3", 0xbf647ae147ae147bu}, + {"1e23", 0x44b52d02c7e14af6u}, + {"1e22", 0x4480f0cf064dd592u}, + {"8.98846567431158e307", 0x7fe0000000000000u}, + {"2.2250738585072011e-308", 0x000fffffffffffffu}, + {"2.2250738585072012e-308", 0x0010000000000000u}, + {"2.2250738585072014e-308", 0x0010000000000000u}, + {"4.9406564584124654e-324", 0x0000000000000001u}, + {"2.4703282292062327e-324", 0x0000000000000000u}, + {"2.4703282292062328e-324", 0x0000000000000001u}, + {"1e-324", 0x0000000000000000u}, + {"3e-324", 0x0000000000000001u}, + {"1.7976931348623157e308", 0x7fefffffffffffffu}, + {"1.7976931348623158e308", 0x7fefffffffffffffu}, + {"1.7976931348623159e308", 0x7ff0000000000000u}, + {"1e308", 0x7fe1ccf385ebc8a0u}, + {"1e309", 0x7ff0000000000000u}, + {"-1e400", 0xfff0000000000000u}, + {"1e-400", 0x0000000000000000u}, + {"9007199254740991", 0x433fffffffffffffu}, + {"9007199254740992", 0x4340000000000000u}, + {"9007199254740993", 0x4340000000000000u}, + {"9007199254740995", 0x4340000000000002u}, + {"18014398509481986", 0x4350000000000000u}, + {"18014398509481990", 0x4350000000000002u}, + {"7.2057594037927933e16", 0x4370000000000000u}, + {"123456789012345678901234567890", 0x45f8ee90ff6c373eu}, + {"1.000000000000000111", 0x3ff0000000000000u}, + {"1.0000000000000001110223", 0x3ff0000000000000u}, + {"1.00000000000000011102230246251565404236316680908203125", 0x3ff0000000000000u}, + {"1.00000000000000011102230246251565404236316680908203126", 0x3ff0000000000001u}, + {"0.00000000000000000000000000000000000000000000000000000000000001", 0x3310747ddddf22a8u}, + {"100000000000000000000000000000000000000000000", 0x4911efc659cf7d4cu}, + {"1234567890123456789", 0x43b12210f47de981u}, + {"12345678901234567890", 0x43e56a95319d63e1u}, + {"1234567890123456789.5", 0x43b12210f47de981u}, + {"0.1234567890123456789012345", 0x3fbf9add3746f65fu}, + {"4.4501477170144023e-308", 0x001fffffffffffffu}, + {"2.4406961166466664e-309", 0x0001c14ae5310a48u}, + {"5e-324", 0x0000000000000001u}, + {"1.0e-307", 0x0031fa182c40c60du}, + {"179769313486231570814527423731704356798070567525844996598917476803157260780028538760589558632766878171540458953514382464234321326889464182768467546703537516986049910576551282076245490090389328944075868508455133942304583236903222948165808559332123348274797826204144723168738177180919299881250404026184124858368", 0x7fefffffffffffffu}, + {"4.9e-324", 0x0000000000000001u}, + {"9007199254740993", 0x4340000000000000u}, + {"18014398509481986", 0x4350000000000000u}, + {"9007199254740995", 0x4340000000000002u}, + {"18014398509481990", 0x4350000000000002u}, + {"9007199254740997", 0x4340000000000002u}, + {"18014398509481994", 0x4350000000000002u}, + {"9007199254740999", 0x4340000000000004u}, + {"18014398509481998", 0x4350000000000004u}, + {"9007199254741001", 0x4340000000000004u}, + {"18014398509482002", 0x4350000000000004u}, + {"9007199254741003", 0x4340000000000006u}, + {"18014398509482006", 0x4350000000000006u}, + {"9007199254741005", 0x4340000000000006u}, + {"18014398509482010", 0x4350000000000006u}, + {"9007199254741007", 0x4340000000000008u}, + {"18014398509482014", 0x4350000000000008u}, + {"9007199254741009", 0x4340000000000008u}, + {"18014398509482018", 0x4350000000000008u}, + {"9007199254741011", 0x434000000000000au}, + {"18014398509482022", 0x435000000000000au}, + {"9007199254741013", 0x434000000000000au}, + {"18014398509482026", 0x435000000000000au}, + {"9007199254741015", 0x434000000000000cu}, + {"18014398509482030", 0x435000000000000cu}, + {"9007199254741017", 0x434000000000000cu}, + {"18014398509482034", 0x435000000000000cu}, + {"9007199254741019", 0x434000000000000eu}, + {"18014398509482038", 0x435000000000000eu}, + {"9007199254741021", 0x434000000000000eu}, + {"18014398509482042", 0x435000000000000eu}, + {"9007199254741023", 0x4340000000000010u}, + {"18014398509482046", 0x4350000000000010u}, + {"9007199254741025", 0x4340000000000010u}, + {"18014398509482050", 0x4350000000000010u}, + {"9007199254741027", 0x4340000000000012u}, + {"18014398509482054", 0x4350000000000012u}, + {"9007199254741029", 0x4340000000000012u}, + {"18014398509482058", 0x4350000000000012u}, + {"9007199254741031", 0x4340000000000014u}, + {"18014398509482062", 0x4350000000000014u}, + {"9007199254741033", 0x4340000000000014u}, + {"18014398509482066", 0x4350000000000014u}, + {"9007199254741035", 0x4340000000000016u}, + {"18014398509482070", 0x4350000000000016u}, + {"9007199254741037", 0x4340000000000016u}, + {"18014398509482074", 0x4350000000000016u}, + {"9007199254741039", 0x4340000000000018u}, + {"18014398509482078", 0x4350000000000018u}, + {"9007199254741041", 0x4340000000000018u}, + {"18014398509482082", 0x4350000000000018u}, + {"9007199254741043", 0x434000000000001au}, + {"18014398509482086", 0x435000000000001au}, + {"9007199254741045", 0x434000000000001au}, + {"18014398509482090", 0x435000000000001au}, + {"9007199254741047", 0x434000000000001cu}, + {"18014398509482094", 0x435000000000001cu}, + {"9007199254741049", 0x434000000000001cu}, + {"18014398509482098", 0x435000000000001cu}, + {"9007199254741051", 0x434000000000001eu}, + {"18014398509482102", 0x435000000000001eu}, + {"9007199254741053", 0x434000000000001eu}, + {"18014398509482106", 0x435000000000001eu}, + {"9007199254741055", 0x4340000000000020u}, + {"18014398509482110", 0x4350000000000020u}, + {"9007199254741057", 0x4340000000000020u}, + {"18014398509482114", 0x4350000000000020u}, + {"9007199254741059", 0x4340000000000022u}, + {"18014398509482118", 0x4350000000000022u}, + {"9007199254741061", 0x4340000000000022u}, + {"18014398509482122", 0x4350000000000022u}, + {"9007199254741063", 0x4340000000000024u}, + {"18014398509482126", 0x4350000000000024u}, + {"9007199254741065", 0x4340000000000024u}, + {"18014398509482130", 0x4350000000000024u}, + {"9007199254741067", 0x4340000000000026u}, + {"18014398509482134", 0x4350000000000026u}, + {"9007199254741069", 0x4340000000000026u}, + {"18014398509482138", 0x4350000000000026u}, + {"9007199254741071", 0x4340000000000028u}, + {"18014398509482142", 0x4350000000000028u}, + {"0.00000000000000000142055942108419951085063380808124102279024543292671443374397544090470546507276594638824462890625", 0x3c3a3466f662d406u}, + {"0.000000000000000001420559421084199510850633808081241022790245432926714433743975440904705465072765946388244628906251", 0x3c3a3466f662d407u}, + {"0.00000000000000000142055942108419951085063380808124102279024543292671443374397444090470546507276594638824462890625", 0x3c3a3466f662d406u}, + {"8656.5250079159513916238211095333099365234375", 0x40c0e84333759a94u}, + {"8656.52500791595139162382110953330993652343751", 0x40c0e84333759a94u}, + {"8656.525007915951391623821109533309936523437499999999999999999", 0x40c0e84333759a93u}, + {"13.07696731650454946560557800694368779659271240234375", 0x402a276842967ef0u}, + {"13.076967316504549465605578006943687796592712402343751", 0x402a276842967ef0u}, + {"13.07696731650454946560557800694368779659271240234374999999999", 0x402a276842967eefu}, + {"74708253715391928", 0x437096ac2cc7ee5cu}, + {"747082537153919281", 0x43a4bc5737f9e9f2u}, + {"74708253715391927.99999999999999999999999999999999999999999999", 0x437096ac2cc7ee5bu}, + {"1809802988.27203977108001708984375", 0x41daf7d9bb11691au}, + {"1809802988.272039771080017089843751", 0x41daf7d9bb11691au}, + {"1809802988.272039771080017089843749999999999999999999999999999", 0x41daf7d9bb116919u}, + {"51.390809684186766759239617385901510715484619140625", 0x4049b2060d3e4568u}, + {"51.3908096841867667592396173859015107154846191406251", 0x4049b2060d3e4569u}, + {"51.39080968418676675923961738590151071548461914062499999999999", 0x4049b2060d3e4568u}, + {"9999807412.59738445281982421875", 0x4202a0479da4c772u}, + {"9999807412.597384452819824218751", 0x4202a0479da4c772u}, + {"9999807412.597384452819824218749999999999999999999999999999999", 0x4202a0479da4c771u}, + {"0.00000000023260971767101600534534272272645127367651785021962496102787554264068603515625", 0x3deff83a135dec10u}, + {"0.000000000232609717671016005345342722726451273676517850219624961027875542640686035156251", 0x3deff83a135dec11u}, + {"0.00000000023260971767101600534534272272645127367651785021962496102787544264068603515625", 0x3deff83a135dec10u}, + {"0.000000000497610482021202508216234789619066523902457532813059515319764614105224609375", 0x3e01190730d1ec48u}, + {"0.0000000004976104820212025082162347896190665239024575328130595153197646141052246093751", 0x3e01190730d1ec48u}, + {"0.000000000497610482021202508216234789619066523902457532813059515319764514105224609375", 0x3e01190730d1ec47u}, + {"0.0000000000291336422596533830691676779231223432149733287843673679162748157978057861328125", 0x3dc004321559736eu}, + {"0.00000000002913364225965338306916767792312234321497332878436736791627481579780578613281251", 0x3dc004321559736fu}, + {"0.0000000000291336422596533830691676779231223432149733287843673679162748057978057861328125", 0x3dc004321559736eu}, + {"0.000000000000000039237155154865396441907405399546892080260012902422940561653064150959835387766361236572265625", 0x3c869e61cfa3b8a4u}, + {"0.0000000000000000392371551548653964419074053995468920802600129024229405616530641509598353877663612365722656251", 0x3c869e61cfa3b8a5u}, + {"0.000000000000000039237155154865396441907405399546892080260012902422940561653054150959835387766361236572265625", 0x3c869e61cfa3b8a4u}, + {"0.00000000000000006496592863767266414092845207857846208631635335985395063307379359685000963509082794189453125", 0x3c92b9a3b219ee84u}, + {"0.000000000000000064965928637672664140928452078578462086316353359853950633073793596850009635090827941894531251", 0x3c92b9a3b219ee85u}, + {"0.00000000000000006496592863767266414092845207857846208631635335985395063307378359685000963509082794189453125", 0x3c92b9a3b219ee84u}, + {"0.0000000000448169607439179753541822628688597626549217078917308754171244800090789794921875", 0x3dc8a36d2e8094dau}, + {"0.00000000004481696074391797535418226286885976265492170789173087541712448000907897949218751", 0x3dc8a36d2e8094dau}, + {"0.0000000000448169607439179753541822628688597626549217078917308754171244700090789794921875", 0x3dc8a36d2e8094d9u}, + {"1800873890234250.875", 0x4319978a820cfe2cu}, + {"1800873890234250.8751", 0x4319978a820cfe2cu}, + {"1800873890234250.874999999999999999999999999999999999999999999", 0x4319978a820cfe2bu}, + {"0.000023941132739153309216405436654628857695570331998169422149658203125", 0x3ef91aa61d42e0e0u}, + {"0.0000239411327391533092164054366546288576955703319981694221496582031251", 0x3ef91aa61d42e0e1u}, + {"0.000023941132739153309216405436654628857695570331998169422149658193125", 0x3ef91aa61d42e0e0u}, + {"8339818978785937.5", 0x433da1056bb8ba92u}, + {"8339818978785937.51", 0x433da1056bb8ba92u}, + {"8339818978785937.499999999999999999999999999999999999999999999", 0x433da1056bb8ba91u}, + {"283649145986385328", 0x438f7dcb29dae42eu}, + {"2836491459863853281", 0x43c3ae9efa28ce9cu}, + {"283649145986385327.9999999999999999999999999999999999999999999", 0x438f7dcb29dae42du}, + {"0.0000000000000000004203729478287971874304476341685497759528753070014375965192388040492232903488911688327789306640625", 0x3c1f049ed78cb8a2u}, + {"0.00000000000000000042037294782879718743044763416854977595287530700143759651923880404922329034889116883277893066406251", 0x3c1f049ed78cb8a3u}, + {"0.0000000000000000004203729478287971874304476341685497759528753070014375965192387040492232903488911688327789306640625", 0x3c1f049ed78cb8a2u}, + {"340031.78635183119331486523151397705078125", 0x4114c0ff25396a18u}, + {"340031.786351831193314865231513977050781251", 0x4114c0ff25396a19u}, + {"340031.7863518311933148652315139770507812499999999999999999999", 0x4114c0ff25396a18u}, + {"0.0000000004543709717853330820053712055931242029538363880192264332436025142669677734375", 0x3dff3960f070bf76u}, + {"0.00000000045437097178533308200537120559312420295383638801922643324360251426696777343751", 0x3dff3960f070bf76u}, + {"0.0000000004543709717853330820053712055931242029538363880192264332436024142669677734375", 0x3dff3960f070bf75u}, + {"0.0000000000007937958898451257082591244100968761704503924570008877026339177973568439483642578125", 0x3d6bede0b40e37c2u}, + {"0.00000000000079379588984512570825912441009687617045039245700088770263391779735684394836425781251", 0x3d6bede0b40e37c3u}, + {"0.0000000000007937958898451257082591244100968761704503924570008877026339176973568439483642578125", 0x3d6bede0b40e37c2u}, + {"0.00000000000000068304843500200357021582520000166071595321259251644419041582523277611471712589263916015625", 0x3cc89c02848f8654u}, + {"0.000000000000000683048435002003570215825200001660715953212592516444190415825232776114717125892639160156251", 0x3cc89c02848f8655u}, + {"0.00000000000000068304843500200357021582520000166071595321259251644419041582513277611471712589263916015625", 0x3cc89c02848f8654u}, + {"0.00000000147705080029041786940146712084494760863773166192913777194917201995849609375", 0x3e1960235bc34d06u}, + {"0.000000001477050800290417869401467120844947608637731661929137771949172019958496093751", 0x3e1960235bc34d06u}, + {"0.00000000147705080029041786940146712084494760863773166192913777194917101995849609375", 0x3e1960235bc34d05u}, + {"2164972979236447104", 0x43be0b87743fb524u}, + {"21649729792364471041", 0x43f2c734a8a7d136u}, + {"2164972979236447103.999999999999999999999999999999999999999999", 0x43be0b87743fb523u}, + {"13124633159586767", 0x4347506464a469e8u}, + {"131246331595867671", 0x437d247d7dcd8461u}, + {"13124633159586766.99999999999999999999999999999999999999999999", 0x4347506464a469e7u}, + {"0.000000000000027605165650764659887597286048864477738779108113853499872902830247767269611358642578125", 0x3d1f14a5b417cb08u}, + {"0.0000000000000276051656507646598875972860488644777387791081138534998729028302477672696113586425781251", 0x3d1f14a5b417cb09u}, + {"0.000000000000027605165650764659887597286048864477738779108113853499872902820247767269611358642578125", 0x3d1f14a5b417cb08u}, + {"0.01727792833395616796388072344825559412129223346710205078125", 0x3f91b14e248c42c8u}, + {"0.017277928333956167963880723448255594121292233467102050781251", 0x3f91b14e248c42c9u}, + {"0.01727792833395616796388072344825559412129223346710205078124999", 0x3f91b14e248c42c8u}, + {"0.00000000000003096211381890547702500862566064822950373937142376501441276559489779174327850341796875", 0x3d216e1c61130b22u}, + {"0.000000000000030962113818905477025008625660648229503739371423765014412765594897791743278503417968751", 0x3d216e1c61130b22u}, + {"0.00000000000003096211381890547702500862566064822950373937142376501441276558489779174327850341796875", 0x3d216e1c61130b21u}, + {"0.0000000000000683414954481284270259359974108983284531919182025472281338807079009711742401123046875", 0x3d333c86137e5170u}, + {"0.00000000000006834149544812842702593599741089832845319191820254722813388070790097117424011230468751", 0x3d333c86137e5170u}, + {"0.0000000000000683414954481284270259359974108983284531919182025472281338806979009711742401123046875", 0x3d333c86137e516fu}, + {"3237539054990129.75", 0x4327010c9aa36664u}, + {"3237539054990129.751", 0x4327010c9aa36664u}, + {"3237539054990129.749999999999999999999999999999999999999999999", 0x4327010c9aa36663u}, + {"0.0000000000000105429301486355911969397173440055240907798901443814809653076736140064895153045654296875", 0x3d07bd95dfb8a1eeu}, + {"0.00000000000001054293014863559119693971734400552409077989014438148096530767361400648951530456542968751", 0x3d07bd95dfb8a1eeu}, + {"0.0000000000000105429301486355911969397173440055240907798901443814809653076636140064895153045654296875", 0x3d07bd95dfb8a1edu}, + {"9460.2061893763575426419265568256378173828125", 0x40c27a1a6469da1eu}, + {"9460.20618937635754264192655682563781738281251", 0x40c27a1a6469da1fu}, + {"9460.206189376357542641926556825637817382812499999999999999999", 0x40c27a1a6469da1eu}, + {"465000373656610.53125", 0x42fa6ea56179c228u}, + {"465000373656610.531251", 0x42fa6ea56179c229u}, + {"465000373656610.5312499999999999999999999999999999999999999999", 0x42fa6ea56179c228u}, + {"0.000000000107709773707743713401260815383950956818093214195641849073581397533416748046875", 0x3ddd9b66c974bb14u}, + {"0.0000000001077097737077437134012608153839509568180932141956418490735813975334167480468751", 0x3ddd9b66c974bb14u}, + {"0.000000000107709773707743713401260815383950956818093214195641849073581297533416748046875", 0x3ddd9b66c974bb13u}, + {"0.012083347821554271152300064073870089487172663211822509765625", 0x3f88bf277dc215f4u}, + {"0.0120833478215542711523000640738700894871726632118225097656251", 0x3f88bf277dc215f5u}, + {"0.01208334782155427115230006407387008948717266321182250976562499", 0x3f88bf277dc215f4u}, + {"2309804058391724800", 0x43c0070946d098e2u}, + {"23098040583917248001", 0x43f408cb9884bf1au}, + {"2309804058391724799.999999999999999999999999999999999999999999", 0x43c0070946d098e1u}, + {"0.000000000078286047058220682019445698291671103911937290575906445155851542949676513671875", 0x3dd584e40ca80638u}, + {"0.0000000000782860470582206820194456982916711039119372905759064451558515429496765136718751", 0x3dd584e40ca80638u}, + {"0.000000000078286047058220682019445698291671103911937290575906445155851532949676513671875", 0x3dd584e40ca80637u}, + {"2940024994425.709228515625", 0x4285643929d3cdacu}, + {"2940024994425.7092285156251", 0x4285643929d3cdadu}, + {"2940024994425.709228515624999999999999999999999999999999999999", 0x4285643929d3cdacu}, + {"9503358.427352792583405971527099609375", 0x4162204fcdacdfc4u}, + {"9503358.4273527925834059715270996093751", 0x4162204fcdacdfc4u}, + {"9503358.427352792583405971527099609374999999999999999999999999", 0x4162204fcdacdfc3u}, + {"0.00005589147834433423614399101542193903924271580763161182403564453125", 0x3f0d4da092aa5d6au}, + {"0.000055891478344334236143991015421939039242715807631611824035644531251", 0x3f0d4da092aa5d6bu}, + {"0.00005589147834433423614399101542193903924271580763161182403564452125", 0x3f0d4da092aa5d6au}, + {"0.0000000164797038506435664295103900420409737126448135313694365322589874267578125", 0x3e51b1e8107bd640u}, + {"0.00000001647970385064356642951039004204097371264481353136943653225898742675781251", 0x3e51b1e8107bd641u}, + {"0.0000000164797038506435664295103900420409737126448135313694365322589774267578125", 0x3e51b1e8107bd640u}, + {"73055.7873927834807545877993106842041015625", 0x40f1d5fc99292ce2u}, + {"73055.78739278348075458779931068420410156251", 0x40f1d5fc99292ce3u}, + {"73055.78739278348075458779931068420410156249999999999999999999", 0x40f1d5fc99292ce2u}, + {"0.00000000002558172364455232122427880575336067736115508441940846751094795763492584228515625", 0x3dbc209d7509115au}, + {"0.000000000025581723644552321224278805753360677361155084419408467510947957634925842285156251", 0x3dbc209d7509115bu}, + {"0.00000000002558172364455232122427880575336067736115508441940846751094794763492584228515625", 0x3dbc209d7509115au}, + {"0.000000000854497507560657937804791333430312443020238077906469698064029216766357421875", 0x3e0d5c3d540cc0d2u}, + {"0.0000000008544975075606579378047913334303124430202380779064696980640292167663574218751", 0x3e0d5c3d540cc0d2u}, + {"0.000000000854497507560657937804791333430312443020238077906469698064029116766357421875", 0x3e0d5c3d540cc0d1u}, + {"26671499731071461376", 0x43f722433b19970eu}, + {"266714997310714613761", 0x442cead409dffcd2u}, + {"26671499731071461375.99999999999999999999999999999999999999999", 0x43f722433b19970eu}, + {"0.0000000002725195963972150049060368088050545186395989816219298518262803554534912109375", 0x3df2ba37271cf5f2u}, + {"0.00000000027251959639721500490603680880505451863959898162192985182628035545349121093751", 0x3df2ba37271cf5f2u}, + {"0.0000000002725195963972150049060368088050545186395989816219298518262802554534912109375", 0x3df2ba37271cf5f1u}, + {"0.0000000000377224663702246439557904325637937886957218314165629635681398212909698486328125", 0x3dc4bcf7157af68eu}, + {"0.00000000003772246637022464395579043256379378869572183141656296356813982129096984863281251", 0x3dc4bcf7157af68fu}, + {"0.0000000000377224663702246439557904325637937886957218314165629635681398112909698486328125", 0x3dc4bcf7157af68eu}, + {"0.00000000000000009977342762593364154535842258994910996905560208471673566688053824691451154649257659912109375", 0x3c9cc1fac312e2a6u}, + {"0.000000000000000099773427625933641545358422589949109969055602084716735666880538246914511546492576599121093751", 0x3c9cc1fac312e2a6u}, + {"0.00000000000000009977342762593364154535842258994910996905560208471673566688052824691451154649257659912109375", 0x3c9cc1fac312e2a5u}, + {"0.0000000003146248171139977444431948290159907662133509376189977047033607959747314453125", 0x3df59ef03588a228u}, + {"0.00000000031462481711399774444319482901599076621335093761899770470336079597473144531251", 0x3df59ef03588a229u}, + {"0.0000000003146248171139977444431948290159907662133509376189977047033606959747314453125", 0x3df59ef03588a228u}, + {"488899209263030304", 0x439b23acf64c5e80u}, + {"4888992092630303041", 0x43d0f64c19efbb10u}, + {"488899209263030303.9999999999999999999999999999999999999999999", 0x439b23acf64c5e80u}, + {"1.88357157350592807620870416940306313335895538330078125", 0x3ffe231bf23e21acu}, + {"1.883571573505928076208704169403063133358955383300781251", 0x3ffe231bf23e21adu}, + {"1.883571573505928076208704169403063133358955383300781249999999", 0x3ffe231bf23e21acu}, + {"0.0000000216458400594294836451424756990254139044083103726734407246112823486328125", 0x3e573df694e72fb8u}, + {"0.00000002164584005942948364514247569902541390440831037267344072461128234863281251", 0x3e573df694e72fb9u}, + {"0.0000000216458400594294836451424756990254139044083103726734407246112723486328125", 0x3e573df694e72fb8u}, + {"5107.79271041116453488939441740512847900390625", 0x40b3f3caef11cb26u}, + {"5107.792710411164534889394417405128479003906251", 0x40b3f3caef11cb27u}, + {"5107.792710411164534889394417405128479003906249999999999999999", 0x40b3f3caef11cb26u}, + {"734059.8035226609208621084690093994140625", 0x412666d79b67527cu}, + {"734059.80352266092086210846900939941406251", 0x412666d79b67527du}, + {"734059.8035226609208621084690093994140624999999999999999999999", 0x412666d79b67527cu}, + {"61431562016722684", 0x436b47f5c40021e0u}, + {"614315620167226841", 0x43a10cf99a80152cu}, + {"61431562016722683.99999999999999999999999999999999999999999999", 0x436b47f5c40021dfu}, + {"2.0060840449445034305853141631814651191234588623046875", 0x40000c75cab08326u}, + {"2.00608404494450343058531416318146511912345886230468751", 0x40000c75cab08326u}, + {"2.006084044944503430585314163181465119123458862304687499999999", 0x40000c75cab08325u}, + {"0.0000001760623599453036952716420489480075861621344301966018974781036376953125", 0x3e87a174e55262cau}, + {"0.00000017606235994530369527164204894800758616213443019660189747810363769531251", 0x3e87a174e55262cbu}, + {"0.0000001760623599453036952716420489480075861621344301966018974781035376953125", 0x3e87a174e55262cau}, + {"0.833085849636964581588216560703585855662822723388671875", 0x3feaa8a3a7de6fb6u}, + {"0.8330858496369645815882165607035858556628227233886718751", 0x3feaa8a3a7de6fb6u}, + {"0.8330858496369645815882165607035858556628227233886718749999999", 0x3feaa8a3a7de6fb5u}, + {"45031428.4182307310402393341064453125", 0x418579002358895au}, + {"45031428.41823073104023933410644531251", 0x418579002358895bu}, + {"45031428.41823073104023933410644531249999999999999999999999999", 0x418579002358895au}, + {"5003361733758455296", 0x43d15be0bf39dd24u}, + {"50033617337584552961", 0x4405b2d8ef08546cu}, + {"5003361733758455295.999999999999999999999999999999999999999999", 0x43d15be0bf39dd23u}, + }; + + for (const auto& c : known) + { + CAPTURE(c.first); + double out = 0; + if (eisel_lemire(c.first, out)) + { + CHECK(bits_of(out) == c.second); + } + else + { + // only tokens with more than 19 significant digits are left to + // strtod: those whose value lies too close to a tie + CHECK(significant_digits(c.first) > 19); + } + } + } + + SECTION("round trip") + { + // every double written by to_chars and read back, also with trailing + // digits that make the token longer than 19 digits + std::uint64_t state = 5295; + std::size_t declined = 0; + for (int i = 0; i < 200000; ++i) + { + state ^= state << 13u; + state ^= state >> 7u; + state ^= state << 17u; + std::uint64_t b = state; + if ((b & 0x7FF0000000000000u) == 0x7FF0000000000000u) + { + continue; // infinity or NaN + } + if (i % 4 == 0) + { + b &= 0x800FFFFFFFFFFFFFu; // subnormals + } + double d = 0; + std::memcpy(&d, &b, sizeof(d)); + + std::array buffer{}; + const char* end = nlohmann::detail::to_chars(buffer.data(), buffer.data() + buffer.size(), d); + const std::string token(buffer.data(), static_cast(end - buffer.data())); + CAPTURE(token); + double out = 0; + REQUIRE(eisel_lemire(token, out)); + CHECK(bits_of(out) == b); + + // insert digits before the exponent: the value moves by far less + // than the distance to the rounding boundary, so it must not change + std::string longer = token; + const std::size_t e = longer.find('e'); + const std::size_t dot = longer.find('.'); + const std::string extra = dot == std::string::npos ? ".000000000000000000001" : "000000000000000000001"; + longer.insert(e == std::string::npos ? longer.size() : e, extra); + CAPTURE(longer); + if (eisel_lemire(longer, out)) + { + CHECK(bits_of(out) == b); + } + else + { + // w and w + 1 round differently: only when the value is very + // close to a rounding boundary + ++declined; + } + } + CHECK(declined < 1000); // 107 of the 200,000 + } + + SECTION("used by the lexer") + { + // 17 significant digits: beyond Clinger's fast path + CHECK(bits_of(json::parse("-65.613616999999977").get()) == bits_of(-65.613616999999977)); + CHECK(bits_of(json::parse("2.2250738585072011e-308").get()) == 0x000FFFFFFFFFFFFFu); + CHECK(bits_of(json::parse("4.9406564584124654e-324").get()) == 1u); + json _; + CHECK_THROWS_WITH_AS(_ = json::parse("1.7976931348623159e308"), + "[json.exception.out_of_range.406] number overflow parsing '1.7976931348623159e308'", json::out_of_range&); + } +} From 65af260117cebe1106e963209e3fb0215a318e1d Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:06:34 +0200 Subject: [PATCH 07/31] Move instead of deep-copy ordered_json values when an object grows (#5609) * Move instead of deep-copy ordered_json values when an object grows ordered_map keeps its elements in a std::vector>. With a std::string key, that pair is not nothrow move constructible (the const key has to be copied), so std::vector copies every element when it reallocates. For ordered_json, this deep-copies every member value an object already holds, including whole nested subtrees, on each growth step. Grow the storage in ordered_map instead, copying the keys and moving the values. This happens in two phases, so the strong exception guarantee is kept without try/catch. The first phase may throw, but only touches a temporary buffer: it copies the keys, value-initializes the values, and constructs the new element. The second phase moves the values (noexcept) and swaps the buffers. Because the new element is constructed before any value is moved, arguments that refer to elements of the container stay valid, as with std::vector. Types that cannot take this path keep the std::vector behavior. Parsing into ordered_json (ParseStringOrdered, Apple M1 Max, clang -O3): twitter 3.20 -> 1.70 ms, citm_catalog 7.73 -> 3.67 ms, jeopardy 219 -> 177 ms, canada unchanged. The number of allocations for twitter and citm_catalog drops by two thirds. Also add ParseStringOrdered rows to the benchmarks, and document the growth behavior and the exception safety of ordered_map. Signed-off-by: Niels Lohmann * Fix CI: skip std::pair noexcept assumptions on EDG-based compilers ci_icpc and ci_nvhpc failed to compile unit-ordered_map.cpp: the static assertion that std::pair is not nothrow move-constructible fails there. The EDG front end (Intel icpc 2021.10, NVIDIA nvc++ 25.5) considers the defaulted move constructor of std::pair noexcept even if copying Key can throw. With these compilers, std::vector already moves such elements itself when it grows, and ordered_map correctly leaves growing to it. The same misjudgement makes std::vector call std::terminate when a key copy throws during growth, so the exception-safety test with throwing_key would abort on these compilers as well. Skip the static assertion and the exception-safety section when __EDG__ is defined. Verified with icpc 2021.10 (-std=gnu++11) and nvc++ 25.5 (C++11 and C++17) on Compiler Explorer: unit-ordered_map and unit-disabled_exceptions build and pass. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/ordered_map.md | 11 + include/nlohmann/ordered_map.hpp | 70 ++++- single_include/nlohmann/json.hpp | 70 ++++- tests/benchmarks/src/benchmarks.cpp | 33 +++ tests/src/unit-disabled_exceptions.cpp | 18 ++ tests/src/unit-ordered_map.cpp | 358 +++++++++++++++++++++++++ 6 files changed, 550 insertions(+), 10 deletions(-) diff --git a/docs/mkdocs/docs/api/ordered_map.md b/docs/mkdocs/docs/api/ordered_map.md index df21175d0..e464d0b1e 100644 --- a/docs/mkdocs/docs/api/ordered_map.md +++ b/docs/mkdocs/docs/api/ordered_map.md @@ -28,6 +28,11 @@ A minimal map-like container that preserves insertion order for use within [`nlo The type uses a `std::vector` to store object elements. Therefore, adding elements can yield a reallocation in which case all iterators (including the `end()` iterator) and all references to the elements are invalidated. +When the storage grows, the keys are copied and the mapped values are moved to the new storage. A plain `std::vector` +would copy the whole elements instead, because their `#!cpp const` keys make them not nothrow move constructible; for +[`ordered_json`](ordered_json.md), this would be a deep copy of every nested value. The values are only copied if +`T` is not default constructible or not nothrow move assignable. + ## Member types - **key_type** - key type (`Key`) @@ -56,6 +61,11 @@ std::equal_to<> // since C++14 - **find** - **insert** +## Exception safety + +**emplace**, **operator\[\]**, and **insert(value)** have the strong exception guarantee: if an exception is thrown (for +instance, because copying a key or allocating memory fails), the contents of the container are unchanged. + ## Complexity Because the elements are stored in a `std::vector` in insertion order, there is no index to look a key up by. Every @@ -122,3 +132,4 @@ This differs from `#!cpp std::map`, where the same operations are O(log n). - Added in version 3.9.0 to implement [`nlohmann::ordered_json`](ordered_json.md). - Added **key_compare** member in version 3.11.0. +- Changed in version 3.13.0: growing the storage moves the mapped values instead of copying them. diff --git a/include/nlohmann/ordered_map.hpp b/include/nlohmann/ordered_map.hpp index 7b8cf70f4..15e52ebb1 100644 --- a/include/nlohmann/ordered_map.hpp +++ b/include/nlohmann/ordered_map.hpp @@ -8,13 +8,15 @@ #pragma once +#include // max, min #include // equal_to, less #include // initializer_list #include // input_iterator_tag, iterator_traits #include // allocator #include // for out_of_range -#include // enable_if, is_convertible -#include // pair +#include // forward_as_tuple +#include // enable_if, integral_constant, is_convertible, is_nothrow_move_constructible +#include // forward, move, pair, piecewise_construct #include // vector #include @@ -79,7 +81,7 @@ template , return {it, false}; } } - Container::emplace_back(key, std::forward(t)); + append(key, std::forward(t)); return {std::prev(this->end()), true}; } @@ -94,7 +96,7 @@ template , return {it, false}; } } - Container::emplace_back(std::forward(key), std::forward(t)); + append(std::forward(key), std::forward(t)); return {std::prev(this->end()), true}; } @@ -368,7 +370,7 @@ template , return {it, false}; } } - Container::push_back(value); + append(value); return {--this->end(), true}; } @@ -386,6 +388,64 @@ template , } private: + /*! + @brief add an element whose key is not yet contained at the end + + A std::vector copies all elements when it grows, because their const keys + make them not nothrow move constructible. For ordered_json, this is a deep + copy of every value. Where the strong exception guarantee can be kept, grow + the storage here instead, copying only the keys and moving the values. + */ + template + void append(Args&& ... args) + { + // evaluated here rather than at class scope, because T is still + // incomplete when basic_json instantiates its object_t + using move_values = std::integral_constant>, + std::is_copy_constructible, + detail::is_default_constructible, + std::is_nothrow_move_assignable>::value>; + append_impl(move_values{}, std::forward(args)...); + } + + template + void append_impl(std::true_type /*unused*/, Args&& ... args) + { + if (this->size() < this->capacity()) + { + Container::emplace_back(std::forward(args)...); + return; + } + + // 1. May throw, but only changes tmp: copy the keys, value-initialize + // the values, and add the new element. The arguments may refer to + // elements of this container, so they are used before any value is + // moved out of it. + Container tmp(this->get_allocator()); // equal allocators, so swap() is valid + tmp.reserve((std::min)(this->max_size(), (std::max)(size_type{1}, 2 * this->size()))); + for (const auto& element : *this) + { + tmp.emplace_back(std::piecewise_construct, std::forward_as_tuple(element.first), std::forward_as_tuple()); + } + tmp.emplace_back(std::forward(args)...); + + // 2. Cannot throw: move the values over and adopt the new storage. + auto it = tmp.begin(); + for (auto& element : *this) + { + it->second = std::move(element.second); + ++it; + } + Container::swap(tmp); + } + + template + void append_impl(std::false_type /*unused*/, Args&& ... args) + { + Container::emplace_back(std::forward(args)...); + } + JSON_NO_UNIQUE_ADDRESS key_compare m_compare = key_compare(); }; diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 1232e184c..a3ba24cf6 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -26495,13 +26495,15 @@ NLOHMANN_JSON_NAMESPACE_END +#include // max, min #include // equal_to, less #include // initializer_list #include // input_iterator_tag, iterator_traits #include // allocator #include // for out_of_range -#include // enable_if, is_convertible -#include // pair +#include // forward_as_tuple +#include // enable_if, integral_constant, is_convertible, is_nothrow_move_constructible +#include // forward, move, pair, piecewise_construct #include // vector // #include @@ -26568,7 +26570,7 @@ template , return {it, false}; } } - Container::emplace_back(key, std::forward(t)); + append(key, std::forward(t)); return {std::prev(this->end()), true}; } @@ -26583,7 +26585,7 @@ template , return {it, false}; } } - Container::emplace_back(std::forward(key), std::forward(t)); + append(std::forward(key), std::forward(t)); return {std::prev(this->end()), true}; } @@ -26857,7 +26859,7 @@ template , return {it, false}; } } - Container::push_back(value); + append(value); return {--this->end(), true}; } @@ -26875,6 +26877,64 @@ template , } private: + /*! + @brief add an element whose key is not yet contained at the end + + A std::vector copies all elements when it grows, because their const keys + make them not nothrow move constructible. For ordered_json, this is a deep + copy of every value. Where the strong exception guarantee can be kept, grow + the storage here instead, copying only the keys and moving the values. + */ + template + void append(Args&& ... args) + { + // evaluated here rather than at class scope, because T is still + // incomplete when basic_json instantiates its object_t + using move_values = std::integral_constant>, + std::is_copy_constructible, + detail::is_default_constructible, + std::is_nothrow_move_assignable>::value>; + append_impl(move_values{}, std::forward(args)...); + } + + template + void append_impl(std::true_type /*unused*/, Args&& ... args) + { + if (this->size() < this->capacity()) + { + Container::emplace_back(std::forward(args)...); + return; + } + + // 1. May throw, but only changes tmp: copy the keys, value-initialize + // the values, and add the new element. The arguments may refer to + // elements of this container, so they are used before any value is + // moved out of it. + Container tmp(this->get_allocator()); // equal allocators, so swap() is valid + tmp.reserve((std::min)(this->max_size(), (std::max)(size_type{1}, 2 * this->size()))); + for (const auto& element : *this) + { + tmp.emplace_back(std::piecewise_construct, std::forward_as_tuple(element.first), std::forward_as_tuple()); + } + tmp.emplace_back(std::forward(args)...); + + // 2. Cannot throw: move the values over and adopt the new storage. + auto it = tmp.begin(); + for (auto& element : *this) + { + it->second = std::move(element.second); + ++it; + } + Container::swap(tmp); + } + + template + void append_impl(std::false_type /*unused*/, Args&& ... args) + { + Container::emplace_back(std::forward(args)...); + } + JSON_NO_UNIQUE_ADDRESS key_compare m_compare = key_compare(); }; diff --git a/tests/benchmarks/src/benchmarks.cpp b/tests/benchmarks/src/benchmarks.cpp index 2ad28a57a..5df7dd473 100644 --- a/tests/benchmarks/src/benchmarks.cpp +++ b/tests/benchmarks/src/benchmarks.cpp @@ -119,6 +119,39 @@ BENCHMARK_CAPTURE(ParseIndented, canada / 4, TEST_DATA_DIRECTORY "/nativej BENCHMARK_CAPTURE(ParseIndented, citm_catalog / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", 4); BENCHMARK_CAPTURE(ParseIndented, twitter / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", 4); +////////////////////////////////////////////////////////////////////////////// +// parse JSON from string into an ordered_json +// +// Same as ParseString above, but with nlohmann::ordered_json, whose objects +// keep their members in a vector: the pair of rows shows what preserving the +// insertion order costs. +////////////////////////////////////////////////////////////////////////////// + +static void ParseStringOrdered(benchmark::State& state, const char* filename) +{ + std::ifstream f(filename); + std::string str((std::istreambuf_iterator(f)), std::istreambuf_iterator()); + + while (state.KeepRunning()) + { + state.PauseTiming(); + auto* j = new nlohmann::ordered_json(); + state.ResumeTiming(); + + *j = nlohmann::ordered_json::parse(str); + + state.PauseTiming(); + delete j; + state.ResumeTiming(); + } + + state.SetBytesProcessed(state.iterations() * str.size()); +} +BENCHMARK_CAPTURE(ParseStringOrdered, jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json"); +BENCHMARK_CAPTURE(ParseStringOrdered, canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json"); +BENCHMARK_CAPTURE(ParseStringOrdered, citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json"); +BENCHMARK_CAPTURE(ParseStringOrdered, twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json"); + ////////////////////////////////////////////////////////////////////////////// // serialize JSON ////////////////////////////////////////////////////////////////////////////// diff --git a/tests/src/unit-disabled_exceptions.cpp b/tests/src/unit-disabled_exceptions.cpp index e4532e234..0b8de64d3 100644 --- a/tests/src/unit-disabled_exceptions.cpp +++ b/tests/src/unit-disabled_exceptions.cpp @@ -46,6 +46,24 @@ TEST_CASE("Tests with disabled exceptions") CHECK(*sax_no_exception::error_string == "[json.exception.parse_error.101] parse error at line 1, column 1: syntax error while parsing value - invalid literal; last read: 'x'"); delete sax_no_exception::error_string; // NOLINT(cppcoreguidelines-owning-memory) } + + SECTION("growing an ordered_json object") + { + auto j = nlohmann::ordered_json::object(); + for (int i = 0; i < 100; ++i) + { + j[std::to_string(i)] = {{"nested", i}}; + } + + CHECK(j.size() == 100); + int i = 0; + for (const auto& element : j.items()) + { + CHECK(element.key() == std::to_string(i)); + CHECK(element.value()["nested"] == i); + ++i; + } + } } DOCTEST_GCC_SUPPRESS_WARNING_POP diff --git a/tests/src/unit-ordered_map.cpp b/tests/src/unit-ordered_map.cpp index f380a9869..98b6fa0d1 100644 --- a/tests/src/unit-ordered_map.cpp +++ b/tests/src/unit-ordered_map.cpp @@ -11,6 +11,97 @@ #include using nlohmann::ordered_map; +#include +#include +#include +#include +#include + +// The EDG front end (Intel icpc, NVIDIA nvc++) considers the defaulted move +// constructor of std::pair noexcept even if copying Key can +// throw. std::vector then moves such elements itself when it grows (and calls +// std::terminate if a key copy throws), so ordered_map leaves growing to it. +#if defined(__EDG__) + #define JSON_TEST_PAIR_MOVE_IS_NOEXCEPT +#endif + +namespace +{ +// number of copies made of counted values +int value_copies = 0; + +// a mapped type that counts its copies; moving from it leaves -1 behind +struct counted // NOLINT(cppcoreguidelines-special-member-functions,hicpp-special-member-functions) +{ + int payload = 0; + + counted() = default; + explicit counted(int p) noexcept : payload(p) {} + counted(const counted& other) : payload(other.payload) + { + ++value_copies; + } + counted(counted&& other) noexcept : payload(other.payload) + { + other.payload = -1; + } + counted& operator=(const counted&) = delete; + counted& operator=(counted&& other) noexcept + { + payload = other.payload; + other.payload = -1; + return *this; + } +}; + +#if !defined(JSON_NOEXCEPTION) && !defined(JSON_TEST_PAIR_MOVE_IS_NOEXCEPT) +// number of throwing_key copies that still succeed; the next one throws +// (a negative value means that copies never throw) +int key_copies_until_throw = -1; + +// a key type whose copy constructor can be made to throw +struct throwing_key // NOLINT(cppcoreguidelines-special-member-functions,hicpp-special-member-functions) +{ + int id = 0; + + explicit throwing_key(int i) noexcept : id(i) {} + throwing_key(const throwing_key& other) : id(other.id) + { + if (key_copies_until_throw == 0) + { + throw std::runtime_error("key copy failed"); + } + if (key_copies_until_throw > 0) + { + --key_copies_until_throw; + } + } + throwing_key& operator=(const throwing_key&) = delete; + + friend bool operator==(const throwing_key& lhs, const throwing_key& rhs) noexcept + { + return lhs.id == rhs.id; + } +}; +#endif + +// a mapped type that cannot be default-constructed +struct no_default +{ + explicit no_default(int v) noexcept : value(v) {} + int value; +}; + +// ordered_json must keep moving its values when an object grows +using ordered_object_t = nlohmann::ordered_json::object_t; +#if !defined(JSON_TEST_PAIR_MOVE_IS_NOEXCEPT) + static_assert(!std::is_nothrow_move_constructible::value, "std::vector would move the elements itself"); +#endif +static_assert(std::is_copy_constructible::value, "keys must be copyable"); +static_assert(std::is_default_constructible::value, "values must be default-constructible"); +static_assert(std::is_nothrow_move_assignable::value, "values must be nothrow move-assignable"); +} // namespace + TEST_CASE("ordered_map") { SECTION("constructor") @@ -313,3 +404,270 @@ TEST_CASE("ordered_map") } } } + +TEST_CASE("ordered_map growth") +{ + SECTION("values are moved, not copied, when the storage grows") + { + ordered_map om; + std::size_t growths = 0; + value_copies = 0; + + // inserts 100 elements with the given function and counts the growths + const auto fill = [&om, &growths](void (*insert)(ordered_map&, int)) + { + for (int i = 0; i < 100; ++i) + { + const auto old_capacity = om.capacity(); + insert(om, i); + if (om.capacity() > old_capacity) + { + ++growths; + } + } + }; + + // checks that the elements are in insertion order with their values + const auto check_contents = [&om] + { + CHECK(om.size() == 100); + int i = 0; + for (const auto& element : om) + { + CHECK(element.first == std::to_string(i)); + CHECK(element.second.payload == i); + ++i; + } + }; + + SECTION("emplace") + { + fill([](ordered_map& m, int i) + { + m.emplace(std::to_string(i), counted(i)); + }); + CHECK(growths >= 3); + CHECK(value_copies == 0); + check_contents(); + } + + SECTION("operator[]") + { + fill([](ordered_map& m, int i) + { + m[std::to_string(i)] = counted(i); + }); + CHECK(growths >= 3); + CHECK(value_copies == 0); + check_contents(); + } + + SECTION("insert(value_type&&)") + { + fill([](ordered_map& m, int i) + { + m.insert({std::to_string(i), counted(i)}); + }); + CHECK(growths >= 3); + CHECK(value_copies == 0); + check_contents(); + } + + SECTION("insert(const value_type&)") + { + fill([](ordered_map& m, int i) + { + const std::pair value(std::to_string(i), counted(i)); + m.insert(value); + }); + CHECK(growths >= 3); + // only the inserted values are copied + CHECK(value_copies == 100); + check_contents(); + } + + SECTION("insert(first, last)") + { + std::vector> values; + values.reserve(100); + for (int i = 0; i < 100; ++i) + { + values.emplace_back(std::to_string(i), counted(i)); + } + value_copies = 0; + + om.insert(values.cbegin(), values.cend()); + // only the inserted values are copied + CHECK(value_copies == 100); + check_contents(); + } + } + + SECTION("elements keep their order and values over many growths") + { + ordered_map om; + for (int i = 0; i < 1000; ++i) + { + om.emplace(std::to_string(i), counted(i)); + } + + CHECK(om.size() == 1000); + int i = 0; + for (const auto& element : om) + { + CHECK(element.first == std::to_string(i)); + CHECK(element.second.payload == i); + ++i; + } + } + + SECTION("arguments may refer to elements of the full container") + { + SECTION("moving a value out of the container") + { + ordered_map om; + om.reserve(4); + while (om.size() < om.capacity()) + { + const auto i = static_cast(om.size()); + om.emplace(std::to_string(i), counted(i)); + } + const auto size = om.size(); + + om.emplace("new", std::move(om.at("0"))); + CHECK(om.size() == size + 1); + CHECK(om.at("new").payload == 0); + CHECK(om.at("0").payload == -1); + } + + SECTION("using a value as key") + { + ordered_map om; + om.reserve(4); + while (om.size() < om.capacity()) + { + const auto i = std::to_string(om.size()); + om.emplace("k" + i, "v" + i); + } + const auto size = om.size(); + + om.emplace(om.at("k0"), std::string("x")); + CHECK(om.size() == size + 1); + CHECK(om.at("k0") == "v0"); + CHECK(om.at("v0") == "x"); + } + + SECTION("ordered_json") + { + auto j = nlohmann::ordered_json::object(); + auto& object = j.get_ref(); + object.reserve(4); + while (object.size() < object.capacity()) + { + const auto i = std::to_string(object.size()); + j[i] = "a value that is too long for the small string optimization " + i; + } + const auto size = j.size(); + + j.emplace("new", std::move(j["0"])); + CHECK(j.size() == size + 1); + CHECK(j["new"] == "a value that is too long for the small string optimization 0"); + CHECK(j["0"].is_null()); + } + } + +#if !defined(JSON_NOEXCEPTION) && !defined(JSON_TEST_PAIR_MOVE_IS_NOEXCEPT) + SECTION("the container is unchanged if growing it throws") + { + ordered_map om; + om.reserve(4); + while (om.size() < om.capacity()) + { + const auto i = static_cast(om.size()); + om.emplace(throwing_key(i), counted(i)); + } + const auto size = om.size(); + const auto capacity = om.capacity(); + + // checks that the elements are unchanged + const auto check_unchanged = [&om, size, capacity] + { + CHECK(om.size() == size); + CHECK(om.capacity() == capacity); + int i = 0; + for (const auto& element : om) + { + CHECK(element.first.id == i); + CHECK(element.second.payload == i); + ++i; + } + }; + + SECTION("emplace") + { + // growing copies the existing keys and then the new one; let each of these copies throw + for (std::size_t k = 0; k <= size; ++k) + { + counted value(100); + key_copies_until_throw = static_cast(k); + CHECK_THROWS_AS(om.emplace(throwing_key(100), std::move(value)), std::runtime_error); + key_copies_until_throw = -1; + + check_unchanged(); + CHECK(value.payload == 100); // NOLINT(bugprone-use-after-move,hicpp-invalid-access-moved) + } + + om.emplace(throwing_key(100), counted(100)); + CHECK(om.size() == size + 1); + CHECK(om.capacity() > capacity); + CHECK(om.at(throwing_key(100)).payload == 100); + } + + SECTION("insert(const value_type&)") + { + const std::pair value(throwing_key(100), counted(100)); + value_copies = 0; + + key_copies_until_throw = static_cast(size / 2); + CHECK_THROWS_AS(om.insert(value), std::runtime_error); + key_copies_until_throw = -1; + + check_unchanged(); + CHECK(value_copies == 0); + } + } +#endif + + SECTION("elements that std::vector moves, or that cannot be moved back") + { + SECTION("nothrow move-constructible elements") + { + ordered_map om; + value_copies = 0; + for (int i = 0; i < 100; ++i) + { + om.emplace(i, counted(i)); + } + CHECK(om.size() == 100); + CHECK(value_copies == 0); + } + + SECTION("mapped type without default constructor") + { + ordered_map om; + for (int i = 0; i < 100; ++i) + { + om.emplace(std::to_string(i), no_default(i)); + } + + CHECK(om.size() == 100); + int i = 0; + for (const auto& element : om) + { + CHECK(element.first == std::to_string(i)); + CHECK(element.second.value == i); + ++i; + } + } + } +} From 6a8a7735ed0a158de407d058cafc342eca393f5c Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:06:39 +0200 Subject: [PATCH 08/31] Do not throw in contains() for an empty array reference token (#5614) * Do not throw in contains() for an empty array reference token json_pointer::contains() rejected malformed array indices, but an empty reference token (e.g. "/a/" where "a" is an array, or "/" on an array) passed every check and reached array_index(), which throws out_of_range.404. contains() must not throw (cf. #5395), so it now returns false for an empty token. at() still throws out_of_range.404. Signed-off-by: Niels Lohmann * Fix CI: bind j_nested_const by reference in the json_pointer test clang-tidy (ci_clang_tidy) flagged the new test with performance-unnecessary-copy-initialization: the local copy j_nested_const of j_nested is never modified. Bind it as a const reference instead; it still exercises the const overloads of at() and contains(). Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/detail/json_pointer.hpp | 6 ++++++ single_include/nlohmann/json.hpp | 6 ++++++ tests/src/unit-json_pointer.cpp | 20 ++++++++++++++++++++ 3 files changed, 32 insertions(+) diff --git a/include/nlohmann/detail/json_pointer.hpp b/include/nlohmann/detail/json_pointer.hpp index 1540a8d6f..0b9f9651a 100644 --- a/include/nlohmann/detail/json_pointer.hpp +++ b/include/nlohmann/detail/json_pointer.hpp @@ -746,6 +746,12 @@ class json_pointer // "-" always fails the range check return false; } + if (JSON_HEDLEY_UNLIKELY(reference_token.empty())) + { + // an empty reference token is not an array index; array_index() + // would throw out_of_range.404 -- contains() must not throw (see #5395) + return false; + } if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !("0" <= reference_token && reference_token <= "9"))) { // invalid char diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index a3ba24cf6..6aff2a985 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -20315,6 +20315,12 @@ class json_pointer // "-" always fails the range check return false; } + if (JSON_HEDLEY_UNLIKELY(reference_token.empty())) + { + // an empty reference token is not an array index; array_index() + // would throw out_of_range.404 -- contains() must not throw (see #5395) + return false; + } if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !("0" <= reference_token && reference_token <= "9"))) { // invalid char diff --git a/tests/src/unit-json_pointer.cpp b/tests/src/unit-json_pointer.cpp index e7d6df530..76ae1f79b 100644 --- a/tests/src/unit-json_pointer.cpp +++ b/tests/src/unit-json_pointer.cpp @@ -380,6 +380,26 @@ TEST_CASE("JSON pointers") DOCTEST_MSVC_SUPPRESS_WARNING_POP + { + // contains() must not throw for an empty reference token if the current + // value is an array (cf. #5395) -- at() still reports out_of_range.404 + json j_nested = {{"a", {1, 2}}}; + const json& j_nested_const = j_nested; + json::json_pointer const jp("/a/"); + std::string const throw_msg = "[json.exception.out_of_range.404] unresolved reference token ''"; + + CHECK_THROWS_WITH_AS(j_nested.at(jp), throw_msg.c_str(), json::out_of_range&); + CHECK_THROWS_WITH_AS(j_nested_const.at(jp), throw_msg.c_str(), json::out_of_range&); + + CHECK(j_nested.contains(json::json_pointer("/a/1"))); + CHECK(!j_nested.contains(jp)); + CHECK(!j_nested_const.contains(jp)); + + // same for an empty reference token on a top-level array + CHECK(!j.contains(json::json_pointer("/"))); + CHECK(!j_const.contains(json::json_pointer("/"))); + } + CHECK_THROWS_WITH_AS(j.at("/one"_json_pointer) = 1, "[json.exception.parse_error.109] parse error: array index 'one' is not a number", json::parse_error&); CHECK_THROWS_WITH_AS(j_const.at("/one"_json_pointer) == 1, From 0050d0f7a4f856d9104cbf8b708e462f5bb2ce8e Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:06:43 +0200 Subject: [PATCH 09/31] Share one nesting depth limit between all bounded descents (#5637) Copying and comparing stopped their descent at basic_json::nesting_depth_limit(), while serializing, hashing and merging used detail::recursion_depth_limit(). Both were 128, but nothing kept them equal. The thread-local count now tests against detail::recursion_depth_limit() as well, and a static_assert keeps the limit small enough for the byte that holds the count. Signed-off-by: Niels Lohmann --- .../nlohmann/detail/recursion_depth_limit.hpp | 8 +++++- include/nlohmann/json.hpp | 18 +++++-------- single_include/nlohmann/json.hpp | 26 ++++++++++--------- 3 files changed, 28 insertions(+), 24 deletions(-) diff --git a/include/nlohmann/detail/recursion_depth_limit.hpp b/include/nlohmann/detail/recursion_depth_limit.hpp index fe3bd8026..6553a4d05 100644 --- a/include/nlohmann/detail/recursion_depth_limit.hpp +++ b/include/nlohmann/detail/recursion_depth_limit.hpp @@ -19,11 +19,17 @@ namespace detail /*! @brief the number of nesting levels an operation recurses into -Operations that walk a value (serializing, hashing, merging, ...) recurse once +Operations that walk a value (copying, comparing, serializing, hashing, merging, +...) recurse once per nesting level, which is fastest, but a value nested deeply enough would exhaust the call stack. So they recurse only this many levels deep and finish whatever lies below with an explicit stack. All of them share this limit. +Most of them pass the depth down as an argument. The copy constructor and the +comparison operators cannot, as their signatures are fixed, so they count it +in basic_json::nesting_depth() instead, a byte per thread; the limit must +therefore stay below 255. + @sa https://github.com/nlohmann/json/issues/5387 */ constexpr std::size_t recursion_depth_limit() noexcept diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 500fcddf2..a9cfcb520 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -898,12 +898,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } #ifndef JSON_NO_THREAD_LOCAL - /// the number of levels an operation descends into before it finishes the - /// value below it without the call stack - static constexpr std::uint8_t nesting_depth_limit() - { - return 128; - } + // nesting_depth() is a byte and may exceed the limit by one level + static_assert(detail::recursion_depth_limit() < 255, "the nesting depth count must fit in a byte"); /*! @brief how many levels the operation going on in this thread has descended into @@ -945,7 +941,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static_cast(may_descend); return true; #else - return !may_descend || nesting_depth() >= nesting_depth_limit(); + return !may_descend || nesting_depth() >= detail::recursion_depth_limit(); #endif } @@ -969,7 +965,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec #ifdef JSON_NO_THREAD_LOCAL : m_okay(false) #else - : m_okay(nesting_depth() < nesting_depth_limit()) + : m_okay(nesting_depth() < detail::recursion_depth_limit()) #endif { #ifndef JSON_NO_THREAD_LOCAL @@ -1173,7 +1169,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec The values whose copy has not been created yet are kept on an explicit worklist rather than on the call stack. This is only reached for values - nested deeper than @ref nesting_depth_limit levels, which is why it copies + nested deeper than @ref detail::recursion_depth_limit levels, which is why it copies every container by hand instead of letting the container do it: the fast ways of doing so would descend into the elements and defeat the purpose. */ @@ -1239,7 +1235,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec Copying a container copies its elements, so a value nested deeply enough used to exhaust the call stack. The descent is bounded here: the first - @ref nesting_depth_limit levels are copied by the containers themselves, just + @ref detail::recursion_depth_limit levels are copied by the containers themselves, just as they always were, and anything below that is copied without the call stack by @ref copy_iteratively. Copying a value can therefore no longer exhaust the stack, however deeply it is nested, just like destroying one @@ -1377,7 +1373,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /*! @brief compare @a lhs and @a rhs without descending into them - Reached once a comparison has descended @ref nesting_depth_limit levels, so + Reached once a comparison has descended @ref detail::recursion_depth_limit levels, so that comparing values cannot exhaust the call stack however deeply they are nested. The two values are walked in lockstep on an explicit stack and compared lexicographically, element by element in the order the containers diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 6aff2a985..73c8ab9be 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7281,11 +7281,17 @@ namespace detail /*! @brief the number of nesting levels an operation recurses into -Operations that walk a value (serializing, hashing, merging, ...) recurse once +Operations that walk a value (copying, comparing, serializing, hashing, merging, +...) recurse once per nesting level, which is fastest, but a value nested deeply enough would exhaust the call stack. So they recurse only this many levels deep and finish whatever lies below with an explicit stack. All of them share this limit. +Most of them pass the depth down as an argument. The copy constructor and the +comparison operators cannot, as their signatures are fixed, so they count it +in basic_json::nesting_depth() instead, a byte per thread; the limit must +therefore stay below 255. + @sa https://github.com/nlohmann/json/issues/5387 */ constexpr std::size_t recursion_depth_limit() noexcept @@ -27772,12 +27778,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } #ifndef JSON_NO_THREAD_LOCAL - /// the number of levels an operation descends into before it finishes the - /// value below it without the call stack - static constexpr std::uint8_t nesting_depth_limit() - { - return 128; - } + // nesting_depth() is a byte and may exceed the limit by one level + static_assert(detail::recursion_depth_limit() < 255, "the nesting depth count must fit in a byte"); /*! @brief how many levels the operation going on in this thread has descended into @@ -27819,7 +27821,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static_cast(may_descend); return true; #else - return !may_descend || nesting_depth() >= nesting_depth_limit(); + return !may_descend || nesting_depth() >= detail::recursion_depth_limit(); #endif } @@ -27843,7 +27845,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec #ifdef JSON_NO_THREAD_LOCAL : m_okay(false) #else - : m_okay(nesting_depth() < nesting_depth_limit()) + : m_okay(nesting_depth() < detail::recursion_depth_limit()) #endif { #ifndef JSON_NO_THREAD_LOCAL @@ -28047,7 +28049,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec The values whose copy has not been created yet are kept on an explicit worklist rather than on the call stack. This is only reached for values - nested deeper than @ref nesting_depth_limit levels, which is why it copies + nested deeper than @ref detail::recursion_depth_limit levels, which is why it copies every container by hand instead of letting the container do it: the fast ways of doing so would descend into the elements and defeat the purpose. */ @@ -28113,7 +28115,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec Copying a container copies its elements, so a value nested deeply enough used to exhaust the call stack. The descent is bounded here: the first - @ref nesting_depth_limit levels are copied by the containers themselves, just + @ref detail::recursion_depth_limit levels are copied by the containers themselves, just as they always were, and anything below that is copied without the call stack by @ref copy_iteratively. Copying a value can therefore no longer exhaust the stack, however deeply it is nested, just like destroying one @@ -28251,7 +28253,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /*! @brief compare @a lhs and @a rhs without descending into them - Reached once a comparison has descended @ref nesting_depth_limit levels, so + Reached once a comparison has descended @ref detail::recursion_depth_limit levels, so that comparing values cannot exhaust the call stack however deeply they are nested. The two values are walked in lockstep on an explicit stack and compared lexicographically, element by element in the order the containers From 68beba727c7e874af945484c7748294f23c5b82f Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:06:46 +0200 Subject: [PATCH 10/31] Fix from_json() for enums with underlying type bool (#5679) get_arithmetic_value() rejects boolean_t, so the default from_json() for enums failed to compile for an enum whose underlying type is bool (e.g. enum class Flag : bool { off, on }), even though the matching to_json() serializes such enums as an unsigned number. Read the underlying value through number_unsigned_t in that case, matching what to_json() writes, then cast back to the underlying type before constructing the enum. Fixes #5671. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/conversions/from_json.hpp | 8 ++++++-- single_include/nlohmann/json.hpp | 8 ++++++-- tests/src/unit-conversions.cpp | 8 ++++++++ 3 files changed, 20 insertions(+), 4 deletions(-) diff --git a/include/nlohmann/detail/conversions/from_json.hpp b/include/nlohmann/detail/conversions/from_json.hpp index 11e40f5f4..6ecf7e657 100644 --- a/include/nlohmann/detail/conversions/from_json.hpp +++ b/include/nlohmann/detail/conversions/from_json.hpp @@ -167,9 +167,13 @@ template::value, int> = 0> inline void from_json(const BasicJsonType& j, EnumType& e) { - typename std::underlying_type::type val; + using underlying_type = typename std::underlying_type::type; + // get_arithmetic_value() does not accept boolean_t; read the number that to_json() wrote instead + using value_type = typename std::conditional::value, + typename BasicJsonType::number_unsigned_t, underlying_type>::type; + value_type val; get_arithmetic_value(j, val); - e = static_cast(val); + e = static_cast(static_cast(val)); } #endif // JSON_DISABLE_ENUM_SERIALIZATION diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 73c8ab9be..9a9d6ae0c 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -5664,9 +5664,13 @@ template::value, int> = 0> inline void from_json(const BasicJsonType& j, EnumType& e) { - typename std::underlying_type::type val; + using underlying_type = typename std::underlying_type::type; + // get_arithmetic_value() does not accept boolean_t; read the number that to_json() wrote instead + using value_type = typename std::conditional::value, + typename BasicJsonType::number_unsigned_t, underlying_type>::type; + value_type val; get_arithmetic_value(j, val); - e = static_cast(val); + e = static_cast(static_cast(val)); } #endif // JSON_DISABLE_ENUM_SERIALIZATION diff --git a/tests/src/unit-conversions.cpp b/tests/src/unit-conversions.cpp index 90d972f71..077ba0e14 100644 --- a/tests/src/unit-conversions.cpp +++ b/tests/src/unit-conversions.cpp @@ -1358,6 +1358,14 @@ TEST_CASE("value conversion") CHECK(json(value_1).get() == value_1); CHECK(json(cpp_enum::value_1).get() == cpp_enum::value_1); } + + SECTION("get an enum with underlying type bool (#5671)") + { + enum class bool_enum : bool { off, on }; + + CHECK(json(bool_enum::off).get() == bool_enum::off); + CHECK(json(bool_enum::on).get() == bool_enum::on); + } #endif SECTION("more involved conversions") From fc4c9c3446e2151448b5cc2fe8b55f23869ba571 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:06:51 +0200 Subject: [PATCH 11/31] Fix clear() to also reset the subtype of a binary value (#5680) * Fix clear() to also reset the subtype of a binary value clear() on a binary value cleared the bytes but left the subtype untouched, so the result was not equal to a default-constructed binary value even though the documentation says clear() has the same effect as *this = basic_json(type()). The fix calls byte_container_with_subtype::clear_subtype() alongside the existing clear() call. Extended the "filled binary" clear() test in unit-modifiers.cpp with a case that uses a subtype, since the existing cases only covered binary values without one. Fixes #5669. Signed-off-by: Niels Lohmann * Fix the table alignment in clear.md Addresses review comment by @gregmarr. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/clear.md | 19 ++++++++++--------- include/nlohmann/json.hpp | 1 + single_include/nlohmann/json.hpp | 1 + tests/src/unit-modifiers.cpp | 12 ++++++++++++ 4 files changed, 24 insertions(+), 9 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/clear.md b/docs/mkdocs/docs/api/basic_json/clear.md index 427a7fc39..a242b678a 100644 --- a/docs/mkdocs/docs/api/basic_json/clear.md +++ b/docs/mkdocs/docs/api/basic_json/clear.md @@ -7,15 +7,15 @@ void clear() noexcept; Clears the content of a JSON value and resets it to the default value as if [`basic_json(value_t)`](basic_json.md) would have been called with the current value type from [`type()`](type.md): -| Value type | initial value | -|------------|----------------------| -| null | `null` | -| boolean | `false` | -| string | `""` | -| number | `0` | -| binary | An empty byte vector | -| object | `{}` | -| array | `[]` | +| Value type | initial value | +|------------|-----------------------------------------| +| null | `null` | +| boolean | `false` | +| string | `""` | +| number | `0` | +| binary | An empty byte vector with no subtype | +| object | `{}` | +| array | `[]` | Has the same effect as calling @@ -56,3 +56,4 @@ All iterators, pointers, and references related to this container are invalidate - Added in version 1.0.0. - Added support for binary types in version 3.8.0. +- Fixed in version 3.13.0 to also clear the subtype of a binary value; before, the subtype was left unchanged. diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index a9cfcb520..28d91600f 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -3829,6 +3829,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::binary: { m_data.m_value.binary->clear(); + m_data.m_value.binary->clear_subtype(); break; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 9a9d6ae0c..a59a4013c 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -30713,6 +30713,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::binary: { m_data.m_value.binary->clear(); + m_data.m_value.binary->clear_subtype(); break; } diff --git a/tests/src/unit-modifiers.cpp b/tests/src/unit-modifiers.cpp index c878ec15c..1ec6016cb 100644 --- a/tests/src/unit-modifiers.cpp +++ b/tests/src/unit-modifiers.cpp @@ -155,6 +155,18 @@ TEST_CASE("modifiers") CHECK(j == json(json::value_t::binary)); CHECK(j == json(k.type())); } + + SECTION("filled binary with subtype") + { + json j = json::binary({1, 2, 3, 4, 5}, 42); + json const k = j; + + j.clear(); + CHECK(!j.empty()); + CHECK(!j.get_binary().has_subtype()); + CHECK(j == json(json::value_t::binary)); + CHECK(j == json(k.type())); + } } SECTION("number (integer)") From 678fd3017b463873e6adf85e3ab6df89af7fe90d Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:06:55 +0200 Subject: [PATCH 12/31] Fix element path for map/unordered_map JSON_DIAGNOSTICS errors (#5681) When converting a JSON array to std::map or std::unordered_map with a non-string key, each element must itself be a [key, value] array. If an element is not an array, from_json() threw type_error 302 with the outer array's value (&j) as the exception context, so with JSON_DIAGNOSTICS enabled the message pointed at the whole array instead of the offending element (e.g. "(/outer/m)" instead of "(/outer/m/2)"), even though the message text already described the element's type. Both from_json() overloads now pass the element (&p) as the context, so the reported JSON Pointer matches the type named in the message, the same way std::vector> and similar conversions already do. Fixes #5668. Signed-off-by: Niels Lohmann --- .../nlohmann/detail/conversions/from_json.hpp | 4 +-- single_include/nlohmann/json.hpp | 4 +-- tests/src/unit-diagnostics.cpp | 25 +++++++++++++++++++ 3 files changed, 29 insertions(+), 4 deletions(-) diff --git a/include/nlohmann/detail/conversions/from_json.hpp b/include/nlohmann/detail/conversions/from_json.hpp index 6ecf7e657..c68f51f69 100644 --- a/include/nlohmann/detail/conversions/from_json.hpp +++ b/include/nlohmann/detail/conversions/from_json.hpp @@ -568,7 +568,7 @@ inline void from_json(const BasicJsonType& j, std::map(), p.at(1).template get()); } @@ -588,7 +588,7 @@ inline void from_json(const BasicJsonType& j, std::unordered_map(), p.at(1).template get()); } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index a59a4013c..f2f8a7ab4 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -6065,7 +6065,7 @@ inline void from_json(const BasicJsonType& j, std::map(), p.at(1).template get()); } @@ -6085,7 +6085,7 @@ inline void from_json(const BasicJsonType& j, std::unordered_map(), p.at(1).template get()); } diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index 3ae649e5b..a46ce5746 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -17,6 +17,9 @@ #include using nlohmann::json; +#include +#include + TEST_CASE("Better diagnostics") { SECTION("empty JSON Pointer") @@ -331,6 +334,28 @@ TEST_CASE("Regression tests for extended diagnostics") } } + SECTION("Regression test for issue #5668 - wrong path for std::map/unordered_map with non-string keys") + { + // a map with non-string keys is read from an array of [key, value] arrays; + // element 2 of "m" is not an array, so the path must point at "m/2", not "m" + json j; + j["outer"]["m"] = json::array({json::array({1, 2}), json::array({3, 4}), 5}); + + SECTION("std::map") + { + CHECK_THROWS_WITH_AS((j["outer"]["m"].get>()), + "[json.exception.type_error.302] (/outer/m/2) type must be array, " + "but is number", json::type_error); + } + + SECTION("std::unordered_map") + { + CHECK_THROWS_WITH_AS((j["outer"]["m"].get>()), + "[json.exception.type_error.302] (/outer/m/2) type must be array, " + "but is number", json::type_error); + } + } + SECTION("Regression test - swap(array_t&)/swap(object_t&) must update JSON_DIAGNOSTICS parent pointers") { // swap(array_t&) From 6d7845d207483adf208ab7fff46aaf29ca90ef39 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:06:59 +0200 Subject: [PATCH 13/31] Fix deprecated json_pointer/string operator== warning in value() (#5683) value(KeyType&&, default) is constrained on is_comparable_with_object_key, which passes KeyType as a reference. is_comparable's dispatch on is_json_pointer_of only matches a json_pointer as a plain type or a plain reference, so a const-qualified reference (as produced when KeyType is deduced from a json_pointer argument) fell through to is_comparable_no_json_pointer, which instantiates the deprecated json_pointer/string comparison operators. This made ordered_json's transparent comparator (and any transparent comparator on a custom string type) warn under -Wdeprecated-declarations when calling value(json_pointer, default), even though no such comparison is ever performed. at() was already fixed for this in #5289, which does not use is_comparable_with_object_key. Strip references and cv-qualifiers with uncvref_t before the is_json_pointer_of dispatch, so any reference-to-json_pointer is recognized regardless of qualifiers. Fixes #5664. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/meta/type_traits.hpp | 2 +- single_include/nlohmann/json.hpp | 2 +- tests/src/unit-json_pointer.cpp | 19 +++++++++++++++++++ 3 files changed, 21 insertions(+), 2 deletions(-) diff --git a/include/nlohmann/detail/meta/type_traits.hpp b/include/nlohmann/detail/meta/type_traits.hpp index 6f8bf2a3d..36573bf6f 100644 --- a/include/nlohmann/detail/meta/type_traits.hpp +++ b/include/nlohmann/detail/meta/type_traits.hpp @@ -725,7 +725,7 @@ std::is_constructible ()(std::declval(), std:: // avoid their instantiation on all compilers, even when the first operand // is false. The dispatch on is_json_pointer_of can be removed once the // deprecated json_pointer comparison operators have been removed. -template::value> +template, uncvref_t>::value> struct is_comparable : std::false_type {}; template diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index f2f8a7ab4..6c98e7a8f 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -4735,7 +4735,7 @@ std::is_constructible ()(std::declval(), std:: // avoid their instantiation on all compilers, even when the first operand // is false. The dispatch on is_json_pointer_of can be removed once the // deprecated json_pointer comparison operators have been removed. -template::value> +template, uncvref_t>::value> struct is_comparable : std::false_type {}; template diff --git a/tests/src/unit-json_pointer.cpp b/tests/src/unit-json_pointer.cpp index 76ae1f79b..86c636c2a 100644 --- a/tests/src/unit-json_pointer.cpp +++ b/tests/src/unit-json_pointer.cpp @@ -878,6 +878,25 @@ TEST_CASE("JSON pointers") } } + SECTION("value(json_pointer, default) with ordered_json #5664") + { + // ordered_json's transparent object comparator made value()'s + // is_comparable_with_object_key check (which passes the pointer as + // a reference) instantiate the deprecated json_pointer/string + // comparison; this must compile without relying on it. The + // deprecation warning itself is not observable here, since the + // unit test build disables -Wdeprecated-declarations (see + // cmake/clang_flags.cmake); it was checked manually instead. + const nlohmann::ordered_json j = {{"n", 1}, {"s", "text"}}; + const nlohmann::ordered_json::json_pointer ptr_n("/n"); + const nlohmann::ordered_json::json_pointer ptr_s("/s"); + const nlohmann::ordered_json::json_pointer ptr_missing("/missing"); + + CHECK(j.value(ptr_n, 0) == 1); + CHECK(j.value(ptr_s, std::string("x")) == "text"); + CHECK(j.value(ptr_missing, 42) == 42); + } + // build with C++20 // JSON_HAS_CPP_20 #if defined(__cpp_char8_t) From 42f89e3130a2b0ad4844198428deb898644002a9 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:02 +0200 Subject: [PATCH 14/31] Let to_json(std::optional) propagate exceptions from T's to_json (#5684) * Let to_json(std::optional) propagate exceptions from T's to_json The overload was marked noexcept even though its body assigns *opt to the JSON value, which calls T's to_json (or allocates for std::string, std::vector, or json). Any exception from there -- a user-defined to_json reporting an error, or std::bad_alloc -- called std::terminate() instead of propagating. The noexcept also made basic_json's converting constructor noexcept(true) for std::optional, so json j = opt; could not report the error either. Fixes #5642. Signed-off-by: Niels Lohmann * Fix CI: make to_json(std::optional) conditionally noexcept GCC's -Wnoexcept (an error in ci_test_gcc) fired at to_json_fn's noexcept(noexcept(to_json(j, val))): after dropping the unconditional noexcept, to_json(std::optional) had no exception specification although GCC could prove its body cannot throw. It also made json(std::optional) lose its noexcept. Declare the overload noexcept exactly when assigning the contained value to the JSON value is (std::is_nothrow_assignable), which is what the body does. std::optional is noexcept again; a T whose to_json may throw still propagates the exception. Static assertions in the test check both cases. The test's throwing to_json triggered -Wmissing-prototypes and -Wmissing-noreturn (clang) and -Wmissing-declarations and -Wsuggest-attribute=noreturn (GCC). Move the type and its to_json into an anonymous namespace, mark the function [[noreturn]], and compile them only without JSON_NOEXCEPTION, like the test that uses them. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .../nlohmann/detail/conversions/to_json.hpp | 2 +- single_include/nlohmann/json.hpp | 2 +- tests/src/unit-conversions.cpp | 32 +++++++++++++++++++ 3 files changed, 34 insertions(+), 2 deletions(-) diff --git a/include/nlohmann/detail/conversions/to_json.hpp b/include/nlohmann/detail/conversions/to_json.hpp index 491bb9873..fb48e4f82 100644 --- a/include/nlohmann/detail/conversions/to_json.hpp +++ b/include/nlohmann/detail/conversions/to_json.hpp @@ -287,7 +287,7 @@ struct external_constructor #ifdef JSON_HAS_CPP_17 template::value, int> = 0> -void to_json(BasicJsonType& j, const std::optional& opt) noexcept +void to_json(BasicJsonType& j, const std::optional& opt) noexcept(std::is_nothrow_assignable::value) { if (opt.has_value()) { diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 6c98e7a8f..a606e1762 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -6820,7 +6820,7 @@ struct external_constructor #ifdef JSON_HAS_CPP_17 template::value, int> = 0> -void to_json(BasicJsonType& j, const std::optional& opt) noexcept +void to_json(BasicJsonType& j, const std::optional& opt) noexcept(std::is_nothrow_assignable::value) { if (opt.has_value()) { diff --git a/tests/src/unit-conversions.cpp b/tests/src/unit-conversions.cpp index 077ba0e14..0d53f2226 100644 --- a/tests/src/unit-conversions.cpp +++ b/tests/src/unit-conversions.cpp @@ -1834,6 +1834,21 @@ TEST_CASE("std::u8string") #endif #endif +#if !defined(JSON_NOEXCEPTION) +namespace +{ +// a type whose to_json reports an error by throwing, used below to check that +// converting a std::optional to JSON propagates an exception thrown while +// converting its contained value instead of calling std::terminate (#5642) +struct throwing_to_json_type {}; + +[[noreturn]] void to_json(json& /*unused*/, const throwing_to_json_type& /*unused*/) +{ + throw std::runtime_error("cannot serialize throwing_to_json_type"); +} +} // namespace +#endif + TEST_CASE("std::optional") { SECTION("null") @@ -1916,6 +1931,23 @@ TEST_CASE("std::optional") CHECK(json(opt_object) == j_object); CHECK(std::map>(j_object) == opt_object); } + +#if !defined(JSON_NOEXCEPTION) + SECTION("exception from contained value's to_json propagates (#5642)") + { + // to_json(BasicJsonType&, const std::optional&) must not be + // noexcept: it calls T's to_json, which may throw (a user-defined + // to_json that reports an error, or std::bad_alloc for T = + // std::string/vector/json). Before the fix, this called + // std::terminate() instead of letting the exception propagate. + const std::optional opt = throwing_to_json_type{}; + CHECK_THROWS_WITH_AS(json(opt), "cannot serialize throwing_to_json_type", std::runtime_error&); + + // the conversion is noexcept exactly when converting the contained value is + static_assert(!std::is_nothrow_constructible&>::value, ""); + static_assert(std::is_nothrow_constructible&>::value, ""); + } +#endif } #endif From 6218aa212b04994fe67ba02086a2d40a74e41921 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:06 +0200 Subject: [PATCH 15/31] Throw std::length_error for operator[](SIZE_MAX) instead of corrupting the array (#5687) For idx == SIZE_MAX, the non-const array operator[] computed the new size as idx + 1, which wraps to 0. resize(0) then emptied the array, and the subsequent operator[](idx) on the now-empty vector wrote one element before its buffer. Every other too-large index (e.g. SIZE_MAX - 1) already went through resize(), which throws std::length_error and leaves the array unchanged; SIZE_MAX was the one value for which the overflow bypassed that safety net. Add a guard that throws std::length_error before computing idx + 1 when idx is the largest representable size_type value, so the array is left unchanged, matching the exception vector::resize() already throws for smaller (but still too large) indices. Fixes #5647. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/operator[].md | 6 +++++- include/nlohmann/json.hpp | 9 +++++++++ single_include/nlohmann/json.hpp | 9 +++++++++ tests/src/unit-element_access1.cpp | 18 ++++++++++++++++++ 4 files changed, 41 insertions(+), 1 deletion(-) diff --git a/docs/mkdocs/docs/api/basic_json/operator[].md b/docs/mkdocs/docs/api/basic_json/operator[].md index 3195870e4..21c4fe8c0 100644 --- a/docs/mkdocs/docs/api/basic_json/operator[].md +++ b/docs/mkdocs/docs/api/basic_json/operator[].md @@ -70,6 +70,9 @@ Strong exception safety: if an exception occurs, the original value stays intact 1. The function can throw the following exceptions: - Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the JSON value is not an array or null; in that case, using the `[]` operator with an index makes no sense. + - Throws `#!cpp std::length_error` if `idx` equals the maximum value of `size_type`; the array is left unchanged. + (This is the one index for which growing the array to hold it cannot be expressed as a `size_type` size, the same + way an oversized [`resize`](https://en.cppreference.com/w/cpp/container/vector/resize) throws.) 2. The function can throw the following exceptions: - Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the JSON value is not an object or null; in that case, using the `[]` operator with a key makes no sense. @@ -257,7 +260,8 @@ Strong exception safety: if an exception occurs, the original value stays intact ## Version history -1. Added in version 1.0.0. +1. Added in version 1.0.0. Fixed in version 3.13.0 to throw `#!cpp std::length_error` instead of emptying the array and + accessing it out of bounds when `idx` equals the maximum value of `size_type`. 2. Added in version 1.0.0. Added overloads for `T* key` in version 1.1.0. Removed overloads for `T* key` (replaced by 3) in version 3.11.0. 3. Added in version 3.11.0. Fixed in version 3.13.0 to consistently accept `std::string_view`-convertible keys, as diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 28d91600f..3bc72805c 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -36,7 +36,9 @@ #include // istream, ostream #endif // JSON_NO_IO #include // make_move_iterator, random_access_iterator_tag +#include // numeric_limits #include // unique_ptr +#include // length_error #include // string, stoi, to_string #include // declval, forward, move, pair, swap #include // vector @@ -2826,6 +2828,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // fill up the array with null values if given idx is outside the range if (idx >= m_data.m_value.array->size()) { + // idx + 1 would overflow size_type and wrap to 0, which would empty + // the array instead of growing it; reject such an idx the same way + // resize() rejects other indices that are too large to represent + if (JSON_HEDLEY_UNLIKELY(idx == (std::numeric_limits::max)())) + { + JSON_THROW(std::length_error(detail::concat("array index ", std::to_string(idx), " exceeds size_type"))); + } #if JSON_DIAGNOSTICS // remember array size & capacity before resizing const auto old_size = m_data.m_value.array->size(); diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index a606e1762..8b442e3db 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -36,7 +36,9 @@ #include // istream, ostream #endif // JSON_NO_IO #include // make_move_iterator, random_access_iterator_tag +#include // numeric_limits #include // unique_ptr +#include // length_error #include // string, stoi, to_string #include // declval, forward, move, pair, swap #include // vector @@ -29710,6 +29712,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // fill up the array with null values if given idx is outside the range if (idx >= m_data.m_value.array->size()) { + // idx + 1 would overflow size_type and wrap to 0, which would empty + // the array instead of growing it; reject such an idx the same way + // resize() rejects other indices that are too large to represent + if (JSON_HEDLEY_UNLIKELY(idx == (std::numeric_limits::max)())) + { + JSON_THROW(std::length_error(detail::concat("array index ", std::to_string(idx), " exceeds size_type"))); + } #if JSON_DIAGNOSTICS // remember array size & capacity before resizing const auto old_size = m_data.m_value.array->size(); diff --git a/tests/src/unit-element_access1.cpp b/tests/src/unit-element_access1.cpp index eccefb3ce..22a515054 100644 --- a/tests/src/unit-element_access1.cpp +++ b/tests/src/unit-element_access1.cpp @@ -147,6 +147,24 @@ TEST_CASE("element access 1") CHECK(j_const[7] == json({1, 2, 3})); } + SECTION("SIZE_MAX index (#5647)") + { + // idx + 1 must not be computed for idx == SIZE_MAX: it wraps to 0, + // which would empty the array and then write out of bounds instead + // of growing it; reject it like an oversized resize() would and + // leave the array unchanged + const auto max_idx = (std::numeric_limits::max)(); + const std::string expected = "array index " + std::to_string(max_idx) + " exceeds size_type"; + const json j_before = j; // NOLINT(performance-unnecessary-copy-initialization) + + CHECK_THROWS_WITH_AS(j[max_idx] = 1, expected.c_str(), std::length_error&); + CHECK(j == j_before); + + json j_empty = json::array(); + CHECK_THROWS_WITH_AS(j_empty[max_idx] = 1, expected.c_str(), std::length_error&); + CHECK(j_empty == json::array()); + } + SECTION("access on non-array type") { SECTION("null") From 115182650851ed3d5a193b7fdc356bfd5552da88 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:10 +0200 Subject: [PATCH 16/31] Take diff()'s fast path unless the object type reorders members (#5691) * Take diff()'s fast path unless the object type reorders members For every object type except an insertion-ordered one like ordered_map, diff() no longer produced a member-by-member patch when target had a key that sorts before a key the two objects share: it fell through to the slow path, which removes every member of source and re-adds every member of target, instead of just adding the new key. #5465 added an order check to require the fast path to also reproduce target's member order, needed because ordered_map's patch()-driven "add" appends a new member at the end. The check compared the common keys' order between source and target and also required that every added key come after every common key in target's order ("new_keys_form_suffix"). The comment above it argued this check is always true for std::map, and that reasoning is correct for the order of the common keys themselves, but not for new_keys_form_suffix: a std::map iterates in sorted key order, so a new key that sorts before an existing common key is enumerated between common keys, making new_keys_form_suffix false even though std::map's own key order does not need reordering at all - it places every member itself, regardless of insertion history, so a member-by-member diff already reproduces target's iteration order. Only require the order check for an object type that keeps insertion order, using the same detail::is_ordered_map trait the library already uses to recognize such an object type in set_parent(). Every other object type - std::map in key order, a hash map in an order its operator== ignores - always takes the fast path. Added a regression test to unit-json_patch.cpp: the issue's example now yields a single "add" op for json, while ordered_json still takes the slow path to reproduce target's member order. Fixes #5639. Signed-off-by: Niels Lohmann * Only track the target key order in diff() for insertion-ordered objects common_keys_target_order and new_keys_form_suffix are only read when object_t keeps its members in insertion order; skip building them otherwise. Addresses review comment by @gregmarr. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 35 +++++++++++++++------ single_include/nlohmann/json.hpp | 35 +++++++++++++++------ tests/src/unit-json_patch.cpp | 52 ++++++++++++++++++++++++++++++++ 3 files changed, 102 insertions(+), 20 deletions(-) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 3bc72805c..7f0f1d7c2 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -6156,12 +6156,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // the same time, determine whether every added key comes // after every common key in target's order (a precondition // for the fast path below, which only ever appends new keys - // at the very end): for an object_t whose iteration order is - // a pure function of the key set (e.g. the default std::map, - // which always iterates in sorted key order), the order - // check further below is always true and this whole - // mechanism is effectively a no-op; it only matters for a - // reorderable object_t such as the one backing `ordered_json`. + // at the very end). Both are only needed for an object_t that + // keeps its members in insertion order, such as the one + // backing `ordered_json`; for any other object_t, the fast + // path is always taken and they are not computed. // patch ops for keys that were added (i.e., in target but not // in source); built here so the fast path below can reuse // them without a second source.find() per target key. Only @@ -6185,15 +6183,32 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } else { - common_keys_target_order.push_back(it.key()); - if (seen_new_key) +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning(push ) +#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr +#endif + if (detail::is_ordered_map::value) { - new_keys_form_suffix = false; + common_keys_target_order.push_back(it.key()); + if (seen_new_key) + { + new_keys_form_suffix = false; + } } +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning( pop ) +#endif } } - if (common_keys_source_order == common_keys_target_order && new_keys_form_suffix) + // Only an object type that keeps its members in insertion + // order, such as nlohmann::ordered_map, can need reordering: + // patch() appends a new member at the end of such an object. + // Any other object type places its members itself - std::map + // in key order, a hash map in an order its operator== ignores - + // so a member-by-member diff always reproduces target there. + if (!detail::is_ordered_map::value + || (common_keys_source_order == common_keys_target_order && new_keys_form_suffix)) { // fast path: order of common keys already matches (or the // object_t's iteration order does not depend on diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 8b442e3db..96a3e67ad 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -33040,12 +33040,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // the same time, determine whether every added key comes // after every common key in target's order (a precondition // for the fast path below, which only ever appends new keys - // at the very end): for an object_t whose iteration order is - // a pure function of the key set (e.g. the default std::map, - // which always iterates in sorted key order), the order - // check further below is always true and this whole - // mechanism is effectively a no-op; it only matters for a - // reorderable object_t such as the one backing `ordered_json`. + // at the very end). Both are only needed for an object_t that + // keeps its members in insertion order, such as the one + // backing `ordered_json`; for any other object_t, the fast + // path is always taken and they are not computed. // patch ops for keys that were added (i.e., in target but not // in source); built here so the fast path below can reuse // them without a second source.find() per target key. Only @@ -33069,15 +33067,32 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } else { - common_keys_target_order.push_back(it.key()); - if (seen_new_key) +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning(push ) +#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr +#endif + if (detail::is_ordered_map::value) { - new_keys_form_suffix = false; + common_keys_target_order.push_back(it.key()); + if (seen_new_key) + { + new_keys_form_suffix = false; + } } +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning( pop ) +#endif } } - if (common_keys_source_order == common_keys_target_order && new_keys_form_suffix) + // Only an object type that keeps its members in insertion + // order, such as nlohmann::ordered_map, can need reordering: + // patch() appends a new member at the end of such an object. + // Any other object type places its members itself - std::map + // in key order, a hash map in an order its operator== ignores - + // so a member-by-member diff always reproduces target there. + if (!detail::is_ordered_map::value + || (common_keys_source_order == common_keys_target_order && new_keys_form_suffix)) { // fast path: order of common keys already matches (or the // object_t's iteration order does not depend on diff --git a/tests/src/unit-json_patch.cpp b/tests/src/unit-json_patch.cpp index 216d00c41..ef239e9f4 100644 --- a/tests/src/unit-json_patch.cpp +++ b/tests/src/unit-json_patch.cpp @@ -1752,6 +1752,58 @@ TEST_CASE("JSON patch - diff emits array removals in descending index order") } } +TEST_CASE("JSON patch - diff() takes the fast path for non-reorderable object types (regression #5639)") +{ + // #5465 added an order check to diff()'s object handling so a + // member-by-member diff is only used when it would also reproduce + // target's member *order* -- needed for ordered_json, whose object_t + // keeps insertion order and whose patch() "add" op appends a new + // member at the end. For json's default object_t (std::map, which + // orders members by key regardless of insertion history), that check + // could still fail: a new key that sorts before an existing common key + // makes target's iteration interleave the new key between common keys, + // even though nothing else about the object changed. That sent the + // whole object through the slow (remove-every-member, + // re-add-every-member) path instead of the minimal one. + SECTION("json: added key sorts before an existing common key") + { + const json source = {{"a", 1}, {"c", {{"x", 1}, {"y", 2}}}}; + const json target = {{"a", 1}, {"b", 0}, {"c", {{"x", 1}, {"y", 2}}}}; + + const json patch = json::diff(source, target); + + // only the new key is added; "a" and "c" are left alone instead of + // being removed and re-added + const json expected = R"([{"op": "add", "path": "/b", "value": 0}])"_json; + CHECK(patch == expected); + CHECK(source.patch(patch) == target); + } + + SECTION("ordered_json: reordering behavior from #5465 is unchanged") + { + using nlohmann::ordered_json; + + // same key/value shape as the json case above, but for ordered_json + // the *target*'s member order must be reproduced, so the slow path + // is still required here. + ordered_json source; + source["a"] = 1; + source["c"] = ordered_json{{"x", 1}, {"y", 2}}; + + ordered_json target; + target["a"] = 1; + target["b"] = 0; + target["c"] = ordered_json{{"x", 1}, {"y", 2}}; + + const ordered_json patch = ordered_json::diff(source, target); + + // unlike the json case: every member is still removed and re-added + // so the result ends up in target's order (2 removes + 3 adds) + CHECK(patch.size() == 5); + CHECK(source.patch(patch) == target); + } +} + TEST_CASE("JSON patch - every operation on ordered_json") { using nlohmann::ordered_json; From e444a662762420681487f369aa5fc8565de8366f Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:16 +0200 Subject: [PATCH 17/31] Copy values before inserting an initializer list into an array (#5693) * Copy values before inserting an initializer list into an array insert(pos, {...}) inserted wrong values when the initializer list contained const references to elements of the array being inserted into. json_ref stores only a pointer for a const lvalue, so the initializer_list_t range passed straight to the array's range insert aliased the array's own storage; std::vector::insert(pos, first, last) may move or shift elements before copying from that range, so the source elements were already stale by the time they were read (different wrong results on libc++ and libstdc++). Copy the referenced values into a temporary array_t first, then move that temporary into place, so the source range never aliases the array being modified. Fixes #5656. Signed-off-by: Niels Lohmann * Use the reserve_array helper in the initializer_list insert fix The previous commit called array_t::reserve() directly on the temporary buffer used to copy an ilist's values before inserting. std::deque, a documented ArrayType (tests/src/unit-custom-array-type.cpp), has no reserve(), so insert(pos, initializer_list) no longer compiled for it. Use the existing detail::reserve_array() SFINAE helper (already used by the SAX DOM parser) instead, which leaves array types without reserve() untouched. Added a regression check that deque_json::insert(pos, {...}) compiles and handles the aliasing case from #5656. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/insert.md | 4 +++- include/nlohmann/json.hpp | 10 +++++++- single_include/nlohmann/json.hpp | 10 +++++++- tests/src/unit-custom-array-type.cpp | 16 +++++++++++++ tests/src/unit-modifiers.cpp | 28 +++++++++++++++++++++++ 5 files changed, 65 insertions(+), 3 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/insert.md b/docs/mkdocs/docs/api/basic_json/insert.md index fcb1e6e44..ff9082b56 100644 --- a/docs/mkdocs/docs/api/basic_json/insert.md +++ b/docs/mkdocs/docs/api/basic_json/insert.md @@ -195,5 +195,7 @@ Strong exception safety: if an exception occurs, the original value stays intact 1. Added in version 1.0.0. 2. Added in version 1.0.0. 3. Added in version 1.0.0. -4. Added in version 1.0.0. +4. Added in version 1.0.0. Fixed in version 3.13.0 to copy the values before inserting; before, an `ilist` that + referred to elements of the array being inserted into could insert wrong values, because the range insert could + move from or shift an element before it was copied. 5. Added in version 3.0.0. diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 7f0f1d7c2..7f9dd8caf 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -4158,8 +4158,16 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(invalid_iterator::create(202, "iterator does not fit current value", this)); } + // copy the values first: ilist may refer to elements of this array + array_t values; + detail::reserve_array(values, ilist.size(), detail::priority_tag<1> {}); + for (const auto& element : ilist) + { + values.push_back(element.moved_or_copied()); + } + // insert to array and return iterator - return insert_iterator(pos, ilist.begin(), ilist.end()); + return insert_iterator(pos, std::make_move_iterator(values.begin()), std::make_move_iterator(values.end())); } /// @brief inserts range of elements into object diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 96a3e67ad..27f65e412 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -31042,8 +31042,16 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(invalid_iterator::create(202, "iterator does not fit current value", this)); } + // copy the values first: ilist may refer to elements of this array + array_t values; + detail::reserve_array(values, ilist.size(), detail::priority_tag<1> {}); + for (const auto& element : ilist) + { + values.push_back(element.moved_or_copied()); + } + // insert to array and return iterator - return insert_iterator(pos, ilist.begin(), ilist.end()); + return insert_iterator(pos, std::make_move_iterator(values.begin()), std::make_move_iterator(values.end())); } /// @brief inserts range of elements into object diff --git a/tests/src/unit-custom-array-type.cpp b/tests/src/unit-custom-array-type.cpp index 00606c6e0..7a374b612 100644 --- a/tests/src/unit-custom-array-type.cpp +++ b/tests/src/unit-custom-array-type.cpp @@ -117,6 +117,22 @@ TEST_CASE("array type without capacity()") CHECK(nested.flatten().unflatten() == nested); } + SECTION("insert(pos, initializer_list) compiles and works without reserve()") + { + // std::deque has no reserve() either; insert(pos, ilist) must not + // require it (regression test for #5656, which also covers an ilist + // that refers to elements of the array being inserted into) + deque_json j = deque_json::array(); + j.push_back("a"); + j.push_back("b"); + j.push_back("c"); + + const deque_json& cj = j; + auto it = j.insert(j.begin(), {cj[0], cj[1]}); + CHECK(*it == deque_json("a")); + CHECK(j == deque_json({"a", "b", "a", "b", "c"})); + } + SECTION("references stay valid while the array grows") { deque_json j = deque_json::array(); diff --git a/tests/src/unit-modifiers.cpp b/tests/src/unit-modifiers.cpp index 1ec6016cb..d36ed5ad2 100644 --- a/tests/src/unit-modifiers.cpp +++ b/tests/src/unit-modifiers.cpp @@ -773,6 +773,34 @@ TEST_CASE("modifiers") } } + SECTION("initializer list referring to the array's own elements (#5656)") + { + SECTION("sufficient capacity (no reallocation)") + { + json j_own = json::array(); + j_own.get_ref().reserve(8); + j_own.push_back("a"); + j_own.push_back("b"); + j_own.push_back("c"); + + const json& j_own_cref = j_own; + auto it = j_own.insert(j_own.begin(), {j_own_cref[0], j_own_cref[1]}); + CHECK(*it == json("a")); + CHECK(j_own == json({"a", "b", "a", "b", "c"})); + } + + SECTION("insufficient capacity (reallocation)") + { + json j_own = {"a", "b", "c"}; + j_own.get_ref().shrink_to_fit(); + + const json& j_own_cref = j_own; + auto it = j_own.insert(j_own.begin(), {j_own_cref[2]}); + CHECK(*it == json("c")); + CHECK(j_own == json({"c", "a", "b", "c"})); + } + } + SECTION("invalid iterator") { // pass iterator to a different array From bfea6f36d3b1f514b1e65841ca99309a25fe5dad Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:19 +0200 Subject: [PATCH 18/31] Fix to_msgpack() reading the inactive number union member (#5694) * Fix to_msgpack() reading the inactive number union member basic_json stores number_integer and number_unsigned in a union, and number_unsigned_t only has to be at least as wide as number_integer_t (with the default types, both are 64-bit and have the same representation). When number_integer_t is narrower, write_msgpack() read the wrong union member in two places: - The number_unsigned case wrote number_integer's bits instead of number_unsigned's, silently writing the wrong value whenever it did not fit in number_integer_t. - The number_integer case (non-negative branch) picked the encoded width by comparing number_unsigned's bits, which is undefined behavior, though the value written was still number_integer's, so at worst a too-wide encoding was chosen. Read the active member in both cases, like the other binary writers (CBOR, UBJSON, BJData, BSON, BON8) already do. Fixes #5644. Signed-off-by: Niels Lohmann * Cast number_integer to number_unsigned_t only once in to_msgpack() Addresses review comment by @gregmarr. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/to_msgpack.md | 3 + .../nlohmann/detail/output/binary_writer.hpp | 19 ++++--- single_include/nlohmann/json.hpp | 19 ++++--- tests/src/unit-msgpack.cpp | 57 +++++++++++++++++++ 4 files changed, 80 insertions(+), 18 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/to_msgpack.md b/docs/mkdocs/docs/api/basic_json/to_msgpack.md index 007fb1914..19f86f9c2 100644 --- a/docs/mkdocs/docs/api/basic_json/to_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/to_msgpack.md @@ -76,3 +76,6 @@ Linear in the size of the JSON value `j`. - Added in version 2.0.9. - Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0. +- Fixed in version 3.13.0 to serialize `number_integer_t`/`number_unsigned_t` pairs of different width correctly; + before, integers could be serialized with the wrong value if `number_integer_t` was narrower than + `number_unsigned_t`. diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 9da4269a8..4f39f0537 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -337,24 +337,25 @@ class binary_writer // MessagePack does not differentiate between positive // signed integers and unsigned integers. Therefore, we used // the code from the value_t::number_unsigned case here. - if (j.m_data.m_value.number_unsigned < 128) + const auto value_as_unsigned = static_cast(j.m_data.m_value.number_integer); + if (value_as_unsigned < 128) { // positive fixnum write_number(static_cast(j.m_data.m_value.number_integer)); } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) + else if (value_as_unsigned <= (std::numeric_limits::max)()) { // uint 8 oa.write_character(to_char_type(0xCC)); write_number(static_cast(j.m_data.m_value.number_integer)); } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) + else if (value_as_unsigned <= (std::numeric_limits::max)()) { // uint 16 oa.write_character(to_char_type(0xCD)); write_number(static_cast(j.m_data.m_value.number_integer)); } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) + else if (value_as_unsigned <= (std::numeric_limits::max)()) { // uint 32 oa.write_character(to_char_type(0xCE)); @@ -410,31 +411,31 @@ class binary_writer if (j.m_data.m_value.number_unsigned < 128) { // positive fixnum - write_number(static_cast(j.m_data.m_value.number_integer)); + write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 8 oa.write_character(to_char_type(0xCC)); - write_number(static_cast(j.m_data.m_value.number_integer)); + write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 16 oa.write_character(to_char_type(0xCD)); - write_number(static_cast(j.m_data.m_value.number_integer)); + write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 32 oa.write_character(to_char_type(0xCE)); - write_number(static_cast(j.m_data.m_value.number_integer)); + write_number(static_cast(j.m_data.m_value.number_unsigned)); } else { // uint 64 oa.write_character(to_char_type(0xCF)); - write_number(static_cast(j.m_data.m_value.number_integer)); + write_number(static_cast(j.m_data.m_value.number_unsigned)); } break; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 27f65e412..6d3f6f4ab 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -21412,24 +21412,25 @@ class binary_writer // MessagePack does not differentiate between positive // signed integers and unsigned integers. Therefore, we used // the code from the value_t::number_unsigned case here. - if (j.m_data.m_value.number_unsigned < 128) + const auto value_as_unsigned = static_cast(j.m_data.m_value.number_integer); + if (value_as_unsigned < 128) { // positive fixnum write_number(static_cast(j.m_data.m_value.number_integer)); } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) + else if (value_as_unsigned <= (std::numeric_limits::max)()) { // uint 8 oa.write_character(to_char_type(0xCC)); write_number(static_cast(j.m_data.m_value.number_integer)); } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) + else if (value_as_unsigned <= (std::numeric_limits::max)()) { // uint 16 oa.write_character(to_char_type(0xCD)); write_number(static_cast(j.m_data.m_value.number_integer)); } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) + else if (value_as_unsigned <= (std::numeric_limits::max)()) { // uint 32 oa.write_character(to_char_type(0xCE)); @@ -21485,31 +21486,31 @@ class binary_writer if (j.m_data.m_value.number_unsigned < 128) { // positive fixnum - write_number(static_cast(j.m_data.m_value.number_integer)); + write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 8 oa.write_character(to_char_type(0xCC)); - write_number(static_cast(j.m_data.m_value.number_integer)); + write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 16 oa.write_character(to_char_type(0xCD)); - write_number(static_cast(j.m_data.m_value.number_integer)); + write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 32 oa.write_character(to_char_type(0xCE)); - write_number(static_cast(j.m_data.m_value.number_integer)); + write_number(static_cast(j.m_data.m_value.number_unsigned)); } else { // uint 64 oa.write_character(to_char_type(0xCF)); - write_number(static_cast(j.m_data.m_value.number_integer)); + write_number(static_cast(j.m_data.m_value.number_unsigned)); } break; } diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index 498dec859..0876e0f9f 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -2475,3 +2475,60 @@ TEST_CASE("MessagePack lengths beyond UINT32_MAX cannot be serialized") } #endif } + +TEST_CASE("MessagePack numbers use the active union member (see #5644)") +{ + // when number_integer_t is narrower than number_unsigned_t, to_msgpack() + // used to read the union member that was not the active one, writing + // wrong bytes for some values; std::int64_t/std::uint64_t (the default + // types, where both members have the same width) were not affected + using int32_json = nlohmann::basic_json; + using int16_json = nlohmann::basic_json; + + SECTION("number_integer_t = std::int32_t") + { + SECTION("6442450944 (uint 64; the low 32 bits used to be sign-extended)") + { + const int32_json j = 6442450944ULL; + CHECK(j.is_number_unsigned()); + + std::vector const expected{0xcf, 0x00, 0x00, 0x00, 0x01, 0x80, 0x00, 0x00, 0x00}; + const auto result = int32_json::to_msgpack(j); + CHECK(result == expected); + CHECK(int32_json::from_msgpack(result) == j); + } + + SECTION("4294967496 (uint 64; the low 32 bits used to be the whole value)") + { + const int32_json j = 4294967496ULL; + CHECK(j.is_number_unsigned()); + + std::vector const expected{0xcf, 0x00, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00, 0xc8}; + const auto result = int32_json::to_msgpack(j); + CHECK(result == expected); + CHECK(int32_json::from_msgpack(result) == j); + } + } + + SECTION("number_integer_t = std::int16_t, 98304 (uint 32)") + { + const int16_json j = 98304ULL; + CHECK(j.is_number_unsigned()); + + std::vector const expected{0xce, 0x00, 0x01, 0x80, 0x00}; + const auto result = int16_json::to_msgpack(j); + CHECK(result == expected); + CHECK(int16_json::from_msgpack(result) == j); + } + + SECTION("default types (std::int64_t/std::uint64_t) are unaffected") + { + const json j = 4294967496ULL; + CHECK(j.is_number_unsigned()); + + std::vector const expected{0xcf, 0x00, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00, 0xc8}; + const auto result = json::to_msgpack(j); + CHECK(result == expected); + CHECK(json::from_msgpack(result) == j); + } +} From 5b11a0282c208a7005214b58848fe12942872ad3 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:24 +0200 Subject: [PATCH 19/31] Exchange the CustomBaseClass subobject in basic_json::swap() (#5697) basic_json::swap() (and the friend swap() and the pre-C++20 std::swap overload that forward to it) only exchanged m_data.m_type/m_data.m_value, leaving each value's json_base_class_t subobject in place. This is inconsistent with the copy and move constructors and copy assignment, which all carry the base class along with the value, so after a.swap(b) any metadata stored in a CustomBaseClass ended up attached to the wrong value. Algorithms that mix swap() with moves, such as std::sort, scrambled the metadata across the whole container. Fix the member swap() to also exchange the json_base_class_t subobject and extend the noexcept specifications of swap() and the friend swap() accordingly. Fixes #5653. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/swap.md | 16 ++++-- include/nlohmann/json.hpp | 15 +++++- single_include/nlohmann/json.hpp | 15 +++++- tests/src/unit-custom-base-class.cpp | 72 +++++++++++++++++++++++++ 4 files changed, 110 insertions(+), 8 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/swap.md b/docs/mkdocs/docs/api/basic_json/swap.md index aa5aa6c4c..3ac74288d 100644 --- a/docs/mkdocs/docs/api/basic_json/swap.md +++ b/docs/mkdocs/docs/api/basic_json/swap.md @@ -6,7 +6,9 @@ void swap(reference other) noexcept ( std::is_nothrow_move_constructible::value && std::is_nothrow_move_assignable::value && std::is_nothrow_move_constructible::value && - std::is_nothrow_move_assignable::value + std::is_nothrow_move_assignable::value && + std::is_nothrow_move_constructible::value && + std::is_nothrow_move_assignable::value ); // (2) @@ -14,7 +16,9 @@ friend void swap(reference left, reference right) noexcept ( std::is_nothrow_move_constructible::value && std::is_nothrow_move_assignable::value && std::is_nothrow_move_constructible::value && - std::is_nothrow_move_assignable::value + std::is_nothrow_move_assignable::value && + std::is_nothrow_move_constructible::value && + std::is_nothrow_move_assignable::value ); // (3) @@ -37,11 +41,15 @@ void swap(typename binary_t::container_type& other); individual elements. All iterators and references remain valid. The past-the-end iterator is invalidated. If macro [`JSON_DIAGNOSTIC_POSITIONS`](../macros/json_diagnostic_positions.md) is defined to `#!cpp 1`, the [`start_pos()`](start_pos.md)/[`end_pos()`](end_pos.md) diagnostic positions are exchanged along with the value. + The [`json_base_class_t`](json_base_class_t.md) subobject is exchanged along with the value as well, the same way it + is copied or moved by the copy/move constructors and assignment operators. 2. Exchanges the contents of the JSON value from `left` with those of `right`. Does not invoke any move, copy, or swap operations on individual elements. All iterators and references remain valid. The past-the-end iterator is invalidated. Implemented as a friend function callable via ADL. If macro [`JSON_DIAGNOSTIC_POSITIONS`](../macros/json_diagnostic_positions.md) is defined to `#!cpp 1`, the [`start_pos()`](start_pos.md)/[`end_pos()`](end_pos.md) diagnostic positions are exchanged along with the value. + The [`json_base_class_t`](json_base_class_t.md) subobject is exchanged along with the value as well, the same way it + is copied or moved by the copy/move constructors and assignment operators. 3. Exchanges the contents of a JSON array with those of `other`. Does not invoke any move, copy, or swap operations on individual elements. All iterators and references remain valid. The past-the-end iterator is invalidated. 4. Exchanges the contents of a JSON object with those of `other`. Does not invoke any move, copy, or swap operations on @@ -164,8 +172,8 @@ Constant. ## Version history -1. Since version 1.0.0. -2. Since version 1.0.0. +1. Since version 1.0.0. Exchanges the `json_base_class_t` subobject along with the value since version 3.13.0. +2. Since version 1.0.0. Exchanges the `json_base_class_t` subobject along with the value since version 3.13.0. 3. Since version 1.0.0. 4. Since version 1.0.0. 5. Since version 1.0.0. diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 7f9dd8caf..13274f902 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -4349,12 +4349,21 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec std::is_nothrow_move_constructible::value&& std::is_nothrow_move_assignable::value&& std::is_nothrow_move_constructible::value&& // NOLINT(cppcoreguidelines-noexcept-swap,performance-noexcept-swap) - std::is_nothrow_move_assignable::value + std::is_nothrow_move_assignable::value&& + std::is_nothrow_move_constructible::value&& + std::is_nothrow_move_assignable::value ) { std::swap(m_data.m_type, other.m_data.m_type); std::swap(m_data.m_value, other.m_data.m_value); + // the custom base class travels with the value when it is copied or + // moved, so it is exchanged along with it + { + using std::swap; + swap(static_cast(*this), static_cast(other)); + } + #if JSON_DIAGNOSTIC_POSITIONS std::swap(start_position, other.start_position); std::swap(end_position, other.end_position); @@ -4371,7 +4380,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec std::is_nothrow_move_constructible::value&& std::is_nothrow_move_assignable::value&& std::is_nothrow_move_constructible::value&& // NOLINT(cppcoreguidelines-noexcept-swap,performance-noexcept-swap) - std::is_nothrow_move_assignable::value + std::is_nothrow_move_assignable::value&& + std::is_nothrow_move_constructible::value&& + std::is_nothrow_move_assignable::value ) { left.swap(right); diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 6d3f6f4ab..f1aad02ce 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -31234,12 +31234,21 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec std::is_nothrow_move_constructible::value&& std::is_nothrow_move_assignable::value&& std::is_nothrow_move_constructible::value&& // NOLINT(cppcoreguidelines-noexcept-swap,performance-noexcept-swap) - std::is_nothrow_move_assignable::value + std::is_nothrow_move_assignable::value&& + std::is_nothrow_move_constructible::value&& + std::is_nothrow_move_assignable::value ) { std::swap(m_data.m_type, other.m_data.m_type); std::swap(m_data.m_value, other.m_data.m_value); + // the custom base class travels with the value when it is copied or + // moved, so it is exchanged along with it + { + using std::swap; + swap(static_cast(*this), static_cast(other)); + } + #if JSON_DIAGNOSTIC_POSITIONS std::swap(start_position, other.start_position); std::swap(end_position, other.end_position); @@ -31256,7 +31265,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec std::is_nothrow_move_constructible::value&& std::is_nothrow_move_assignable::value&& std::is_nothrow_move_constructible::value&& // NOLINT(cppcoreguidelines-noexcept-swap,performance-noexcept-swap) - std::is_nothrow_move_assignable::value + std::is_nothrow_move_assignable::value&& + std::is_nothrow_move_constructible::value&& + std::is_nothrow_move_assignable::value ) { left.swap(right); diff --git a/tests/src/unit-custom-base-class.cpp b/tests/src/unit-custom-base-class.cpp index 7dab5c576..a7f466e36 100644 --- a/tests/src/unit-custom-base-class.cpp +++ b/tests/src/unit-custom-base-class.cpp @@ -6,9 +6,11 @@ // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT +#include #include #include #include +#include #include "doctest_compatibility.h" @@ -180,6 +182,76 @@ TEST_CASE("JSON Node Metadata") CHECK(val.metadata().at(1) == 2); } } + SECTION("member swap") + { + using json = json_with_metadata; + json a = 1; + a.metadata() = 100; + json b = 2; + b.metadata() = 200; + + a.swap(b); + + CHECK(a.get() == 2); + CHECK(b.get() == 1); + CHECK(a.metadata() == 200); + CHECK(b.metadata() == 100); + } + SECTION("nonmember swap") + { + using json = json_with_metadata; + json a = 1; + a.metadata() = 100; + json b = 2; + b.metadata() = 200; + + using std::swap; + swap(a, b); + + CHECK(a.get() == 2); + CHECK(b.get() == 1); + CHECK(a.metadata() == 200); + CHECK(b.metadata() == 100); + } + SECTION("std::swap") + { + using json = json_with_metadata; + json a = 1; + a.metadata() = 100; + json b = 2; + b.metadata() = 200; + + std::swap(a, b); + + CHECK(a.get() == 2); + CHECK(b.get() == 1); + CHECK(a.metadata() == 200); + CHECK(b.metadata() == 100); + } + SECTION("std::sort keeps metadata attached to its value") + { + // std::sort mixes swap() with moves; each value's metadata must + // travel with it, just as it does for copy, move, and assignment + using json = json_with_metadata; + std::vector values; + for (int v : + { + 5, 3, 9, 1, 7, 2, 8, 4, 6, 0, 15, 13, 19, 11, 17, 12, 18, 14, 16, 10, + 25, 23, 29, 21, 27, 22, 28, 24, 26, 20, 35, 33 + }) + { + json value = v; + value.metadata() = v; + values.push_back(value); + } + + std::sort(values.begin(), values.end()); + + for (const auto& value : values) + { + CHECK(value.metadata() == value.get()); + } + } } // Test extending nlohmann::json by using a custom base class. From 44a88d85bee2ca160ee732263eb0c9f313e5edbc Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:29 +0200 Subject: [PATCH 20/31] Fix NLOHMANN_JSON_SERIALIZE_ENUM_STRICT's from_json message (#5698) from_json built its out_of_range.410 message with "..." + j.dump(). If the unmatched value is (or contains) a string with invalid UTF-8, that dump() itself throws type_error.316, so the caller got type_error.316 instead of the documented out_of_range.410; such strings can reach get() unvalidated, e.g. from from_cbor()/from_msgpack(). With a custom string_t, j.dump() returns that type, and "const char*" + string_t does not compile unless the type happens to provide operator+, so the macro failed to compile for such types. Build the message with detail::concat(), which appends any type exposing data()/size() and always yields a std::string, and dump with error_handler_t::replace so building the message itself cannot throw. Added regression tests: an invalid-UTF-8 case in the existing strict-enum test in unit-conversions.cpp, and a strict-enum use with alt_string (the custom string_t from unit-alt-string.cpp) to cover the compile failure. Fixes #5667. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/macro_scope.hpp | 2 +- single_include/nlohmann/json.hpp | 2 +- tests/src/unit-alt-string.cpp | 24 ++++++++++++++++++++++++ tests/src/unit-conversions.cpp | 6 ++++++ 4 files changed, 32 insertions(+), 2 deletions(-) diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index 3a4eb79f4..9c35cc3b7 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -336,7 +336,7 @@ void templated_json_throw(ExceptionType exception) return ej_pair.second == j; \ }); \ if (it != std::end(m)) e = it->first; \ - else templated_json_throw(nlohmann::detail::out_of_range::create(410,"enum value out of range for " #ENUM_TYPE ": " + j.dump(), &j)); \ + else templated_json_throw(nlohmann::detail::out_of_range::create(410, nlohmann::detail::concat("enum value out of range for " #ENUM_TYPE ": ", j.dump(-1, ' ', false, nlohmann::detail::error_handler_t::replace)), &j)); \ } // Ugly macros to avoid uglier copy-paste when specializing basic_json. They diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index f1aad02ce..16c2cdca6 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -2749,7 +2749,7 @@ void templated_json_throw(ExceptionType exception) return ej_pair.second == j; \ }); \ if (it != std::end(m)) e = it->first; \ - else templated_json_throw(nlohmann::detail::out_of_range::create(410,"enum value out of range for " #ENUM_TYPE ": " + j.dump(), &j)); \ + else templated_json_throw(nlohmann::detail::out_of_range::create(410, nlohmann::detail::concat("enum value out of range for " #ENUM_TYPE ": ", j.dump(-1, ' ', false, nlohmann::detail::error_handler_t::replace)), &j)); \ } // Ugly macros to avoid uglier copy-paste when specializing basic_json. They diff --git a/tests/src/unit-alt-string.cpp b/tests/src/unit-alt-string.cpp index cddaadb7e..f95d2213c 100644 --- a/tests/src/unit-alt-string.cpp +++ b/tests/src/unit-alt-string.cpp @@ -174,6 +174,15 @@ bool operator<(const char* op1, const alt_string& op2) noexcept return op1 < op2.str_impl; } +enum class alt_color { red, green }; + +// NOLINTNEXTLINE(misc-use-internal-linkage,misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - false positive +NLOHMANN_JSON_SERIALIZE_ENUM_STRICT(alt_color, +{ + {alt_color::red, "red"}, + {alt_color::green, "green"}, +}) + TEST_CASE("alternative string type") { SECTION("binary formats") @@ -374,4 +383,19 @@ TEST_CASE("alternative string type") const auto j2 = j.flatten(); CHECK(j2.dump() == R"({"/foo/0":"bar","/foo/1":"baz"})"); } + + SECTION("strict enum") + { + // regression test for #5667: NLOHMANN_JSON_SERIALIZE_ENUM_STRICT's from_json + // built its exception message with "..." + j.dump(), which does not compile + // when j.dump() returns a custom string_t (here alt_string) instead of + // std::string + alt_json doc; + doc = "red"; + CHECK(doc.get() == alt_color::red); + + alt_json _; + doc = "blue"; + CHECK_THROWS_WITH_AS(_ = doc.get(), "[json.exception.out_of_range.410] enum value out of range for alt_color: \"blue\"", alt_json::out_of_range&); + } } diff --git a/tests/src/unit-conversions.cpp b/tests/src/unit-conversions.cpp index 0d53f2226..2fce40fa9 100644 --- a/tests/src/unit-conversions.cpp +++ b/tests/src/unit-conversions.cpp @@ -1748,6 +1748,12 @@ TEST_CASE("Strict JSON to enum mapping") // conversion of unmapped enum -> exception thrown CHECK_THROWS_WITH_AS(json(strict_cards::andere), "[json.exception.out_of_range.410] enum value out of range for strict_cards", json::out_of_range&); + + // invalid UTF-8 -> out_of_range.410, not the type_error.316 thrown while building the + // message (regression test for #5667); such strings can reach get() unvalidated, + // e.g. from from_cbor()/from_msgpack() (#5529) + const json j_invalid_utf8 = "\xFF"; + CHECK_THROWS_WITH_AS(_ = j_invalid_utf8.get(), "[json.exception.out_of_range.410] enum value out of range for strict_cards: \"\xEF\xBF\xBD\"", json::out_of_range&); } SECTION("traditional enum") From bfe0f32d71fdc65dcddd672c7f9bd8f310c5c762 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:33 +0200 Subject: [PATCH 21/31] Fix std::terminate and null pointer access in input_stream_adapter (#5699) Parsing from a std::istream crashed in two unusual but valid stream states, both in input_stream_adapter: - With eofbit in the stream's exceptions() mask, get_character() sets eofbit via is->clear(), which throws std::ios_base::failure. While that exception unwinds, ~input_stream_adapter() called clear() again to reset eofbit, which is still set and still in the exception mask, so it throws a second time out of the (implicitly noexcept) destructor and std::terminate() is called. The destructor now only calls clear() if a bit other than eofbit remains set, so the first exception can propagate normally. - For an std::istream without a stream buffer (rdbuf() == nullptr, e.g. std::istream(nullptr)), the constructor stored the null pointer without checking it, and get_character() dereferenced it. input_adapter(std::istream&) now throws parse_error.101 for such a stream, the same as it already does for a null FILE* or char*. Added regression tests to unit-deserialization.cpp and, for the JSON_PRECISE_STREAM_POSITION variant of get_character(), to unit-precise-stream-position.cpp; both crashed before this fix. Documented the two exceptions in parse.md and operator_gtgt.md. Fixes #5646. Signed-off-by: Niels Lohmann Co-authored-by: Claude Sonnet 5 --- docs/mkdocs/docs/api/basic_json/parse.md | 9 +++++- docs/mkdocs/docs/api/operator_gtgt.md | 9 +++++- .../nlohmann/detail/input/input_adapters.hpp | 14 +++++++-- single_include/nlohmann/json.hpp | 14 +++++++-- tests/src/unit-deserialization.cpp | 31 +++++++++++++++++++ tests/src/unit-precise-stream-position.cpp | 29 +++++++++++++++++ 6 files changed, 100 insertions(+), 6 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/parse.md b/docs/mkdocs/docs/api/basic_json/parse.md index 20bb1c708..554065717 100644 --- a/docs/mkdocs/docs/api/basic_json/parse.md +++ b/docs/mkdocs/docs/api/basic_json/parse.md @@ -88,7 +88,12 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va ## Exceptions - Throws [`parse_error.101`](../../home/exceptions.md#jsonexceptionparse_error101) in case of an unexpected token, or - empty input like a null `FILE*` or `char*` pointer. + empty input like a null `FILE*` or `char*` pointer, or an `std::istream` without a stream buffer + (`#!cpp i.rdbuf() == nullptr`, for instance `#!cpp std::istream(nullptr)`). +- If reading from an `std::istream` reaches the end of the input and `eofbit` is part of the stream's + [`exceptions()`](https://en.cppreference.com/w/cpp/io/basic_ios/exceptions) mask, the `std::ios_base::failure` + thrown by the stream itself propagates instead of a `parse_error`, the same as it would for the standard library's + own extraction operators. ## Complexity @@ -254,6 +259,8 @@ outside of a string, invalid) byte; see the [FAQ entry](../../home/faq.md#nul-by - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. - `JSON_STRICT_NUL_HANDLING` added in version 3.13.0 to optionally reject a NUL byte in the input instead of treating it as end of input; planned to become the default in version 4.0.0. +- Extended empty-input detection to also cover an `std::istream` without a stream buffer, and fixed a crash + (`std::terminate`) when parsing from an `std::istream` with `eofbit` in its exception mask, in version 3.13.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/operator_gtgt.md b/docs/mkdocs/docs/api/operator_gtgt.md index 0173b9fb3..b68889af9 100644 --- a/docs/mkdocs/docs/api/operator_gtgt.md +++ b/docs/mkdocs/docs/api/operator_gtgt.md @@ -20,7 +20,12 @@ the stream `i` ## Exceptions -- Throws [`parse_error.101`](../home/exceptions.md#jsonexceptionparse_error101) in case of an unexpected token. +- Throws [`parse_error.101`](../home/exceptions.md#jsonexceptionparse_error101) in case of an unexpected token, or if + `i` has no stream buffer (`#!cpp i.rdbuf() == nullptr`, for instance `#!cpp std::istream(nullptr)`). +- If reading from `i` reaches the end of the input and `eofbit` is part of `i`'s + [`exceptions()`](https://en.cppreference.com/w/cpp/io/basic_ios/exceptions) mask, the `std::ios_base::failure` + thrown by `i` itself propagates instead of a `parse_error`, the same as it would for the standard library's own + extraction operators. ## Complexity @@ -118,3 +123,5 @@ being read. it as end of input; planned to become the default in version 4.0.0. - `JSON_PRECISE_STREAM_POSITION` added in version 3.13.0 to optionally leave the character that terminates a number in the stream; planned to become the default in version 4.0.0. +- Fixed a null pointer dereference for an `std::istream` without a stream buffer (now throws `parse_error.101`), and a + crash (`std::terminate`) when `i` has `eofbit` in its exception mask, in version 3.13.0. diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index e174775c5..f06713700 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -106,7 +106,13 @@ class input_stream_adapter // was given back with release_lookahead() commit_lookahead(); #endif - is->clear(is->rdstate() & std::ios::eofbit); + // only call clear() if there is something to clear: it throws + // std::ios_base::failure if the stream has exceptions() enabled + // for a state bit that remains set, and a destructor must not throw + if ((is->rdstate() & ~std::ios::eofbit) != 0) + { + is->clear(is->rdstate() & std::ios::eofbit); + } } } @@ -811,12 +817,16 @@ inline file_input_adapter input_adapter(std::FILE* file) inline input_stream_adapter input_adapter(std::istream& stream) { + if (stream.rdbuf() == nullptr) + { + JSON_THROW(parse_error::create(101, 0, "attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr)); + } return input_stream_adapter(stream); } inline input_stream_adapter input_adapter(std::istream&& stream) { - return input_stream_adapter(stream); + return input_adapter(stream); } #endif // JSON_NO_IO diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 16c2cdca6..3735a1925 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7662,7 +7662,13 @@ class input_stream_adapter // was given back with release_lookahead() commit_lookahead(); #endif - is->clear(is->rdstate() & std::ios::eofbit); + // only call clear() if there is something to clear: it throws + // std::ios_base::failure if the stream has exceptions() enabled + // for a state bit that remains set, and a destructor must not throw + if ((is->rdstate() & ~std::ios::eofbit) != 0) + { + is->clear(is->rdstate() & std::ios::eofbit); + } } } @@ -8367,12 +8373,16 @@ inline file_input_adapter input_adapter(std::FILE* file) inline input_stream_adapter input_adapter(std::istream& stream) { + if (stream.rdbuf() == nullptr) + { + JSON_THROW(parse_error::create(101, 0, "attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr)); + } return input_stream_adapter(stream); } inline input_stream_adapter input_adapter(std::istream&& stream) { - return input_stream_adapter(stream); + return input_adapter(stream); } #endif // JSON_NO_IO diff --git a/tests/src/unit-deserialization.cpp b/tests/src/unit-deserialization.cpp index 4202644f6..c18fe0c94 100644 --- a/tests/src/unit-deserialization.cpp +++ b/tests/src/unit-deserialization.cpp @@ -388,6 +388,37 @@ TEST_CASE("deserialization") })); } + SECTION("stream with eofbit in its exception mask (issue #5646)") + { + // reaching EOF while parsing a value that fills the whole input + // (e.g., a number, or any value under strict parsing) makes + // get_character() call std::istream::clear() to record eofbit; + // with eofbit in the exception mask, that clear() itself throws + // std::ios_base::failure - it must propagate to the caller instead + // of ~input_stream_adapter() throwing a second exception while the + // first is still unwinding, which would call std::terminate + json _; + + std::istringstream is1("1"); + is1.exceptions(std::ios::eofbit); + CHECK_THROWS_AS(_ = json::parse(is1), std::ios_base::failure&); + + // the same holds for the common std::ifstream::exceptions(failbit | + // badbit | eofbit) pattern, because only eofbit ends up set + std::istringstream is2("1"); + is2.exceptions(std::ios::failbit | std::ios::badbit | std::ios::eofbit); + CHECK_THROWS_AS(_ = json::parse(is2), std::ios_base::failure&); + } + + SECTION("stream without a streambuf (issue #5646)") + { + // std::istream(nullptr) has badbit set and rdbuf() == nullptr; + // get_character() must not dereference that null streambuf + std::istream is(nullptr); + json _; + CHECK_THROWS_WITH_AS(_ = json::parse(is), "[json.exception.parse_error.101] parse error: attempting to parse an empty input; check that your input string or stream contains the expected JSON", json::parse_error&); + } + SECTION("string") { json::string_t const s = R"(["foo",1,2,3,false,{"one":1})"; diff --git a/tests/src/unit-precise-stream-position.cpp b/tests/src/unit-precise-stream-position.cpp index 5b6bff682..be8cabd99 100644 --- a/tests/src/unit-precise-stream-position.cpp +++ b/tests/src/unit-precise-stream-position.cpp @@ -234,4 +234,33 @@ TEST_CASE("JSON_PRECISE_STREAM_POSITION") CHECK(j == json(1)); CHECK(remaining(is) == "true"); } + + SECTION("stream with eofbit in its exception mask (issue #5646)") + { + // with JSON_PRECISE_STREAM_POSITION, get_character() peeks via + // sb->sgetc() rather than consuming via sb->sbumpc(), but it still + // calls std::istream::clear() to record eofbit once the streambuf is + // exhausted; with eofbit in the exception mask, that clear() itself + // throws std::ios_base::failure, which must propagate to the caller + // instead of ~input_stream_adapter() throwing a second exception + // while the first is still unwinding (which would call std::terminate) + json _; + + std::istringstream is1("1"); + is1.exceptions(std::ios::eofbit); + CHECK_THROWS_AS(_ = json::parse(is1), std::ios_base::failure&); + + std::istringstream is2("1"); + is2.exceptions(std::ios::failbit | std::ios::badbit | std::ios::eofbit); + CHECK_THROWS_AS(_ = json::parse(is2), std::ios_base::failure&); + } + + SECTION("stream without a streambuf (issue #5646)") + { + // std::istream(nullptr) has badbit set and rdbuf() == nullptr; + // get_character() must not dereference that null streambuf + std::istream is(nullptr); + json _; + CHECK_THROWS_WITH_AS(_ = json::parse(is), "[json.exception.parse_error.101] parse error: attempting to parse an empty input; check that your input string or stream contains the expected JSON", json::parse_error&); + } } From 4bb1b14b068d17bce1853297a857d04851cba3dd Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:37 +0200 Subject: [PATCH 22/31] Trim the compiler-appended NUL from wide/UTF string literals too (#5702) With JSON_STRICT_NUL_HANDLING defined to 1, parsing a wide, UTF-16, UTF-32, or (C++20) UTF-8 string literal (e.g. json::parse(L"[1]")) failed with parse_error.101 at the terminating NUL of the literal, and accept() returned false. The array overload of input_adapter() only dropped the compiler-added trailing '\0' for arrays of char, so for wchar_t, char16_t, char32_t, and char8_t arrays that terminator was passed to the parser as data, which the macro then rejected. Broaden the trimming to every character type that a string literal can use (char, wchar_t, char16_t, char32_t, and, since C++20, char8_t). Arrays of any other element type (unsigned char, std::uint8_t, ...), as used for CBOR/MessagePack, are unaffected: a trailing zero byte there is still read as genuine data. Fixes #5658. Signed-off-by: Niels Lohmann --- .../api/macros/json_strict_nul_handling.md | 19 ++++++----- .../nlohmann/detail/input/input_adapters.hpp | 32 +++++++++++++------ single_include/nlohmann/json.hpp | 32 +++++++++++++------ tests/src/unit-class_parser.cpp | 24 ++++++++++++++ 4 files changed, 79 insertions(+), 28 deletions(-) diff --git a/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md b/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md index f22395d3b..3bad2dea1 100644 --- a/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md +++ b/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md @@ -16,7 +16,8 @@ byte is still not rejected: [`from_msgpack`](../basic_json/from_msgpack.md), [`from_ubjson`](../basic_json/from_ubjson.md)) are never affected: there, `0x00` is ordinary data. - A bare `const char*` pointer has no length of its own, so its length is still determined with `strlen()`. The first NUL byte therefore still marks the end of the input, and nothing after it is read. -- One trailing `'\0'` at the end of a `char` array (e.g., a string literal) is trimmed; see the warning below. +- One trailing `'\0'` at the end of a `char`, `wchar_t`, `char16_t`, `char32_t`, or (C++20) `char8_t` array (e.g., a + string literal) is trimmed; see the warning below. ## Default definition @@ -57,13 +58,15 @@ The default value is `0` (disabled — existing behavior is preserved). This macro must be defined **before** including ``. Defining it after the include has no effect. - Enabling it also changes how a `char` array (including a string literal, e.g. `json::parse("123")`) is read: such - an array normally carries a trailing `'\0'` contributed by the compiler, not by the source text. With this macro - enabled, that one trailing byte is trimmed if present so that parsing a string literal keeps working; every other - byte in the array - including any `'\0'` that is not the very last element - is read as real data and rejected - like any other unexpected byte. Arrays of any other element type (`unsigned char`, `std::uint8_t`, ...), as used - for CBOR or MessagePack, are never affected by this trimming; their full extent - including a genuine trailing - `0x00` - is always preserved, in both states of this macro. + Enabling it also changes how an array of a text-literal element type (`char`, `wchar_t`, `char16_t`, `char32_t`, + or, since C++20, `char8_t` - including a string literal, e.g. `json::parse("123")` or `json::parse(L"123")`) is + read: such an array normally carries a trailing `'\0'` contributed by the compiler, not by the source text. With + this macro enabled, that one trailing element is trimmed if present so that parsing a string literal keeps + working, for any of these character types; every other element in the array - including any `'\0'` that is not + the very last element - is read as real data and rejected like any other unexpected byte. Arrays of any other + element type (`unsigned char`, `std::uint8_t`, ...), as used for CBOR or MessagePack, are never affected by this + trimming; their full extent - including a genuine trailing `0x00` - is always preserved, in both states of this + macro. !!! note "ABI compatibility" diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index f06713700..3608a836e 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -855,16 +855,28 @@ template auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) { #if JSON_STRICT_NUL_HANDLING - // A `char` array from string-literal initialization (e.g. json::parse("123")) - // carries a trailing '\0' contributed by the compiler, not by the source - // text; drop exactly that one byte so it is not mistaken for real trailing - // data. Every other element type (unsigned char, std::uint8_t, ...) keeps - // the full extent unconditionally, since a trailing zero byte there is - // genuine data (e.g. CBOR/MessagePack). This intentionally does not - // strlen()-scan the array (as the pointer overload above does for a - // null-delimited string): for a `char` array that is not NUL-terminated - // within its bounds, that would read past the end of the array. - if (std::is_same::type, char>::value && N > 0 && array[N - 1] == 0) + // A text-literal array from string-literal initialization (e.g. + // json::parse("123") or json::parse(L"123")) carries a trailing '\0' + // contributed by the compiler, not by the source text; drop exactly that + // one byte so it is not mistaken for real trailing data. This covers all + // character types that string literals can use: char, wchar_t, char16_t, + // char32_t, and (C++20) char8_t. Every other element type (unsigned char, + // std::uint8_t, ...) keeps the full extent unconditionally, since a + // trailing zero byte there is genuine data (e.g. CBOR/MessagePack). This + // intentionally does not strlen()-scan the array (as the pointer overload + // above does for a null-delimited string): for an array that is not + // NUL-terminated within its bounds, that would read past the end of the + // array. + using char_t = typename std::remove_cv::type; + constexpr bool is_text_literal_type = std::is_same::value + || std::is_same::value + || std::is_same::value + || std::is_same::value +#if defined(__cpp_char8_t) + || std::is_same::value +#endif + ; + if (is_text_literal_type && N > 0 && array[N - 1] == 0) { return input_adapter(array, array + N - 1); } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 3735a1925..4fadfb382 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -8411,16 +8411,28 @@ template auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) { #if JSON_STRICT_NUL_HANDLING - // A `char` array from string-literal initialization (e.g. json::parse("123")) - // carries a trailing '\0' contributed by the compiler, not by the source - // text; drop exactly that one byte so it is not mistaken for real trailing - // data. Every other element type (unsigned char, std::uint8_t, ...) keeps - // the full extent unconditionally, since a trailing zero byte there is - // genuine data (e.g. CBOR/MessagePack). This intentionally does not - // strlen()-scan the array (as the pointer overload above does for a - // null-delimited string): for a `char` array that is not NUL-terminated - // within its bounds, that would read past the end of the array. - if (std::is_same::type, char>::value && N > 0 && array[N - 1] == 0) + // A text-literal array from string-literal initialization (e.g. + // json::parse("123") or json::parse(L"123")) carries a trailing '\0' + // contributed by the compiler, not by the source text; drop exactly that + // one byte so it is not mistaken for real trailing data. This covers all + // character types that string literals can use: char, wchar_t, char16_t, + // char32_t, and (C++20) char8_t. Every other element type (unsigned char, + // std::uint8_t, ...) keeps the full extent unconditionally, since a + // trailing zero byte there is genuine data (e.g. CBOR/MessagePack). This + // intentionally does not strlen()-scan the array (as the pointer overload + // above does for a null-delimited string): for an array that is not + // NUL-terminated within its bounds, that would read past the end of the + // array. + using char_t = typename std::remove_cv::type; + constexpr bool is_text_literal_type = std::is_same::value + || std::is_same::value + || std::is_same::value + || std::is_same::value +#if defined(__cpp_char8_t) + || std::is_same::value +#endif + ; + if (is_text_literal_type && N > 0 && array[N - 1] == 0) { return input_adapter(array, array + N - 1); } diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index 3d825162e..52b49c5d9 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -645,6 +645,30 @@ TEST_CASE("parser class") // carries a compiler-appended trailing '\0') still works, // even though a NUL byte is now rejected everywhere else CHECK(json::parse("123") == json(123)); + + // regression test for issue #5658: the same holds for wide, + // UTF-16, UTF-32, and (C++20) UTF-8 string literals, whose + // compiler-appended trailing '\0' is not of type `char` + CHECK(json::parse(L"[1]") == json({1})); + CHECK(json::accept(L"[1]")); + CHECK(json::parse(u"[1]") == json({1})); + CHECK(json::accept(u"[1]")); + CHECK(json::parse(U"[1]") == json({1})); + CHECK(json::accept(U"[1]")); +#if defined(__cpp_char8_t) + CHECK(json::parse(u8"[1]") == json({1})); + CHECK(json::accept(u8"[1]")); +#endif + + // a NUL byte inside such a literal, as opposed to the single + // compiler-appended trailing one, is still rejected + { + json _; // NOLINT(readability-identifier-naming) + CHECK_THROWS_WITH_AS(_ = json::parse(L"[1\0]"), + "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing array - invalid literal; last read: '1'; expected ']'", + json::parse_error&); + CHECK_FALSE(json::accept(L"[1\0]")); + } } #endif } From 5bd766aa505d0788422412c8d9360eb04a0e2120 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:41 +0200 Subject: [PATCH 23/31] Move to_bson's binary subtype check into calc_bson_sizes (#5703) to_bson() rejected a binary value's subtype above 255 (out_of_range.415) in write_bson_binary(), which only has the binary_t, not the basic_json value that holds it, so the exception was created with no JSON_DIAGNOSTICS context even though the equivalent to_msgpack() check names the value's path. The check also ran after the document size, all preceding elements, and this element's header and length had already reached the output adapter, so a caller-provided std::vector or std::string ended up holding a truncated document. calc_bson_sizes() already walks every value before anything is written, to size embedded documents and arrays and to reject invalid keys (out_of_range.409) up front. The subtype check now runs there instead, in calc_bson_binary_size(), which is given the basic_json value so the exception can use it as context. The now-redundant check in write_bson_binary() is removed, since calc_bson_sizes() always throws first if any binary value in the document has an oversized subtype. Fixes #5675. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/to_bson.md | 4 +++ .../nlohmann/detail/output/binary_writer.hpp | 26 +++++++++++++------ single_include/nlohmann/json.hpp | 26 +++++++++++++------ tests/src/unit-bson.cpp | 21 ++++++++++++++- tests/src/unit-diagnostics.cpp | 6 +++++ 5 files changed, 66 insertions(+), 17 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/to_bson.md b/docs/mkdocs/docs/api/basic_json/to_bson.md index fb02c51e1..5ba3c8bb4 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bson.md +++ b/docs/mkdocs/docs/api/basic_json/to_bson.md @@ -43,6 +43,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`out_of_range.412`](../../home/exceptions.md#jsonexceptionout_of_range412) if the length of a document, array, string, or binary value exceeds the range of the 32-bit BSON length field; example: `"BSON length 2147483661 exceeds maximum of 2147483647"` +- Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value + exceeds 255, the maximum of the BSON binary subtype; example: + `"subtype 70000 is too large for the BSON binary subtype (max 255)"` ## Complexity @@ -78,3 +81,4 @@ pass before anything is written. - Added in version 3.4.0. - Linear in the size of `j`, and no longer limited by the call stack for deeply nested values, since version 3.13.0. +- `out_of_range.415` is now detected before anything is written, like the other exceptions above, since version 3.13.0. diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 4f39f0537..1ebd3c8ab 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -1056,15 +1056,26 @@ class binary_writer } /*! - @return The size of the BSON-encoded binary array @a value + @return The size of the BSON-encoded binary array in @a j + @throw out_of_range.415 if the subtype of @a j does not fit into a byte, + before anything is written */ - static std::size_t calc_bson_binary_size(const typename BasicJsonType::binary_t& value) + static std::size_t calc_bson_binary_size(const BasicJsonType& j) { + const auto& value = *j.m_data.m_value.binary; + + if (value.has_subtype() && JSON_HEDLEY_UNLIKELY(value.subtype() > (std::numeric_limits::max)())) + { + JSON_THROW(out_of_range::create(415, concat("subtype ", std::to_string(value.subtype()), " is too large for the BSON binary subtype (max 255)"), &j)); + } + return sizeof(std::int32_t) + value.size() + 1ul; } /*! @brief Writes a BSON element with key @a name and binary value @a value + @pre @a value's subtype, if any, fits into a byte; @ref calc_bson_sizes + checks this for every binary value in the document beforehand. */ void write_bson_binary(const string_t& name, const binary_t& value) @@ -1073,11 +1084,6 @@ class binary_writer write_number(to_bson_length(value.size()), true); - if (value.has_subtype() && JSON_HEDLEY_UNLIKELY(value.subtype() > (std::numeric_limits::max)())) - { - JSON_THROW(out_of_range::create(415, concat("subtype ", std::to_string(value.subtype()), " is too large for the BSON binary subtype (max 255)"), nullptr)); - } - write_number(value.has_subtype() ? static_cast(value.subtype()) : static_cast(0x00)); oa.write_characters(reinterpret_cast(value.data()), value.size()); @@ -1086,13 +1092,15 @@ class binary_writer /*! @return The size of the value of the BSON document entry for @a j, which is neither an object nor an array + @throw out_of_range.415 if @a j is binary with a subtype that does not fit + into a byte, before anything is written */ static std::size_t calc_bson_value_size(const BasicJsonType& j) { switch (j.type()) { case value_t::binary: - return calc_bson_binary_size(*j.m_data.m_value.binary); + return calc_bson_binary_size(j); case value_t::boolean: return 1ul; @@ -1218,6 +1226,8 @@ class binary_writer @return the size of @a document @throw out_of_range.409 if a key contains U+0000, before anything is written + @throw out_of_range.415 if a binary value's subtype does not fit into a + byte, before anything is written */ static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4fadfb382..02065bdf8 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -22153,15 +22153,26 @@ class binary_writer } /*! - @return The size of the BSON-encoded binary array @a value + @return The size of the BSON-encoded binary array in @a j + @throw out_of_range.415 if the subtype of @a j does not fit into a byte, + before anything is written */ - static std::size_t calc_bson_binary_size(const typename BasicJsonType::binary_t& value) + static std::size_t calc_bson_binary_size(const BasicJsonType& j) { + const auto& value = *j.m_data.m_value.binary; + + if (value.has_subtype() && JSON_HEDLEY_UNLIKELY(value.subtype() > (std::numeric_limits::max)())) + { + JSON_THROW(out_of_range::create(415, concat("subtype ", std::to_string(value.subtype()), " is too large for the BSON binary subtype (max 255)"), &j)); + } + return sizeof(std::int32_t) + value.size() + 1ul; } /*! @brief Writes a BSON element with key @a name and binary value @a value + @pre @a value's subtype, if any, fits into a byte; @ref calc_bson_sizes + checks this for every binary value in the document beforehand. */ void write_bson_binary(const string_t& name, const binary_t& value) @@ -22170,11 +22181,6 @@ class binary_writer write_number(to_bson_length(value.size()), true); - if (value.has_subtype() && JSON_HEDLEY_UNLIKELY(value.subtype() > (std::numeric_limits::max)())) - { - JSON_THROW(out_of_range::create(415, concat("subtype ", std::to_string(value.subtype()), " is too large for the BSON binary subtype (max 255)"), nullptr)); - } - write_number(value.has_subtype() ? static_cast(value.subtype()) : static_cast(0x00)); oa.write_characters(reinterpret_cast(value.data()), value.size()); @@ -22183,13 +22189,15 @@ class binary_writer /*! @return The size of the value of the BSON document entry for @a j, which is neither an object nor an array + @throw out_of_range.415 if @a j is binary with a subtype that does not fit + into a byte, before anything is written */ static std::size_t calc_bson_value_size(const BasicJsonType& j) { switch (j.type()) { case value_t::binary: - return calc_bson_binary_size(*j.m_data.m_value.binary); + return calc_bson_binary_size(j); case value_t::boolean: return 1ul; @@ -22315,6 +22323,8 @@ class binary_writer @return the size of @a document @throw out_of_range.409 if a key contains U+0000, before anything is written + @throw out_of_range.415 if a binary value's subtype does not fit into a + byte, before anything is written */ static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index 292ff4dc1..5bfaac7c2 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -797,7 +797,11 @@ TEST_CASE("regression test - BSON binary subtype rejects a value that doesn't fi CHECK(json::from_bson(json::to_bson(doc255))["b"].get_binary().subtype() == 255); CHECK_THROWS_AS(json::to_bson(json{{"b", json::binary({1, 2}, 256)}}), json::out_of_range); - CHECK_THROWS_WITH_AS(json::to_bson(json{{"b", json::binary({1, 2}, 300)}}), "[json.exception.out_of_range.415] subtype 300 is too large for the BSON binary subtype (max 255)", json::out_of_range); +#if JSON_DIAGNOSTICS + CHECK_THROWS_WITH_AS(json::to_bson(json {{"b", json::binary({1, 2}, 300)}}), "[json.exception.out_of_range.415] (/b) subtype 300 is too large for the BSON binary subtype (max 255)", json::out_of_range); +#else + CHECK_THROWS_WITH_AS(json::to_bson(json {{"b", json::binary({1, 2}, 300)}}), "[json.exception.out_of_range.415] subtype 300 is too large for the BSON binary subtype (max 255)", json::out_of_range); +#endif } TEST_CASE("BSON input/output_adapters") @@ -1805,6 +1809,21 @@ value = depth % 2 == 0 ? json{{"a", std::move(value)}, {"b", {1, "x"}}} : CHECK(output.empty()); } + SECTION("a binary subtype that doesn't fit a byte is rejected before anything is written (#5675)") + { + // the offending value is nested, so this also covers that the check + // is not limited to a directly written value's own document + json const j = {{"a", {{"b", json::binary({1, 2}, 300)}}}}; + + std::vector vector_output; + CHECK_THROWS_AS(json::to_bson(j, vector_output), json::out_of_range&); + CHECK(vector_output.empty()); + + std::string string_output; + CHECK_THROWS_AS(json::to_bson(j, string_output), json::out_of_range&); + CHECK(string_output.empty()); + } + SECTION("values nested too deeply for the call stack (#5392)") { // serializing recursed once per nesting level, and computed every diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index a46ce5746..46f41f252 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -104,6 +104,12 @@ TEST_CASE("Regression tests for extended diagnostics") CHECK_THROWS_WITH_AS(j.unflatten(), "[json.exception.type_error.315] (/~1foo) values in object must be primitive", json::type_error); } + SECTION("Regression test for issue #5675 - to_bson: out_of_range.415 has no diagnostics context") + { + json const j = {{"a", {{"b", json::binary({1, 2}, 300)}}}}; + CHECK_THROWS_WITH_AS(json::to_bson(j), "[json.exception.out_of_range.415] (/a/b) subtype 300 is too large for the BSON binary subtype (max 255)", json::out_of_range); + } + SECTION("Regression test for issue #2838 - Assertion failure when inserting into arrays with JSON_DIAGNOSTICS set") { // void push_back(basic_json&& val) From 6ae17630a4fab15f5b456c08e3cf27dbaf1eef82 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:45 +0200 Subject: [PATCH 24/31] Reject integral keys for contains(), find(), and count() at compile time (#5705) j.contains(0), j.find(0), and j.count(0) used to compile: the literal 0 is a null pointer constant, so it converts to a null const char*, and the overloads taking const typename object_t::key_type& accepted it by constructing a std::string from that null pointer, which is undefined behavior (a crash with both libc++ and libstdc++). value(0, default_value) had the same problem in C++11, where the object comparator is not transparent. Add deleted overloads for integral arguments to contains(), find() (const and non-const), count(), and value() so that these calls are compile errors in every supported language mode instead of crashing. Calls with string, string_view, json_pointer, and size-typed element access (at(), operator[](), erase()) are unaffected. Fixes #5657. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/contains.md | 6 ++ docs/mkdocs/docs/api/basic_json/count.md | 8 +- docs/mkdocs/docs/api/basic_json/find.md | 8 +- docs/mkdocs/docs/api/basic_json/value.md | 12 ++- include/nlohmann/json.hpp | 14 ++++ single_include/nlohmann/json.hpp | 14 ++++ tests/src/unit-element_access2.cpp | 81 +++++++++++++++++++++ 7 files changed, 140 insertions(+), 3 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/contains.md b/docs/mkdocs/docs/api/basic_json/contains.md index 7f865f83c..84b3c443c 100644 --- a/docs/mkdocs/docs/api/basic_json/contains.md +++ b/docs/mkdocs/docs/api/basic_json/contains.md @@ -58,6 +58,10 @@ Logarithmic in the size of the JSON object. - This method always returns `#!cpp false` when executed on a JSON type that is not an object. - This method can be executed on any JSON value type. +- Calling this function with an integer argument (for example, `#!cpp contains(0)`) does not compile: such an argument + would otherwise implicitly convert to a null `#!cpp const char*` and, from there, cause undefined behavior when + constructing a `#!cpp std::string` for the object key. To check for an array element instead, use [`at`](at.md), + [`operator[]`](operator%5B%5D.md), or compare against [`size`](size.md). !!! info "Postconditions" @@ -117,3 +121,5 @@ Logarithmic in the size of the JSON object. 1. Added in version 3.11.0. 2. Added in version 3.6.0. Extended template `KeyType` to support comparable types in version 3.11.0. 3. Added in version 3.7.0. +4. Deleted overloads for integral key types added in version 3.13.0 to reject such calls at compile time instead of + causing undefined behavior at runtime. diff --git a/docs/mkdocs/docs/api/basic_json/count.md b/docs/mkdocs/docs/api/basic_json/count.md index ce51addbd..bffc46534 100644 --- a/docs/mkdocs/docs/api/basic_json/count.md +++ b/docs/mkdocs/docs/api/basic_json/count.md @@ -40,7 +40,11 @@ Logarithmic in the size of the JSON object. ## Notes -This method always returns `0` when executed on a JSON type that is not an object. +- This method always returns `0` when executed on a JSON type that is not an object. +- Calling this function with an integer argument (for example, `#!cpp count(0)`) does not compile: such an argument + would otherwise implicitly convert to a null `#!cpp const char*` and, from there, cause undefined behavior when + constructing a `#!cpp std::string` for the object key. To check for an array element instead, use [`at`](at.md), + [`operator[]`](operator%5B%5D.md), or compare against [`size`](size.md). ## Examples @@ -81,3 +85,5 @@ This method always returns `0` when executed on a JSON type that is not an objec 1. Added in version 3.11.0. 2. Added in version 1.0.0. Changed parameter `key` type to `KeyType&&` in version 3.11.0. +3. Deleted overload for integral key types added in version 3.13.0 to reject such calls at compile time instead of + causing undefined behavior at runtime. diff --git a/docs/mkdocs/docs/api/basic_json/find.md b/docs/mkdocs/docs/api/basic_json/find.md index 35ff9dcb2..bc746ee2f 100644 --- a/docs/mkdocs/docs/api/basic_json/find.md +++ b/docs/mkdocs/docs/api/basic_json/find.md @@ -44,7 +44,11 @@ Logarithmic in the size of the JSON object. ## Notes -This method always returns `end()` when executed on a JSON type that is not an object. +- This method always returns `end()` when executed on a JSON type that is not an object. +- Calling this function with an integer argument (for example, `#!cpp find(0)`) does not compile: such an argument + would otherwise implicitly convert to a null `#!cpp const char*` and, from there, cause undefined behavior when + constructing a `#!cpp std::string` for the object key. To access an array element instead, use [`at`](at.md), + [`operator[]`](operator%5B%5D.md), or compare against [`size`](size.md). ## Examples @@ -85,3 +89,5 @@ This method always returns `end()` when executed on a JSON type that is not an o 1. Added in version 3.11.0. 2. Added in version 1.0.0. Changed to support comparable types in version 3.11.0. +3. Deleted overloads for integral key types added in version 3.13.0 to reject such calls at compile time instead of + causing undefined behavior at runtime. diff --git a/docs/mkdocs/docs/api/basic_json/value.md b/docs/mkdocs/docs/api/basic_json/value.md index 2e9b85a2b..c0e4b9949 100644 --- a/docs/mkdocs/docs/api/basic_json/value.md +++ b/docs/mkdocs/docs/api/basic_json/value.md @@ -52,6 +52,14 @@ This is equivalent to Python's `dict.get(key, default)`. - Unlike [`operator[]`](operator[].md), this function does not implicitly add an element to the position defined by `key`/`ptr` key. This function is furthermore also applicable to const objects. +!!! note "Integer keys" + + Calling this function with an integer `key` argument (for example, `#!cpp value(0, 1)`) does not compile in + C++11, where `object_comparator_t` is not transparent: such an argument would otherwise implicitly convert to a + null `#!cpp const char*` and, from there, cause undefined behavior when constructing a `#!cpp std::string` for the + object key. To access an array element with a default value, use [`at`](at.md) together with a `#!cpp try`/`#!cpp + catch` block, or compare against [`size`](size.md) instead. + ## Template parameters `KeyType` @@ -184,7 +192,9 @@ changes to any JSON value. ## Version history -1. Added in version 1.0.0. Changed parameter `default_value` type from `const ValueType&` to `ValueType&&` in version 3.11.0. +1. Added in version 1.0.0. Changed parameter `default_value` type from `const ValueType&` to `ValueType&&` in version + 3.11.0. Deleted overload for integral key types added in version 3.13.0 to reject such calls at compile time + instead of causing undefined behavior at runtime. 2. Added in version 3.11.0. Made `ValueType` the first template parameter in version 3.11.2. 3. Added in version 2.0.2. Extended to work with arrays in version 3.13.0, including fixing an issue where resolving `ptr` through an array unexpectedly threw `out_of_range` instead of returning the resolved element (or diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 13274f902..f394af567 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -2980,6 +2980,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec string_t, typename std::decay::type >; public: + // an integer literal 0 would otherwise convert to a null const char* and from there to key_type + template::value, int> = 0> + ValueType value(T, ValueType&&) const = delete; + /// @brief access specified object element with default value /// @sa https://json.nlohmann.me/api/basic_json/value/ template < class ValueType, detail::enable_if_t < @@ -3414,6 +3418,16 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @name lookup /// @{ + // an integer literal 0 would otherwise convert to a null const char* and from there to key_type + template::value, int> = 0> + iterator find(T) = delete; + template::value, int> = 0> + const_iterator find(T) const = delete; + template::value, int> = 0> + size_type count(T) const = delete; + template::value, int> = 0> + bool contains(T) const = delete; + /// @brief find an element in a JSON object /// @sa https://json.nlohmann.me/api/basic_json/find/ iterator find(const typename object_t::key_type& key) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 02065bdf8..8d48e643a 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -29897,6 +29897,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec string_t, typename std::decay::type >; public: + // an integer literal 0 would otherwise convert to a null const char* and from there to key_type + template::value, int> = 0> + ValueType value(T, ValueType&&) const = delete; + /// @brief access specified object element with default value /// @sa https://json.nlohmann.me/api/basic_json/value/ template < class ValueType, detail::enable_if_t < @@ -30331,6 +30335,16 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @name lookup /// @{ + // an integer literal 0 would otherwise convert to a null const char* and from there to key_type + template::value, int> = 0> + iterator find(T) = delete; + template::value, int> = 0> + const_iterator find(T) const = delete; + template::value, int> = 0> + size_type count(T) const = delete; + template::value, int> = 0> + bool contains(T) const = delete; + /// @brief find an element in a JSON object /// @sa https://json.nlohmann.me/api/basic_json/find/ iterator find(const typename object_t::key_type& key) diff --git a/tests/src/unit-element_access2.cpp b/tests/src/unit-element_access2.cpp index 40e8216d5..04974c251 100644 --- a/tests/src/unit-element_access2.cpp +++ b/tests/src/unit-element_access2.cpp @@ -16,6 +16,40 @@ // build test with C++14 // JSON_HAS_CPP_14 +// used to check at compile time (via is_detected) whether a call is well-formed; see +// https://github.com/nlohmann/json/issues/5657 +// +// note: an integer *literal* (rather than a std::declval() of integral type T) is required to +// reproduce the bug, because only a null pointer constant--an integer literal with value zero, not +// merely a runtime value that happens to be zero--implicitly converts to a null const char*; that is +// why can_call_*_with_0 below hard-code the literal 0 instead of taking it as a template argument +template +using can_call_find = decltype(std::declval().find(std::declval())); + +template +using can_call_count = decltype(std::declval().count(std::declval())); + +template +using can_call_contains = decltype(std::declval().contains(std::declval())); + +template +using can_call_value = decltype(std::declval().value(std::declval(), std::declval())); + +template +using can_call_find_with_0 = decltype(std::declval().find(0)); + +template +using can_call_count_with_0 = decltype(std::declval().count(0)); + +template +using can_call_contains_with_0 = decltype(std::declval().contains(0)); + +template +using can_call_contains_with_0L = decltype(std::declval().contains(0L)); + +template +using can_call_value_with_0 = decltype(std::declval().value(0, 1)); + TEST_CASE_TEMPLATE("element access 2", Json, nlohmann::json, nlohmann::ordered_json) // NOLINT(readability-math-missing-parentheses, bugprone-throwing-static-initialization) { SECTION("object") @@ -1493,6 +1527,53 @@ TEST_CASE_TEMPLATE("element access 2", Json, nlohmann::json, nlohmann::ordered_j } } } + + SECTION("integral keys for object lookup are rejected at compile time") + { + // https://github.com/nlohmann/json/issues/5657: an integer literal like 0 is a null pointer + // constant, which used to convert to a null const char* and from there--via undefined + // behavior in the std::string constructor--to key_type, so contains(0), find(0), and + // count(0) used to compile and then crash instead of failing to compile + using nlohmann::detail::is_detected; + + CHECK_FALSE(is_detected::value); + CHECK_FALSE(is_detected::value); + CHECK_FALSE(is_detected::value); + CHECK_FALSE(is_detected::value); + // value(0, ...) is only affected in C++11, where the comparator is not transparent; with a + // transparent comparator (C++14 and later), int is already rejected for lacking a + // comparison with the key type, independently of this fix + CHECK_FALSE(is_detected::value); + + // another integral literal type must be rejected as well, not just int + CHECK_FALSE(is_detected::value); + + // the valid overloads must remain callable + CHECK(is_detected::value); + CHECK(is_detected::value); + CHECK(is_detected::value); + CHECK(is_detected::value); + CHECK(is_detected::value); + CHECK(is_detected::value); + CHECK(is_detected::value); + +#ifdef JSON_HAS_CPP_17 + CHECK(is_detected::value); + CHECK(is_detected::value); + CHECK(is_detected::value); +#endif + + // the neighboring size_type overloads for array access are unaffected by the new + // integral-key overloads above (at(), operator[](), and erase() take a size_type) + Json arr = {10, 20, 30}; + const Json arr_const = arr; + CHECK(arr.at(0) == 10); + CHECK(arr_const.at(0) == 10); + CHECK(arr[0] == 10); + CHECK(arr_const[0] == 10); + arr.erase(0); + CHECK(arr.size() == 2); + } } #if !defined(JSON_NOEXCEPTION) From 7fd68957886258edeec2bd14a367f33ab56002ac Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:48 +0200 Subject: [PATCH 25/31] Hide a discarded container's content from the parser callback (#5706) When a parser callback rejects an object's or array's start event, json_sax_dom_callback_parser kept calling it for everything inside that container anyway: nested keys, values, and the start/end events of containers below it. This contradicts parser_callback_t's own documentation, which promises that discarding a container at its start event also hides its content from the callback. The same code path also kept a full copy of every key inside such a discarded container in key_stack until the whole parse finished, because the early return for values that are not stored skipped the matching pop. Filtering out a large subtree is the main reason to use a callback, so this made peak memory during the parse scale with the size of the very subtree the callback was trying to skip. Fix start_object(), start_array(), and key() so that a container whose own start event was discarded, or that is nested inside one, is never handed to the callback, and no longer pushes onto the key stacks. A container whose start event was accepted but whose key was rejected still gets its content reported, as documented ("the callback is still called for the associated value, but its return value has no further effect"); only its own bookkeeping is skipped since it will not be stored. Fixes #5643. Signed-off-by: Niels Lohmann --- .../docs/api/basic_json/parser_callback_t.md | 3 + include/nlohmann/detail/input/json_sax.hpp | 18 +++- single_include/nlohmann/json.hpp | 18 +++- tests/src/unit-class_parser.cpp | 96 +++++++++++++++++++ 4 files changed, 129 insertions(+), 6 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/parser_callback_t.md b/docs/mkdocs/docs/api/basic_json/parser_callback_t.md index da23e9bc2..c1a35afb9 100644 --- a/docs/mkdocs/docs/api/basic_json/parser_callback_t.md +++ b/docs/mkdocs/docs/api/basic_json/parser_callback_t.md @@ -100,3 +100,6 @@ the latter case, it is skipped completely, or replaced by `null` if it is the to - Added in version 1.0.0. - Fixed in version 3.13.0 to also remove discarded values from a parent object; before, discarding an array or a value stored under an object key left a discarded member behind, which made the parse result serialize to invalid JSON. +- Fixed in version 3.13.0 so that discarding an array or object at its start event also hides its content from the + callback, as documented above; before, the callback was still called for the content, and the key of every member of + a discarded object was kept in memory until the parse ended. diff --git a/include/nlohmann/detail/input/json_sax.hpp b/include/nlohmann/detail/input/json_sax.hpp index 962913610..8b8f544eb 100644 --- a/include/nlohmann/detail/input/json_sax.hpp +++ b/include/nlohmann/detail/input/json_sax.hpp @@ -582,8 +582,8 @@ class json_sax_dom_callback_parser bool start_object(std::size_t len) { - // check callback for object start - const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::object_start, discarded); + // check callback for object start; not called inside a discarded container + const bool keep = keep_stack.back() && callback(static_cast(ref_stack.size()), parse_event_t::object_start, discarded); keep_stack.push_back(keep); // the key this object will be stored under, read before handle_value() @@ -619,6 +619,18 @@ class json_sax_dom_callback_parser bool key(string_t& val) { + if (!keep_stack.back() || !ref_stack.back()) + { + // the object is not stored: the value of this key is dropped in + // handle_value() without touching the key stacks + if (keep_stack.back()) + { + BasicJsonType k = BasicJsonType(val); + static_cast(callback(static_cast(ref_stack.size()), parse_event_t::key, k)); + } + return true; + } + BasicJsonType k = BasicJsonType(val); // check callback for the key @@ -704,7 +716,7 @@ class json_sax_dom_callback_parser bool start_array(std::size_t len) { - const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::array_start, discarded); + const bool keep = keep_stack.back() && callback(static_cast(ref_stack.size()), parse_event_t::array_start, discarded); keep_stack.push_back(keep); // see start_object() diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 8d48e643a..5a4595956 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -12757,8 +12757,8 @@ class json_sax_dom_callback_parser bool start_object(std::size_t len) { - // check callback for object start - const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::object_start, discarded); + // check callback for object start; not called inside a discarded container + const bool keep = keep_stack.back() && callback(static_cast(ref_stack.size()), parse_event_t::object_start, discarded); keep_stack.push_back(keep); // the key this object will be stored under, read before handle_value() @@ -12794,6 +12794,18 @@ class json_sax_dom_callback_parser bool key(string_t& val) { + if (!keep_stack.back() || !ref_stack.back()) + { + // the object is not stored: the value of this key is dropped in + // handle_value() without touching the key stacks + if (keep_stack.back()) + { + BasicJsonType k = BasicJsonType(val); + static_cast(callback(static_cast(ref_stack.size()), parse_event_t::key, k)); + } + return true; + } + BasicJsonType k = BasicJsonType(val); // check callback for the key @@ -12879,7 +12891,7 @@ class json_sax_dom_callback_parser bool start_array(std::size_t len) { - const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::array_start, discarded); + const bool keep = keep_stack.back() && callback(static_cast(ref_stack.size()), parse_event_t::array_start, discarded); keep_stack.push_back(keep); // see start_object() diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index 52b49c5d9..2d4ff6a38 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -1990,6 +1990,102 @@ TEST_CASE("parser class") } } + SECTION("no callback for the content of a discarded container (#5643)") + { + // discarding a container at its start event must also hide + // everything inside it from the callback: none of the nested + // keys, values, or nested containers' own start/end events may + // be reported + std::vector log; + bool first = true; + const json j = json::parse(R"({"skip": {"k1": 1, "k2": [2, {"k3": 3}]}, "keep": 1})", + [&](int depth, json::parse_event_t event, json & parsed) + { + static const char* const names[] = {"object_start", "object_end", "array_start", "array_end", "key", "value"}; + log.push_back(std::to_string(depth) + " " + names[static_cast(event)] + " " + parsed.dump()); + + if (depth == 1 && event == json::parse_event_t::object_start && first) + { + // discard "skip" right at its object_start event + first = false; + return false; + } + return true; + }); + + CHECK(log == std::vector + { + "0 object_start ", + "1 key \"skip\"", + "1 object_start ", + "1 key \"keep\"", + "1 value 1", + "0 object_end {\"keep\":1}" + }); + CHECK(j == json({{"keep", 1}})); + } + + SECTION("callback still called inside a container whose key was rejected (#5643)") + { + // rejecting a key does not discard its value's container at the + // container's own start event, so the callback is still called + // for that container's content; only storing the container + // under the rejected key is skipped + // (documented for parser_callback_t: "the callback is still + // called for the associated value, but its return value has no + // further effect") + const auto record = [](std::vector& log, int depth, json::parse_event_t event, const json & parsed) + { + static const char* const names[] = {"object_start", "object_end", "array_start", "array_end", "key", "value"}; + log.push_back(std::to_string(depth) + " " + names[static_cast(event)] + " " + parsed.dump()); + }; + + std::vector log_object; + const json j_object = json::parse(R"({"skip": {"k1": 1}, "keep": 2})", + [&](int depth, json::parse_event_t event, json & parsed) + { + record(log_object, depth, event, parsed); + return !(event == json::parse_event_t::key && parsed == json("skip")); + }); + + CHECK(log_object == std::vector + { + "0 object_start ", + "1 key \"skip\"", + "1 object_start ", + "2 key \"k1\"", + "2 value 1", + "1 key \"keep\"", + "1 value 2", + "0 object_end {\"keep\":2}" + }); + CHECK(j_object == json({{"keep", 2}})); + + // same for a rejected key whose value is an array rather than an object + std::vector log_array; + const json j_array = json::parse(R"({"skip": [1, {"k1": 2}], "keep": 2})", + [&](int depth, json::parse_event_t event, json & parsed) + { + record(log_array, depth, event, parsed); + return !(event == json::parse_event_t::key && parsed == json("skip")); + }); + + CHECK(log_array == std::vector + { + "0 object_start ", + "1 key \"skip\"", + "1 array_start ", + "2 value 1", + "2 object_start ", + "3 key \"k1\"", + "3 value 2", + "1 key \"keep\"", + "1 value 2", + "0 object_end {\"keep\":2}" + }); + CHECK(j_array == json({{"keep", 2}})); + } + SECTION("special cases") { // the following test cases cover the situation in which an empty From 66877675b1fca34f009b3f11db786ef62c5071c7 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:52 +0200 Subject: [PATCH 26/31] Check the iterator range for binary values in basic_json(first, last) (#5719) basic_json(first, last) treated value_t::binary like the structured types (array, object) in the range check, so it always copied the whole binary value regardless of the iterators, even for an empty range such as (b.end(), b.end()). The other primitive types (number, boolean, string) already reject such a range with invalid_iterator.204, and erase(first, last) already does the same for binary values, so this made the constructor inconsistent with both. Move case value_t::binary into the group of checked primitive types. Also update the two matching passages in basic_json.md that describe overload 7, and add a version-history note. Fixes #5670. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/basic_json.md | 12 +++++++----- include/nlohmann/json.hpp | 2 +- single_include/nlohmann/json.hpp | 2 +- tests/src/unit-constructor1.cpp | 14 ++++++++++++++ 4 files changed, 23 insertions(+), 7 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/basic_json.md b/docs/mkdocs/docs/api/basic_json/basic_json.md index 6aaab23c9..3a7a6eb8b 100644 --- a/docs/mkdocs/docs/api/basic_json/basic_json.md +++ b/docs/mkdocs/docs/api/basic_json/basic_json.md @@ -139,8 +139,8 @@ basic_json(basic_json&& other) noexcept; - In case of a `#!json null` type, [invalid_iterator.206](../../home/exceptions.md#jsonexceptioninvalid_iterator206) is thrown. - - In case of other primitive types (number, boolean, or string), `first` must be `begin()` and `last` must be - `end()`. In this case, the value is copied. Otherwise, + - In case of other primitive types (number, boolean, string, or binary), `first` must be `begin()` and `last` + must be `end()`. In this case, the value is copied. Otherwise, [`invalid_iterator.204`](../../home/exceptions.md#jsonexceptioninvalid_iterator204) is thrown. - In case of structured types (array, object), the constructor behaves as similar versions for `std::vector` or `std::map`; that is, a JSON array or object is constructed from the values in the range. @@ -242,8 +242,8 @@ basic_json(basic_json&& other) noexcept; and `last` are not compatible (i.e., do not belong to the same JSON value). In this case, the range `[first, last)` is undefined. - Throws [`invalid_iterator.204`](../../home/exceptions.md#jsonexceptioninvalid_iterator204) if iterators `first` - and `last` belong to a primitive type (number, boolean, or string), but `first` does not point to the first - element anymore. In this case, the range `[first, last)` is undefined. See the example code below. + and `last` belong to a primitive type (number, boolean, string, or binary), but `first` does not point to the + first element anymore. In this case, the range `[first, last)` is undefined. See the example code below. - Throws [`invalid_iterator.206`](../../home/exceptions.md#jsonexceptioninvalid_iterator206) if iterators `first` and `last` belong to a `#!json null` value. In this case, the range `[first, last)` is undefined. 8. (none) @@ -423,6 +423,8 @@ basic_json(basic_json&& other) noexcept; 4. Since version 3.2.0. 5. Since version 1.0.0. 6. Since version 1.0.0. -7. Since version 1.0.0. +7. Since version 1.0.0. Fixed in version 3.13.0 to also check the iterator range for binary values; before, a range + that did not cover the whole value (such as `(end(), end())`) was accepted and the whole binary value was copied, + unlike the other primitive types. 8. Since version 1.0.0. 9. Since version 1.0.0. diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index f394af567..cd731c290 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -1827,6 +1827,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::number_integer: case value_t::number_unsigned: case value_t::string: + case value_t::binary: { if (JSON_HEDLEY_UNLIKELY(!first.m_it.primitive_iterator.is_begin() || !last.m_it.primitive_iterator.is_end())) @@ -1839,7 +1840,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::null: case value_t::object: case value_t::array: - case value_t::binary: case value_t::discarded: default: break; diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 5a4595956..c54656207 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -28756,6 +28756,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::number_integer: case value_t::number_unsigned: case value_t::string: + case value_t::binary: { if (JSON_HEDLEY_UNLIKELY(!first.m_it.primitive_iterator.is_begin() || !last.m_it.primitive_iterator.is_end())) @@ -28768,7 +28769,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::null: case value_t::object: case value_t::array: - case value_t::binary: case value_t::discarded: default: break; diff --git a/tests/src/unit-constructor1.cpp b/tests/src/unit-constructor1.cpp index 631d1a212..600eb0069 100644 --- a/tests/src/unit-constructor1.cpp +++ b/tests/src/unit-constructor1.cpp @@ -1648,6 +1648,20 @@ TEST_CASE("constructors") CHECK_THROWS_WITH_AS(json(j.cbegin(), j.cbegin()), "[json.exception.invalid_iterator.204] iterators out of range", json::invalid_iterator&); } } + + SECTION("binary") + { + { + json j = json::binary({1, 2, 3}); + CHECK_THROWS_WITH_AS(json(j.end(), j.end()), "[json.exception.invalid_iterator.204] iterators out of range", json::invalid_iterator&); + CHECK_THROWS_WITH_AS(json(j.begin(), j.begin()), "[json.exception.invalid_iterator.204] iterators out of range", json::invalid_iterator&); + } + { + json const j = json::binary({1, 2, 3}); + CHECK_THROWS_WITH_AS(json(j.cend(), j.cend()), "[json.exception.invalid_iterator.204] iterators out of range", json::invalid_iterator&); + CHECK_THROWS_WITH_AS(json(j.cbegin(), j.cbegin()), "[json.exception.invalid_iterator.204] iterators out of range", json::invalid_iterator&); + } + } } } } From 2ea6d8c127b73453237d842e5cbe0b904bd02344 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:07:56 +0200 Subject: [PATCH 27/31] Require the found key to equal the looked-up key when comparing objects (#5720) For an object type whose comparator treats unequal keys as equivalent (for example a std::map with a case-insensitive comparator), compare_iteratively() looked up a mismatched left key in the right object with find(), which uses the object's own comparator, and accepted whatever entry it found without checking that the keys are actually equal. A case-insensitive comparator then found "KEY" for "key", so two objects nested past the recursion bound (or at every depth with JSON_NO_THREAD_LOCAL) could compare equal even though the object type's own operator== - and basic_json itself, below the bound - consider them different. Accept the found entry only if its key equals (not just compares equivalent to) the looked-up key. Fixes #5655. Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 4 ++- single_include/nlohmann/json.hpp | 4 ++- tests/src/unit-comparison.cpp | 46 ++++++++++++++++++++++++++++++++ 3 files changed, 52 insertions(+), 2 deletions(-) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index cd731c290..99ec53585 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -1505,7 +1505,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const auto found = (!Ordered && !detail::is_ordered_map::value) ? rhs_object->find(current.lhs_object_it->first) : rhs_object->cend(); - if (found == rhs_object->cend()) + // the object's comparator may find an entry whose + // key is only equivalent, not equal, to this one + if (found == rhs_object->cend() || !(found->first == current.lhs_object_it->first)) { return key_result; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index c54656207..913e9937b 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -28434,7 +28434,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const auto found = (!Ordered && !detail::is_ordered_map::value) ? rhs_object->find(current.lhs_object_it->first) : rhs_object->cend(); - if (found == rhs_object->cend()) + // the object's comparator may find an entry whose + // key is only equivalent, not equal, to this one + if (found == rhs_object->cend() || !(found->first == current.lhs_object_it->first)) { return key_result; } diff --git a/tests/src/unit-comparison.cpp b/tests/src/unit-comparison.cpp index 69c0103c9..6dcec8390 100644 --- a/tests/src/unit-comparison.cpp +++ b/tests/src/unit-comparison.cpp @@ -17,6 +17,7 @@ #include +#include #include #include #include @@ -825,6 +826,24 @@ Json nest(Json j, const std::size_t depth) } return j; } + +// orders keys case-insensitively, so "key" and "KEY" compare equivalent +// (neither less than the other) although they are not equal +struct case_insensitive_less +{ + bool operator()(const std::string& a, const std::string& b) const + { + return std::lexicographical_compare(a.begin(), a.end(), b.begin(), b.end(), + [](unsigned char x, unsigned char y) + { + return std::tolower(x) < std::tolower(y); + }); + } +}; + +template +using case_insensitive_map = std::map; +using ci_json = nlohmann::basic_json; } // namespace TEST_CASE("equality of objects whose entries have no fixed order") @@ -872,6 +891,33 @@ TEST_CASE("equality of objects whose entries have no fixed order") } } +TEST_CASE("equality of an object whose comparator treats different keys as equivalent") +{ + // https://github.com/nlohmann/json/issues/5655: past the nesting bound, + // the entries are compared without the call stack, and a key that finds + // no counterpart at the same position is looked up with find(), which + // uses the object's own comparator. A case-insensitive comparator then + // finds "KEY" for "key" and must not accept that pair as a match - the + // object type's own operator==, like std::map's, compares keys with ==. + ci_json a = ci_json::object(); + a["key"] = 1; + ci_json b = ci_json::object(); + b["KEY"] = 1; + + // sanity check: the object type's own comparison already disagrees + CHECK_FALSE(a.get_ref() == b.get_ref()); + + for (const std::size_t depth : std::vector {0, 127, 128, 200}) + { + CAPTURE(depth); + + const ci_json x = nest(a, depth); + const ci_json y = nest(b, depth); + CHECK_FALSE(x == y); + CHECK(x != y); + } +} + TEST_CASE("containers are compared element by element") { // Containers nested deeper than a bound are compared without the call From 67435c9c7eca1b975116de3d1e0b63edb50a4e9a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:08:00 +0200 Subject: [PATCH 28/31] Give a deep copy its type only after its container exists (#5721) When copying a value nested deeper than 128 levels, and an allocation fails while an inner array or object is being copied, the partially built copy ended up with an element typed array/object but holding a null pointer. That element was already a fully constructed member of its parent's container, so destroying the parent during stack unwinding dereferenced the null pointer (release builds) or failed assert_invariant() (debug builds), instead of letting std::bad_alloc reach the caller. copy_iteratively() set a pending worklist element's type right after popping it, before the next loop iteration created its container in copy_array_level()/copy_object_level(). Move that type assignment into those two functions, right after the container is successfully created, and drop the premature one in copy_iteratively(), so a half-built element stays a null value - as copy_shallow()'s comment already promised - until it can safely hold one. Add a regression test to tests/src/unit-allocator.cpp that copies a value nested 130 levels deep (both arrays and objects, with a std::map- and an ordered_map-backed object_t) and fails every allocation of the copy in turn: each attempt must throw std::bad_alloc without crashing, and the source must stay unchanged. Fixes #5640. Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 7 +- single_include/nlohmann/json.hpp | 7 +- tests/src/unit-allocator.cpp | 127 +++++++++++++++++++++++++++++++ 3 files changed, 135 insertions(+), 6 deletions(-) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 99ec53585..03574f5bf 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -1116,6 +1116,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // provides the latter (e.g., ones without a matching allocator-aware // fill constructor) dst.m_data.m_value.array = create(); + // only now that the array exists may dst stop being a null value + dst.m_data.m_type = value_t::array; dst.m_data.m_value.array->resize(src_array.size()); auto dst_it = dst.m_data.m_value.array->begin(); @@ -1144,6 +1146,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec dst.m_data.m_value.object = create(std::make_move_iterator(scratch.begin()), std::make_move_iterator(scratch.end())); + // only now that the object exists may dst stop being a null value + dst.m_data.m_type = value_t::object; scratch.clear(); // pair every value of the copy with its counterpart in the original; @@ -1206,9 +1210,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec src_value = next.first; dst_value = next.second; worklist.pop_back(); - - // the value stops being a null value exactly here - dst_value->m_data.m_type = src_value->m_data.m_type; } } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 913e9937b..ef98ba71d 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -28045,6 +28045,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // provides the latter (e.g., ones without a matching allocator-aware // fill constructor) dst.m_data.m_value.array = create(); + // only now that the array exists may dst stop being a null value + dst.m_data.m_type = value_t::array; dst.m_data.m_value.array->resize(src_array.size()); auto dst_it = dst.m_data.m_value.array->begin(); @@ -28073,6 +28075,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec dst.m_data.m_value.object = create(std::make_move_iterator(scratch.begin()), std::make_move_iterator(scratch.end())); + // only now that the object exists may dst stop being a null value + dst.m_data.m_type = value_t::object; scratch.clear(); // pair every value of the copy with its counterpart in the original; @@ -28135,9 +28139,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec src_value = next.first; dst_value = next.second; worklist.pop_back(); - - // the value stops being a null value exactly here - dst_value->m_data.m_type = src_value->m_data.m_type; } } diff --git a/tests/src/unit-allocator.cpp b/tests/src/unit-allocator.cpp index 9dde143c2..ce498558d 100644 --- a/tests/src/unit-allocator.cpp +++ b/tests/src/unit-allocator.cpp @@ -276,6 +276,133 @@ TEST_CASE("controlled bad_alloc") } } +namespace +{ +// counts every allocation made on behalf of a basic_json value (of its own +// object_t/array_t/string_t/binary_t or of its own type), and can be told to +// fail one of them: the n-th call to allocate() throws std::bad_alloc instead +// of allocating, whichever type it is allocating for +std::size_t alloc_call_count = 0; +long fail_at_alloc_call = -1; // -1: never fail + +template +struct nth_alloc_fails_allocator : std::allocator +{ + using std::allocator::allocator; + + T* allocate(std::size_t n) + { + const auto index = alloc_call_count++; + if (fail_at_alloc_call >= 0 && index == static_cast(fail_at_alloc_call)) + { + throw std::bad_alloc(); + } + return std::allocator::allocate(n); + } + + template + struct rebind + { + using other = nth_alloc_fails_allocator; + }; +}; + +// builds a value nested more than 128 levels deep - the bound the copy +// constructor descends into before it continues without the call stack - and +// checks that a copy survives any single allocation of it failing: every +// attempt either throws std::bad_alloc, without crashing or leaving the +// source altered, or completes the copy +template +void check_deep_copy_survives_failing_allocation(bool nest_objects) +{ + CAPTURE(nest_objects); + + fail_at_alloc_call = -1; + + // [[[ ... [1] ... ]]], or the same nesting with objects, 130 levels deep + BasicJsonType src = 1; + for (std::size_t i = 0; i < 130; ++i) + { + if (nest_objects) + { + BasicJsonType wrapper = BasicJsonType::object(); + wrapper["a"] = std::move(src); + src = std::move(wrapper); + } + else + { + src = BasicJsonType::array({std::move(src)}); + } + } + + const std::string original_dump = src.dump(); + + // first measure how many allocations an unhindered copy takes + alloc_call_count = 0; + { + // NOLINTNEXTLINE(performance-unnecessary-copy-initialization): the copy is what is measured + const BasicJsonType measure(src); + } + const std::size_t total_allocations = alloc_call_count; + REQUIRE(total_allocations > 0); + REQUIRE(src.dump() == original_dump); + + // let the 0th, 1st, 2nd, ... allocation of the copy fail in turn; every + // such copy must throw std::bad_alloc rather than crash, and the source + // must come out exactly as it went in + for (std::size_t n = 0; n < total_allocations; ++n) + { + CAPTURE(n); + alloc_call_count = 0; + fail_at_alloc_call = static_cast(n); + + CHECK_THROWS_AS(BasicJsonType(src), std::bad_alloc&); + + fail_at_alloc_call = -1; + CHECK(src.dump() == original_dump); + } + + // once no allocation is made to fail, the copy itself must succeed + fail_at_alloc_call = -1; + const BasicJsonType copy(src); + CHECK(copy.dump() == original_dump); + CHECK(src.dump() == original_dump); +} +} // namespace + +TEST_CASE("copy of a deeply nested value survives a failing allocation (#5640)") +{ + SECTION("std::map-backed object_t") + { + using bad_alloc_json = nlohmann::basic_json; + + check_deep_copy_survives_failing_allocation(false); + check_deep_copy_survives_failing_allocation(true); + } + + SECTION("ordered_map-backed object_t") + { + using bad_alloc_ordered_json = nlohmann::basic_json; + + check_deep_copy_survives_failing_allocation(false); + check_deep_copy_survives_failing_allocation(true); + } +} + namespace { // counts the allocations of pairs with a non-const first member: the object From 6fc0d501f37a6e58669a718527d295a847f7b942 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:08:17 +0200 Subject: [PATCH 29/31] Move the user-defined string literals to and add JSON_NO_AUTOMATIC_UDLS (#5610) * Add JSON_NO_UDLS to leave out the user-defined string literals The bodies of operator""_json and operator""_json_pointer call the parser, so every translation unit including the library instantiates it, even if it never parses anything. Defining JSON_NO_UDLS leaves the literals out entirely, which saves 15-35% compile time for such translation units (#5294). Nothing changes if the macro is not defined. Signed-off-by: Niels Lohmann * Mention JSON_NO_UDLS in the list of exported module symbols Signed-off-by: Niels Lohmann * Move the user-defined string literals to Following the review in #5294, the literals now live in their own header instead of being removed entirely: includes it at the end unless JSON_NO_AUTOMATIC_UDLS (renamed from JSON_NO_UDLS) is defined, so a project can opt out globally and include the header only where the literals are used. The header only uses public and standard macros, because the library's internal macros are undefined at the end of json.hpp and the amalgamation inlines macro_scope.hpp only once. For the same reason, the library no longer defines and undefines JSON_USE_GLOBAL_UDLS, so a user's definition is still visible to the header. The single-header copy is identical to the multi-header one, as it only includes . The module always exports the literals. Signed-off-by: Niels Lohmann * Fix CI: include cycle, GCC 4.8 literal operator spacing, and global UDLs off in the JSON_NO_AUTOMATIC_UDLS test - Suppress clang-tidy misc-header-include-cycle on the intentional mutual include of json.hpp and json_literals.hpp. - Use operator"" _json with a space for GCC 4.8 in the test's detection aliases, as the header does. - Only test the global literal operators when JSON_USE_GLOBAL_UDLS is on. Signed-off-by: Niels Lohmann * Declare the literal operators through a local macro The GCC 4.8 spacing condition was repeated for both operator definitions and the global using-declarations. NLOHMANN_JSON_LITERAL_OPERATOR(suffix) now selects operator""##suffix or operator"" suffix in one place and is undefined at the end of json_literals.hpp. Suggested by gregmarr in review. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/workflows/check_amalgamation.yml | 1 + BUILD.bazel | 1 + Makefile | 21 ++- cmake/ci.cmake | 5 +- docs/mkdocs/docs/api/macros/index.md | 1 + .../docs/api/macros/json_no_automatic_udls.md | 68 ++++++++ .../docs/api/macros/json_use_global_udls.md | 8 +- docs/mkdocs/docs/api/operator_literal_json.md | 6 +- .../docs/api/operator_literal_json_pointer.md | 6 +- docs/mkdocs/docs/features/macros.md | 9 + docs/mkdocs/docs/features/modules.md | 3 + docs/mkdocs/docs/integration/index.md | 5 +- docs/mkdocs/mkdocs.yml | 1 + include/nlohmann/detail/macro_scope.hpp | 4 - include/nlohmann/detail/macro_unscope.hpp | 1 - include/nlohmann/json.hpp | 69 +------- include/nlohmann/json_literals.hpp | 78 +++++++++ meson.build | 1 + single_include/nlohmann/json.hpp | 154 ++++++++++-------- single_include/nlohmann/json_literals.hpp | 78 +++++++++ src/modules/json.cppm | 1 + tests/src/unit-no_automatic_udls.cpp | 108 ++++++++++++ 22 files changed, 489 insertions(+), 140 deletions(-) create mode 100644 docs/mkdocs/docs/api/macros/json_no_automatic_udls.md create mode 100644 include/nlohmann/json_literals.hpp create mode 100644 single_include/nlohmann/json_literals.hpp create mode 100644 tests/src/unit-no_automatic_udls.cpp diff --git a/.github/workflows/check_amalgamation.yml b/.github/workflows/check_amalgamation.yml index 670269cbb..35f3a57f7 100644 --- a/.github/workflows/check_amalgamation.yml +++ b/.github/workflows/check_amalgamation.yml @@ -67,6 +67,7 @@ jobs: python3 $TOOL_DIR/amalgamate.py -c $TOOL_DIR/config_json.json -s . python3 $TOOL_DIR/amalgamate.py -c $TOOL_DIR/config_json_fwd.json -s . + cp include/nlohmann/json_literals.hpp $INCLUDE_DIR/json_literals.hpp # the header list of the Bazel "json" target must match the files in include/ cmake -P cmake/scripts/gen_bazel_build_file.cmake diff --git a/BUILD.bazel b/BUILD.bazel index 899f9196d..03f73fa92 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -67,6 +67,7 @@ cc_library( "include/nlohmann/detail/value_t.hpp", "include/nlohmann/json.hpp", "include/nlohmann/json_fwd.hpp", + "include/nlohmann/json_literals.hpp", "include/nlohmann/ordered_map.hpp", "include/nlohmann/thirdparty/hedley/hedley.hpp", "include/nlohmann/thirdparty/hedley/hedley_undef.hpp", diff --git a/Makefile b/Makefile index 196f87243..ccea759af 100644 --- a/Makefile +++ b/Makefile @@ -21,6 +21,8 @@ TESTS_SRCS=$(shell find tests -type f \( -name '*.hpp' -o -name '*.cpp' -o -name # the single headers (amalgamated from the source files) AMALGAMATED_FILE=single_include/nlohmann/json.hpp AMALGAMATED_FWD_FILE=single_include/nlohmann/json_fwd.hpp +# json_literals.hpp only includes , so it is copied verbatim +AMALGAMATED_LITERALS_FILE=single_include/nlohmann/json_literals.hpp ########################################################################## @@ -29,7 +31,7 @@ AMALGAMATED_FWD_FILE=single_include/nlohmann/json_fwd.hpp # main target all: - @echo "amalgamate - amalgamate files single_include/nlohmann/json{,_fwd}.hpp from the include/nlohmann sources" + @echo "amalgamate - amalgamate files single_include/nlohmann/json{,_fwd,_literals}.hpp from the include/nlohmann sources" @echo "BUILD.bazel - regenerate the Bazel BUILD file from the include/nlohmann sources" @echo "ChangeLog.md - generate ChangeLog file" @echo "check-amalgamation - check whether sources have been amalgamated and BUILD.bazel is up to date" @@ -154,14 +156,14 @@ install_astyle: # call the Artistic Style pretty printer on all source files pretty: install_astyle - $(ASTYLE) --project=tools/astyle/.astylerc $(SRCS) $(TESTS_SRCS) $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) docs/mkdocs/docs/examples/*.cpp + $(ASTYLE) --project=tools/astyle/.astylerc $(SRCS) $(TESTS_SRCS) $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) docs/mkdocs/docs/examples/*.cpp # call the Clang-Format on all source files pretty_format: for FILE in $(SRCS) $(TESTS_SRCS) $(AMALGAMATED_FILE) docs/mkdocs/docs/examples/*.cpp; do echo $$FILE; clang-format -i $$FILE; done # create single header files and pretty print -amalgamate: $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) +amalgamate: $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) $(MAKE) pretty # call the amalgamation tool for json.hpp @@ -172,16 +174,23 @@ $(AMALGAMATED_FILE): $(SRCS) $(AMALGAMATED_FWD_FILE): $(SRCS) tools/amalgamate/amalgamate.py -c tools/amalgamate/config_json_fwd.json -s . --verbose=yes +# copy json_literals.hpp +$(AMALGAMATED_LITERALS_FILE): include/nlohmann/json_literals.hpp + cp include/nlohmann/json_literals.hpp $(AMALGAMATED_LITERALS_FILE) + # check if file single_include/nlohmann/json.hpp has been amalgamated from the nlohmann sources # Note: this target is called by Travis check-amalgamation: @mv $(AMALGAMATED_FILE) $(AMALGAMATED_FILE)~ @mv $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_FWD_FILE)~ + @mv $(AMALGAMATED_LITERALS_FILE) $(AMALGAMATED_LITERALS_FILE)~ @$(MAKE) amalgamate @diff $(AMALGAMATED_FILE) $(AMALGAMATED_FILE)~ || (echo "===================================================================\n Amalgamation required! Please read the contribution guidelines\n in file .github/CONTRIBUTING.md.\n===================================================================" ; mv $(AMALGAMATED_FILE)~ $(AMALGAMATED_FILE) ; false) @diff $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_FWD_FILE)~ || (echo "===================================================================\n Amalgamation required! Please read the contribution guidelines\n in file .github/CONTRIBUTING.md.\n===================================================================" ; mv $(AMALGAMATED_FWD_FILE)~ $(AMALGAMATED_FWD_FILE) ; false) + @diff $(AMALGAMATED_LITERALS_FILE) $(AMALGAMATED_LITERALS_FILE)~ || (echo "===================================================================\n Amalgamation required! Please read the contribution guidelines\n in file .github/CONTRIBUTING.md.\n===================================================================" ; mv $(AMALGAMATED_LITERALS_FILE)~ $(AMALGAMATED_LITERALS_FILE) ; false) @mv $(AMALGAMATED_FILE)~ $(AMALGAMATED_FILE) @mv $(AMALGAMATED_FWD_FILE)~ $(AMALGAMATED_FWD_FILE) + @mv $(AMALGAMATED_LITERALS_FILE)~ $(AMALGAMATED_LITERALS_FILE) @mv BUILD.bazel BUILD.bazel~ @$(MAKE) BUILD.bazel @diff BUILD.bazel BUILD.bazel~ || (echo "===================================================================\n BUILD.bazel is out of date! Please run 'make BUILD.bazel'.\n===================================================================" ; mv BUILD.bazel~ BUILD.bazel ; false) @@ -222,7 +231,7 @@ json.tar.xz: # We use `-X` to make the resulting ZIP file reproducible, see # . include.zip: BUILD.bazel - zip -9 --recurse-paths -X include.zip $(SRCS) $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) BUILD.bazel MODULE.bazel meson.build LICENSE.MIT + zip -9 --recurse-paths -X include.zip $(SRCS) $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) BUILD.bazel MODULE.bazel meson.build LICENSE.MIT # Create the files for a release and add signatures and hashes. release: include.zip json.tar.xz @@ -231,10 +240,12 @@ release: include.zip json.tar.xz gpg --armor --detach-sig include.zip gpg --armor --detach-sig $(AMALGAMATED_FILE) gpg --armor --detach-sig $(AMALGAMATED_FWD_FILE) + gpg --armor --detach-sig $(AMALGAMATED_LITERALS_FILE) gpg --armor --detach-sig json.tar.xz cp $(AMALGAMATED_FILE) release_files cp $(AMALGAMATED_FWD_FILE) release_files - mv $(AMALGAMATED_FILE).asc $(AMALGAMATED_FWD_FILE).asc json.tar.xz json.tar.xz.asc include.zip include.zip.asc release_files + cp $(AMALGAMATED_LITERALS_FILE) release_files + mv $(AMALGAMATED_FILE).asc $(AMALGAMATED_FWD_FILE).asc $(AMALGAMATED_LITERALS_FILE).asc json.tar.xz json.tar.xz.asc include.zip include.zip.asc release_files cd release_files ; shasum -a 256 json.hpp include.zip json.tar.xz > hashes.txt diff --git a/cmake/ci.cmake b/cmake/ci.cmake index f7695c36e..a001a05ae 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -373,9 +373,10 @@ file(GLOB_RECURSE INDENT_FILES set(include_dir ${PROJECT_SOURCE_DIR}/single_include/nlohmann) set(tool_dir ${PROJECT_SOURCE_DIR}/tools/amalgamate) add_custom_target(ci_test_amalgamation - COMMAND rm -fr ${include_dir}/json.hpp~ ${include_dir}/json_fwd.hpp~ + COMMAND rm -fr ${include_dir}/json.hpp~ ${include_dir}/json_fwd.hpp~ ${include_dir}/json_literals.hpp~ COMMAND cp ${include_dir}/json.hpp ${include_dir}/json.hpp~ COMMAND cp ${include_dir}/json_fwd.hpp ${include_dir}/json_fwd.hpp~ + COMMAND cp ${include_dir}/json_literals.hpp ${include_dir}/json_literals.hpp~ COMMAND ${Python3_EXECUTABLE} -mvenv venv_astyle COMMAND venv_astyle/bin/pip3 --quiet install -r ${CMAKE_SOURCE_DIR}/tools/astyle/requirements.txt @@ -383,10 +384,12 @@ add_custom_target(ci_test_amalgamation COMMAND ${Python3_EXECUTABLE} ${tool_dir}/amalgamate.py -c ${tool_dir}/config_json.json -s . COMMAND ${Python3_EXECUTABLE} ${tool_dir}/amalgamate.py -c ${tool_dir}/config_json_fwd.json -s . + COMMAND cp ${PROJECT_SOURCE_DIR}/include/nlohmann/json_literals.hpp ${include_dir}/json_literals.hpp COMMAND venv_astyle/bin/astyle --project=tools/astyle/.astylerc --suffix=none ${include_dir}/json.hpp ${include_dir}/json_fwd.hpp COMMAND diff ${include_dir}/json.hpp~ ${include_dir}/json.hpp COMMAND diff ${include_dir}/json_fwd.hpp~ ${include_dir}/json_fwd.hpp + COMMAND diff ${include_dir}/json_literals.hpp~ ${include_dir}/json_literals.hpp COMMAND venv_astyle/bin/astyle --project=tools/astyle/.astylerc --suffix=orig ${INDENT_FILES} COMMAND for FILE in `find . -name '*.orig'`\; do false \; done diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index bf773b5c4..9d2636eb2 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -28,6 +28,7 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_HAS_RANGES**](json_has_ranges.md) - control `std::ranges` support - [**JSON_HAS_STD_FORMAT**](json_has_std_format.md) - control `std::format`/`std::formatter` support - [**JSON_HAS_THREE_WAY_COMPARISON**](json_has_three_way_comparison.md) - control 3-way comparison support +- [**JSON_NO_AUTOMATIC_UDLS**](json_no_automatic_udls.md) - do not include the user-defined string literals (UDLs) automatically - [**JSON_NO_IO**](json_no_io.md) - switch off functions relying on certain C++ I/O headers - [**JSON_NO_THREAD_LOCAL**](json_no_thread_local.md) - switch off the use of `thread_local` storage - [**JSON_SKIP_UNSUPPORTED_COMPILER_CHECK**](json_skip_unsupported_compiler_check.md) - do not warn about unsupported compilers diff --git a/docs/mkdocs/docs/api/macros/json_no_automatic_udls.md b/docs/mkdocs/docs/api/macros/json_no_automatic_udls.md new file mode 100644 index 000000000..acb6a356e --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_no_automatic_udls.md @@ -0,0 +1,68 @@ +# JSON_NO_AUTOMATIC_UDLS + +```cpp +#define JSON_NO_AUTOMATIC_UDLS +``` + +When defined, `` does not include ``, so the user-defined string +literals [`operator""_json`](../operator_literal_json.md) and +[`operator""_json_pointer`](../operator_literal_json_pointer.md) are not declared. Include +`` in the files that use them. + +The literals are ordinary inline functions whose bodies call the parser, so every translation unit that includes them +instantiates the parser — even if it never parses anything itself. Defining `JSON_NO_AUTOMATIC_UDLS` for a whole project +avoids this cost in translation units that do not parse (e.g., ones that only define types and conversions or pass +`json` values around) and reduces their compile time. + +## Default definition + +By default, `#!cpp JSON_NO_AUTOMATIC_UDLS` is not defined, and `` includes +``. + +```cpp +#undef JSON_NO_AUTOMATIC_UDLS +``` + +## Notes + +!!! info "Header ``" + + The header includes `` itself and places the literals according to + [`JSON_USE_GLOBAL_UDLS`](json_use_global_udls.md). It is part of the multi-header sources (`include/nlohmann`) + and of the single-header sources (`single_include/nlohmann`), next to `json.hpp`. + +!!! info "C++ modules" + + The `nlohmann.json` [module](../../features/modules.md) always exports the literals, regardless of this macro. + +## Examples + +??? example + + The code below includes the library without the literals and adds them in a single translation unit. + + ```cpp + // compiled with -DJSON_NO_AUTOMATIC_UDLS for the whole project + #include + + // this file uses the literals, so it includes them explicitly + #include + + int main() + { + auto j = R"({"foo": 42})"_json; + return j.at("/foo"_json_pointer) == 42 ? 0 : 1; + } + ``` + + Without the include of ``, the code would fail to compile. + +## See also + +- [`operator""_json`](../operator_literal_json.md) +- [`operator""_json_pointer`](../operator_literal_json_pointer.md) +- [`JSON_USE_GLOBAL_UDLS`](json_use_global_udls.md) - place user-defined string literals (UDLs) into the global namespace + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/macros/json_use_global_udls.md b/docs/mkdocs/docs/api/macros/json_use_global_udls.md index 3110d4662..0b055640e 100644 --- a/docs/mkdocs/docs/api/macros/json_use_global_udls.md +++ b/docs/mkdocs/docs/api/macros/json_use_global_udls.md @@ -15,7 +15,7 @@ The default value is `1`. #define JSON_USE_GLOBAL_UDLS 1 ``` -When the macro is not defined, the library will define it to its default value. +When the macro is not defined, the library behaves as if it were defined to its default value. ## Notes @@ -32,6 +32,11 @@ When the macro is not defined, the library will define it to its default value. [`JSON_GlobalUDLs`](../../integration/cmake.md#json_globaludls) (`ON` by default) which defines `JSON_USE_GLOBAL_UDLS` accordingly. +!!! info "Leaving out the literals" + + If [`JSON_NO_AUTOMATIC_UDLS`](json_no_automatic_udls.md) is defined, the literals are only declared where + `` is included; this macro then applies to that header. + ## Examples ??? example "Example 1: Default behavior" @@ -92,6 +97,7 @@ When the macro is not defined, the library will define it to its default value. - [`operator""_json`](../operator_literal_json.md) - [`operator""_json_pointer`](../operator_literal_json_pointer.md) +- [`JSON_NO_AUTOMATIC_UDLS`](json_no_automatic_udls.md) - do not include the user-defined string literals automatically - [:simple-cmake: JSON_GlobalUDLs](../../integration/cmake.md#json_globaludls) - CMake option to control the macro ## Version history diff --git a/docs/mkdocs/docs/api/operator_literal_json.md b/docs/mkdocs/docs/api/operator_literal_json.md index babce5799..a909d8165 100644 --- a/docs/mkdocs/docs/api/operator_literal_json.md +++ b/docs/mkdocs/docs/api/operator_literal_json.md @@ -18,7 +18,9 @@ using namespace nlohmann; ``` This is suggested to ease migration to the next major version release of the library. See -[`JSON_USE_GLOBAL_UDLS`](macros/json_use_global_udls.md#notes) for details. +[`JSON_USE_GLOBAL_UDLS`](macros/json_use_global_udls.md#notes) for details. The operator is declared in header +``, which `` includes unless +[`JSON_NO_AUTOMATIC_UDLS`](macros/json_no_automatic_udls.md) is defined. ## Parameters @@ -59,6 +61,8 @@ Linear. ## See also - [Creating JSON values](../features/creating_values.md) - the article on creating JSON values +- [JSON_NO_AUTOMATIC_UDLS](macros/json_no_automatic_udls.md) - do not include the user-defined string literals + automatically ## Version history diff --git a/docs/mkdocs/docs/api/operator_literal_json_pointer.md b/docs/mkdocs/docs/api/operator_literal_json_pointer.md index e1b729467..0494439aa 100644 --- a/docs/mkdocs/docs/api/operator_literal_json_pointer.md +++ b/docs/mkdocs/docs/api/operator_literal_json_pointer.md @@ -17,7 +17,9 @@ using namespace nlohmann::literals::json_literals; using namespace nlohmann; ``` This is suggested to ease migration to the next major version release of the library. See -[`JSON_USE_GLOBAL_UDLS`](macros/json_use_global_udls.md#notes) for details. +[`JSON_USE_GLOBAL_UDLS`](macros/json_use_global_udls.md#notes) for details. The operator is declared in header +``, which `` includes unless +[`JSON_NO_AUTOMATIC_UDLS`](macros/json_no_automatic_udls.md) is defined. ## Parameters @@ -58,6 +60,8 @@ Linear. ## See also - [json_pointer](json_pointer/index.md) - type to represent JSON Pointers +- [JSON_NO_AUTOMATIC_UDLS](macros/json_no_automatic_udls.md) - do not include the user-defined string literals + automatically ## Version history diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index 2bc8a4a5b..e4a628c3e 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -83,6 +83,15 @@ When defined, default parse and serialize functions for enums are excluded and h See [full documentation of `JSON_DISABLE_ENUM_SERIALIZATION`](../api/macros/json_disable_enum_serialization.md). +## `JSON_NO_AUTOMATIC_UDLS` + +When defined, `` does not include `` with the user-defined string literals +`operator""_json` and `operator""_json_pointer`. This reduces the compile time of translation units that do not use +them, because the literals instantiate the parser in every translation unit that includes them. Include +`` where the literals are needed. + +See [full documentation of `JSON_NO_AUTOMATIC_UDLS`](../api/macros/json_no_automatic_udls.md). + ## `JSON_NO_IO` When defined, headers ``, ``, ``, ``, and `` are not included and parse functions diff --git a/docs/mkdocs/docs/features/modules.md b/docs/mkdocs/docs/features/modules.md index 6708f1d81..32e9ba7fd 100644 --- a/docs/mkdocs/docs/features/modules.md +++ b/docs/mkdocs/docs/features/modules.md @@ -40,6 +40,9 @@ Only the following symbols are exported from `nlohmann.json`: - `nlohmann::literals::json_literals::operator""_json` - `nlohmann::literals::json_literals::operator""_json_pointer` +The module always exports the two user-defined string literals, even if +[`JSON_NO_AUTOMATIC_UDLS`](../api/macros/json_no_automatic_udls.md) is defined when building it. + Additionally, the following `nlohmann::detail` symbols are exported, solely to work around an MSVC compilation issue ([#3970](https://github.com/nlohmann/json/issues/3970)). They are implementation details, not part of the public API, and should not be used directly: diff --git a/docs/mkdocs/docs/integration/index.md b/docs/mkdocs/docs/integration/index.md index 2bbaa8604..99bf088b6 100644 --- a/docs/mkdocs/docs/integration/index.md +++ b/docs/mkdocs/docs/integration/index.md @@ -15,4 +15,7 @@ Clang). You can further use file [`single_include/nlohmann/json_fwd.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json_fwd.hpp) -for forward declarations. +for forward declarations, and file +[`single_include/nlohmann/json_literals.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json_literals.hpp) +for the user-defined string literals if you define +[`JSON_NO_AUTOMATIC_UDLS`](../api/macros/json_no_automatic_udls.md). diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 70537bd27..21fad832a 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -293,6 +293,7 @@ nav: - 'JSON_HAS_STATIC_RTTI': api/macros/json_has_static_rtti.md - 'JSON_HAS_STD_FORMAT': api/macros/json_has_std_format.md - 'JSON_HAS_THREE_WAY_COMPARISON': api/macros/json_has_three_way_comparison.md + - 'JSON_NO_AUTOMATIC_UDLS': api/macros/json_no_automatic_udls.md - 'JSON_NOEXCEPTION': api/macros/json_noexception.md - 'JSON_NO_IO': api/macros/json_no_io.md - 'JSON_NO_THREAD_LOCAL': api/macros/json_no_thread_local.md diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index 9c35cc3b7..0d524c710 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -915,7 +915,3 @@ void templated_json_throw(ExceptionType exception) #ifndef JSON_DISABLE_ENUM_SERIALIZATION #define JSON_DISABLE_ENUM_SERIALIZATION 0 #endif - -#ifndef JSON_USE_GLOBAL_UDLS - #define JSON_USE_GLOBAL_UDLS 1 -#endif diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index a73951a22..d2675d3f1 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -25,7 +25,6 @@ #undef JSON_INLINE_VARIABLE #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION -#undef JSON_USE_GLOBAL_UDLS #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 03574f5bf..cb08777bc 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -6480,55 +6480,6 @@ std::string format_as(const NLOHMANN_BASIC_JSON_TPL& j) return j.dump(); } -inline namespace literals -{ -inline namespace json_literals -{ - -/// @brief user-defined string literal for JSON values -/// @sa https://json.nlohmann.me/api/basic_json/operator_literal_json/ -JSON_HEDLEY_NON_NULL(1) -#if !defined(JSON_HEDLEY_GCC_VERSION) || JSON_HEDLEY_GCC_VERSION_CHECK(4,9,0) - inline nlohmann::json operator""_json(const char* s, std::size_t n) -#else - // GCC 4.8 requires a space between "" and suffix - inline nlohmann::json operator"" _json(const char* s, std::size_t n) -#endif -{ - return nlohmann::json::parse(s, s + n); -} - -#if defined(__cpp_char8_t) -JSON_HEDLEY_NON_NULL(1) -inline nlohmann::json operator""_json(const char8_t* s, std::size_t n) -{ - return nlohmann::json::parse(reinterpret_cast(s), - reinterpret_cast(s) + n); -} -#endif - -/// @brief user-defined string literal for JSON pointer -/// @sa https://json.nlohmann.me/api/basic_json/operator_literal_json_pointer/ -JSON_HEDLEY_NON_NULL(1) -#if !defined(JSON_HEDLEY_GCC_VERSION) || JSON_HEDLEY_GCC_VERSION_CHECK(4,9,0) - inline nlohmann::json::json_pointer operator""_json_pointer(const char* s, std::size_t n) -#else - // GCC 4.8 requires a space between "" and suffix - inline nlohmann::json::json_pointer operator"" _json_pointer(const char* s, std::size_t n) -#endif -{ - return nlohmann::json::json_pointer(std::string(s, n)); -} - -#if defined(__cpp_char8_t) -inline nlohmann::json::json_pointer operator""_json_pointer(const char8_t* s, std::size_t n) -{ - return nlohmann::json::json_pointer(std::string(reinterpret_cast(s), n)); -} -#endif - -} // namespace json_literals -} // namespace literals NLOHMANN_JSON_NAMESPACE_END /////////////////////// @@ -6663,17 +6614,6 @@ struct formatter // NOLINT(cert-dcl58-c } // namespace std -#if JSON_USE_GLOBAL_UDLS - #if !defined(JSON_HEDLEY_GCC_VERSION) || JSON_HEDLEY_GCC_VERSION_CHECK(4,9,0) - using nlohmann::literals::json_literals::operator""_json; // NOLINT(misc-unused-using-decls,google-global-names-in-headers) - using nlohmann::literals::json_literals::operator""_json_pointer; //NOLINT(misc-unused-using-decls,google-global-names-in-headers) - #else - // GCC 4.8 requires a space between "" and suffix - using nlohmann::literals::json_literals::operator"" _json; // NOLINT(misc-unused-using-decls,google-global-names-in-headers) - using nlohmann::literals::json_literals::operator"" _json_pointer; //NOLINT(misc-unused-using-decls,google-global-names-in-headers) - #endif -#endif - #include // End of GCC diagnostic pragmas for C++ modules support @@ -6681,4 +6621,13 @@ struct formatter // NOLINT(cert-dcl58-c #pragma GCC diagnostic pop #endif +// The user-defined string literals are in a separate header, because their +// bodies instantiate the parser in every translation unit that includes them. +// Define JSON_NO_AUTOMATIC_UDLS to include only +// where needed. +#ifndef JSON_NO_AUTOMATIC_UDLS + // NOLINTNEXTLINE(misc-header-include-cycle): json_literals.hpp includes this header + #include +#endif + #endif // INCLUDE_NLOHMANN_JSON_HPP_ diff --git a/include/nlohmann/json_literals.hpp b/include/nlohmann/json_literals.hpp new file mode 100644 index 000000000..a865b6c82 --- /dev/null +++ b/include/nlohmann/json_literals.hpp @@ -0,0 +1,78 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#ifndef INCLUDE_NLOHMANN_JSON_LITERALS_HPP_ +#define INCLUDE_NLOHMANN_JSON_LITERALS_HPP_ + +#include // size_t +#include // string + +// NOLINTNEXTLINE(misc-header-include-cycle): json.hpp includes this header at its end +#include + +// This header is included at the end of unless +// JSON_NO_AUTOMATIC_UDLS is defined, and can be included on its own after that. +// Either way, the library's internal macros are no longer defined here (and the +// amalgamation inlines macro_scope.hpp only once), so only standard and public +// macros may be used below. + +// declares the literal operator for the given suffix; GCC 4.8 requires a space +// between "" and the suffix, which newer compilers deprecate (CWG 2521) +#if !defined(__GNUC__) || defined(__clang__) || __GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 9) + #define NLOHMANN_JSON_LITERAL_OPERATOR(suffix) operator""##suffix +#else + #define NLOHMANN_JSON_LITERAL_OPERATOR(suffix) operator"" suffix +#endif + +NLOHMANN_JSON_NAMESPACE_BEGIN +inline namespace literals +{ +inline namespace json_literals +{ + +/// @brief user-defined string literal for JSON values +/// @sa https://json.nlohmann.me/api/operator_literal_json/ +inline nlohmann::json NLOHMANN_JSON_LITERAL_OPERATOR(_json)(const char* s, std::size_t n) +{ + return nlohmann::json::parse(s, s + n); +} + +#if defined(__cpp_char8_t) +inline nlohmann::json operator""_json(const char8_t* s, std::size_t n) +{ + return nlohmann::json::parse(reinterpret_cast(s), + reinterpret_cast(s) + n); +} +#endif + +/// @brief user-defined string literal for JSON pointer +/// @sa https://json.nlohmann.me/api/operator_literal_json_pointer/ +inline nlohmann::json::json_pointer NLOHMANN_JSON_LITERAL_OPERATOR(_json_pointer)(const char* s, std::size_t n) +{ + return nlohmann::json::json_pointer(std::string(s, n)); +} + +#if defined(__cpp_char8_t) +inline nlohmann::json::json_pointer operator""_json_pointer(const char8_t* s, std::size_t n) +{ + return nlohmann::json::json_pointer(std::string(reinterpret_cast(s), n)); +} +#endif + +} // namespace json_literals +} // namespace literals +NLOHMANN_JSON_NAMESPACE_END + +#if !defined(JSON_USE_GLOBAL_UDLS) || JSON_USE_GLOBAL_UDLS + using nlohmann::literals::json_literals::NLOHMANN_JSON_LITERAL_OPERATOR(_json); // NOLINT(misc-unused-using-decls,google-global-names-in-headers) + using nlohmann::literals::json_literals::NLOHMANN_JSON_LITERAL_OPERATOR(_json_pointer); //NOLINT(misc-unused-using-decls,google-global-names-in-headers) +#endif + +#undef NLOHMANN_JSON_LITERAL_OPERATOR + +#endif // INCLUDE_NLOHMANN_JSON_LITERALS_HPP_ diff --git a/meson.build b/meson.build index a2d6e31a2..2800957ec 100644 --- a/meson.build +++ b/meson.build @@ -15,6 +15,7 @@ nlohmann_json_multiple_headers = declare_dependency( if not meson.is_subproject() install_headers('single_include/nlohmann/json.hpp', subdir: 'nlohmann') install_headers('single_include/nlohmann/json_fwd.hpp', subdir: 'nlohmann') +install_headers('single_include/nlohmann/json_literals.hpp', subdir: 'nlohmann') pkgc = import('pkgconfig') pkgc.generate(name: 'nlohmann_json', diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index ef98ba71d..cf94c53a9 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -3329,10 +3329,6 @@ void templated_json_throw(ExceptionType exception) #define JSON_DISABLE_ENUM_SERIALIZATION 0 #endif -#ifndef JSON_USE_GLOBAL_UDLS - #define JSON_USE_GLOBAL_UDLS 1 -#endif - #if JSON_HAS_THREE_WAY_COMPARISON #include // partial_ordering #endif @@ -33409,55 +33405,6 @@ std::string format_as(const NLOHMANN_BASIC_JSON_TPL& j) return j.dump(); } -inline namespace literals -{ -inline namespace json_literals -{ - -/// @brief user-defined string literal for JSON values -/// @sa https://json.nlohmann.me/api/basic_json/operator_literal_json/ -JSON_HEDLEY_NON_NULL(1) -#if !defined(JSON_HEDLEY_GCC_VERSION) || JSON_HEDLEY_GCC_VERSION_CHECK(4,9,0) - inline nlohmann::json operator""_json(const char* s, std::size_t n) -#else - // GCC 4.8 requires a space between "" and suffix - inline nlohmann::json operator"" _json(const char* s, std::size_t n) -#endif -{ - return nlohmann::json::parse(s, s + n); -} - -#if defined(__cpp_char8_t) -JSON_HEDLEY_NON_NULL(1) -inline nlohmann::json operator""_json(const char8_t* s, std::size_t n) -{ - return nlohmann::json::parse(reinterpret_cast(s), - reinterpret_cast(s) + n); -} -#endif - -/// @brief user-defined string literal for JSON pointer -/// @sa https://json.nlohmann.me/api/basic_json/operator_literal_json_pointer/ -JSON_HEDLEY_NON_NULL(1) -#if !defined(JSON_HEDLEY_GCC_VERSION) || JSON_HEDLEY_GCC_VERSION_CHECK(4,9,0) - inline nlohmann::json::json_pointer operator""_json_pointer(const char* s, std::size_t n) -#else - // GCC 4.8 requires a space between "" and suffix - inline nlohmann::json::json_pointer operator"" _json_pointer(const char* s, std::size_t n) -#endif -{ - return nlohmann::json::json_pointer(std::string(s, n)); -} - -#if defined(__cpp_char8_t) -inline nlohmann::json::json_pointer operator""_json_pointer(const char8_t* s, std::size_t n) -{ - return nlohmann::json::json_pointer(std::string(reinterpret_cast(s), n)); -} -#endif - -} // namespace json_literals -} // namespace literals NLOHMANN_JSON_NAMESPACE_END /////////////////////// @@ -33592,17 +33539,6 @@ struct formatter // NOLINT(cert-dcl58-c } // namespace std -#if JSON_USE_GLOBAL_UDLS - #if !defined(JSON_HEDLEY_GCC_VERSION) || JSON_HEDLEY_GCC_VERSION_CHECK(4,9,0) - using nlohmann::literals::json_literals::operator""_json; // NOLINT(misc-unused-using-decls,google-global-names-in-headers) - using nlohmann::literals::json_literals::operator""_json_pointer; //NOLINT(misc-unused-using-decls,google-global-names-in-headers) - #else - // GCC 4.8 requires a space between "" and suffix - using nlohmann::literals::json_literals::operator"" _json; // NOLINT(misc-unused-using-decls,google-global-names-in-headers) - using nlohmann::literals::json_literals::operator"" _json_pointer; //NOLINT(misc-unused-using-decls,google-global-names-in-headers) - #endif -#endif - // #include // __ _____ _____ _____ // __| | __| | | | JSON for Modern C++ @@ -33631,7 +33567,6 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_INLINE_VARIABLE #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION -#undef JSON_USE_GLOBAL_UDLS #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH @@ -33824,4 +33759,93 @@ struct formatter // NOLINT(cert-dcl58-c #pragma GCC diagnostic pop #endif +// The user-defined string literals are in a separate header, because their +// bodies instantiate the parser in every translation unit that includes them. +// Define JSON_NO_AUTOMATIC_UDLS to include only +// where needed. +#ifndef JSON_NO_AUTOMATIC_UDLS +// NOLINTNEXTLINE(misc-header-include-cycle): json_literals.hpp includes this header +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#ifndef INCLUDE_NLOHMANN_JSON_LITERALS_HPP_ +#define INCLUDE_NLOHMANN_JSON_LITERALS_HPP_ + +#include // size_t +#include // string + +// NOLINTNEXTLINE(misc-header-include-cycle): json.hpp includes this header at its end +// #include + + +// This header is included at the end of unless +// JSON_NO_AUTOMATIC_UDLS is defined, and can be included on its own after that. +// Either way, the library's internal macros are no longer defined here (and the +// amalgamation inlines macro_scope.hpp only once), so only standard and public +// macros may be used below. + +// declares the literal operator for the given suffix; GCC 4.8 requires a space +// between "" and the suffix, which newer compilers deprecate (CWG 2521) +#if !defined(__GNUC__) || defined(__clang__) || __GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 9) + #define NLOHMANN_JSON_LITERAL_OPERATOR(suffix) operator""##suffix +#else + #define NLOHMANN_JSON_LITERAL_OPERATOR(suffix) operator"" suffix +#endif + +NLOHMANN_JSON_NAMESPACE_BEGIN +inline namespace literals +{ +inline namespace json_literals +{ + +/// @brief user-defined string literal for JSON values +/// @sa https://json.nlohmann.me/api/operator_literal_json/ +inline nlohmann::json NLOHMANN_JSON_LITERAL_OPERATOR(_json)(const char* s, std::size_t n) +{ + return nlohmann::json::parse(s, s + n); +} + +#if defined(__cpp_char8_t) +inline nlohmann::json operator""_json(const char8_t* s, std::size_t n) +{ + return nlohmann::json::parse(reinterpret_cast(s), + reinterpret_cast(s) + n); +} +#endif + +/// @brief user-defined string literal for JSON pointer +/// @sa https://json.nlohmann.me/api/operator_literal_json_pointer/ +inline nlohmann::json::json_pointer NLOHMANN_JSON_LITERAL_OPERATOR(_json_pointer)(const char* s, std::size_t n) +{ + return nlohmann::json::json_pointer(std::string(s, n)); +} + +#if defined(__cpp_char8_t) +inline nlohmann::json::json_pointer operator""_json_pointer(const char8_t* s, std::size_t n) +{ + return nlohmann::json::json_pointer(std::string(reinterpret_cast(s), n)); +} +#endif + +} // namespace json_literals +} // namespace literals +NLOHMANN_JSON_NAMESPACE_END + +#if !defined(JSON_USE_GLOBAL_UDLS) || JSON_USE_GLOBAL_UDLS + using nlohmann::literals::json_literals::NLOHMANN_JSON_LITERAL_OPERATOR(_json); // NOLINT(misc-unused-using-decls,google-global-names-in-headers) + using nlohmann::literals::json_literals::NLOHMANN_JSON_LITERAL_OPERATOR(_json_pointer); //NOLINT(misc-unused-using-decls,google-global-names-in-headers) +#endif + +#undef NLOHMANN_JSON_LITERAL_OPERATOR + +#endif // INCLUDE_NLOHMANN_JSON_LITERALS_HPP_ + +#endif + #endif // INCLUDE_NLOHMANN_JSON_HPP_ diff --git a/single_include/nlohmann/json_literals.hpp b/single_include/nlohmann/json_literals.hpp new file mode 100644 index 000000000..a865b6c82 --- /dev/null +++ b/single_include/nlohmann/json_literals.hpp @@ -0,0 +1,78 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#ifndef INCLUDE_NLOHMANN_JSON_LITERALS_HPP_ +#define INCLUDE_NLOHMANN_JSON_LITERALS_HPP_ + +#include // size_t +#include // string + +// NOLINTNEXTLINE(misc-header-include-cycle): json.hpp includes this header at its end +#include + +// This header is included at the end of unless +// JSON_NO_AUTOMATIC_UDLS is defined, and can be included on its own after that. +// Either way, the library's internal macros are no longer defined here (and the +// amalgamation inlines macro_scope.hpp only once), so only standard and public +// macros may be used below. + +// declares the literal operator for the given suffix; GCC 4.8 requires a space +// between "" and the suffix, which newer compilers deprecate (CWG 2521) +#if !defined(__GNUC__) || defined(__clang__) || __GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 9) + #define NLOHMANN_JSON_LITERAL_OPERATOR(suffix) operator""##suffix +#else + #define NLOHMANN_JSON_LITERAL_OPERATOR(suffix) operator"" suffix +#endif + +NLOHMANN_JSON_NAMESPACE_BEGIN +inline namespace literals +{ +inline namespace json_literals +{ + +/// @brief user-defined string literal for JSON values +/// @sa https://json.nlohmann.me/api/operator_literal_json/ +inline nlohmann::json NLOHMANN_JSON_LITERAL_OPERATOR(_json)(const char* s, std::size_t n) +{ + return nlohmann::json::parse(s, s + n); +} + +#if defined(__cpp_char8_t) +inline nlohmann::json operator""_json(const char8_t* s, std::size_t n) +{ + return nlohmann::json::parse(reinterpret_cast(s), + reinterpret_cast(s) + n); +} +#endif + +/// @brief user-defined string literal for JSON pointer +/// @sa https://json.nlohmann.me/api/operator_literal_json_pointer/ +inline nlohmann::json::json_pointer NLOHMANN_JSON_LITERAL_OPERATOR(_json_pointer)(const char* s, std::size_t n) +{ + return nlohmann::json::json_pointer(std::string(s, n)); +} + +#if defined(__cpp_char8_t) +inline nlohmann::json::json_pointer operator""_json_pointer(const char8_t* s, std::size_t n) +{ + return nlohmann::json::json_pointer(std::string(reinterpret_cast(s), n)); +} +#endif + +} // namespace json_literals +} // namespace literals +NLOHMANN_JSON_NAMESPACE_END + +#if !defined(JSON_USE_GLOBAL_UDLS) || JSON_USE_GLOBAL_UDLS + using nlohmann::literals::json_literals::NLOHMANN_JSON_LITERAL_OPERATOR(_json); // NOLINT(misc-unused-using-decls,google-global-names-in-headers) + using nlohmann::literals::json_literals::NLOHMANN_JSON_LITERAL_OPERATOR(_json_pointer); //NOLINT(misc-unused-using-decls,google-global-names-in-headers) +#endif + +#undef NLOHMANN_JSON_LITERAL_OPERATOR + +#endif // INCLUDE_NLOHMANN_JSON_LITERALS_HPP_ diff --git a/src/modules/json.cppm b/src/modules/json.cppm index 15207535f..a32b97603 100644 --- a/src/modules/json.cppm +++ b/src/modules/json.cppm @@ -18,6 +18,7 @@ module; // See: https://github.com/nlohmann/json/issues/5103 #include +#include export module nlohmann.json; diff --git a/tests/src/unit-no_automatic_udls.cpp b/tests/src/unit-no_automatic_udls.cpp new file mode 100644 index 000000000..1a708da50 --- /dev/null +++ b/tests/src/unit-no_automatic_udls.cpp @@ -0,0 +1,108 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// This translation unit checks JSON_NO_AUTOMATIC_UDLS, which keeps +// from including and thereby +// leaves out the user-defined string literals operator""_json and +// operator""_json_pointer (see #5294), and that including +// afterwards brings them back. +#define JSON_NO_AUTOMATIC_UDLS 1 + +#include "doctest_compatibility.h" + +#include +#include + +#include +using json = nlohmann::json; + +// An argument type whose associated namespace is the library namespace, so +// argument-dependent lookup of a literal operator called by its function name +// also searches the inline namespaces nlohmann::literals::json_literals. +NLOHMANN_JSON_NAMESPACE_BEGIN +struct no_automatic_udls_probe +{ + operator const char* () const // NOLINT(google-explicit-constructor,hicpp-explicit-conversions) + { + return ""; + } +}; +NLOHMANN_JSON_NAMESPACE_END + +namespace +{ +// The calls below use a dependent argument, so a literal operator that is not +// declared at all is a substitution failure rather than a hard error: lookup is +// deferred to the point of instantiation, where it considers the declarations +// visible from here (the global using-declarations of JSON_USE_GLOBAL_UDLS) plus +// argument-dependent lookup (the literals in the library namespace). +#if !defined(__GNUC__) || defined(__clang__) || __GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 9) + template + using json_udl_t = decltype(operator""_json(std::declval(), std::size_t())); + + template + using json_pointer_udl_t = decltype(operator""_json_pointer(std::declval(), std::size_t())); +#else + // GCC 4.8 requires a space between "" and suffix + template + using json_udl_t = decltype(operator"" _json(std::declval(), std::size_t())); + + template + using json_pointer_udl_t = decltype(operator"" _json_pointer(std::declval(), std::size_t())); +#endif + +template +using has_json_udl = nlohmann::detail::is_detected; + +template +using has_json_pointer_udl = nlohmann::detail::is_detected; +} // namespace + +TEST_CASE("JSON_NO_AUTOMATIC_UDLS") +{ + SECTION("literals are not declared") + { + // global namespace (JSON_USE_GLOBAL_UDLS defaults to 1) + CHECK_FALSE(has_json_udl::value); + CHECK_FALSE(has_json_pointer_udl::value); + + // nlohmann::literals::json_literals + CHECK_FALSE(has_json_udl::value); + CHECK_FALSE(has_json_pointer_udl::value); + } + + SECTION("the rest of the library keeps working") + { + const json j = json::parse(R"({"foo": {"bar": 42}})"); + CHECK(j.dump() == R"({"foo":{"bar":42}})"); + + const json::json_pointer ptr("/foo/bar"); + CHECK(j.at(ptr) == 42); + CHECK(j.contains(ptr)); + } +} + +// the literals can still be added where they are needed +#include + +TEST_CASE("JSON_NO_AUTOMATIC_UDLS with ") +{ +#if !defined(JSON_USE_GLOBAL_UDLS) || JSON_USE_GLOBAL_UDLS + SECTION("global namespace") + { + CHECK("[1,2]"_json == json({1, 2})); + CHECK("/a/0"_json_pointer == json::json_pointer("/a/0")); + } +#endif + + SECTION("nlohmann::literals::json_literals") + { + using namespace nlohmann::literals::json_literals; // NOLINT(google-build-using-namespace) + CHECK(R"({"a":[42]})"_json.at("/a/0"_json_pointer) == 42); + } +} From fdcc569eee979cb8237c5a2c61e290c527b2d487 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:13:31 +0200 Subject: [PATCH 30/31] Use only documented StringType members in json_pointer (#5692) contains(const json_pointer&) and operator/=(std::size_t) (and hence operator/(std::size_t)) used string_t operations that the StringType template parameter documentation explicitly does not require: comparing string_t with a const char* literal, c_str(), and constructibility from std::string. This made both functions fail to compile for a conforming custom StringType, even though the documentation's own reference StringType satisfies the requirements. Fix contains() to compare individual chars ('0'..'9') instead of comparing string_t with const char* literals, and to call data() (documented to be null-terminated) instead of c_str(). Fix operator/=(std::size_t) to build the array-index token via the existing detail::to_string helper (ADL int_to_string() or assignment from std::to_string()) instead of via std::to_string() directly, matching how diff(), items(), and std::hash already convert a std::size_t to a StringType. Add regression tests to tests/src/unit-alt-string.cpp: contains() for present/missing keys and indices, "-", a leading zero, and a non-numeric token on an array, plus json_pointer::operator/(std::size_t). Fixes #5666. Signed-off-by: Niels Lohmann --- .../features/types/template_parameters.md | 1 + include/nlohmann/detail/json_pointer.hpp | 7 ++-- single_include/nlohmann/json.hpp | 8 +++-- tests/src/unit-alt-string.cpp | 35 +++++++++++++++++++ 4 files changed, 45 insertions(+), 6 deletions(-) diff --git a/docs/mkdocs/docs/features/types/template_parameters.md b/docs/mkdocs/docs/features/types/template_parameters.md index 8ed23e156..5b528fd79 100644 --- a/docs/mkdocs/docs/features/types/template_parameters.md +++ b/docs/mkdocs/docs/features/types/template_parameters.md @@ -389,6 +389,7 @@ using array_t = ArrayType>; | Functionality | Additional requirement | |-----------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| | [`diff`](../../api/basic_json/diff.md), [`items`](../../api/basic_json/items.md), [`std::hash`](../../api/basic_json/std_hash.md) | conversion of a `#!cpp std::size_t` to `StringType`: either assignability from the result of `#!cpp std::to_string`, or an ADL overload `#!cpp void int_to_string(StringType&, std::size_t)` | +| [`operator/(std::size_t)`](../../api/json_pointer/operator_slash.md) | the same conversion of a `#!cpp std::size_t` to `StringType` as `diff`, `items`, and `std::hash` above | | [`std::hash`](../../api/basic_json/std_hash.md) | additionally a specialization of `#!cpp std::hash` | | [`to_bson`](../../api/basic_json/to_bson.md) | `find(value_type)` and `npos` | | [`parse`](../../api/basic_json/parse.md) from a `string_t` | the input adapters must accept it; otherwise pass a character range | diff --git a/include/nlohmann/detail/json_pointer.hpp b/include/nlohmann/detail/json_pointer.hpp index 0b9f9651a..78074988e 100644 --- a/include/nlohmann/detail/json_pointer.hpp +++ b/include/nlohmann/detail/json_pointer.hpp @@ -26,6 +26,7 @@ #include #include #include +#include #include NLOHMANN_JSON_NAMESPACE_BEGIN @@ -116,7 +117,7 @@ class json_pointer /// @sa https://json.nlohmann.me/api/json_pointer/operator_slasheq/ json_pointer& operator/=(std::size_t array_idx) { - return *this /= std::to_string(array_idx); + return *this /= detail::to_string(array_idx); } /// @brief create a new JSON pointer by appending the right JSON pointer at the end of the left JSON pointer @@ -752,7 +753,7 @@ class json_pointer // would throw out_of_range.404 -- contains() must not throw (see #5395) return false; } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !("0" <= reference_token && reference_token <= "9"))) + if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !('0' <= reference_token[0] && reference_token[0] <= '9'))) { // invalid char return false; @@ -780,7 +781,7 @@ class json_pointer // not throw (see #5395), so such a reference token is treated as "not found" errno = 0; // strtoull() does not reset errno on success char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.c_str(), &p_end, 10); // NOLINT(runtime/int) + const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) { diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index cf94c53a9..88c29a4f1 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -19636,6 +19636,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include @@ -19727,7 +19729,7 @@ class json_pointer /// @sa https://json.nlohmann.me/api/json_pointer/operator_slasheq/ json_pointer& operator/=(std::size_t array_idx) { - return *this /= std::to_string(array_idx); + return *this /= detail::to_string(array_idx); } /// @brief create a new JSON pointer by appending the right JSON pointer at the end of the left JSON pointer @@ -20363,7 +20365,7 @@ class json_pointer // would throw out_of_range.404 -- contains() must not throw (see #5395) return false; } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !("0" <= reference_token && reference_token <= "9"))) + if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !('0' <= reference_token[0] && reference_token[0] <= '9'))) { // invalid char return false; @@ -20391,7 +20393,7 @@ class json_pointer // not throw (see #5395), so such a reference token is treated as "not found" errno = 0; // strtoull() does not reset errno on success char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.c_str(), &p_end, 10); // NOLINT(runtime/int) + const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) { diff --git a/tests/src/unit-alt-string.cpp b/tests/src/unit-alt-string.cpp index f95d2213c..cac2d183d 100644 --- a/tests/src/unit-alt-string.cpp +++ b/tests/src/unit-alt-string.cpp @@ -352,6 +352,41 @@ TEST_CASE("alternative string type") CHECK(j2.flatten().unflatten() == j2); } + SECTION("contains(json_pointer)") + { + // contains(json_pointer) must compile and work with a string_t that has + // no c_str() and no comparison with const char* (see #5666) + auto j = alt_json::parse(R"({"foo": ["bar", "baz"]})"); + + // present: object key and array indices + CHECK(j.contains(alt_json::json_pointer("/foo"))); + CHECK(j.contains(alt_json::json_pointer("/foo/0"))); + CHECK(j.contains(alt_json::json_pointer("/foo/1"))); + + // missing: absent object key and out-of-range array index + CHECK_FALSE(j.contains(alt_json::json_pointer("/bar"))); + CHECK_FALSE(j.contains(alt_json::json_pointer("/foo/2"))); + + // "-" always fails the range check + CHECK_FALSE(j.contains(alt_json::json_pointer("/foo/-"))); + + // an array index must not have a leading zero + CHECK_FALSE(j.contains(alt_json::json_pointer("/foo/01"))); + + // a reference token that is not a number + CHECK_FALSE(j.contains(alt_json::json_pointer("/foo/bar"))); + } + + SECTION("operator/(std::size_t)") + { + // json_pointer::operator/=(std::size_t) must compile without string_t + // being constructible from std::string (see #5666) + auto j = alt_json::parse(R"({"foo": ["bar", "baz"]})"); + + CHECK(j.at(alt_json::json_pointer("/foo") / std::size_t(0)) == j["foo"][0]); + CHECK(j.at(alt_json::json_pointer("/foo") / std::size_t(1)) == j["foo"][1]); + } + SECTION("patch") { alt_json const patch1 = alt_json::parse(R"([{ "op": "add", "path": "/a/b", "value": [ "foo", "bar" ] }])"); From 6a073dbae4a2cce5ade27546a8ca365b0fef0786 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:13:34 +0200 Subject: [PATCH 31/31] Give operator>> a strong exception-safety guarantee (#5695) operator>> parsed directly into its basic_json& target, so a parse error left the target holding whatever was parsed before the error instead of its previous value. With JSON_DIAGNOSTICS=1, that partial value also violated the class invariant, because the parent pointers of an array or object's elements are only set when the container is closed, which a failed parse never reaches; copying such a value then aborted in assert_invariant(). Fix it the way basic_json::parse() already handles this: parse into a temporary and move it into the target only once parsing succeeds, so the target is left unchanged if an exception is thrown. Fixes #5652. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/operator_gtgt.md | 6 ++++++ include/nlohmann/json.hpp | 5 ++++- single_include/nlohmann/json.hpp | 5 ++++- tests/src/unit-diagnostics.cpp | 16 ++++++++++++++++ 4 files changed, 30 insertions(+), 2 deletions(-) diff --git a/docs/mkdocs/docs/api/operator_gtgt.md b/docs/mkdocs/docs/api/operator_gtgt.md index b68889af9..35b8c1016 100644 --- a/docs/mkdocs/docs/api/operator_gtgt.md +++ b/docs/mkdocs/docs/api/operator_gtgt.md @@ -18,6 +18,10 @@ Deserializes an input stream to a JSON value. the stream `i` +## Exception safety + +Strong guarantee: if an exception is thrown, there are no changes in `j`. + ## Exceptions - Throws [`parse_error.101`](../home/exceptions.md#jsonexceptionparse_error101) in case of an unexpected token, or if @@ -125,3 +129,5 @@ being read. the stream; planned to become the default in version 4.0.0. - Fixed a null pointer dereference for an `std::istream` without a stream buffer (now throws `parse_error.101`), and a crash (`std::terminate`) when `i` has `eofbit` in its exception mask, in version 3.13.0. +- Changed to the strong exception safety guarantee in version 3.13.0: `j` is no longer left with a partially parsed + value if parsing throws. diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index cb08777bc..78c4e4f62 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -5092,7 +5092,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/operator_gtgt/ friend std::istream& operator>>(std::istream& i, basic_json& j) { - parser(detail::input_adapter(i)).parse(false, j); + // parse into a temporary so that j is left unchanged if parsing fails + basic_json result; + parser(detail::input_adapter(i)).parse(false, result); + j = std::move(result); return i; } #endif // JSON_NO_IO diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 88c29a4f1..5ef598150 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -32019,7 +32019,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/operator_gtgt/ friend std::istream& operator>>(std::istream& i, basic_json& j) { - parser(detail::input_adapter(i)).parse(false, j); + // parse into a temporary so that j is left unchanged if parsing fails + basic_json result; + parser(detail::input_adapter(i)).parse(false, result); + j = std::move(result); return i; } #endif // JSON_NO_IO diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index 46f41f252..b3778e802 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -19,6 +19,7 @@ using nlohmann::json; #include #include +#include TEST_CASE("Better diagnostics") { @@ -492,6 +493,21 @@ TEST_CASE("Regression tests for extended diagnostics") CHECK(copy == j); } } + + SECTION("Regression test for issue #5652 - operator>> leaves a partial value in its target on a parse error") + { + json j = "old value"; + std::istringstream is("[1, x"); + CHECK_THROWS_WITH_AS(is >> j, "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: '1, x'", json::parse_error); + + // j must be left unchanged, as json::parse() guarantees for its result + CHECK(j == "old value"); + + // copying j must not trigger assert_invariant(): a failed parse must + // not leave array/object elements without a parent pointer + json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + } } TEST_CASE("Better diagnostics past the descent bound of update() and merge_patch()")