Merge remote-tracking branch 'origin/develop' into claude/issue-5340-restore-unget

Keep both sides in lexer.hpp (lookahead detection next to the new bulk-scan
detection) and in the operator>> version history.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-25 08:41:14 +02:00
188 changed files with 22152 additions and 5087 deletions
+145 -1
View File
@@ -2,6 +2,9 @@ cmake_minimum_required(VERSION 3.13...4.0)
option(JSON_Valgrind "Execute test suite with Valgrind." OFF)
option(JSON_FastTests "Skip expensive/slow tests." OFF)
option(JSON_TestSimdutf "Build the unit tests against the simdutf UTF-8 validation backend." OFF)
set(JSON_SIMDUTF_VERSION 9.1.0 CACHE STRING "The simdutf version used by JSON_TestSimdutf.")
set(JSON_32bitTest AUTO CACHE STRING "Enable the 32bit unit test (ON/OFF/AUTO/ONLY).")
set(JSON_TestStandards "" CACHE STRING "The list of standards to test explicitly.")
@@ -72,7 +75,12 @@ target_compile_options(test_main PUBLIC
# is annotated JSON_HEDLEY_NO_RETURN (it always throws), which
# makes MSVC flag the code following its call in binary_reader.hpp
# as unreachable for that instantiation, in both Debug and Release
$<$<CXX_COMPILER_ID:MSVC>:/W4;/wd4566;/wd4996;/wd4702>
# Disable warning C4503: decorated name length exceeded, name was truncated; the deep
# copy support added for #5387 pushes the mangled name of
# std::allocator_traits<...>::construct for the custom-base-class
# test's map type past VS2015's limit. The name is only used for
# debug info, so truncation does not affect the build.
$<$<CXX_COMPILER_ID:MSVC>:/W4;/wd4566;/wd4996;/wd4702;/wd4503>
# https://github.com/nlohmann/json/issues/1114
$<$<CXX_COMPILER_ID:MSVC>:/bigobj> $<$<BOOL:${MINGW}>:-Wa,-mbig-obj>
@@ -125,6 +133,51 @@ json_test_set_test_options(test-unicode4 TEST_PROPERTIES TIMEOUT 3000)
# add unit tests
#############################################################################
# Generate the leak checks for every JSON_HEDLEY_* macro defined in
# hedley.hpp; tests/src/unit-no-macro-leak.cpp #include-s the result after
# nlohmann/json.hpp (see issue #5408). Using the shared
# cmake/scripts/gen_hedley_undef_check.cmake script (also used by `make
# update_hedley_undef`) instead of a hand-maintained list of macro names
# means this test can never go stale after a future `make update_hedley`.
set(hedley_hpp "${PROJECT_SOURCE_DIR}/include/nlohmann/thirdparty/hedley/hedley.hpp")
set(hedley_undef_check_script "${PROJECT_SOURCE_DIR}/cmake/scripts/gen_hedley_undef_check.cmake")
set(hedley_undef_checks "${PROJECT_BINARY_DIR}/include/hedley_undef_checks.inc")
# Reconfigure whenever the vendored header or the generator script changes,
# so a `cmake --build` after `make update_hedley` does not silently keep a
# stale generated file around.
set_property(DIRECTORY APPEND PROPERTY CMAKE_CONFIGURE_DEPENDS
"${hedley_hpp}"
"${hedley_undef_check_script}")
# Generate once at configure time, so the very first build (before any
# custom-command build step has run) already has an up-to-date file.
execute_process(
COMMAND ${CMAKE_COMMAND}
"-DHEDLEY_HPP=${hedley_hpp}"
"-DOUTPUT=${hedley_undef_checks}"
-DMODE=checks
-P "${hedley_undef_check_script}"
RESULT_VARIABLE hedley_undef_check_result
)
if(NOT hedley_undef_check_result EQUAL 0)
message(FATAL_ERROR "Failed to generate ${hedley_undef_checks}")
endif()
# Also (re)generate as a build step, so an incremental build after editing
# hedley.hpp without a full reconfigure still picks up the change.
add_custom_command(
OUTPUT "${hedley_undef_checks}"
COMMAND ${CMAKE_COMMAND}
"-DHEDLEY_HPP=${hedley_hpp}"
"-DOUTPUT=${hedley_undef_checks}"
-DMODE=checks
-P "${hedley_undef_check_script}"
DEPENDS "${hedley_hpp}" "${hedley_undef_check_script}"
COMMENT "Generating Hedley undef leak checks"
VERBATIM)
add_custom_target(generate_hedley_undef_checks DEPENDS "${hedley_undef_checks}")
if("${JSON_TestStandards}" STREQUAL "")
set(test_cxx_standards 11 14 17 20 23)
unset(test_force)
@@ -149,6 +202,71 @@ if(test_force)
endif()
message(STATUS "${msg}")
#############################################################################
# optionally validate UTF-8 with simdutf (JSON_USE_SIMDUTF)
#############################################################################
# The simdutf backend is opt-in and not vendored, so it is fetched here rather
# than being a checked-in dependency. Everything below hangs off test_main,
# whose usage requirements every test target inherits; the library target and
# the installed CMake package are deliberately left untouched.
if (JSON_TestSimdutf)
# simdutf requires C++17, both to compile itself and to be reachable from
# the library, which keeps its scalar validator below that. Find a tested
# standard that satisfies it.
set(simdutf_standard "")
foreach(cxx_standard ${test_cxx_standards})
if(NOT cxx_standard LESS 17 AND compiler_supports_cpp_${cxx_standard})
set(simdutf_standard ${cxx_standard})
break()
endif()
endforeach()
if("${simdutf_standard}" STREQUAL "")
# Building simdutf would fail outright without a C++17 compiler, and
# even with one it would go unused if no C++17-or-later standard is
# tested. Say so and fall back to the scalar validator rather than
# failing the build.
if(NOT compiler_supports_cpp_17)
set(simdutf_reason "the compiler does not support C++17")
else()
set(simdutf_reason "no tested standard is C++17 or later (testing ${msg_standards})")
endif()
message(WARNING
"JSON_TestSimdutf is enabled, but ${simdutf_reason}. simdutf requires C++17, so it "
"is not fetched and JSON_USE_SIMDUTF is not defined: the tests run against the "
"built-in scalar UTF-8 validator instead. Set JSON_TestStandards to include 17 or "
"later, or build with a compiler that supports C++17.")
else()
if (CMAKE_VERSION VERSION_LESS 3.18)
message(FATAL_ERROR "JSON_TestSimdutf requires CMake 3.18 or later (simdutf's minimum).")
endif()
include(FetchContent)
# simdutf builds its tests and tools by default, and its tests pull
# further dependencies of their own; only the library is needed here
set(SIMDUTF_TESTS OFF CACHE BOOL "" FORCE)
set(SIMDUTF_TOOLS OFF CACHE BOOL "" FORCE)
set(SIMDUTF_BENCHMARKS OFF CACHE BOOL "" FORCE)
set(SIMDUTF_ICONV OFF CACHE BOOL "" FORCE)
FetchContent_Declare(simdutf
URL https://github.com/simdutf/simdutf/archive/refs/tags/v${JSON_SIMDUTF_VERSION}.tar.gz
DOWNLOAD_EXTRACT_TIMESTAMP TRUE
)
FetchContent_MakeAvailable(simdutf)
target_compile_definitions(test_main PUBLIC JSON_USE_SIMDUTF)
target_link_libraries(test_main PUBLIC simdutf::simdutf)
# simdutf.h requires C++17; below that the library keeps its scalar
# validator, so any C++11/14 test targets exercise the fallback and the
# C++17-and-later ones exercise simdutf. Both must agree.
message(STATUS "UTF-8 validation delegated to simdutf ${JSON_SIMDUTF_VERSION} for C++17 and later (JSON_USE_SIMDUTF)")
endif()
endif()
# *DO* use json_test_set_test_options() above this line
json_test_should_build_32bit_test(json_32bit_test json_32bit_test_only "${JSON_32bitTest}")
@@ -163,6 +281,14 @@ foreach(file ${files})
json_test_add_test_for(${file} MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force})
endforeach()
# tests/src/unit-no-macro-leak.cpp #include-s the generated leak-check file,
# so its test targets must be built after generate_hedley_undef_checks.
foreach(cxx_standard ${test_cxx_standards})
if(TARGET test-no-macro-leak_cpp${cxx_standard})
add_dependencies(test-no-macro-leak_cpp${cxx_standard} generate_hedley_undef_checks)
endif()
endforeach()
if(json_32bit_test_only)
# Skip all other tests in this file
return()
@@ -177,6 +303,24 @@ json_test_add_test_for(src/unit-comparison.cpp
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
)
# test the parser again with JSON_DIAGNOSTIC_POSITIONS enabled
json_test_set_test_options(test-class_parser_diagnostic_positions
COMPILE_DEFINITIONS JSON_DIAGNOSTIC_POSITIONS=1
)
json_test_add_test_for(src/unit-class_parser.cpp
NAME test-class_parser_diagnostic_positions
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
)
# test diagnostic positions again without regular diagnostics (JSON pointer paths)
json_test_set_test_options(test-diagnostic-positions_only
COMPILE_DEFINITIONS JSON_DIAGNOSTICS=0
)
json_test_add_test_for(src/unit-diagnostic-positions.cpp
NAME test-diagnostic-positions_only
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
)
# *DO NOT* use json_test_set_test_options() below this line
#############################################################################
+14
View File
@@ -14,6 +14,20 @@ add_test(
NAME test-abi_config_noversion
COMMAND abi_config_noversion ${DOCTEST_TEST_FILTER})
# test default and no version namespace with all ABI tags enabled, so the
# expected tag order is checked regardless of the JSON_* CMake options
foreach(test default noversion)
add_executable(abi_config_${test}_all_tags ${test}.cpp)
target_compile_definitions(abi_config_${test}_all_tags PRIVATE
JSON_DIAGNOSTICS=1
JSON_DIAGNOSTIC_POSITIONS=1
JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1)
target_link_libraries(abi_config_${test}_all_tags PRIVATE abi_compat_main)
add_test(
NAME test-abi_config_${test}_all_tags
COMMAND abi_config_${test}_all_tags ${DOCTEST_TEST_FILTER})
endforeach()
# test custom namespace
add_executable(abi_config_custom custom.cpp)
target_link_libraries(abi_config_custom PRIVATE abi_compat_main)
+6 -2
View File
@@ -24,12 +24,16 @@ TEST_CASE("default namespace")
expected += "_diag";
#endif
#if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
expected += "_ldvcmp";
#endif
#if JSON_DIAGNOSTIC_POSITIONS
expected += "_dp";
#endif
#if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
expected += "_ldvcmp";
#if JSON_BRACE_INIT_COPY_SEMANTICS
expected += "_bics";
#endif
expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR);
+6 -2
View File
@@ -25,12 +25,16 @@ TEST_CASE("default namespace without version component")
expected += "_diag";
#endif
#if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
expected += "_ldvcmp";
#endif
#if JSON_DIAGNOSTIC_POSITIONS
expected += "_dp";
#endif
#if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
expected += "_ldvcmp";
#if JSON_BRACE_INIT_COPY_SEMANTICS
expected += "_bics";
#endif
expected += "::basic_json";
+357
View File
@@ -81,6 +81,44 @@ BENCHMARK_CAPTURE(ParseString, signed_ints, TEST_DATA_DIRECTORY "/regressi
BENCHMARK_CAPTURE(ParseString, unsigned_ints, TEST_DATA_DIRECTORY "/regression/unsigned_ints.json");
BENCHMARK_CAPTURE(ParseString, small_signed_ints, TEST_DATA_DIRECTORY "/regression/small_signed_ints.json");
//////////////////////////////////////////////////////////////////////////////
// parse pretty-printed JSON from string
//
// Every file in the corpus above is minified or only lightly spaced, so none of
// them exercise the lexer's whitespace handling. Real-world JSON is frequently
// indented - configuration files, pretty-printed API responses, anything kept
// under version control - where insignificant whitespace can outweigh the data.
// Re-serializing a document with an indentation and parsing that keeps the
// content identical to the ParseString row above, so the pair isolates the cost
// of the whitespace alone.
//////////////////////////////////////////////////////////////////////////////
static void ParseIndented(benchmark::State& state, const char* filename, int indent)
{
std::ifstream f(filename);
std::string str((std::istreambuf_iterator<char>(f)), std::istreambuf_iterator<char>());
const std::string indented = json::parse(str).dump(indent);
while (state.KeepRunning())
{
state.PauseTiming();
auto* j = new json();
state.ResumeTiming();
*j = json::parse(indented);
state.PauseTiming();
delete j;
state.ResumeTiming();
}
state.SetBytesProcessed(state.iterations() * indented.size());
}
BENCHMARK_CAPTURE(ParseIndented, jeopardy / 4, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", 4);
BENCHMARK_CAPTURE(ParseIndented, canada / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", 4);
BENCHMARK_CAPTURE(ParseIndented, citm_catalog / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", 4);
BENCHMARK_CAPTURE(ParseIndented, twitter / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", 4);
//////////////////////////////////////////////////////////////////////////////
// serialize JSON
//////////////////////////////////////////////////////////////////////////////
@@ -214,4 +252,323 @@ static void BinaryToCbor(benchmark::State& state)
}
BENCHMARK(BinaryToCbor)->RangeMultiplier(2)->Range(8, 8 << 12);
//////////////////////////////////////////////////////////////////////////////
// parse binary formats
//////////////////////////////////////////////////////////////////////////////
// Only MessagePack had a read benchmark (FromMsgpack above, left untouched so
// its numbers stay comparable across releases). The benchmarks below cover the
// other formats, and read from a contiguous buffer as well as from a FILE*:
// most callers pass a container, and the two adapters compile to different
// code. The test data repository ships JSON only, so the input for each is
// derived at setup time by serializing a parsed test file.
/// binary format to benchmark; the _optimized variants add UBJSON/BJData size
/// and type annotations, which the readers handle in a separate code path
enum class binary_format
{
cbor,
msgpack,
ubjson,
ubjson_optimized,
bjdata,
bjdata_optimized,
bson
};
static std::vector<std::uint8_t> to_binary(const json& j, const binary_format format)
{
switch (format)
{
case binary_format::cbor:
return json::to_cbor(j);
case binary_format::msgpack:
return json::to_msgpack(j);
case binary_format::ubjson:
return json::to_ubjson(j);
case binary_format::ubjson_optimized:
return json::to_ubjson(j, true, true);
case binary_format::bjdata:
return json::to_bjdata(j);
case binary_format::bjdata_optimized:
return json::to_bjdata(j, true, true);
case binary_format::bson:
default:
return json::to_bson(j);
}
}
static json from_binary(const std::vector<std::uint8_t>& bytes, const binary_format format)
{
switch (format)
{
case binary_format::cbor:
return json::from_cbor(bytes);
case binary_format::msgpack:
return json::from_msgpack(bytes);
case binary_format::ubjson:
case binary_format::ubjson_optimized:
return json::from_ubjson(bytes);
case binary_format::bjdata:
case binary_format::bjdata_optimized:
return json::from_bjdata(bytes);
case binary_format::bson:
default:
return json::from_bson(bytes);
}
}
static json from_binary(std::FILE* file, const binary_format format)
{
switch (format)
{
case binary_format::cbor:
return json::from_cbor(file);
case binary_format::msgpack:
return json::from_msgpack(file);
case binary_format::ubjson:
case binary_format::ubjson_optimized:
return json::from_ubjson(file);
case binary_format::bjdata:
case binary_format::bjdata_optimized:
return json::from_bjdata(file);
case binary_format::bson:
default:
return json::from_bson(file);
}
}
/*!
@brief serialize a parsed test file to @a format
Returns an empty vector and marks the benchmark as skipped if the file cannot
be represented in the format, rather than letting the exception escape: BSON
requires an object at the top level, and several test files are arrays.
*/
static std::vector<std::uint8_t> binary_input(benchmark::State& state, const char* filename, const binary_format format)
{
std::ifstream f(filename);
std::string const str((std::istreambuf_iterator<char>(f)), std::istreambuf_iterator<char>());
const json j = json::parse(str);
if (format == binary_format::bson && !j.is_object())
{
state.SkipWithError("BSON requires an object at the top level");
return {};
}
return to_binary(j, format);
}
static void FromBinaryBuffer(benchmark::State& state, const char* filename, const binary_format format)
{
const std::vector<std::uint8_t> bytes = binary_input(state, filename, format);
if (bytes.empty())
{
return;
}
for (auto _ : state)
{
// the value is destroyed outside the timed section, because destroying
// a large DOM is not what this benchmark measures
state.PauseTiming();
auto* j = new json();
state.ResumeTiming();
*j = from_binary(bytes, format);
state.PauseTiming();
delete j;
state.ResumeTiming();
}
state.SetBytesProcessed(state.iterations() * bytes.size());
}
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::cbor);
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::cbor);
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::cbor);
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::cbor);
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / floats, TEST_DATA_DIRECTORY "/regression/floats.json", binary_format::cbor);
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / signed_ints, TEST_DATA_DIRECTORY "/regression/signed_ints.json", binary_format::cbor);
BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::msgpack);
BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::msgpack);
BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::msgpack);
BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::msgpack);
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::ubjson);
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson);
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::ubjson);
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson);
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson_optimized / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson_optimized);
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson_optimized / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson_optimized);
BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bjdata);
BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata);
BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata_optimized / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bjdata_optimized);
BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata_optimized / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata_optimized);
// BSON requires an object at the top level, so the array-rooted test files
// (jeopardy and the regression files) cannot be captured here
BENCHMARK_CAPTURE(FromBinaryBuffer, bson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bson);
BENCHMARK_CAPTURE(FromBinaryBuffer, bson / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::bson);
BENCHMARK_CAPTURE(FromBinaryBuffer, bson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bson);
static void FromBinaryFile(benchmark::State& state, const char* filename, const binary_format format)
{
const std::vector<std::uint8_t> bytes = binary_input(state, filename, format);
if (bytes.empty())
{
return;
}
const char* tmp = "benchmark_input.bin";
std::ofstream o(tmp, std::ios::binary);
o.write(reinterpret_cast<const char*>(bytes.data()), static_cast<std::streamsize>(bytes.size()));
o.flush();
o.close();
for (auto _ : state)
{
state.PauseTiming();
auto* j = new json();
auto* file = std::fopen(tmp, "rb");
state.ResumeTiming();
*j = from_binary(file, format);
state.PauseTiming();
std::fclose(file);
delete j;
state.ResumeTiming();
}
state.SetBytesProcessed(state.iterations() * bytes.size());
}
BENCHMARK_CAPTURE(FromBinaryFile, cbor / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::cbor);
BENCHMARK_CAPTURE(FromBinaryFile, cbor / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::cbor);
BENCHMARK_CAPTURE(FromBinaryFile, ubjson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson);
BENCHMARK_CAPTURE(FromBinaryFile, ubjson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson);
BENCHMARK_CAPTURE(FromBinaryFile, bjdata / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata);
BENCHMARK_CAPTURE(FromBinaryFile, bson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bson);
//////////////////////////////////////////////////////////////////////////////
// parse binary formats: value shapes
//////////////////////////////////////////////////////////////////////////////
// The test files above are wide and shallow, but the readers' cost is per
// container, so these cover the shapes that stress the container handling
// itself. Every shape is wrapped in an object so that BSON, which requires an
// object at the top level, measures the same value as the other formats.
/// deeply nested arrays: one container per level, no other work
static json make_nested()
{
json nested = json::array();
json* p = &nested;
for (std::size_t i = 1; i < 1000; ++i)
{
p->push_back(json::array());
p = &p->operator[](0);
}
json j = json::object();
j["data"] = std::move(nested);
return j;
}
/// many sibling containers: maximum container churn, minimum nesting
static json make_containers()
{
json data = json::array();
for (std::size_t i = 0; i < 100000; ++i)
{
data.push_back(json::array({1, 2}));
}
json j = json::object();
j["data"] = std::move(data);
return j;
}
/// one flat array of numbers: the scalar decoding path, which must not move
static json make_scalars()
{
json data = json::array();
for (std::size_t i = 0; i < 1000000; ++i)
{
data.push_back(i);
}
json j = json::object();
j["data"] = std::move(data);
return j;
}
static void FromBinaryShape(benchmark::State& state, json (*build)(), const binary_format format)
{
const std::vector<std::uint8_t> bytes = to_binary(build(), format);
for (auto _ : state)
{
state.PauseTiming();
auto* j = new json();
state.ResumeTiming();
*j = from_binary(bytes, format);
state.PauseTiming();
delete j;
state.ResumeTiming();
}
state.SetBytesProcessed(state.iterations() * bytes.size());
}
BENCHMARK_CAPTURE(FromBinaryShape, nested / cbor, make_nested, binary_format::cbor);
BENCHMARK_CAPTURE(FromBinaryShape, nested / msgpack, make_nested, binary_format::msgpack);
BENCHMARK_CAPTURE(FromBinaryShape, nested / ubjson, make_nested, binary_format::ubjson);
BENCHMARK_CAPTURE(FromBinaryShape, nested / bjdata, make_nested, binary_format::bjdata);
BENCHMARK_CAPTURE(FromBinaryShape, nested / bson, make_nested, binary_format::bson);
BENCHMARK_CAPTURE(FromBinaryShape, containers / cbor, make_containers, binary_format::cbor);
BENCHMARK_CAPTURE(FromBinaryShape, containers / msgpack, make_containers, binary_format::msgpack);
BENCHMARK_CAPTURE(FromBinaryShape, containers / ubjson, make_containers, binary_format::ubjson);
BENCHMARK_CAPTURE(FromBinaryShape, containers / ubjson_optimized, make_containers, binary_format::ubjson_optimized);
BENCHMARK_CAPTURE(FromBinaryShape, containers / bjdata, make_containers, binary_format::bjdata);
BENCHMARK_CAPTURE(FromBinaryShape, containers / bson, make_containers, binary_format::bson);
// BSON names every array element, so a large array measures key generation
// rather than scalar decoding and is left out here
BENCHMARK_CAPTURE(FromBinaryShape, scalars / cbor, make_scalars, binary_format::cbor);
BENCHMARK_CAPTURE(FromBinaryShape, scalars / msgpack, make_scalars, binary_format::msgpack);
BENCHMARK_CAPTURE(FromBinaryShape, scalars / ubjson, make_scalars, binary_format::ubjson);
BENCHMARK_CAPTURE(FromBinaryShape, scalars / bjdata, make_scalars, binary_format::bjdata);
/*!
@brief parse an indefinite-length CBOR string
The writer never emits this form, so the input is assembled by hand: 0x7F
opens the string, each chunk is a one-character string, and 0xFF closes it.
*/
static void FromCborChunkedString(benchmark::State& state, const std::size_t chunks)
{
std::vector<std::uint8_t> bytes;
bytes.reserve(2 * chunks + 2);
bytes.push_back(0x7F);
for (std::size_t i = 0; i < chunks; ++i)
{
bytes.push_back(0x61); // string of length 1
bytes.push_back(0x61); // 'a'
}
bytes.push_back(0xFF);
for (auto _ : state)
{
json j = json::from_cbor(bytes);
benchmark::DoNotOptimize(j);
}
state.SetBytesProcessed(state.iterations() * bytes.size());
}
BENCHMARK_CAPTURE(FromCborChunkedString, 10000 chunks, 10000);
BENCHMARK_MAIN();
+23
View File
@@ -79,3 +79,26 @@ the same `fuzzers` target as above and also relies on the `FUZZER_ENGINE` variab
[build script](https://github.com/google/oss-fuzz/blob/master/projects/json/build.sh) for more information.
In case the build at OSS-Fuzz fails, an issue will be created automatically.
### Handling OSS-Fuzz reports
OSS-Fuzz files the crashes it finds in its own [issue tracker](https://issues.oss-fuzz.com), not on GitHub. So that
each report can be traced to the change that fixed it, and each fix to the report it answers, fixes follow these
conventions:
- **Reference the OSS-Fuzz issue in the pull request**, next to any GitHub issue it closes, as `OSS-Fuzz: <id>` (for
example, `OSS-Fuzz: 563659413`), and in the commit message. The ID alone does not disclose the crash. If the report
was triaged into a GitHub issue, link the OSS-Fuzz issue there too.
- **Turn the reproducer into a unit test.** Download the testcase from the OSS-Fuzz report, reduce it if possible, and
add it as a regression test to the unit test of the affected format (e.g., `tests/src/unit-bjdata.cpp`), with a
comment naming the OSS-Fuzz issue. This way the input is checked by every CI run rather than only by OSS-Fuzz, and
it stays covered even if OSS-Fuzz later closes the report as not reproducible.
- **Keep the fuzzer drivers and the unit tests in sync.** The round-trip checks of the UBJSON and BJData drivers are
also run on a fixed corpus in the unit tests (see `tests/src/round_trip_corpus.hpp` and the "round-trip invariants"
test cases), so a regression shows up in CI first. When a driver's checks change, change the unit tests with them.
- **Record in the report whether the bug shipped.** OSS-Fuzz asks whether a crash was a short-lived regression or
affects a released version; answer it when the fix is merged, as it decides whether the fix needs a release note or
a security advisory (see the [security policy](../.github/SECURITY.md)).
After the fix is merged, OSS-Fuzz re-runs the reproducer on its next build and marks the report as verified and
closed. If it does not, the fix is incomplete.
+43 -4
View File
@@ -21,16 +21,53 @@ array data, it performs the following steps:
- j4 = from_bjdata(vec3)
- assert(j1 == j4)
Re-serializing j2/j3/j4 with the same use_size/use_type settings is checked
for value-stability rather than byte-exact stability: from_bjdata(to_bjdata(j2))
must equal j2 (and likewise for j3, j4). Byte-exact stability does not hold in
general, because a BJData value can lose type fidelity across a round trip
(e.g. a binary_t value serialized without the optimized "$U#" array header is
parsed back as a plain array of numbers, see #5398 and the discussion on
PR #5494) - the numeric value is preserved, but the writer's smallest-type
selection for the now-plain numbers may legitimately pick a different, but
equally valid, single-byte type marker than the dedicated binary-data writer
would have. Both encodings are valid BJData and both decode to the same
value, so this is not treated as a round-trip failure here.
"Value-stable" is checked by comparing dump()s rather than with operator==
directly: a BJData/UBJSON payload can decode to a non-finite double (NaN or
+-Infinity), and IEEE 754 NaN is never equal to itself, so operator== would
report two structurally-identical trees as different whenever a NaN is
involved -- not a round-trip bug, just NaN's ordinary (non-)reflexivity.
dump() serializes any non-finite double the same deterministic way (as JSON
`null`, since JSON itself cannot represent NaN/Infinity), so comparing
dumps is stable under exactly the same values that break operator==.
The unit tests run the same checks on a fixed corpus (see the "BJData round-trip
invariants" test case), so keep both in sync.
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
drivers.
*/
#include <cassert>
#include <iostream>
#include <sstream>
#include <nlohmann/json.hpp>
// the round-trip checks below are assertions; NDEBUG would compile them away
#ifdef NDEBUG
#error "the fuzzer drivers must be built without NDEBUG"
#endif
using json = nlohmann::json;
// value-stable comparison for the round-trip checks below; see the note
// above on why this compares dump()s rather than the json values directly
static bool is_value_stable(const json& lhs, const json& rhs)
{
return lhs.dump() == rhs.dump();
}
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
{
@@ -56,10 +93,12 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
json const j3 = json::from_bjdata(vec3);
json const j4 = json::from_bjdata(vec4);
// serializations must match
assert(json::to_bjdata(j2, false, false) == vec2);
assert(json::to_bjdata(j3, true, false) == vec3);
assert(json::to_bjdata(j4, true, true) == vec4);
// re-serializing must be value-stable (see the notes above on
// why byte-exact stability is not guaranteed in general, and
// why this compares dump()s rather than the values directly)
assert(is_value_stable(json::from_bjdata(json::to_bjdata(j2, false, false)), j2));
assert(is_value_stable(json::from_bjdata(json::to_bjdata(j3, true, false)), j3));
assert(is_value_stable(json::from_bjdata(json::to_bjdata(j4, true, true)), j4));
}
catch (const json::parse_error&)
{
+6
View File
@@ -19,10 +19,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
drivers.
*/
#include <cassert>
#include <iostream>
#include <sstream>
#include <nlohmann/json.hpp>
// the round-trip checks below are assertions; NDEBUG would compile them away
#ifdef NDEBUG
#error "the fuzzer drivers must be built without NDEBUG"
#endif
using json = nlohmann::json;
// see http://llvm.org/docs/LibFuzzer.html
+6
View File
@@ -19,10 +19,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
drivers.
*/
#include <cassert>
#include <iostream>
#include <sstream>
#include <nlohmann/json.hpp>
// the round-trip checks below are assertions; NDEBUG would compile them away
#ifdef NDEBUG
#error "the fuzzer drivers must be built without NDEBUG"
#endif
using json = nlohmann::json;
// see http://llvm.org/docs/LibFuzzer.html
+6
View File
@@ -20,10 +20,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
drivers.
*/
#include <cassert>
#include <iostream>
#include <sstream>
#include <nlohmann/json.hpp>
// the round-trip checks below are assertions; NDEBUG would compile them away
#ifdef NDEBUG
#error "the fuzzer drivers must be built without NDEBUG"
#endif
using json = nlohmann::json;
// see http://llvm.org/docs/LibFuzzer.html
+6
View File
@@ -19,10 +19,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
drivers.
*/
#include <cassert>
#include <iostream>
#include <sstream>
#include <nlohmann/json.hpp>
// the round-trip checks below are assertions; NDEBUG would compile them away
#ifdef NDEBUG
#error "the fuzzer drivers must be built without NDEBUG"
#endif
using json = nlohmann::json;
// see http://llvm.org/docs/LibFuzzer.html
+9
View File
@@ -21,14 +21,23 @@ array data, it performs the following steps:
- j4 = from_ubjson(vec3)
- assert(j1 == j4)
The unit tests run the same checks on a fixed corpus (see the "UBJSON round-trip
invariants" test case), so keep both in sync.
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
drivers.
*/
#include <cassert>
#include <iostream>
#include <sstream>
#include <nlohmann/json.hpp>
// the round-trip checks below are assertions; NDEBUG would compile them away
#ifdef NDEBUG
#error "the fuzzer drivers must be built without NDEBUG"
#endif
using json = nlohmann::json;
// see http://llvm.org/docs/LibFuzzer.html
+213
View File
@@ -0,0 +1,213 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#pragma once
#include <cmath> // nan
#include <cstddef> // size_t
#include <cstdint> // int32_t, int64_t, uint32_t, uint64_t
#include <limits> // numeric_limits
#include <random> // mt19937
#include <string> // string, to_string
#include <utility> // move
#include <vector> // vector
#include <nlohmann/json.hpp>
// Values for the round-trip property tests of the UBJSON and BJData writers.
//
// The fuzzer drivers (tests/src/fuzzer-parse_ubjson.cpp and
// fuzzer-parse_bjdata.cpp) check that anything the library parses can be
// serialized, parsed back, and serialized again without loss. Those checks
// only run at OSS-Fuzz, so a regression used to surface days later as an
// external report. The unit tests run the same checks on this corpus in CI.
//
// The corpus is deterministic: std::mt19937's output sequence is fixed by
// the standard, and it is used directly rather than through a distribution
// (whose results are implementation-defined).
namespace utils
{
class round_trip_corpus
{
public:
using json = nlohmann::json;
static std::vector<json> values()
{
round_trip_corpus corpus;
return corpus.build();
}
// whether a value contains a binary value, which a BJData or UBJSON round
// trip may turn into an array of integers
static bool contains_binary(const json& j)
{
if (j.is_binary())
{
return true;
}
if (j.is_structured())
{
for (const auto& element : j)
{
if (contains_binary(element))
{
return true;
}
}
}
return false;
}
private:
std::vector<json> atoms;
// a fixed seed is the point: the corpus must be the same in every run
std::mt19937 generator{42}; // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed)
round_trip_corpus()
: atoms
{
nullptr, true, false,
// integers at the boundaries of every UBJSON/BJData integer type
0, 1, -1, 127, 128, 255, 256, -128, -129,
32767, 32768, 65535, 65536, -32768, -32769,
(std::numeric_limits<std::int32_t>::min)(), (std::numeric_limits<std::int32_t>::max)(),
(std::numeric_limits<std::uint32_t>::max)(),
(std::numeric_limits<std::int64_t>::min)(), (std::numeric_limits<std::int64_t>::max)(),
static_cast<std::uint64_t>((std::numeric_limits<std::int64_t>::max)()) + 1u,
(std::numeric_limits<std::uint64_t>::max)(),
// floating-point numbers, including non-finite ones
0.0, -0.0, 1.5, -2.25, 3.4e38, (std::numeric_limits<double>::max)(),
std::nan(""), std::numeric_limits<double>::infinity(), -std::numeric_limits<double>::infinity(),
// strings, including a non-ASCII one and one longer than 255 bytes
"", "a", "\xC3\xA4", std::string(300, 'x'),
// binary values with and without subtype
json::binary({}), json::binary({1, 2, 255}), json::binary({0x80, 0x7F}, 42), json::binary({1}, 0)
}
{}
std::vector<json> build()
{
std::vector<json> result = atoms;
// each atom inside containers, including homogeneous ones that the
// writers encode as optimized (typed) containers
result.emplace_back(json::array());
result.emplace_back(json::object());
for (const auto& atom : atoms)
{
result.push_back(json::array({atom}));
result.push_back(json::array({atom, atom, atom}));
result.push_back(json::array({json::array({atom})}));
result.push_back(json::object({{"key", atom}}));
}
result.push_back(json::array({1, 1.5}));
result.push_back(json::array({-1, 255}));
result.push_back(json::array({"a", "b"}));
// deep, but well below any recursion or depth limit
json nested_array = 1;
json nested_object = 1;
for (int i = 0; i < 300; ++i)
{
nested_array = json::array({nested_array});
nested_object = json::object({{"key", nested_object}});
}
result.push_back(nested_array);
result.push_back(nested_object);
add_annotated_arrays(result);
add_random_values(result);
return result;
}
// objects in the JData annotated array format, which the BJData writer
// encodes as ND-arrays when the annotation describes a packed array, and
// as plain objects otherwise (see #5398, #5399, #5403, #5404, and #5542)
static void add_annotated_arrays(std::vector<json>& result)
{
const std::vector<json> types =
{
"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64",
"single", "double", "char", "byte", "bool", "unknown", 5, nullptr
};
const std::vector<json> sizes =
{
json::array(), {3}, {1, 3}, {3, 1}, {2, 3}, {2, 0}, {0, 2}, {2, 2, 2}, {-1, 2}, {2, 1.5},
"3", 3, nullptr, json::binary({})
};
const std::vector<json> data =
{
nullptr, 5, "s", json::object({{"a", 1}}), json::array(),
{1, 2, 3}, {1, 2, 3, 4, 5, 6}, {1, 2, 3, 4, 5, 6, 7, 8},
{1.5, 2.5, 3.5, 4.5, 5.5, 6.5}, {300, -300, 70000, -70000, 1, 2},
{"a", "b", "c", "d", "e", "f"}, {json::array({1, 2, 3}), json::array({4, 5, 6})}
};
for (const auto& type : types)
{
for (const auto& size : sizes)
{
for (const auto& d : data)
{
result.push_back({{"_ArrayType_", type}, {"_ArraySize_", size}, {"_ArrayData_", d}});
}
}
}
// incomplete annotations and annotations with an extra key
result.push_back({{"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}});
result.push_back({{"_ArrayType_", "uint8"}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}});
result.push_back({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}});
result.push_back({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}, {"extra", 1}});
}
// random containers of atoms, both homogeneous and mixed
void add_random_values(std::vector<json>& result)
{
for (int i = 0; i < 1000; ++i)
{
result.push_back(random_value(0));
}
}
std::size_t random_below(std::size_t bound)
{
return generator() % bound;
}
json random_value(int depth)
{
const auto kind = random_below(10);
if (depth > 3 || kind < 5)
{
return atoms[random_below(atoms.size())];
}
json result = kind < 8 ? json::array() : json::object();
const auto count = random_below(5);
const bool homogeneous = random_below(2) == 0;
const json fixed = atoms[random_below(atoms.size())];
for (std::size_t i = 0; i < count; ++i)
{
json element = homogeneous ? fixed : random_value(depth + 1);
if (result.is_array())
{
result.push_back(std::move(element));
}
else
{
result[std::to_string(i)] = std::move(element);
}
}
return result;
}
};
} // namespace utils
+61
View File
@@ -0,0 +1,61 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
// Standalone compile-and-run check for the JSON_SKIP_LIBRARY_VERSION_CHECK
// configuration macro, which (per #5423) was never exercised anywhere in the
// test matrix.
//
// include/nlohmann/detail/abi_macros.hpp normally emits a #warning if
// NLOHMANN_JSON_VERSION_MAJOR/MINOR/PATCH are already defined (as they would
// be by an earlier inclusion of a different version of the library) with
// values that mismatch the version about to be defined -- unless
// JSON_SKIP_LIBRARY_VERSION_CHECK is defined, in which case the check (and
// that #warning) is skipped.
//
// This file deliberately is not named tests/src/unit-*.cpp: it is compiled
// directly (with a modest, non-strict warning set) by the dedicated
// ci_test_skiplibraryversioncheck target in cmake/ci.cmake, rather than being
// folded into the library's own -Weverything/-Werror unit test matrix. That
// is because the scenario simulated here -- mixing two different, already
// differently-versioned inclusions of the library in one translation unit --
// unavoidably also triggers the *compiler's own* "macro redefined" warning,
// independent of (and unaffected by) JSON_SKIP_LIBRARY_VERSION_CHECK, which
// only ever silences the library's own #warning. Building this file under
// -Weverything -Werror would therefore fail for a reason unrelated to the
// macro under test.
#define NLOHMANN_JSON_VERSION_MAJOR 0
#define NLOHMANN_JSON_VERSION_MINOR 0
#define NLOHMANN_JSON_VERSION_PATCH 0
#define JSON_SKIP_LIBRARY_VERSION_CHECK 1
#include <nlohmann/json.hpp>
int main()
{
// reaching this point at all already proves that the mismatched,
// pre-defined version macros above did not stop compilation -- which is
// exactly what JSON_SKIP_LIBRARY_VERSION_CHECK is for. The library must
// also still be fully usable.
const nlohmann::json j = {{"a", 1}, {"b", {1, 2, 3}}};
if (j.dump() != "{\"a\":1,\"b\":[1,2,3]}")
{
return 1;
}
// include/nlohmann/detail/abi_macros.hpp unconditionally (re)defines the
// version macros to the library's real, current version right after the
// (here, skipped) mismatch check, regardless of the deliberately wrong
// stand-in values defined above.
if (NLOHMANN_JSON_VERSION_MAJOR == 0 && NLOHMANN_JSON_VERSION_MINOR == 0 && NLOHMANN_JSON_VERSION_PATCH == 0)
{
return 1;
}
return 0;
}
+9
View File
@@ -15,6 +15,15 @@
namespace utils
{
// Some tests intentionally discard the [[nodiscard]]/JSON_HEDLEY_WARN_UNUSED_RESULT
// return value of a call they only make to exercise its side effects (e.g. checking
// that it does not throw). A plain (void) cast on the call expression does not
// suppress GCC's warning for functions using the GNU __attribute__((warn_unused_result))
// form (as opposed to the C++17 [[nodiscard]] attribute) -- passing the value into an
// ordinary function call does.
template<typename T>
inline void ignore_return_value(T&& /*unused*/) noexcept {}
inline std::vector<std::uint8_t> read_binary_file(const std::string& filename)
{
std::ifstream file(filename, std::ios::binary);
+51
View File
@@ -216,6 +216,57 @@ TEST_CASE("controlled bad_alloc")
CHECK_THROWS_AS(my_json(s), std::bad_alloc&);
next_construct_fails = false;
}
SECTION("basic_json(const basic_json&) of a deeply nested value (#5387)")
{
// Copying a value nested deeper than the descent bound builds the
// copy from the top down: every value whose own copy has not been
// made yet stays a null value until it is. Failing an allocation
// part-way through is what proves such a half-built copy can still
// be destroyed.
//
// Which path the failure lands in depends on the build: the first
// allocation of a copy belongs to the outermost level, so here it
// is the descending one. Built with JSON_NO_THREAD_LOCAL - as the
// ci_test_no_thread_local target builds the whole suite - no
// descent is made at all and the very same failure lands in the
// iterative path instead, part-way through its worklist.
const auto check_deep_copy = [](bool objects)
{
CAPTURE(objects);
next_construct_fails = false;
// deeper than the 128 levels the copy constructor descends into
const std::size_t depth = 300;
my_json j = 1;
for (std::size_t i = 0; i < depth; ++i)
{
if (objects)
{
my_json wrapper = my_json::object();
wrapper["a"] = std::move(j);
j = std::move(wrapper);
}
else
{
j = my_json::array({std::move(j)});
}
}
// NOLINTNEXTLINE(performance-unnecessary-copy-initialization): the copy is what is tested
CHECK_NOTHROW(my_json(j));
next_construct_fails = true;
// NOLINTNEXTLINE(performance-unnecessary-copy-initialization): the copy is what is tested
CHECK_THROWS_AS(my_json(j), std::bad_alloc&);
next_construct_fails = false;
};
check_deep_copy(false);
check_deep_copy(true);
}
}
}
+40 -32
View File
@@ -11,8 +11,10 @@
#include <nlohmann/json.hpp>
#include <cstdint>
#include <string>
#include <utility>
#include <vector>
/* forward declarations */
class alt_string;
@@ -22,6 +24,10 @@ void int_to_string(alt_string& target, std::size_t value); // NOLINT(misc-use-in
/*
* This is virtually a string class.
* It covers std::string under the hood.
*
* It deliberately does not provide c_str(), back(), find(str, pos), replace(),
* or substr(): the library must not rely on them. Do not add members here
* without checking that the library actually needs them.
*/
class alt_string
{
@@ -106,11 +112,6 @@ class alt_string
return str_impl < op.str_impl;
}
const char* c_str() const
{
return str_impl.c_str();
}
char& operator[](std::size_t index)
{
return str_impl[index];
@@ -121,16 +122,6 @@ class alt_string
return str_impl[index];
}
char& back()
{
return str_impl.back();
}
const char& back() const
{
return str_impl.back();
}
void clear()
{
str_impl.clear();
@@ -146,28 +137,11 @@ class alt_string
return str_impl.empty();
}
std::size_t find(const alt_string& str, std::size_t pos = 0) const
{
return str_impl.find(str.str_impl, pos);
}
std::size_t find_first_of(char c, std::size_t pos = 0) const
{
return str_impl.find_first_of(c, pos);
}
alt_string substr(std::size_t pos = 0, std::size_t count = npos) const
{
const std::string s = str_impl.substr(pos, count);
return {s.data(), s.size()};
}
alt_string& replace(std::size_t pos, std::size_t count, const alt_string& str)
{
str_impl.replace(pos, count, str.str_impl);
return *this;
}
void reserve( std::size_t new_cap = 0 )
{
str_impl.reserve(new_cap);
@@ -202,6 +176,31 @@ bool operator<(const char* op1, const alt_string& op2) noexcept
TEST_CASE("alternative string type")
{
SECTION("binary formats")
{
alt_json doc;
doc["pi"] = 3.141;
doc["happy"] = true;
doc["list"] = {1, 2, 3};
CHECK(alt_json::from_cbor(alt_json::to_cbor(doc)) == doc);
CHECK(alt_json::from_msgpack(alt_json::to_msgpack(doc)) == doc);
// BSON is not covered: it additionally needs string_t::find(value_type),
// which alt_string does not provide
CHECK(alt_json::from_ubjson(alt_json::to_ubjson(doc)) == doc);
// a UBJSON high-precision number is parsed into a std::string that the
// reader has to hand to the SAX interface as an alt_string
const std::vector<uint8_t> high_precision =
{
'H', 'i', 0x16, '3', '.', '1', '4', '1', '5', '9', '2', '6', '5', '3',
'5', '8', '9', '7', '9', '3', '2', '3', '8', '4', '6'
};
const auto number = alt_json::from_ubjson(high_precision);
CHECK(number.is_number_float());
CHECK(number.get<double>() == doctest::Approx(3.14159265358979323846));
}
SECTION("dump")
{
{
@@ -332,6 +331,15 @@ TEST_CASE("alternative string type")
CHECK(j.at(alt_json::json_pointer("/foo/0")) == j["foo"][0]);
CHECK(j.at(alt_json::json_pointer("/foo/1")) == j["foo"][1]);
// RFC 6901 escaping works without string_t::find(str, pos), replace(),
// and substr()
auto j2 = alt_json::parse(R"({"a/b": 1, "m~n": 2, "~/~~//": 3})");
CHECK(j2.at(alt_json::json_pointer("/a~1b")) == 1);
CHECK(j2.at(alt_json::json_pointer("/m~0n")) == 2);
CHECK(j2.at(alt_json::json_pointer("/~0~1~0~0~1~1")) == 3);
CHECK(alt_json::json_pointer("/~0~1~0~0~1~1").to_string() == alt_string("/~0~1~0~0~1~1"));
CHECK(j2.flatten().unflatten() == j2);
}
SECTION("patch")
+198
View File
@@ -0,0 +1,198 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <cstdint>
#include <string>
#include <vector>
namespace
{
// a spread of values exercising every writer path: scalars of each width, the
// float paths, strings, binary, and containers big enough to reallocate
std::vector<json> test_values()
{
json big_array = json::array();
for (int i = 0; i < 5000; ++i)
{
big_array.push_back(i);
}
json big_object = json::object();
for (int i = 0; i < 1000; ++i)
{
big_object[std::to_string(i)] = i;
}
return
{
json(nullptr), json(true), json(false),
json(0), json(-1), json(255), json(-129), json(65535), json(-32769),
json(4294967295U), json(-2147483649LL), json(18446744073709551615ULL),
json(0.0), json(-0.5), json(3.1415926535897932),
json(""), json("hello"), json(std::string(1000, 'x')),
json::binary({0x00, 0x01, 0x02}, 42),
json::array(), json::object(),
json::array({1, 2, 3}), json({{"a", 1}, {"b", nullptr}}),
json({{"nested", {{"deep", json::array({1, "two", 3.0, nullptr})}}}}),
big_array, big_object
};
}
// values to_bson() accepts: the document must be an object
std::vector<json> bson_values()
{
json big_object = json::object();
for (int i = 0; i < 1000; ++i)
{
big_object[std::to_string(i)] = i;
}
return
{
json::object(),
json({{"a", 1}, {"b", nullptr}, {"c", true}, {"d", 2.5}, {"e", "text"}}),
json({{"arr", json::array({1, 2, 3})}, {"obj", {{"k", "v"}}}}),
big_object
};
}
} // namespace
// The vector-returning to_*(j) overloads write through the non-virtual
// output_vector_sink, while to_*(j, adapter) goes through output_adapter_sink.
// The two are separate code paths that must stay byte-for-byte identical; these
// checks fail if either overload is ever changed without the other.
TEST_CASE("binary writer output sinks")
{
SECTION("vector sink and adapter sink agree")
{
// note: no SUBCASE inside these loops - doctest keys subcases by
// name/file/line, so a subcase in a loop body would only ever run for
// the first iteration
for (const auto& j : test_values())
{
CAPTURE(j.dump(-1, ' ', false, json::error_handler_t::replace));
std::vector<std::uint8_t> cbor;
json::to_cbor(j, cbor);
CHECK(json::to_cbor(j) == cbor);
std::vector<std::uint8_t> msgpack;
json::to_msgpack(j, msgpack);
CHECK(json::to_msgpack(j) == msgpack);
for (const bool use_size :
{
false, true
})
{
for (const bool use_type :
{
false, true
})
{
if (use_type && !use_size)
{
continue; // not a supported combination
}
CAPTURE(use_size);
CAPTURE(use_type);
std::vector<std::uint8_t> ubjson;
json::to_ubjson(j, ubjson, use_size, use_type);
CHECK(json::to_ubjson(j, use_size, use_type) == ubjson);
}
}
for (const auto version :
{
json::bjdata_version_t::draft2, json::bjdata_version_t::draft3
})
{
std::vector<std::uint8_t> bjdata;
json::to_bjdata(j, bjdata, false, false, version);
CHECK(json::to_bjdata(j, false, false, version) == bjdata);
}
}
for (const auto& j : bson_values())
{
CAPTURE(j.dump());
std::vector<std::uint8_t> bson;
json::to_bson(j, bson);
CHECK(json::to_bson(j) == bson);
}
}
SECTION("the char adapter produces the same bytes")
{
for (const auto& j : test_values())
{
CAPTURE(j.dump(-1, ' ', false, json::error_handler_t::replace));
const std::vector<std::uint8_t> expected = json::to_cbor(j);
std::vector<char> as_char;
json::to_cbor(j, as_char);
REQUIRE(as_char.size() == expected.size());
std::vector<std::uint8_t> as_bytes;
as_bytes.reserve(as_char.size());
for (const char c : as_char)
{
as_bytes.push_back(static_cast<std::uint8_t>(c));
}
CHECK(as_bytes == expected);
}
}
}
// binary_reserve_hint() is documented as a *lower* bound on the serialized size,
// so that reserving it up front can never leave the returned vector holding
// capacity beyond what the value actually needs.
TEST_CASE("binary_reserve_hint never over-reserves")
{
for (const auto& j : test_values())
{
CAPTURE(j.dump(-1, ' ', false, json::error_handler_t::replace));
const std::size_t hint = nlohmann::detail::binary_reserve_hint(j);
CHECK(hint <= json::to_cbor(j).size());
CHECK(hint <= json::to_msgpack(j).size());
CHECK(hint <= json::to_ubjson(j).size());
CHECK(hint <= json::to_ubjson(j, true, true).size());
CHECK(hint <= json::to_bjdata(j).size());
}
for (const auto& j : bson_values())
{
CAPTURE(j.dump());
CHECK(nlohmann::detail::binary_reserve_hint(j) <= json::to_bson(j).size());
}
SECTION("scalars get no hint")
{
CHECK(nlohmann::detail::binary_reserve_hint(json(nullptr)) == 0);
CHECK(nlohmann::detail::binary_reserve_hint(json(42)) == 0);
CHECK(nlohmann::detail::binary_reserve_hint(json("a string")) == 0);
CHECK(nlohmann::detail::binary_reserve_hint(json::binary({0x01})) == 0);
}
SECTION("containers are hinted from their element count")
{
CHECK(nlohmann::detail::binary_reserve_hint(json::array()) == 1);
CHECK(nlohmann::detail::binary_reserve_hint(json::array({1, 2, 3})) == 4);
CHECK(nlohmann::detail::binary_reserve_hint(json::object()) == 1);
CHECK(nlohmann::detail::binary_reserve_hint(json({{"a", 1}, {"b", 2}})) == 5);
}
}
+547 -14
View File
@@ -19,6 +19,7 @@ using nlohmann::json;
#include <fstream>
#include <set>
#include "make_test_data_available.hpp"
#include "round_trip_corpus.hpp"
#include "test_utils.hpp"
namespace
@@ -2586,7 +2587,12 @@ TEST_CASE("BJData")
CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d);
CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D);
CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C);
CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true) == v_B);
// v_B uses the Draft-3-only 'B' marker, so it round-trips only when
// Draft 3 is explicitly selected (see GitHub issue #5404); the
// default Draft 2 falls back to a plain object instead, covered by
// the "ndarray with _ArrayType_ "byte" is gated by the BJData draft
// version" section below
CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true, json::bjdata_version_t::draft3) == v_B);
}
SECTION("ndarray with data not matching _ArrayType_ is written as an object")
@@ -2599,25 +2605,25 @@ TEST_CASE("BJData")
// that still round-trips.
// string data declared as a uint64 array
json const j_str = json({{"_ArrayType_", "uint64"}, {"_ArraySize_", {1}}, {"_ArrayData_", {"pointer"}}});
json const j_str = json({{"_ArrayType_", "uint64"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {"pointer", "value"}}});
const auto out_str = json::to_bjdata(j_str);
CHECK(out_str.at(0) == '{');
CHECK(json::from_bjdata(out_str) == j_str);
// integer data declared as a double array
json const j_float = json({{"_ArrayType_", "double"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
json const j_float = json({{"_ArrayType_", "double"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, 2}}});
const auto out_float = json::to_bjdata(j_float);
CHECK(out_float.at(0) == '{');
CHECK(json::from_bjdata(out_float) == j_float);
// a non-integer shape entry is likewise not treated as an ndarray
json const j_size = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {"x"}}, {"_ArrayData_", {1}}});
json const j_size = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {"x", 1}}, {"_ArrayData_", {1}}});
const auto out_size = json::to_bjdata(j_size);
CHECK(out_size.at(0) == '{');
CHECK(json::from_bjdata(out_size) == j_size);
// a negative shape entry is not a usable dimension either
json const j_neg = json::parse(R"({"_ArrayType_":"uint8","_ArraySize_":[-1],"_ArrayData_":[1]})");
json const j_neg = json::parse(R"({"_ArrayType_":"uint8","_ArraySize_":[-1,1],"_ArrayData_":[1]})");
const auto out_neg = json::to_bjdata(j_neg);
CHECK(out_neg.at(0) == '{');
CHECK(json::from_bjdata(out_neg) == j_neg);
@@ -2629,8 +2635,10 @@ TEST_CASE("BJData")
// the C++ API stores an int literal as number_integer, so _ArrayType_
// names the wire type rather than the storage. Both storages have to
// produce the same typed array for every type.
// "byte" is checked separately below since it additionally requires
// BJData Draft 3 to be selected explicitly (see GitHub issue #5404).
for (const char* type :
{"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char", "byte"
{"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char"
})
{
CAPTURE(type);
@@ -2641,15 +2649,23 @@ TEST_CASE("BJData")
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}})));
}
{
const std::string text = R"({"_ArrayType_":"byte","_ArraySize_":[2,3],"_ArrayData_":[1,2,3,4,5,6]})";
const auto from_text = json::to_bjdata(json::parse(text), true, true, json::bjdata_version_t::draft3);
CHECK(from_text.at(0) == '[');
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}),
true, true, json::bjdata_version_t::draft3));
}
// negative values under a signed type behave the same way
const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})"));
const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2,1],"_ArrayData_":[-5,7]})"));
CHECK(from_neg.at(0) == '[');
CHECK(from_neg == json::to_bjdata(json({{"_ArrayType_", "int32"}, {"_ArraySize_", {2}}, {"_ArrayData_", {-5, 7}}})));
CHECK(from_neg == json::to_bjdata(json({{"_ArrayType_", "int32"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {-5, 7}}})));
// and so do the floating point types
const auto from_float = json::to_bjdata(json::parse(R"({"_ArrayType_":"double","_ArraySize_":[2],"_ArrayData_":[1.5,2.5]})"));
const auto from_float = json::to_bjdata(json::parse(R"({"_ArrayType_":"double","_ArraySize_":[2,1],"_ArrayData_":[1.5,2.5]})"));
CHECK(from_float.at(0) == '[');
CHECK(from_float == json::to_bjdata(json({{"_ArrayType_", "double"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1.5, 2.5}}})));
CHECK(from_float == json::to_bjdata(json({{"_ArrayType_", "double"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 2.5}}})));
}
SECTION("optimized ndarray (type and vector-size as 1D array)")
@@ -2731,6 +2747,83 @@ TEST_CASE("BJData")
CHECK(json::from_bjdata(json::to_bjdata(j_size), true, true) == j_size);
}
SECTION("ndarray whose _ArrayType_ is not a string stays as object")
{
// the type name is looked up as a string below the annotation
// check; a non-string _ArrayType_ cannot name a known dtype,
// so calling get<string_t>() on it would throw type_error.302
// instead of falling back like an unrecognized type name
// already does (see GitHub issue #5398)
json const j_number = json({{"_ArrayType_", 1}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
const auto out_number = json::to_bjdata(j_number);
CHECK(out_number.at(0) == '{');
CHECK(json::from_bjdata(out_number) == j_number);
json const j_null = json({{"_ArrayType_", nullptr}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
const auto out_null = json::to_bjdata(j_null);
CHECK(out_null.at(0) == '{');
CHECK(json::from_bjdata(out_null) == j_null);
json const j_bool = json({{"_ArrayType_", true}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
const auto out_bool = json::to_bjdata(j_bool);
CHECK(out_bool.at(0) == '{');
CHECK(json::from_bjdata(out_bool) == j_bool);
json const j_array = json({{"_ArrayType_", {"uint8"}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
const auto out_array = json::to_bjdata(j_array);
CHECK(out_array.at(0) == '{');
CHECK(json::from_bjdata(out_array) == j_array);
json const j_object = json({{"_ArrayType_", {{"a", 1}}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
const auto out_object = json::to_bjdata(j_object);
CHECK(out_object.at(0) == '{');
CHECK(json::from_bjdata(out_object) == j_object);
}
SECTION("re-serializing a value containing a plain-array-of-bytes is value-stable but not byte-stable")
{
// OSS-Fuzz found this input (an array whose first element is a
// binary_t byte, followed by an object whose _ArrayType_ is
// not a string) while exercising the fix for #5398 above: once
// the fix stops to_bjdata() from throwing type_error.302 for
// the third element, serialization proceeds far enough to
// reach a pre-existing, unrelated round-trip quirk in how a
// single-byte binary_t value is re-encoded.
std::vector<std::uint8_t> const input
{
0x5b, 0x5b, 0x24, 0x42, 0x23, 0x5b, 0x69, 0x01, 0x5d, 0x5b, 0x5b, 0x5d, 0x7b, 0x55, 0x0b,
0x5f, 0x41, 0x72, 0x72, 0x61, 0x79, 0x44, 0x61, 0x74, 0x61, 0x5f, 0x54, 0x55, 0x0b, 0x5f,
0x41, 0x72, 0x72, 0x61, 0x79, 0x53, 0x69, 0x7a, 0x65, 0x5f, 0x5a, 0x55, 0x0b, 0x5f, 0x41,
0x72, 0x72, 0x61, 0x79, 0x54, 0x79, 0x70, 0x65, 0x5f, 0x54, 0x7d, 0x5d
};
json const j1 = json::from_bjdata(input);
// to_bjdata() must not throw (this is what #5398 fixes)
std::vector<std::uint8_t> vec2;
CHECK_NOTHROW(vec2 = json::to_bjdata(j1, false, false));
// parsing back a plain (non-optimized) array of bytes cannot
// recover that it used to be a binary_t: from_bjdata() has no
// way to distinguish "array of uint8 numbers" from "array of
// bytes" unless the compact "$U#" array header is used, so
// the binary_t collapses into a plain JSON array
json const j2 = json::from_bjdata(vec2);
CHECK(j1 != j2);
CHECK(j2 == json({{91}, json::array(), {{"_ArrayData_", true}, {"_ArraySize_", nullptr}, {"_ArrayType_", true}}}));
// re-serializing j2 no longer goes through the dedicated
// binary_t writer (which always uses the 'U' marker for raw
// bytes); the now-plain number 91 goes through the generic
// smallest-type writer instead, which - like the rest of the
// UBJSON/BJData writer, and unchanged by this fix - prefers
// the 'i' (int8) marker over 'U' (uint8) for values that fit
// both. Both markers are valid BJData and both decode back to
// 91, so this is not byte-for-byte identical to vec2, but it
// is value-stable: parsing it again reproduces j2 exactly.
std::vector<std::uint8_t> const vec3 = json::to_bjdata(j2, false, false);
CHECK(json::from_bjdata(vec3) == j2);
}
SECTION("ndarray whose dimensions overflow stays as object")
{
// the product of the dimensions wraps around std::size_t to 0
@@ -2743,7 +2836,7 @@ TEST_CASE("BJData")
// a single dimension that does not fit into std::size_t is
// rejected for the same reason (only observable where
// std::size_t is narrower than 64 bit)
json j_huge = json({{"_ArrayData_", json::array()}, {"_ArraySize_", {18446744073709551615ull}}, {"_ArrayType_", "uint8"}});
json j_huge = json({{"_ArrayData_", json::array()}, {"_ArraySize_", {18446744073709551615ull, 2}}, {"_ArrayType_", "uint8"}});
CHECK(json::from_bjdata(json::to_bjdata(j_huge), true, true) == j_huge);
// a well-formed ndarray is still encoded as one
@@ -2751,6 +2844,197 @@ TEST_CASE("BJData")
CHECK(json::to_bjdata(j_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, 'i', 3, ']', 1, 2, 3, 4, 5, 6}));
CHECK(json::from_bjdata(json::to_bjdata(j_ok), true, true) == j_ok);
}
SECTION("ndarray whose _ArraySize_ is not an array stays as object")
{
// the shape is written verbatim as the header length, so a
// value that is not an array cannot produce a valid one: null
// would emit 'Z' and an object '{', neither of which a reader
// accepts after '#'. Both have to stay plain objects.
json const j_null = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", nullptr}, {"_ArrayData_", json::array()}});
const auto out_null = json::to_bjdata(j_null);
CHECK(out_null.at(0) == '{');
CHECK(json::from_bjdata(out_null) == j_null);
// an object shape passes the per-entry check by iterating its
// values rather than dimensions, so it needs rejecting too
json const j_obj = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {{"a", 1}}}, {"_ArrayData_", {1}}});
const auto out_obj = json::to_bjdata(j_obj);
CHECK(out_obj.at(0) == '{');
CHECK(json::from_bjdata(out_obj) == j_obj);
// a scalar shape is not a dimension list either
json const j_num = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", 1}, {"_ArrayData_", {1}}});
const auto out_num = json::to_bjdata(j_num);
CHECK(out_num.at(0) == '{');
CHECK(json::from_bjdata(out_num) == j_num);
// OSS-Fuzz issue 474400817: an empty object _ArraySize_ was
// written as the ND-array header length, which from_bjdata()
// could not read back
const std::vector<uint8_t> input =
{
'[', '{', 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'D', 'a', 't', 'a', '_', 'Z',
'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'T', 'y', 'p', 'e', '_', 'S', 'i', 5, 'i', 'n', 't', '1', '6',
'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'S', 'i', 'z', 'e', '_', '{', '}', '}', ']'
};
const json j1 = json::from_bjdata(input);
CHECK(j1 == json::parse(R"([{"_ArrayType_":"int16","_ArraySize_":{},"_ArrayData_":null}])"));
json j2;
CHECK_NOTHROW(j2 = json::from_bjdata(json::to_bjdata(j1, false, false)));
CHECK(j2 == j1);
}
SECTION("ndarray with out-of-range _ArrayData_ elements stays as object")
{
// each element is cast to the (possibly narrower) C++ type
// named by _ArrayType_ before being written; a value that
// does not fit that type would silently wrap instead of
// being reported, so such an object falls back to a plain
// object encoding that still round-trips (see GitHub issue #5403)
// an unsigned element that does not fit uint8
json const j_uint8 = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, 256}}});
const auto out_uint8 = json::to_bjdata(j_uint8);
CHECK(out_uint8.at(0) == '{');
CHECK(json::from_bjdata(out_uint8) == j_uint8);
// a signed element that does not fit int8
json const j_int8 = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, 200}}});
const auto out_int8 = json::to_bjdata(j_int8);
CHECK(out_int8.at(0) == '{');
CHECK(json::from_bjdata(out_int8) == j_int8);
// a negative element is likewise out of range for an
// unsigned _ArrayType_
json const j_uint16_neg = json({{"_ArrayType_", "uint16"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, -1}}});
const auto out_uint16_neg = json::to_bjdata(j_uint16_neg);
CHECK(out_uint16_neg.at(0) == '{');
CHECK(json::from_bjdata(out_uint16_neg) == j_uint16_neg);
// a double element that overflows to infinity when narrowed
// to the "single" (float) precision named by _ArrayType_
json const j_single = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 1e40}}});
const auto out_single = json::to_bjdata(j_single);
CHECK(out_single.at(0) == '{');
CHECK(json::from_bjdata(out_single) == j_single);
// in-range boundary values still use the compact ndarray encoding
json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {0, 255}}});
CHECK(json::to_bjdata(j_uint8_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, 'i', 1, ']', 0, 255}));
json const j_int8_ok = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {-128, 127}}});
CHECK(json::to_bjdata(j_int8_ok) == std::vector<uint8_t>({'[', '$', 'i', '#', '[', 'i', 2, 'i', 1, ']', 0x80, 0x7F}));
json const j_single_ok = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, -1.5}}});
const auto out_single_ok = json::to_bjdata(j_single_ok);
CHECK(out_single_ok.at(0) == '[');
CHECK(json::from_bjdata(out_single_ok) == json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5f, -1.5f}}}));
}
SECTION("ndarray that would not be read back as an annotated object stays as object")
{
// the reader only restores an annotated object from an ND-array
// with at least two non-zero dimensions that is not a 1xN row
// vector; any other shape is read back as a plain array. Writing
// such an object as an ND-array would drop its annotation, so it
// falls back to a plain object encoding that round-trips.
for (const char* text :
{
R"({"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":[]})",
R"({"_ArrayType_":"int16","_ArraySize_":[2],"_ArrayData_":[1,2]})",
R"({"_ArrayType_":"int16","_ArraySize_":[1,2],"_ArrayData_":[1,2]})",
R"({"_ArrayType_":"int16","_ArraySize_":[0],"_ArrayData_":[]})",
R"({"_ArrayType_":"int16","_ArraySize_":[2,0],"_ArrayData_":[]})",
R"({"_ArrayType_":"int16","_ArraySize_":[0,2],"_ArrayData_":[]})"
})
{
CAPTURE(text);
const json j = json::parse(text);
for (const bool use_size :
{
false, true
})
{
const auto out = json::to_bjdata(j, use_size, use_size);
CHECK(out.at(0) == '{');
CHECK(json::from_bjdata(out) == j);
}
}
// a genuine ND-array still uses the compact encoding and round-trips
const json j_2d = json::parse(R"({"_ArrayType_":"int16","_ArraySize_":[2,1],"_ArrayData_":[1,2]})");
const auto out_2d = json::to_bjdata(j_2d);
CHECK(out_2d.at(0) == '[');
CHECK(json::from_bjdata(out_2d) == j_2d);
}
SECTION("ndarray with non-array _ArrayData_ stays as object")
{
// the elements are written from _ArrayData_ as a flat list, so it
// has to be an array: null has size 0, any other scalar has size 1,
// and iterating an object visits its values, so each of these could
// match the dimensions and be encoded as an unrelated ND-array
for (const char* text :
{
R"({"_ArrayType_":"int16","_ArraySize_":[2,1],"_ArrayData_":null})",
R"({"_ArrayType_":"int16","_ArraySize_":[2,1],"_ArrayData_":{"a":1,"b":2}})",
R"({"_ArrayType_":"int16","_ArraySize_":[1],"_ArrayData_":5})",
R"({"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":null})"
})
{
CAPTURE(text);
const json j = json::parse(text);
const auto out = json::to_bjdata(j);
CHECK(out.at(0) == '{');
CHECK(json::from_bjdata(out) == j);
}
// OSS-Fuzz issue 563659413: an empty binary _ArraySize_ is written
// as a plain object and read back as an empty array, after which
// the object with a null _ArrayData_ was encoded as an empty
// ND-array and re-read as [], so a second round trip lost the value
const std::vector<uint8_t> input =
{
'{', 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'D', 'a', 't', 'a', '_', 'Z',
'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'T', 'y', 'p', 'e', '_', 'S', 'i', 5, 'i', 'n', 't', '1', '6',
'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'S', 'i', 'z', 'e', '_', '[', '$', 'B', '#', '[', ']', '}'
};
const json j1 = json::from_bjdata(input);
const json j2 = json::from_bjdata(json::to_bjdata(j1, false, false));
CHECK(j2 == json::parse(R"({"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":null})"));
CHECK(json::from_bjdata(json::to_bjdata(j2, false, false)) == j2);
}
SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version")
{
// the 'B' (byte) marker used by _ArrayType_ "byte" is only defined
// by BJData Draft 3; Draft 2 (the default) has no such marker, so
// emitting it unconditionally produced a stream that a Draft 2
// reader could not parse as intended (see GitHub issue #5404).
// Two dimensions are used so that a successfully written ndarray
// round-trips back into the annotated object (a single dimension
// is, by the BJData ndarray convention, read back as a plain
// binary value rather than the annotated object, same as every
// other single-dimension ndarray of a non-"byte" type is read
// back as a plain array instead of the annotated object).
json const j_byte = json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}});
// default (Draft 2): falls back to a plain object and round-trips
const auto out_draft2 = json::to_bjdata(j_byte);
CHECK(out_draft2.at(0) == '{');
CHECK(json::from_bjdata(out_draft2) == j_byte);
// explicit Draft 2: same as the default
const auto out_draft2_explicit = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft2);
CHECK(out_draft2_explicit.at(0) == '{');
CHECK(json::from_bjdata(out_draft2_explicit) == j_byte);
// Draft 3 explicitly selected: still uses the compact 'B' ndarray encoding
const auto out_draft3 = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft3);
CHECK(out_draft3 == std::vector<uint8_t>({'[', '$', 'B', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6}));
CHECK(json::from_bjdata(out_draft3) == j_byte);
}
}
}
@@ -3263,8 +3547,10 @@ TEST_CASE("BJData")
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR1), "[json.exception.parse_error.113] parse error at byte 6: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
CHECK(json::from_bjdata(vR1, true, false).is_discarded());
// a dimension vector that opens another one is rejected where the
// nested '[' is read, rather than after it has been descended into
std::vector<uint8_t> const vR2 = {'[', '$', 'i', '#', '[', '#', '[', 'i', 1, ']', ']', 1};
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR2), "[json.exception.parse_error.113] parse error at byte 11: syntax error while parsing BJData size: expected length type specification (U, i, u, I, m, l, M, L) after '#'; last byte: 0x5D", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR2), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
CHECK(json::from_bjdata(vR2, true, false).is_discarded());
std::vector<uint8_t> const vR3 = {'[', '#', '[', 'i', '2', 'i', 2, ']'};
@@ -3272,7 +3558,7 @@ TEST_CASE("BJData")
CHECK(json::from_bjdata(vR3, true, false).is_discarded());
std::vector<uint8_t> const vR4 = {'[', '$', 'i', '#', '[', '$', 'i', '#', '[', 'i', 1, ']', 1};
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR4), "[json.exception.parse_error.110] parse error at byte 14: syntax error while parsing BJData number: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR4), "[json.exception.parse_error.113] parse error at byte 9: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
CHECK(json::from_bjdata(vR4, true, false).is_discarded());
std::vector<uint8_t> const vR5 = {'[', '$', 'i', '#', '[', '[', '[', ']', ']', ']'};
@@ -3280,12 +3566,25 @@ TEST_CASE("BJData")
CHECK(json::from_bjdata(vR5, true, false).is_discarded());
std::vector<uint8_t> const vR6 = {'[', '$', 'i', '#', '[', '$', 'i', '#', '[', 'i', '2', 'i', 2, ']'};
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR6), "[json.exception.parse_error.112] parse error at byte 14: syntax error while parsing BJData size: ndarray can not be recursive", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR6), "[json.exception.parse_error.113] parse error at byte 9: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
CHECK(json::from_bjdata(vR6, true, false).is_discarded());
std::vector<uint8_t> const vH = {'[', 'H', '[', '#', '[', '$', 'i', '#', '[', 'i', '2', 'i', 2, ']'};
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vH), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
CHECK(json::from_bjdata(vH, true, false).is_discarded());
// Every "#[" of this chain used to open another dimension vector
// and cost several stack frames before anything was rejected, so a
// long enough chain crashed the process (see #5104). The nested
// vector is refused where it is read, so the length is irrelevant.
std::vector<uint8_t> vRdeep = {'['};
for (std::size_t i = 0; i < 100000; ++i)
{
vRdeep.push_back('#');
vRdeep.push_back('[');
}
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vRdeep), "[json.exception.parse_error.113] parse error at byte 5: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&);
CHECK(json::from_bjdata(vRdeep, true, false).is_discarded());
}
SECTION("objects")
@@ -3464,6 +3763,111 @@ TEST_CASE("BJData")
}
}
TEST_CASE("issue #5405 - array reserve for definite-length BJData arrays")
{
#if !defined(JSON_NOEXCEPTION)
// this SECTION relies on catching a thrown exception to distinguish
// which of two acceptable, bounded rejections a hostile header took;
// under JSON_NOEXCEPTION, JSON_THROW never produces a catchable C++
// exception (it aborts instead), so this cannot be tested that way here
SECTION("a huge claimed length with no element data must not over-allocate")
{
// optimized form [$type#count: type 'i' (int8), count as a four-byte
// little-endian 'l' (int32) of 0x7FFFFFFF (2147483647), but no
// element data at all. max_size() for a std::vector is far larger
// than this count, so it does not reject the header outright; the
// (capped) reservation must not attempt to allocate space for
// billions of elements before the missing data is detected.
json _;
const std::vector<uint8_t> input = {'[', '$', 'i', '#', 'l', 0xFF, 0xFF, 0xFF, 0x7F};
// On a platform where std::vector<json>::max_size() is smaller than
// the claimed count (e.g. 32-bit, where max_size() is bounded by a
// 32-bit SIZE_MAX divided by sizeof(json)), the SAX consumer's own
// check rejects the header outright (out_of_range.408, with the
// claimed count in the message) instead of accepting it and only
// finding it short of data once the (capped) reservation looks for
// element bytes that were never provided (parse_error.110). Either
// is an acceptable, bounded rejection of the hostile header -- the
// property under test is that no path attempts to allocate space
// for billions of elements.
bool threw = false;
try
{
_ = json::from_bjdata(input);
}
catch (const json::parse_error& e)
{
threw = true;
CHECK(e.id == 110);
CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 10: syntax error while parsing BJData number: unexpected end of input");
}
catch (const json::out_of_range& e)
{
threw = true;
CHECK(e.id == 408);
CHECK(std::string(e.what()).find("excessive array size") != std::string::npos);
}
CHECK(threw);
// json_sax_dom_parser::start_array()'s max_size() check (unlike the
// scanner's own parse_error path) throws unconditionally via
// JSON_THROW rather than going through sax->parse_error(), so it is
// not gated by allow_exceptions=false on a platform where this
// header hits that check (e.g. 32-bit, see above) -- allow either
// a discarded result or the same out_of_range it throws with
// exceptions enabled.
try
{
CHECK(json::from_bjdata(input, true, false).is_discarded());
}
catch (const json::out_of_range& e)
{
CHECK(e.id == 408);
}
}
#endif
SECTION("arrays of various sizes decode to the same value as before the reserve optimization")
{
for (const auto size :
{
std::size_t{0}, std::size_t{1}, std::size_t{5}, // small
std::size_t{16384}, // exactly at the reserve cap
std::size_t{20000} // above the reserve cap
})
{
CAPTURE(size)
json j = json::array();
for (std::size_t i = 0; i < size; ++i)
{
j.push_back(static_cast<int>(i % 1000));
}
// exercise both the plain and the optimized [$type#count encoding
const auto packed_plain = json::to_bjdata(j);
CHECK(json::from_bjdata(packed_plain) == j);
const auto packed_optimized = json::to_bjdata(j, true, true);
CHECK(json::from_bjdata(packed_optimized) == j);
}
}
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
{
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
// a custom SAX consumer that does not touch a DOM array sees identical events
json j = json::array();
for (int i = 0; i < 100; ++i)
{
j.push_back(i);
}
const auto packed = json::to_bjdata(j, true, true);
SaxCountdown scp(1000000); // large enough to never trigger an abort
CHECK(json::sax_parse(packed, &scp, json::input_format_t::bjdata));
}
}
TEST_CASE("Universal Binary JSON Specification Examples 1")
{
SECTION("Null Value")
@@ -3843,6 +4247,135 @@ TEST_CASE("all BJData first bytes")
}
#endif
TEST_CASE("BJData use_type requires use_size")
{
SECTION("non-empty object throws other_error.502")
{
const json j = {{"a", 1}, {"b", 2}};
CHECK_THROWS_WITH_AS(json::to_bjdata(j, false, true),
"[json.exception.other_error.502] use_type requires use_size = true",
json::other_error&);
}
SECTION("non-empty array throws other_error.502")
{
const json j = {1, 2, 3};
CHECK_THROWS_WITH_AS(json::to_bjdata(j, false, true),
"[json.exception.other_error.502] use_type requires use_size = true",
json::other_error&);
}
SECTION("scalars do not throw with use_type=true, use_count=false")
{
CHECK_NOTHROW(json::to_bjdata(42, false, true));
CHECK_NOTHROW(json::to_bjdata(3.14, false, true));
CHECK_NOTHROW(json::to_bjdata("hello", false, true));
CHECK_NOTHROW(json::to_bjdata(true, false, true));
CHECK_NOTHROW(json::to_bjdata(nullptr, false, true));
}
SECTION("empty containers do not throw with use_type=true, use_count=false")
{
CHECK_NOTHROW(json::to_bjdata(json::array(), false, true));
CHECK_NOTHROW(json::to_bjdata(json::object(), false, true));
}
SECTION("valid combinations on non-empty containers")
{
const json j = {{"a", 1}, {"b", 2}};
CHECK_NOTHROW(json::to_bjdata(j, false, false));
CHECK_NOTHROW(json::to_bjdata(j, true, false));
CHECK_NOTHROW(json::to_bjdata(j, true, true));
}
}
TEST_CASE("BJData round-trip invariants")
{
// This checks what the parse_bjdata_fuzzer driver checks (see
// tests/src/fuzzer-parse_bjdata.cpp), so that a regression shows up in CI
// rather than as an OSS-Fuzz report: every value from_bjdata() returns
// (j1) can be serialized with any combination of options, the result can
// be parsed back (j2), and serializing j2 again with the same options
// yields a value-equal result.
//
// Beyond the driver, this also checks that j2 equals j1 and that
// serializing j2 reproduces the exact bytes, both except for values that
// contain a binary value: a binary value is only written as a binary
// value with Draft 3's optimized binary array, and otherwise read back as
// an array of integers, for which the writer may choose different (but
// equally valid) type markers when it is serialized again (see #5494).
//
// Values are compared with dump() rather than operator==, because a NaN
// never compares equal to itself.
struct options
{
bool use_size;
bool use_type;
json::bjdata_version_t version;
};
const std::vector<options> all_options =
{
{false, false, json::bjdata_version_t::draft2},
{true, false, json::bjdata_version_t::draft2},
{true, true, json::bjdata_version_t::draft2},
{false, false, json::bjdata_version_t::draft3},
{true, false, json::bjdata_version_t::draft3},
{true, true, json::bjdata_version_t::draft3},
};
for (const auto& j0 : utils::round_trip_corpus::values())
{
// turn the corpus value into a value as from_bjdata() returns it
for (const auto& initial : all_options)
{
const json j1 = json::from_bjdata(json::to_bjdata(j0, initial.use_size, initial.use_type, initial.version));
const bool has_binary = utils::round_trip_corpus::contains_binary(j1);
for (const auto& o : all_options)
{
INFO("j1 = " << j1.dump() << ", use_size = " << o.use_size << ", use_type = " << o.use_type
<< ", draft3 = " << (o.version == json::bjdata_version_t::draft3));
const std::vector<std::uint8_t> vec = json::to_bjdata(j1, o.use_size, o.use_type, o.version);
json j2;
// anything the library writes must be parsable by the library
REQUIRE_NOTHROW(j2 = json::from_bjdata(vec));
const std::vector<std::uint8_t> vec2 = json::to_bjdata(j2, o.use_size, o.use_type, o.version);
CHECK(json::from_bjdata(vec2).dump() == j2.dump());
if (!has_binary)
{
CHECK(j2.dump() == j1.dump());
CHECK(vec2 == vec);
}
}
}
}
}
TEST_CASE("BJData round trip of a binary value is value-stable, not byte-stable")
{
// OSS-Fuzz issue 474480402: a Draft 3 optimized binary array is read as a
// binary value, which to_bjdata() writes in the default Draft 2 mode as a
// plain array of uint8 numbers. That is read back as an array of numbers,
// for which the writer then picks the smallest type marker, int8 ('i'),
// so re-serializing changes the bytes, but not the value. This is the
// exception described in the "Round trips" note of the BJData
// documentation, and why the fuzzer checks value stability (see #5494).
const std::vector<uint8_t> input = {'[', '$', 'B', '#', 'U', 1, 0x20};
const json j1 = json::from_bjdata(input);
CHECK(j1 == json::binary({0x20}));
const std::vector<uint8_t> vec = json::to_bjdata(j1, false, false);
CHECK(vec == std::vector<uint8_t>({'[', 'U', 0x20, ']'}));
const json j2 = json::from_bjdata(vec);
CHECK(j2 == json::array({0x20}));
const std::vector<uint8_t> vec2 = json::to_bjdata(j2, false, false);
CHECK(vec2 == std::vector<uint8_t>({'[', 'i', 0x20, ']'}));
CHECK(json::from_bjdata(vec2) == j2);
}
TEST_CASE("BJData roundtrips" * doctest::skip())
{
SECTION("input from self-generated BJData files")
@@ -0,0 +1,167 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
// This file tests the opt-in JSON_BRACE_INIT_COPY_SEMANTICS, so it defines the
// macro itself rather than relying on a -D flag, and runs in every build.
#ifdef JSON_BRACE_INIT_COPY_SEMANTICS
#undef JSON_BRACE_INIT_COPY_SEMANTICS
#endif
#define JSON_BRACE_INIT_COPY_SEMANTICS 1
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <array>
#include <list>
#include <map>
#include <string>
#include <tuple>
#include <utility>
#include <vector>
#define STRINGIZE_EX(x) #x
#define STRINGIZE(x) STRINGIZE_EX(x)
TEST_CASE("JSON_BRACE_INIT_COPY_SEMANTICS")
{
SECTION("the macro is part of the ABI tag")
{
const std::string ns = STRINGIZE(NLOHMANN_JSON_NAMESPACE);
// other tags may come before it, e.g. json_abi_ldvcmp_bics
CHECK(ns.find("_bics") != std::string::npos);
}
SECTION("single-element brace initialization copies the element (#5074)")
{
json const j_obj = {{"key", "value"}, {"num", 42}};
json const j_arr = {1, 2, 3};
// object: brace init copies instead of wrapping
json const j1{j_obj};
CHECK(j1.is_object());
CHECK(j1 == j_obj);
// array: brace init copies instead of wrapping
json const j2{j_arr};
CHECK(j2.is_array());
CHECK(j2.size() == 3);
CHECK(j2 == j_arr);
// this applies to any single element, not only to JSON values
json const j3{true};
CHECK(j3.is_boolean());
json const j4{42};
CHECK(j4.is_number_integer());
json const j5 = {1};
CHECK(j5 == 1);
json const j6 = {"text"};
CHECK(j6 == "text");
json const j7 = {{1, 2}};
CHECK(j7 == json::array({1, 2}));
}
SECTION("what the macro does not change")
{
// lists with more than one element are unaffected
json const j1 = {1, 2};
CHECK(j1.is_array());
CHECK(j1.size() == 2);
// a single [string, value] pair still describes an object
json const j2 = {{"key", "value"}};
CHECK(j2.is_object());
CHECK(j2["key"] == "value");
// json::array() always creates an array
json const j3 = json::array({1});
CHECK(j3.is_array());
CHECK(j3.size() == 1);
CHECK(j3[0] == 1);
json const j_obj = {{"key", "value"}};
json const j4 = json::array({j_obj});
CHECK(j4.is_array());
CHECK(j4.size() == 1);
CHECK(j4[0] == j_obj);
}
SECTION("conversions build the same values as without the macro")
{
SECTION("one-element std::tuple")
{
json const j1 = std::tuple<int> {5};
CHECK(j1.dump() == "[5]");
CHECK(std::get<0>(j1.get<std::tuple<int>>()) == 5);
json const j2 = std::tuple<std::string> {"text"};
CHECK(j2.dump() == "[\"text\"]");
CHECK(std::get<0>(j2.get<std::tuple<std::string>>()) == "text");
json const j3 = std::tuple<json> {json::array({1, 2})};
CHECK(j3.dump() == "[[1,2]]");
// as without the macro, a [string, value] pair becomes an object
// member (see the known limitation documented for std::pair)
json const j4 = std::tuple<std::pair<std::string, int>> {{"a", 1}};
CHECK(j4.dump() == "{\"a\":1}");
}
SECTION("tuples with more elements")
{
json const j1 = std::tuple<int, std::string> {1, "a"};
CHECK(j1.dump() == "[1,\"a\"]");
json const j2 = std::tuple<> {};
CHECK(j2.dump() == "[]");
}
SECTION("one-element containers")
{
json const j1 = std::vector<int> {1};
CHECK(j1.dump() == "[1]");
CHECK(j1.get<std::vector<int>>() == std::vector<int> {1});
std::array<int, 1> const arr = {{1}};
json const j2 = arr;
CHECK(j2.dump() == "[1]");
json const j3 = std::list<std::string> {"a"};
CHECK(j3.dump() == "[\"a\"]");
json const j4 = std::map<std::string, int> {{"a", 1}};
CHECK(j4.dump() == "{\"a\":1}");
json const j5 = std::map<int, int> {{1, 2}};
CHECK(j5.dump() == "[[1,2]]");
}
SECTION("std::pair")
{
json const j = std::pair<int, int> {1, 2};
CHECK(j.dump() == "[1,2]");
CHECK((j.get<std::pair<int, int>>() == std::pair<int, int> {1, 2}));
}
SECTION("items()")
{
json j_obj = {{"key", 1}};
for (const auto& el : j_obj.items())
{
json const j = el;
CHECK(j.dump() == "{\"key\":1}");
}
}
}
}
+236 -3
View File
@@ -38,6 +38,54 @@ class huge_binary_t : public std::vector<std::uint8_t>
using huge_binary_json = nlohmann::basic_json <
std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t,
double, std::allocator, nlohmann::adl_serializer, huge_binary_t, void >;
// a string type that can be made to report a size beyond INT32_MAX without
// allocating that much memory, so BSON length overflow can be tested for
// strings and (embedded) documents as well, following the same idea as
// huge_binary_t.
//
// Unlike huge_binary_t (which is only ever used as the BSON *value* type),
// this type doubles as basic_json's StringType and is therefore also used
// for *object keys* (e.g. "s" or "nested" below). Only the designated test
// value is meant to lie about its size - if every huge_string_t (including
// keys) reported a huge size, the running totals computed while walking the
// BSON document (see calc_bson_object_size & friends in binary_writer.hpp)
// would need more than 32 bits, and on platforms where std::size_t is only
// 32 bits wide that arithmetic would silently wrap around, producing wrong
// (or even unguarded) lengths. The fake size is therefore opt-in via
// as_huge(), and plain strings - in particular object keys - keep reporting
// their real, small size.
class huge_string_t : public std::string
{
public:
using std::string::string;
huge_string_t(const std::string& s) : std::string(s) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions)
// returns a copy of @a s whose size() pretends to be huge
static huge_string_t as_huge(const std::string& s)
{
huge_string_t result(s);
result.pretend_huge = true;
return result;
}
size_type size() const noexcept
{
if (pretend_huge)
{
// one byte more than the BSON length field can represent
return static_cast<size_type>((std::numeric_limits<std::int32_t>::max)()) + 1;
}
return std::string::size();
}
private:
bool pretend_huge = false;
};
using huge_string_json = nlohmann::basic_json <
std::map, std::vector, huge_string_t, bool, std::int64_t, std::uint64_t,
double, std::allocator, nlohmann::adl_serializer, std::vector<std::uint8_t>, void >;
} // namespace
TEST_CASE("BSON")
@@ -105,10 +153,36 @@ TEST_CASE("BSON")
SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON")
{
huge_binary_json j;
j["b"] = huge_binary_json::binary(huge_binary_t{});
// out_of_range.412 is thrown from a single shared helper
// (to_bson_length) that guards the BSON length fields of binary
// values, strings, and (embedded) documents alike
SECTION("binary")
{
huge_binary_json j;
j["b"] = huge_binary_json::binary(huge_binary_t{});
CHECK_THROWS_WITH_AS(huge_binary_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483661 exceeds maximum of 2147483647", huge_binary_json::out_of_range&);
CHECK_THROWS_WITH_AS(huge_binary_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483661 exceeds maximum of 2147483647", huge_binary_json::out_of_range&);
}
SECTION("string")
{
huge_string_json j;
j["s"] = huge_string_t::as_huge("value");
CHECK_THROWS_WITH_AS(huge_string_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483661 exceeds maximum of 2147483647", huge_string_json::out_of_range&);
}
SECTION("document")
{
// an oversized string nested one level deep makes the
// *embedded* document's own length exceed INT32_MAX as well
huge_string_json nested;
nested["s"] = huge_string_t::as_huge("value");
huge_string_json j;
j["nested"] = nested;
CHECK_THROWS_WITH_AS(huge_string_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483674 exceeds maximum of 2147483647", huge_string_json::out_of_range&);
}
}
SECTION("string length must be at least 1")
@@ -193,6 +267,23 @@ TEST_CASE("BSON")
CHECK(json::from_bson(result, true, false) == j);
}
SECTION("non-empty object with bool from a non-0/1 byte (lenient parsing)")
{
// documented lenient behavior (see gh-5333): any non-zero byte
// is accepted as `true`, not just 0x01
std::vector<std::uint8_t> const input =
{
0x0D, 0x00, 0x00, 0x00, // size (little endian)
0x08, // entry: boolean
'e', 'n', 't', 'r', 'y', '\x00',
0x02, // value = 0x02 (neither 0x00 nor 0x01)
0x00 // end marker
};
const json expected = { { "entry", true } };
CHECK(json::from_bson(input) == expected);
}
SECTION("non-empty object with double")
{
json const j =
@@ -499,6 +590,29 @@ TEST_CASE("BSON")
CHECK(json::from_bson(result, true, false) == j);
}
SECTION("array elements with non-conforming keys (lenient parsing)")
{
// documented lenient behavior (see gh-5333): BSON array element
// keys are not checked against the required decimal sequence
// "0", "1", "2", ... - elements are taken in encoded order
std::vector<std::uint8_t> const input =
{
0x26, 0x00, 0x00, 0x00, // size (little endian)
0x04, 'e', 'n', 't', 'r', 'y', '\x00', // entry: embedded array
0x1A, 0x00, 0x00, 0x00, // size (little endian)
0x10, '5', 0x00, 0x0A, 0x00, 0x00, 0x00, // key "5" (bogus) -> 10
0x10, 'x', 0x00, 0x14, 0x00, 0x00, 0x00, // key "x" (non-numeric) -> 20
0x10, '1', 0x00, 0x1E, 0x00, 0x00, 0x00, // key "1" (out of order) -> 30
0x00, // end marker (embedded array)
0x00 // end marker
};
const json expected = { { "entry", json::array({10, 20, 30}) } };
CHECK(json::from_bson(input) == expected);
}
SECTION("non-empty object with binary member")
{
const size_t N = 10;
@@ -594,6 +708,31 @@ TEST_CASE("BSON")
CHECK(json::from_bson(result, true, false) == j);
}
SECTION("binary member with subtype 0x02 (old binary) keeps its inner length prefix (lenient parsing)")
{
// documented lenient behavior (see gh-5333): the payload for
// binary subtype 0x02 ("old binary") is returned as-is,
// including its own inner 4-byte length prefix; it is not
// stripped or reinterpreted
std::vector<std::uint8_t> const input =
{
0x17, 0x00, 0x00, 0x00, // size (little endian)
0x05, 'e', 'n', 't', 'r', 'y', '\x00', // entry: binary
0x06, 0x00, 0x00, 0x00, // size of binary (little endian)
0x02, // "old binary" subtype
0x02, 0x00, 0x00, 0x00, // inner length prefix (part of the old-binary payload)
0x68, 0x69, // payload ('h', 'i')
0x00 // end marker
};
// the inner length prefix is part of the (unmodified) payload
const std::vector<std::uint8_t> expected_payload = {0x02, 0x00, 0x00, 0x00, 0x68, 0x69};
const json expected = { { "entry", json::binary(expected_payload, 0x02) } };
CHECK(json::from_bson(input) == expected);
}
SECTION("Some more complex document")
{
json const j =
@@ -652,6 +791,15 @@ TEST_CASE("BSON")
}
}
TEST_CASE("regression test - BSON binary subtype rejects a value that doesn't fit a single byte")
{
json const doc255 = {{"b", json::binary({1, 2}, 255)}};
CHECK(json::from_bson(json::to_bson(doc255))["b"].get_binary().subtype() == 255);
CHECK_THROWS_AS(json::to_bson(json{{"b", json::binary({1, 2}, 256)}}), json::out_of_range);
CHECK_THROWS_WITH_AS(json::to_bson(json{{"b", json::binary({1, 2}, 300)}}), "[json.exception.out_of_range.415] subtype 300 is too large for the BSON binary subtype (max 255)", json::out_of_range);
}
TEST_CASE("BSON input/output_adapters")
{
const json json_representation =
@@ -1011,6 +1159,91 @@ TEST_CASE("BSON document size mismatch")
}
}
TEST_CASE("BSON nesting does not consume the call stack")
{
// An embedded document or array used to be read by calling back into the
// document reader, so the native call stack grew with the nesting depth of
// the input (#5104). The open documents are kept on a heap stack now.
//
// Deeply nested values must not be compared, copied or dumped here: those
// operations are still recursive and would reintroduce the crash.
// A document nested deeply enough to have crashed. The bytes are built
// here rather than with to_bson(), because the writer still recurses once
// per level and would overflow the stack before the reader is ever
// reached. Every level is
// <int32 size> 0x03 'a' 0x00 <inner document> 0x00
// so a level is eight bytes larger than the one it holds, and the sizes
// can be filled in from the outside in.
const std::size_t depth = 30000;
std::vector<uint8_t> input;
input.reserve(5 + (8 * depth));
for (std::size_t i = 0; i < depth; ++i)
{
const auto size = static_cast<std::uint32_t>(5 + (8 * (depth - i)));
input.push_back(static_cast<uint8_t>(size & 0xFF));
input.push_back(static_cast<uint8_t>((size >> 8) & 0xFF));
input.push_back(static_cast<uint8_t>((size >> 16) & 0xFF));
input.push_back(static_cast<uint8_t>((size >> 24) & 0xFF));
input.push_back(0x03); // embedded document
input.push_back('a');
input.push_back(0x00);
}
// the innermost document is empty, then one terminator closes each level
input.insert(input.end(), {0x05, 0x00, 0x00, 0x00, 0x00});
input.insert(input.end(), depth, 0x00);
SECTION("a well-formed deep document is read through the SAX interface")
{
SaxCountdown accept_all(1000000);
CHECK(json::sax_parse(input, &accept_all, json::input_format_t::bson));
}
SECTION("a well-formed deep document is read into a value")
{
json j = json::from_bson(input);
// walked rather than compared: comparing, copying or dumping a value
// this deep is still recursive
std::size_t measured = 0;
const json* q = &j;
while (q->is_object() && !q->empty())
{
q = &q->begin().value();
++measured;
}
CHECK(measured == depth);
}
SECTION("embedded documents and arrays are still read the same way")
{
const json values = {{"a", {{"b", {{"c", 1}}}}}};
CHECK(json::from_bson(json::to_bson(values)) == values);
const json array = {{"a", {1, 2, 3}}};
CHECK(json::from_bson(json::to_bson(array)) == array);
const json mixed = {{"a", {json{{"x", 1}}, json{{"y", 2}}}}};
CHECK(json::from_bson(json::to_bson(mixed)) == mixed);
CHECK(json::from_bson(json::to_bson(json::object())) == json::object());
}
SECTION("a size that does not match is still reported per document")
{
// the embedded document claims one byte too many
std::vector<uint8_t> const bad =
{
0x15, 0x00, 0x00, 0x00, 0x03, 'a', 0x00,
0x0D, 0x00, 0x00, 0x00, 0x08, 'b', 0x00, 0x01, 0x00,
0x00
};
json _;
CHECK_THROWS_AS(_ = json::from_bson(bad), json::parse_error&);
CHECK(json::from_bson(bad, true, false).is_discarded());
}
}
TEST_CASE("BSON numerical data")
{
SECTION("number")
@@ -42,6 +42,39 @@ TEST_CASE("byte_container_with_subtype")
CHECK(container.subtype() == static_cast<subtype_type>(-1));
}
SECTION("move semantics")
{
// the rvalue-reference constructor (without a subtype) must actually move
// the passed-in container rather than copy it; comparing the buffer address
// before and after is a stronger check than just observing the source is
// empty afterward, since a copy-then-clear could also leave it empty
{
std::vector<std::uint8_t> bytes = {{0xCA, 0xFE, 0xBA, 0xBE}};
const auto* const data_ptr = bytes.data();
nlohmann::byte_container_with_subtype<std::vector<std::uint8_t>> container(std::move(bytes));
CHECK(container.size() == 4);
CHECK(container.data() == data_ptr);
CHECK(!container.has_subtype());
CHECK(bytes.empty()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved)
}
// same check for the rvalue-reference constructor that also takes a subtype
{
std::vector<std::uint8_t> bytes = {{0xCA, 0xFE, 0xBA, 0xBE}};
const auto* const data_ptr = bytes.data();
nlohmann::byte_container_with_subtype<std::vector<std::uint8_t>> container(std::move(bytes), 42);
CHECK(container.size() == 4);
CHECK(container.data() == data_ptr);
CHECK(container.has_subtype());
CHECK(container.subtype() == 42);
CHECK(bytes.empty()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved)
}
}
SECTION("comparisons")
{
std::vector<std::uint8_t> const bytes = {{0xCA, 0xFE, 0xBA, 0xBE}};
+225
View File
@@ -2035,6 +2035,231 @@ TEST_CASE("CBOR definite length equal to the indefinite-length sentinel")
}
}
TEST_CASE("CBOR nesting does not consume the call stack")
{
// Containers used to be read by calling back into the value reader once
// per element, and a tag by calling it for the tagged value, so the native
// call stack grew with the nesting depth of the input. Each of the three
// costs a single byte to encode -- 0x9F, 0x81 and 0xC2 -- so a payload of
// repeated bytes crashed the process (#5104). The containers are kept on a
// heap stack now, and a tag is read in a loop.
//
// Deeply nested values must not be compared, copied or dumped here: those
// operations are still recursive and would reintroduce the crash.
json _;
SECTION("indefinite-length containers")
{
const std::vector<uint8_t> input(500000, 0x9F);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.110] parse error at byte 500001: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&);
CHECK(json::from_cbor(input, true, false).is_discarded());
}
SECTION("definite-length containers")
{
const std::vector<uint8_t> input(500000, 0x81);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.110] parse error at byte 500001: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&);
CHECK(json::from_cbor(input, true, false).is_discarded());
}
SECTION("tags")
{
// a tag is not a value of its own, so a chain of them used to recurse
const std::vector<uint8_t> input(500000, 0xC2);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input, true, true, json::cbor_tag_handler_t::ignore), "[json.exception.parse_error.110] parse error at byte 500001: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&);
CHECK(json::from_cbor(input, true, false, json::cbor_tag_handler_t::ignore).is_discarded());
}
SECTION("a well-formed deep value is read through the SAX interface")
{
std::vector<uint8_t> input(200000, 0x9F);
input.insert(input.end(), 200000, 0xFF);
SaxCountdown accept_all(1000000);
CHECK(json::sax_parse(input, &accept_all, json::input_format_t::cbor));
}
SECTION("a well-formed deep value is read into a value")
{
const std::size_t depth = 10000;
std::vector<uint8_t> input(depth, 0x81);
input.push_back(0x00);
json j = json::from_cbor(input);
std::size_t measured = 0;
const json* p = &j;
while (p->is_array() && !p->empty())
{
p = &p->front();
++measured;
}
CHECK(measured == depth);
CHECK(p->is_number());
}
SECTION("containers are still read the same way")
{
CHECK(json::from_cbor(std::vector<uint8_t>({0x80})) == json::array());
CHECK(json::from_cbor(std::vector<uint8_t>({0xA0})) == json::object());
CHECK(json::from_cbor(std::vector<uint8_t>({0x9F, 0xFF})) == json::array());
CHECK(json::from_cbor(std::vector<uint8_t>({0xBF, 0xFF})) == json::object());
CHECK(json::from_cbor(std::vector<uint8_t>({0x9F, 0x01, 0x02, 0xFF})) == json({1, 2}));
CHECK(json::from_cbor(std::vector<uint8_t>({0xBF, 0x61, 'a', 0x01, 0xFF})) == json({{"a", 1}}));
// definite and indefinite forms nested inside each other
CHECK(json::from_cbor(std::vector<uint8_t>({0x9F, 0x82, 0x01, 0x02, 0xA1, 0x61, 'k', 0xBF, 0xFF, 0xFF})) == json({{1, 2}, {{"k", json::object()}}}));
}
SECTION("tagged values are still read the same way")
{
const auto ignore = json::cbor_tag_handler_t::ignore;
CHECK(json::from_cbor(std::vector<uint8_t>({0xC2, 0x01}), true, true, ignore) == json(1));
// a chain of tags resolves to the value that follows it
CHECK(json::from_cbor(std::vector<uint8_t>({0xC2, 0xC2, 0xC2, 0x01}), true, true, ignore) == json(1));
// a tag inside a container, and one in front of a container
CHECK(json::from_cbor(std::vector<uint8_t>({0x82, 0xC2, 0x01, 0x02}), true, true, ignore) == json({1, 2}));
CHECK(json::from_cbor(std::vector<uint8_t>({0xC2, 0x82, 0x01, 0x02}), true, true, ignore) == json({1, 2}));
}
}
TEST_CASE("CBOR indefinite-length strings do not recurse per chunk")
{
// Reading an indefinite-length string or byte array used to call itself
// once per chunk, so a payload of repeated 0x7F (or 0x5F) bytes exhausted
// the call stack before any of the input was rejected. The open levels are
// counted now, and the levels below prove the reader still reads the same
// values and reports the same errors at the same byte offsets.
json _;
SECTION("many open levels are reported, not crashed on")
{
const std::vector<uint8_t> input(200000, 0x7F);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.110] parse error at byte 200001: syntax error while parsing CBOR string: unexpected end of input", json::parse_error&);
CHECK(json::from_cbor(input, true, false).is_discarded());
}
SECTION("many open levels are reported, not crashed on (binary)")
{
const std::vector<uint8_t> input(200000, 0x5F);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.110] parse error at byte 200001: syntax error while parsing CBOR binary: unexpected end of input", json::parse_error&);
CHECK(json::from_cbor(input, true, false).is_discarded());
}
SECTION("chunks are still concatenated")
{
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0xFF})) == json(""));
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0x61, 0x61, 0xFF})) == json("a"));
// nested indefinite-length strings are concatenated across levels
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0x7F, 0x61, 0x61, 0xFF, 0x61, 0x62, 0xFF})) == json("ab"));
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0x7F, 0x7F, 0x61, 0x7A, 0xFF, 0xFF, 0xFF})) == json("z"));
CHECK(json::from_cbor(std::vector<uint8_t>({0xA1, 0x7F, 0x61, 0x61, 0xFF, 0x01})) == json({{"a", 1}}));
}
SECTION("chunks are still concatenated (binary)")
{
CHECK(json::from_cbor(std::vector<uint8_t>({0x5F, 0x41, 0x61, 0xFF})) == json::binary({0x61}));
CHECK(json::from_cbor(std::vector<uint8_t>({0x5F, 0x5F, 0x41, 0x61, 0xFF, 0x41, 0x62, 0xFF})) == json::binary({0x61, 0x62}));
}
SECTION("a chunk that is not a string is still rejected")
{
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7F, 0x7F, 0x00})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x00", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x5F, 0x5F, 0x00})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR binary: expected length specification (0x40-0x5B) or indefinite binary array type (0x5F); last byte: 0x00", json::parse_error&);
}
SECTION("a break marker outside an indefinite-length string is not a string")
{
// 0xFF only closes a string that was opened; on its own it is not one
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0xFF, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xFF", json::parse_error&);
}
}
TEST_CASE("issue #5405 - array reserve for definite-length CBOR arrays")
{
#if !defined(JSON_NOEXCEPTION)
// this SECTION relies on catching a thrown exception to distinguish
// which of two acceptable, bounded rejections a hostile header took;
// under JSON_NOEXCEPTION, JSON_THROW never produces a catchable C++
// exception (it aborts instead), so this cannot be tested that way here
SECTION("a huge claimed length with no element data must not over-allocate")
{
// 0x9A: array with a four-byte length; claims 0xFFFFFFFF (4294967295)
// elements but provides none. max_size() for a std::vector is far
// larger than this count, so it does not reject the header outright;
// the (capped) reservation must not attempt to allocate space for
// billions of elements before the missing data is detected.
json _;
const std::vector<uint8_t> input = {0x9A, 0xFF, 0xFF, 0xFF, 0xFF};
// On a platform where std::size_t is narrower than 64 bits (e.g.
// 32-bit), the claimed count 0xFFFFFFFF coincides with that
// platform's detail::unknown_size() sentinel (SIZE_MAX), so the
// format-level size check rejects it outright (out_of_range.408,
// "excessive ... size") before the SAX consumer's own max_size()
// check would even run; on a 64-bit platform it passes both of
// those checks and is only found short of data once the (capped)
// reservation looks for element bytes that were never provided
// (parse_error.110). Either is an acceptable, bounded rejection of
// the hostile header -- the property under test is that no path
// attempts to allocate space for billions of elements.
bool threw = false;
try
{
_ = json::from_cbor(input);
}
catch (const json::parse_error& e)
{
threw = true;
CHECK(e.id == 110);
CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 6: syntax error while parsing CBOR value: unexpected end of input");
}
catch (const json::out_of_range& e)
{
threw = true;
CHECK(e.id == 408);
CHECK(std::string(e.what()).find("excessive") != std::string::npos);
}
CHECK(threw);
CHECK(json::from_cbor(input, true, false).is_discarded());
}
#endif
SECTION("arrays of various sizes decode to the same value as before the reserve optimization")
{
for (const auto size :
{
std::size_t{0}, std::size_t{1}, std::size_t{5}, // small
std::size_t{16384}, // exactly at the reserve cap
std::size_t{20000} // above the reserve cap
})
{
CAPTURE(size)
json j = json::array();
for (std::size_t i = 0; i < size; ++i)
{
j.push_back(static_cast<int>(i % 1000));
}
const auto packed = json::to_cbor(j);
CHECK(json::from_cbor(packed) == j);
}
}
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
{
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
// a custom SAX consumer that does not touch a DOM array sees identical events
json j = json::array();
for (int i = 0; i < 100; ++i)
{
j.push_back(i);
}
const auto packed = json::to_cbor(j);
SaxCountdown scp(1000000); // large enough to never trigger an abort
CHECK(json::sax_parse(packed, &scp, json::input_format_t::cbor));
}
}
TEST_CASE("CBOR roundtrips" * doctest::skip())
{
SECTION("input from flynn")
+433
View File
@@ -12,6 +12,11 @@
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <cstdlib> // strtod
#include <sstream> // stringstream
#include <string> // string
#include <vector> // vector
namespace
{
// shortcut to scan a string literal
@@ -224,3 +229,431 @@ TEST_CASE("lexer class")
CHECK((scan_string("/**//**//**/", true) == json::lexer::token_type::end_of_input));
}
}
TEST_CASE("lexer number fast path")
{
// The contiguous fast path (used for pointer/string input) must agree with
// the streaming byte path (used for std::istream) on token type, numeric
// value, and round-trip text for every well-formed number, and reject the
// same malformed numbers with the same message.
SECTION("contiguous vs streaming parity")
{
const std::vector<std::string> numbers =
{
"0", "-0", "1", "-1", "42", "-42", "10", "100", "1234567890",
"0.0", "-0.0", "3.14", "-3.14", "0.5", "-0.001", "123.456789",
"1e0", "1E0", "1e10", "1e-10", "1e+10", "1.5e3", "-2.5E-4",
"9223372036854775807", // INT64_MAX -> unsigned
"9223372036854775808", // INT64_MAX + 1 -> unsigned
"18446744073709551615", // UINT64_MAX -> unsigned
"18446744073709551616", // UINT64_MAX + 1 -> float
"-9223372036854775808", // INT64_MIN -> integer
"-9223372036854775809", // INT64_MIN - 1 -> float
"123456789012345678901234567890", // huge -> float
"0.30000000000000004", "2.2250738585072014e-308", "1e308",
// high-precision / wide-exponent values that exercise the
// std::from_chars (Eisel-Lemire) path beyond the Clinger subset
"1.7976931348623157e308", "1.2345678901234567e-250",
"9007199254740993", "5e-324", "1e-320"
};
for (const auto& n : numbers)
{
const std::string doc = "[" + n + "]";
// contiguous fast path
const json a = json::parse(doc);
// streaming byte path
std::stringstream ss(doc);
const json b = json::parse(ss);
CAPTURE(n);
CHECK(a == b);
CHECK(a.dump() == b.dump());
CHECK(a[0].type() == b[0].type());
}
}
SECTION("significant-digit gate for the Clinger fast path")
{
// Clinger's fast path needs a significand below 2^53, so it cannot
// succeed once the mantissa has 17 or more significant digits (the
// significand would be at least 10^16). The lexer skips the attempt
// there. That is only allowed to save work: every value must still come
// out bit-exactly, and both scanners must agree. In particular the gate
// must not fire for tokens whose leading zeros merely look like extra
// digits - "0.1234567890123456" has 16 significant digits, not 17.
const std::vector<std::string> numbers =
{
"1234567890123456", // 16 significant digits
"12345678901234567", // 17 -> attempt skipped
"123456789012345678", // 18 -> attempt skipped
"0.1234567890123456", // 16: the leading "0" is not significant
"0.12345678901234567", // 17
"0.00000000000000001", // 1, in a long token
"0.000000000000000012345678901234", // 14, in a long token
"-0.0000000000000000000001", // 1, negative
"1.0000000000000000", // 17: trailing zeros are significant here
"10000000000000000", // 17
"9007199254740992", // 2^53
"9007199254740993", // 2^53 + 1
"-65.613616999999977", // canada.json shape
"1.2345678901234567e-250", // 17 with an exponent
"1.234567890123456e-250", // 16 with an exponent
"1e10", "0.0", "-0.0", "0e0", "0.000123"
};
for (const auto& n : numbers)
{
CAPTURE(n);
const std::string doc = "[" + n + "]";
const json a = json::parse(doc); // contiguous fast path
std::stringstream ss(doc);
const json b = json::parse(ss); // streaming byte path
CHECK(a[0].type() == b[0].type());
CHECK(a == b);
if (a[0].is_number_float())
{
const double expected = std::strtod(n.c_str(), nullptr);
CHECK(a[0].get<double>() == expected);
CHECK(b[0].get<double>() == expected);
}
}
}
SECTION("token type classification")
{
CHECK((scan_string("0") == json::lexer::token_type::value_unsigned));
CHECK((scan_string("-1") == json::lexer::token_type::value_integer));
CHECK((scan_string("1.5") == json::lexer::token_type::value_float));
CHECK((scan_string("1e5") == json::lexer::token_type::value_float));
CHECK((scan_string("18446744073709551615") == json::lexer::token_type::value_unsigned));
CHECK((scan_string("18446744073709551616") == json::lexer::token_type::value_float));
CHECK((scan_string("-9223372036854775808") == json::lexer::token_type::value_integer));
CHECK((scan_string("-9223372036854775809") == json::lexer::token_type::value_float));
}
SECTION("malformed numbers are rejected identically")
{
for (const char* bad :
{"-", "1.", "1e", "1e+", "1.2e", "01", "-01", "1..2", "1.2.3"
})
{
CAPTURE(bad);
// the contiguous fast path must decline and let the byte path report
const std::string doc = std::string("[") + bad + "]";
CHECK_FALSE(json::accept(doc));
std::stringstream ss(doc);
CHECK_FALSE(json::accept(ss));
}
}
#if !defined(JSON_NOEXCEPTION)
// these sections parse invalid input, which aborts when exceptions are off
SECTION("exhaustive grammar parity with the streaming path")
{
// The JSON number grammar is encoded twice: once as the scan_number()
// state machine and once as the contiguous fast path. Enumerate every
// short string over the number alphabet and require the two encodings to
// agree exactly - on acceptance, on the reported error, and on the parsed
// value - so they cannot drift apart.
const std::string alphabet = "01.eE+-";
// full outcome of parsing @a doc, so a mismatch in type, value, or error
// message is caught, not just a mismatch in acceptance
const auto outcome = [](const std::string & doc, bool streaming) -> std::string
{
try
{
if (streaming)
{
std::stringstream ss(doc);
const json j = json::parse(ss);
return std::string(j[0].type_name()) + '|' + j.dump();
}
const json j = json::parse(doc);
return std::string(j[0].type_name()) + '|' + j.dump();
}
catch (const json::parse_error& e)
{
return {e.what()};
}
};
std::vector<std::string> mismatches;
std::vector<std::string> tokens{""};
for (std::size_t length = 1; length <= 4; ++length)
{
std::vector<std::string> next;
next.reserve(tokens.size() * alphabet.size());
for (const auto& prefix : tokens)
{
for (const char c : alphabet)
{
next.push_back(prefix + c);
}
}
tokens = next;
for (const auto& token : tokens)
{
const std::string doc = "[" + token + "]";
if (outcome(doc, false) != outcome(doc, true))
{
mismatches.push_back(doc);
}
}
}
// 7 + 49 + 343 + 2401 tokens
CHECK(tokens.size() == 2401);
CAPTURE(mismatches);
CHECK(mismatches.empty());
}
SECTION("error positions match the streaming path")
{
// Rejecting identically is not enough: the fast path must also report the
// error at the same position as the byte path. A number directly followed
// by a newline is the interesting case, because the byte path reaches the
// newline (which resets the column) and then ungets it.
// returns the parse_error message, or "" if the document parsed
const auto contiguous_error = [](const std::string & doc) -> std::string
{
try
{
const json j = json::parse(doc);
static_cast<void>(j);
}
catch (const json::parse_error& e)
{
return {e.what()};
}
return {};
};
const auto streaming_error = [](const std::string & doc) -> std::string
{
try
{
std::stringstream ss(doc);
const json j = json::parse(ss);
static_cast<void>(j);
}
catch (const json::parse_error& e)
{
return {e.what()};
}
return {};
};
for (const char* bad :
{"[01\n]", "[00\n]", "[-01\n]", "{1\n}", "[1\n2]", "[1.2.3\n]",
"[1 \n2]", "[\n1\n2]", "1\n2", "[01\r\n]", "[1e\n]", "[-\n]"
})
{
CAPTURE(bad);
const std::string doc = bad;
const std::string contiguous_what = contiguous_error(doc);
CHECK_FALSE(contiguous_what.empty());
CHECK(contiguous_what == streaming_error(doc));
}
// A number terminated by a newline must report the same position as the
// same number terminated by anything else: scan_number() reads the
// terminator and ungets it, so the reported column is the one reached
// after the number's last character - not the 0 that an unget() across
// the newline used to leave behind.
CHECK(contiguous_error("[01\n]") == contiguous_error("[01 ]"));
CHECK(contiguous_error("[01\n]") ==
"[json.exception.parse_error.101] parse error at line 1, column 3: "
"syntax error while parsing array - unexpected number literal; expected ']'");
// the same for a multi-character token, where the column of the last
// character (the '3' of "-2.5e3") differs from the column it starts at
CHECK(contiguous_error("null -2.5e3\nfalse") == contiguous_error("null -2.5e3 false"));
CHECK(contiguous_error("null -2.5e3\nfalse") ==
"[json.exception.parse_error.101] parse error at line 1, column 11: "
"syntax error while parsing value - unexpected number literal; expected end of input");
}
#endif
}
TEST_CASE("lexer string fast path")
{
// Build a byte string from explicit values: a hex escape in a string
// literal swallows every following hex digit, which makes sequences like
// "\xC3\xA9b" mean something other than they look like.
const auto bytes = [](std::initializer_list<int> values)
{
std::string result;
for (const int value : values)
{
result.push_back(static_cast<char>(value));
}
return result;
};
#if !defined(JSON_NOEXCEPTION)
// the full outcome of parsing @a doc: the parsed value, or the exact error
// message, so a mismatch in either is caught. Only usable with exceptions
// on: parsing invalid input aborts when they are off.
const auto outcome = [](const std::string & doc, bool streaming) -> std::string
{
try
{
if (streaming)
{
std::stringstream ss(doc);
const json j = json::parse(ss);
return j.dump();
}
const json j = json::parse(doc);
return j.dump();
}
// not just parse_error: if a bulk scanner ever let ill-formed UTF-8
// through, dump() would throw type_error.316, and that has to surface
// as a reported mismatch rather than as an uncaught exception
catch (const json::exception& e)
{
return {e.what()};
}
};
#endif
// once at the start of the string, once past the first 8-byte SWAR word, so
// the bulk scanner sees each case with and without a run behind it
const std::vector<std::size_t> offsets{0, 9};
#if !defined(JSON_NOEXCEPTION)
SECTION("exhaustive contiguous vs streaming parity")
{
// ordinary ASCII, both specials, a control byte, characters that make
// the preceding backslash a valid escape, a UTF-8 lead byte of each
// length, a continuation byte, and a byte that is never valid
const std::vector<std::string> alphabet =
{
"a", "\"", "\\", "n", "u", "0", bytes({0x01}),
bytes({0xC3}), bytes({0xA9}), bytes({0xE4}), bytes({0xF0}),
bytes({0x80}), bytes({0xFF})
};
std::vector<std::string> mismatches;
std::vector<std::string> tokens{""};
for (std::size_t length = 1; length <= 3; ++length)
{
std::vector<std::string> next;
next.reserve(tokens.size() * alphabet.size());
for (const auto& prefix : tokens)
{
for (const auto& symbol : alphabet)
{
next.push_back(prefix + symbol);
}
}
tokens = next;
for (const auto& token : tokens)
{
for (const std::size_t offset : offsets)
{
const std::string doc = "[\"" + std::string(offset, 'a') + token + "\"]";
if (outcome(doc, false) != outcome(doc, true))
{
mismatches.push_back(doc);
}
}
}
}
// 13 + 169 + 2197 tokens, each at two offsets
CHECK(tokens.size() == 2197);
CAPTURE(mismatches);
CHECK(mismatches.empty());
}
SECTION("special bytes at every offset of the SWAR stride")
{
// The bulk scanner consumes 8 bytes at a time and then a tail; place
// every kind of byte that ends a run at each offset across two words,
// so multibyte sequences also straddle the word boundary.
const std::vector<std::string> specials =
{
"\"", "\\", bytes({0x01}), bytes({0x1F}), bytes({0x7F}),
bytes({0xC3, 0xA9}), bytes({0xE4, 0xB8, 0xAD}), bytes({0xF0, 0x9F, 0x98, 0x80}),
bytes({0xFF}), bytes({0xC3}), bytes({0xE4, 0xB8})
};
std::vector<std::string> mismatches;
for (std::size_t offset = 0; offset <= 17; ++offset)
{
for (const auto& special : specials)
{
const std::string doc = "[\"" + std::string(offset, 'a') + special + "\"]";
if (outcome(doc, false) != outcome(doc, true))
{
mismatches.push_back(doc);
}
}
}
CAPTURE(mismatches);
CHECK(mismatches.empty());
}
#endif
// json::accept() never throws, so the ranges stay covered without exceptions
SECTION("UTF-8 ranges are accepted and rejected as documented")
{
// The bulk validator must accept exactly what the byte-at-a-time
// scanner accepts, so pin the boundaries of every range it recognizes.
// aggregate, only ever brace-initialized below; default member
// initializers would stop it being an aggregate in C++11
struct utf8_case // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init)
{
std::string sequence;
bool valid;
const char* description;
};
const std::vector<utf8_case> cases =
{
{bytes({0xC2, 0x80}), true, "U+0080, shortest two-byte"},
{bytes({0xDF, 0xBF}), true, "U+07FF, longest two-byte"},
{bytes({0xC1, 0xBF}), false, "overlong two-byte"},
{bytes({0xC2, 0x7F}), false, "two-byte with bad continuation"},
{bytes({0xE0, 0xA0, 0x80}), true, "U+0800, shortest three-byte"},
{bytes({0xE0, 0x9F, 0xBF}), false, "overlong three-byte"},
{bytes({0xED, 0x9F, 0xBF}), true, "U+D7FF, just below the surrogates"},
{bytes({0xED, 0xA0, 0x80}), false, "surrogate U+D800"},
{bytes({0xED, 0xBF, 0xBF}), false, "surrogate U+DFFF"},
{bytes({0xEE, 0x80, 0x80}), true, "U+E000, just above the surrogates"},
{bytes({0xEF, 0xBF, 0xBF}), true, "U+FFFF"},
{bytes({0xF0, 0x90, 0x80, 0x80}), true, "U+10000, shortest four-byte"},
{bytes({0xF0, 0x8F, 0xBF, 0xBF}), false, "overlong four-byte"},
{bytes({0xF4, 0x8F, 0xBF, 0xBF}), true, "U+10FFFF, highest code point"},
{bytes({0xF4, 0x90, 0x80, 0x80}), false, "above U+10FFFF"},
{bytes({0xF5, 0x80, 0x80, 0x80}), false, "lead byte out of range"},
{bytes({0x80}), false, "bare continuation byte"},
{bytes({0xFF}), false, "byte that never appears in UTF-8"},
{bytes({0xC3}), false, "truncated two-byte"},
{bytes({0xE4, 0xB8}), false, "truncated three-byte"},
{bytes({0xF0, 0x9F, 0x98}), false, "truncated four-byte"}
};
for (const auto& test_case : cases)
{
CAPTURE(test_case.description);
for (const std::size_t offset : offsets)
{
CAPTURE(offset);
const std::string doc = "[\"" + std::string(offset, 'a') + test_case.sequence + "\"]";
CHECK(json::accept(doc) == test_case.valid);
#if !defined(JSON_NOEXCEPTION)
CHECK(outcome(doc, false) == outcome(doc, true));
#endif
}
}
}
}
+914 -2
View File
@@ -8,6 +8,14 @@
#include "doctest_compatibility.h"
// capture whether JSON_STRICT_NUL_HANDLING was enabled on the command line
// (e.g. -DJSON_STRICT_NUL_HANDLING=1) *before* including json.hpp, since the
// library #undefs JSON_STRICT_NUL_HANDLING itself once the header has been
// fully processed (see include/nlohmann/detail/macro_unscope.hpp)
#if defined(JSON_STRICT_NUL_HANDLING) && (JSON_STRICT_NUL_HANDLING == 1)
#define JSON_TEST_STRICT_NUL_HANDLING_ENABLED 1
#endif
#define JSON_TESTS_PRIVATE
#include <nlohmann/json.hpp>
using nlohmann::json;
@@ -17,12 +25,16 @@ using nlohmann::json;
#include <valarray>
#include <algorithm>
#include <cstdio>
#include <fstream>
#include <list>
#include <sstream>
#include <string>
#include <utility>
#include <vector>
#include "test_utils.hpp"
namespace
{
class SaxEventLogger
@@ -344,6 +356,50 @@ void trailing_comma_helper(const std::string& s)
}
}
#if JSON_DIAGNOSTIC_POSITIONS
/**
* Validates that the generated JSON object is the same as expected
* Validates that the start position and end position match the start and end of the string
*
* This check assumes that there is no whitespace around the json object in the original string.
*/
void validate_generated_json_and_start_end_pos_helper(const std::string& original_string, const json& j, const json& check)
{
CHECK(j == check);
CHECK(j.start_pos() == 0);
CHECK(j.end_pos() == original_string.size());
}
/**
* Parses the root object from the given root string and validates that the start and end positions for the nested object are correct.
*
* This checks that whitespace around the nested object is included in the start and end positions of the root object.
*/
void validate_start_end_pos_for_nested_obj_helper(const std::string& nested_type_json_str, const std::string& root_type_json_str, const json& expected_json, const json::parser_callback_t& cb = nullptr)
{
json j;
// 1. If callback is provided, use callback version of parse()
if (cb)
{
j = json::parse(root_type_json_str, cb);
}
else
{
j = json::parse(root_type_json_str);
}
// 2. Check if the generated JSON is as expected
// Assumptions: The root_type_json_str does not have any whitespace around the json object
validate_generated_json_and_start_end_pos_helper(root_type_json_str, j, expected_json);
// 3. Get the nested object
const auto& nested = j["nested"];
// 4. Check if the start and end positions are generated correctly for nested objects and arrays
CHECK(nested_type_json_str == root_type_json_str.substr(nested.start_pos(), nested.end_pos() - nested.start_pos()));
}
#endif
} // namespace
TEST_CASE("parser class")
@@ -497,6 +553,88 @@ TEST_CASE("parser class")
}
}
SECTION("NUL byte handling (issue #5530, JSON_STRICT_NUL_HANDLING)")
{
// by default, a NUL byte anywhere in the input (not inside a quoted
// string, which is covered above) is silently treated the same as
// real end of input; JSON_STRICT_NUL_HANDLING (off by default, see
// docs/mkdocs/docs/api/macros/json_strict_nul_handling.md) makes a
// NUL byte an error like any other unexpected byte instead.
//
// The two sections below are mutually exclusive: this whole test
// binary is compiled once, with JSON_STRICT_NUL_HANDLING either
// left at its default or forced to 1 (e.g. by the dedicated
// ci_test_strict_nul_handling CI target), so only the section
// matching the actual, compiled-in behavior can pass.
#if !defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED)
SECTION("default behavior (macro not enabled)")
{
// a NUL byte after a complete value silently truncates the input
std::string s = "123";
s.push_back('\0');
s += "4";
CHECK(json::parse(s) == json(123));
CHECK(json::accept(s));
// parsing from a string literal is unaffected either way
CHECK(json::parse("123") == json(123));
}
#endif
#if defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED)
SECTION("opt-in strict behavior (JSON_STRICT_NUL_HANDLING == 1)")
{
// a NUL byte after a complete value is now a parse error,
// instead of silently truncating the input
{
std::string s = "123";
s.push_back('\0');
json _; // NOLINT(readability-identifier-naming)
CHECK_THROWS_WITH_AS(_ = json::parse(s),
"[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '123<U+0000>'; expected end of input",
json::parse_error&);
CHECK_FALSE(json::accept(s));
}
// a NUL byte where a value is expected is now a parse error,
// instead of being treated the same as an empty input
{
const std::string s(1, '\0');
json _; // NOLINT(readability-identifier-naming)
CHECK_THROWS_WITH_AS(_ = json::parse(s),
"[json.exception.parse_error.101] parse error at line 1, column 1: syntax error while parsing value - invalid literal; last read: '<U+0000>'",
json::parse_error&);
CHECK_FALSE(json::accept(s));
}
// a NUL byte inside a // comment no longer stops the comment
// scan early; scanning continues correctly past it
{
std::string s = "1 // a";
s.push_back('\0');
s += "b\n";
CHECK(json::parse(s, nullptr, true, true) == json(1));
CHECK(json::accept(s, true, true));
}
// a NUL byte inside a /* */ comment no longer stops the
// comment scan early either
{
std::string s = "1 /* a";
s.push_back('\0');
s += "b */ ";
CHECK(json::parse(s, nullptr, true, true) == json(1));
CHECK(json::accept(s, true, true));
}
// regression guard: parsing from a string literal (which
// carries a compiler-appended trailing '\0') still works,
// even though a NUL byte is now rejected everywhere else
CHECK(json::parse("123") == json(123));
}
#endif
}
SECTION("number")
{
SECTION("integers")
@@ -624,7 +762,8 @@ TEST_CASE("parser class")
SECTION("overflow")
{
// overflows during parsing yield an exception
CHECK_THROWS_WITH_AS(parser_helper("1.18973e+4932").empty(), "[json.exception.out_of_range.406] number overflow parsing '1.18973e+4932'", json::out_of_range&);
// empty() is nodiscard; the exception is thrown by parser_helper() itself, before empty() would run
CHECK_THROWS_WITH_AS(utils::ignore_return_value(parser_helper("1.18973e+4932").empty()), "[json.exception.out_of_range.406] number overflow parsing '1.18973e+4932'", json::out_of_range&);
}
SECTION("invalid numbers")
@@ -930,6 +1069,98 @@ TEST_CASE("parser class")
CHECK(accept_helper("+1") == false);
CHECK(accept_helper("+0") == false);
}
SECTION("issue #5411 - skip conversion when accept() does not need the numeric value")
{
// lexer::scan_number() may skip strtoull()/strtoll() for
// value_unsigned/value_integer tokens when the caller (e.g.
// json::accept()) does not need the converted value, as long
// as the digit count alone guarantees no 64-bit overflow (see
// the "safe_digit_count" fast path in scan_number()). This
// differential test checks that json::accept() (which enables
// the fast path) and json::parse() (which never does) always
// agree, over a corpus that exercises both the fast path
// (<=18 digits) and the untouched, exact fallback path (>=19
// digits) -- including reclassification of huge digit-only
// integers to a (possibly non-finite) floating-point value.
const std::vector<std::pair<std::string, bool>> cases =
{
// normal small/large integers, both signs
{"0", true}, {"1", true}, {"-1", true}, {"42", true}, {"-42", true},
{"123456789", true}, {"-123456789", true},
// digit-count boundary around the 18-digit safe cutoff (both signs)
{std::string(17, '9'), true},
{std::string(18, '9'), true},
{std::string(19, '9'), true},
{std::string(20, '9'), true},
{"-" + std::string(17, '9'), true},
{"-" + std::string(18, '9'), true},
{"-" + std::string(19, '9'), true},
{"-" + std::string(20, '9'), true},
// 64-bit boundaries
{"9223372036854775807", true}, // INT64_MAX
{"-9223372036854775808", true}, // INT64_MIN
{"18446744073709551615", true}, // UINT64_MAX
{"18446744073709551616", true}, // UINT64_MAX + 1 (overflows uint64_t, finite double)
// the 28-digit example from the issue: overflows uint64_t
// but is finite as a double, so the scanner reclassifies
// it to value_float and it is accepted
{"9999999999999999999999999999", true},
// huge digit-only integers that overflow even a double -> rejected
{std::string(309, '9'), false},
{std::string(400, '9'), false},
{"1" + std::string(400, '0'), false},
// 1e999 / 1e400 style overflow -> rejected
{"1e999", false},
{"1e400", false},
{"-1e999", false},
{"1E999", false},
// values straddling DBL_MAX
{"1.7976931348623157e308", true}, // <= DBL_MAX, finite
{"1.7976931348623159e308", false}, // > DBL_MAX, overflows to inf
// a mix of other valid/invalid numeric syntax
{"3.14159", true},
{"-0.0", true},
{"1.0e10", true},
{"01", false},
{"-", false},
{"1.", false},
{"1e", false},
{"+1", false},
};
for (const auto& c : cases)
{
const std::string& number = c.first;
const bool expected = c.second;
CAPTURE(number)
CAPTURE(expected)
// accept() takes the fast path (skips conversion when possible)
CHECK(json::accept(number) == expected);
// parse() always performs the full conversion; it must agree
json j;
CHECK_NOTHROW(json::parser(nlohmann::detail::input_adapter(number), nullptr, false).parse(true, j));
CHECK(!j.is_discarded() == expected);
// wrap in an array so get_token() is exercised beyond the
// very first (constructor-time) scan as well
std::string wrapped = "[";
wrapped += number;
wrapped += ",";
wrapped += number;
wrapped += "]";
CHECK(json::accept(wrapped) == expected);
}
}
}
}
@@ -1394,6 +1625,71 @@ TEST_CASE("parser class")
CHECK(accept_helper("\"\\uD80C\\uFFFF\"") == false);
}
#if !defined(JSON_NOEXCEPTION)
SECTION("issue #5412 - whitespace skipping bookkeeping (compact vs. pretty-printed)")
{
// lexer::skip_whitespace() reads its first character with get() (to
// honor a possibly pending unget() from the previous token) and every
// further whitespace character with get_ignoring_pending_unget() (a
// get() variant that skips the then-always-false next_unget check).
// This must not change the reported byte offset, line, or column of
// a syntax error, even when a long run of whitespace containing
// multiple newlines is skipped beforehand (as with pretty-printed
// input). The expected values below were captured from the
// unmodified do-while(get()) loop, so any regression that miscounts
// characters or newlines while skipping whitespace changes them.
const auto check_error = [](const std::string & input, std::size_t expected_byte,
const std::string & expected_what)
{
CAPTURE(input)
try
{
json _ = json::parse(input);
FAIL_CHECK("expected a parse_error, but parsing succeeded");
}
catch (const json::parse_error& e)
{
CHECK(e.byte == expected_byte);
CHECK(std::string(e.what()) == expected_what);
}
};
// a nested document, serialized both compactly and pretty-printed
// (dump(4)), each truncated right before the final closing '}' so
// that the parser hits EOF after skipping all of the (in the
// pretty-printed case, substantial) indentation whitespace
const json doc =
{
{"a", 1},
{"b", json::array({true, false, nullptr, "x"})},
{"c", json::object({{"d", 3.14}, {"e", json::array({1, 2, 3})}})}
};
const std::string compact = doc.dump();
const std::string pretty = doc.dump(4);
check_error(compact.substr(0, compact.size() - 1), 60,
"[json.exception.parse_error.101] parse error at line 1, column 60: syntax error while parsing object - unexpected end of input; expected '}'");
check_error(pretty.substr(0, pretty.size() - 1), 193,
"[json.exception.parse_error.101] parse error at line 17, column 1: syntax error while parsing object - unexpected end of input; expected '}'");
// an invalid token appearing after several indented, multi-line
// whitespace runs vs. the same document without any of that
// whitespace
check_error(R"({
"a": 1,
"b": [
true,
false
],
"c": @
})", 70,
"[json.exception.parse_error.101] parse error at line 7, column 10: syntax error while parsing value - invalid literal; last read: '\"c\": @'");
check_error(R"({"a":1,"b":[true,false],"c":@})", 29,
"[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing value - invalid literal; last read: '\"c\":@'");
}
#endif
SECTION("tests found by mutate++")
{
// test case to make sure no comma precedes the first key
@@ -1564,6 +1860,58 @@ TEST_CASE("parser class")
CHECK (j_filtered2 == json({{"foo", {1, 2}}}));
}
SECTION("filter many members of one container")
{
// Rejecting a value makes the parser remove the placeholder its key
// event stored. Locating that placeholder used to be a scan of the
// whole parent, which made filtering a large container quadratic:
// 128k members took ~25 s. These cases keep many members alive
// while discarding many others, so the removal cost is the whole
// point; they run in milliseconds when the placeholder is erased
// directly.
constexpr int count = 20000;
std::string s = "{";
for (int i = 0; i < count; ++i)
{
// "a<i>" is kept, "z<i>" is discarded
s += "\"a" + std::to_string(i) + "\":" + std::to_string(i) + ",";
s += "\"z" + std::to_string(i) + "\":-1,";
}
s.back() = '}';
const json j_values = json::parse(s, [](int /*unused*/, json::parse_event_t e, const json & parsed) noexcept
{
return !(e == json::parse_event_t::value && parsed == json(-1));
});
CHECK(j_values.size() == count);
CHECK(j_values.at("a0") == json(0));
CHECK(j_values.at("a" + std::to_string(count - 1)) == json(count - 1));
CHECK_FALSE(j_values.contains("z0"));
CHECK_FALSE(j_values.contains("z" + std::to_string(count - 1)));
// the same, but discarding whole containers rather than values,
// which takes the end_object()/end_array() removal path
std::string s_nested = "{";
for (int i = 0; i < count; ++i)
{
s_nested += "\"a" + std::to_string(i) + "\":" + std::to_string(i) + ",";
s_nested += "\"z" + std::to_string(i) + "\":[1,2],";
}
s_nested.back() = '}';
const json j_arrays = json::parse(s_nested, [](int /*unused*/, json::parse_event_t e, const json& /*unused*/) noexcept
{
return e != json::parse_event_t::array_end;
});
CHECK(j_arrays.size() == count);
CHECK(j_arrays.at("a0") == json(0));
CHECK_FALSE(j_arrays.contains("z0"));
CHECK_FALSE(j_arrays.contains("z" + std::to_string(count - 1)));
}
SECTION("filter specific events")
{
SECTION("first closing event")
@@ -1636,7 +1984,13 @@ TEST_CASE("parser class")
SECTION("from std::array")
{
std::array<uint8_t, 5> v { {'t', 'r', 'u', 'e'} };
// NOTE: this array is sized to exactly the length of "true" (unlike
// the trailing-NUL-tolerant default behavior elsewhere in this file,
// see the "NUL byte handling" section above); a size of 5 here would
// leave a value-initialized trailing 0x00 element that is only
// silently accepted as end-of-input by default and would fail under
// JSON_STRICT_NUL_HANDLING
std::array<uint8_t, 4> v { {'t', 'r', 'u', 'e'} };
json j;
json::parser(nlohmann::detail::input_adapter(std::begin(v), std::end(v))).parse(true, j);
CHECK(j == json(true));
@@ -1777,8 +2131,240 @@ TEST_CASE("parser class")
{
json _;
CHECK_THROWS_WITH_AS(_ = json::parse("/a", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid comment; expecting '/' or '*' after '/'; last read: '/a'", json::parse_error);
// "/*" is a string literal, so it carries a compiler-appended trailing
// '\0'; by default that NUL is read like any other byte and shows up
// in "last read", but JSON_STRICT_NUL_HANDLING trims exactly that one
// trailing byte from a char array (see
// docs/mkdocs/docs/api/macros/json_strict_nul_handling.md), so it no
// longer appears in the message in that state
#if defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED)
CHECK_THROWS_WITH_AS(_ = json::parse("/*", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid comment; missing closing '*/'; last read: '/*'", json::parse_error);
#else
CHECK_THROWS_WITH_AS(_ = json::parse("/*", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid comment; missing closing '*/'; last read: '/*<U+0000>'", json::parse_error);
#endif
}
#if JSON_DIAGNOSTIC_POSITIONS
// Macro for all test cases for start_pos and end_pos
#define SETUP_TESTCASES() \
SECTION("with callback") \
{ \
SECTION("filter nothing") \
{ \
json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t /*unused*/, json& /*unused*/) noexcept \
{ \
return true; \
}; \
validate_start_end_pos_for_nested_obj_helper(nested_type_json_str, root_type_json_str, expected, cb); \
} \
SECTION("filter element") \
{ \
json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t event, json& j) noexcept \
{ \
return (event != json::parse_event_t::key && event != json::parse_event_t::value) || j != json("a"); \
}; \
validate_start_end_pos_for_nested_obj_helper(nested_type_json_str, root_type_json_str, filteredExpected, cb); \
} \
} \
SECTION("without callback") \
{ \
validate_start_end_pos_for_nested_obj_helper(nested_type_json_str, root_type_json_str, expected); \
}
SECTION("retrieve start position and end position")
{
SECTION("for object")
{
// Create an object with spaces to test the start and end positions. Spaces will not be included in the
// JSON object, however, the start and end positions should include the spaces from the input JSON string.
const std::string nested_type_json_str = R"({ "a": 1,"b" : "test1"})";
const std::string root_type_json_str = R"({ "nested": )" + nested_type_json_str + R"(, "anotherValue": "test2"})";
auto expected = json({{"nested", {{"a", 1}, {"b", "test1"}}}, {"anotherValue", "test2"}});
auto filteredExpected = expected;
filteredExpected["nested"].erase("a");
SETUP_TESTCASES()
}
SECTION("for array")
{
const std::string nested_type_json_str = R"(["a", "test", 45])";
const std::string root_type_json_str = R"({ "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
auto expected = json({{"nested", {"a", "test", 45}}, {"anotherValue", "test"}});
auto filteredExpected = expected;
filteredExpected["nested"] = json({"test", 45});
SETUP_TESTCASES()
}
SECTION("for array with objects")
{
const std::string nested_type_json_str = R"([{"a": 1, "b": "test"}, {"c": 2, "d": "test2"}])";
const std::string root_type_json_str = R"({ "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
auto expected = json({{"nested", {{{"a", 1}, {"b", "test"}}, {{"c", 2}, {"d", "test2"}}}}, {"anotherValue", "test"}});
auto filteredExpected = expected;
filteredExpected["nested"][0].erase("a");
SETUP_TESTCASES()
auto j = json::parse(root_type_json_str);
auto nested_array = j["nested"];
const auto& nested_obj = nested_array[0];
CHECK(nested_type_json_str.substr(1, 21) == root_type_json_str.substr(nested_obj.start_pos(), nested_obj.end_pos() - nested_obj.start_pos()));
CHECK(nested_type_json_str.substr(24, 22) == root_type_json_str.substr(nested_array[1].start_pos(), nested_array[1].end_pos() - nested_array[1].start_pos()));
}
SECTION("for two levels of nesting objects")
{
const std::string nested_type_json_str = R"({"nested2": {"b": "test"}})";
const std::string root_type_json_str = R"({ "a": 2, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
auto expected = json({{"a", 2}, {"nested", {{"nested2", {{"b", "test"}}}}}, {"anotherValue", "test"}});
auto filteredExpected = expected;
filteredExpected.erase("a");
SETUP_TESTCASES()
auto j = json::parse(root_type_json_str);
auto nested_obj = j["nested"]["nested2"];
CHECK(nested_type_json_str.substr(12, 13) == root_type_json_str.substr(nested_obj.start_pos(), nested_obj.end_pos() - nested_obj.start_pos()));
}
SECTION("for simple types")
{
SECTION("no nested")
{
SECTION("with callback")
{
json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t /*unused*/, json& /*unused*/) noexcept
{
return true;
};
// 1. string type
std::string json_str = R"("test")";
auto j = json::parse(json_str, cb);
validate_generated_json_and_start_end_pos_helper(json_str, j, "test");
// 2. number type
json_str = R"(1)";
j = json::parse(json_str, cb);
validate_generated_json_and_start_end_pos_helper(json_str, j, 1);
// 3. boolean type
json_str = R"(true)";
j = json::parse(json_str, cb);
validate_generated_json_and_start_end_pos_helper(json_str, j, true);
// 4. null type
json_str = R"(null)";
j = json::parse(json_str, cb);
validate_generated_json_and_start_end_pos_helper(json_str, j, nullptr);
}
SECTION("without callback")
{
// 1. string type
std::string json_str = R"("test")";
auto j = json::parse(json_str);
validate_generated_json_and_start_end_pos_helper(json_str, j, "test");
// 2. number type
json_str = R"(1)";
j = json::parse(json_str);
validate_generated_json_and_start_end_pos_helper(json_str, j, 1);
json_str = R"(1.001239923)";
j = json::parse(json_str);
validate_generated_json_and_start_end_pos_helper(json_str, j, 1.001239923);
json_str = R"(1.123812389000000)";
j = json::parse(json_str);
validate_generated_json_and_start_end_pos_helper(json_str, j, 1.123812389);
// 3. boolean type
json_str = R"(true)";
j = json::parse(json_str);
validate_generated_json_and_start_end_pos_helper(json_str, j, true);
json_str = R"(false)";
j = json::parse(json_str);
validate_generated_json_and_start_end_pos_helper(json_str, j, false);
// 4. null type
json_str = R"(null)";
j = json::parse(json_str);
validate_generated_json_and_start_end_pos_helper(json_str, j, nullptr);
}
}
SECTION("string type")
{
const std::string nested_type_json_str = R"("test")";
const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
auto expected = json({{"nested", "test"}, {"anotherValue", "test"}, {"a", 1}});
auto filteredExpected = expected;
filteredExpected.erase("a");
SETUP_TESTCASES()
}
SECTION("number type")
{
const std::string nested_type_json_str = R"(2)";
const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
auto expected = json({{"nested", 2}, {"anotherValue", "test"}, {"a", 1}});
auto filteredExpected = expected;
filteredExpected.erase("a");
SETUP_TESTCASES()
}
SECTION("boolean type")
{
const std::string nested_type_json_str = R"(true)";
const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
auto expected = json({{"nested", true}, {"anotherValue", "test"}, {"a", 1}});
auto filteredExpected = expected;
filteredExpected.erase("a");
SETUP_TESTCASES()
}
SECTION("null type")
{
const std::string nested_type_json_str = R"(null)";
const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
auto expected = json({{"nested", nullptr}, {"anotherValue", "test"}, {"a", 1}});
auto filteredExpected = expected;
filteredExpected.erase("a");
SETUP_TESTCASES()
}
}
SECTION("with leading whitespace and newlines around root JSON")
{
const std::string initial_whitespace = R"(
)";
const std::string nested_type_json_str = R"({
"a": 1,
"nested": {
"b": "test"
},
"anotherValue": "test"
})";
const std::string end_whitespace = R"(
)";
const std::string root_type_json_str = initial_whitespace + nested_type_json_str + end_whitespace;
auto expected = json({{"a", 1}, {"nested", {{"b", "test"}}}, {"anotherValue", "test"}});
auto j = json::parse(root_type_json_str);
// 2. Check if the generated JSON is as expected
CHECK(j == expected);
// 3. Check if the start and end positions do not include the surrounding whitespace
CHECK(j.start_pos() == initial_whitespace.size());
CHECK(j.end_pos() == root_type_json_str.size() - end_whitespace.size());
}
}
#undef SETUP_TESTCASES
#endif
}
// this test relies on parse errors being thrown, so it is skipped when
@@ -1887,3 +2473,329 @@ TEST_CASE("last-read diagnostics are identical across input adapters")
}
}
#endif // !defined(JSON_NOEXCEPTION)
// this test characterizes the current (documented-by-example, not otherwise
// specified) behavior of JSON_DIAGNOSTIC_POSITIONS positions with respect to
// value lifetime (copy/move/swap/mutation), the various input adapters, and
// user-driven SAX usage. It is regression protection, not a behavior
// specification: if any of these checks fail after a change to json.hpp,
// that change deliberately altered observable behavior and the test (and
// this comment) should be updated accordingly, rather than "fixed" blindly.
#if JSON_DIAGNOSTIC_POSITIONS
TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX")
{
SECTION("value lifetime")
{
SECTION("copy constructor copies positions, recursively")
{
// basic_json(const basic_json&) (json.hpp, around line 1192) copies
// start_position/end_position for the value itself; nested values
// are copied via their own copy constructor (through the copied
// object/array container), so positions are preserved throughout
// the whole tree.
const std::string s = R"({"a":1,"b":[1,2,3]})";
const json a = json::parse(s);
const json b = a; // NOLINT(performance-unnecessary-copy-initialization)
CHECK(b.start_pos() == a.start_pos());
CHECK(b.end_pos() == a.end_pos());
CHECK(b["b"].start_pos() == a["b"].start_pos());
CHECK(b["b"].end_pos() == a["b"].end_pos());
CHECK(b["b"][0].start_pos() == a["b"][0].start_pos());
CHECK(b["b"][0].end_pos() == a["b"][0].end_pos());
// sanity: the positions are meaningful (not all npos)
CHECK(b.start_pos() == 0);
CHECK(b.end_pos() == s.size());
}
SECTION("move constructor resets the moved-from value to npos")
{
// basic_json(basic_json&&) (json.hpp, around line 1265) copies
// other's start_position/end_position into *this and then resets
// other's to npos (see the cppcheck-suppress[accessForwarded]
// annotation there, which flags this reset as worth a second
// look). Only the top-level moved-from value is affected; its
// (moved-away) children are gone along with it.
const std::string s = R"({"a":1,"b":[1,2,3]})";
json a = json::parse(s);
const auto a_start = a.start_pos();
const auto a_end = a.end_pos();
const auto nested_start = a["b"].start_pos();
const auto nested_end = a["b"].end_pos();
const json b(std::move(a));
// the destination retains the original positions, recursively
CHECK(b.start_pos() == a_start);
CHECK(b.end_pos() == a_end);
CHECK(b["b"].start_pos() == nested_start);
CHECK(b["b"].end_pos() == nested_end);
// the moved-from value is reset to a null and reports npos
CHECK(a.is_null()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move)
CHECK(a.start_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move)
CHECK(a.end_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move)
}
SECTION("swap() exchanges positions along with values")
{
// basic_json::swap() (json.hpp, around line 3626, and the friend
// swap() that forwards to it) swaps start_position/end_position
// together with m_data.m_type and m_data.m_value, so after
// swap(a, b) each variable's position describes its own new
// content, consistent with copy-assignment's
// operator=(basic_json) (json.hpp, around line 1291), which also
// swaps positions as part of its copy-and-swap implementation.
json a = json::parse(R"({"a":1})");
json b = json::parse(R"([1,2,3,4,5])");
const auto a_start = a.start_pos();
const auto a_end = a.end_pos();
const auto b_start = b.start_pos();
const auto b_end = b.end_pos();
// both start at 0 (root values start right away), but their
// lengths (and thus end positions) differ, which is enough to
// tell after the swap whether positions actually moved with
// the values
CHECK(a_end != b_end);
using std::swap;
swap(a, b);
// values were exchanged as expected ...
CHECK(a == json::parse(R"([1,2,3,4,5])"));
CHECK(b == json::parse(R"({"a":1})"));
// ... and so were positions: each variable now carries the
// other's original position, describing its own new content
CHECK(a.start_pos() == b_start);
CHECK(a.end_pos() == b_end);
CHECK(b.start_pos() == a_start);
CHECK(b.end_pos() == a_end);
// member swap() behaves the same as the free function
json c = json::parse(R"({"a":1})");
json d = json::parse(R"([1,2,3,4,5])");
const auto c_start = c.start_pos();
const auto c_end = c.end_pos();
const auto d_start = d.start_pos();
const auto d_end = d.end_pos();
c.swap(d);
CHECK(c.start_pos() == d_start);
CHECK(c.end_pos() == d_end);
CHECK(d.start_pos() == c_start);
CHECK(d.end_pos() == c_end);
}
SECTION("mutating a parsed document leaves positions of unrelated values untouched")
{
// Positions are recorded once, during parsing, and are not
// recomputed on mutation. As a consequence, after a mutation the
// parent's own recorded span may no longer describe its current
// (serialized) content -- it still describes what was originally
// parsed. This is characterized here as current behavior, not
// asserted to be desirable or specified.
SECTION("operator[] adding a new object key")
{
const std::string s = R"({"a":1})";
json j = json::parse(s);
const auto root_start = j.start_pos();
const auto root_end = j.end_pos();
const auto a_start = j["a"].start_pos();
const auto a_end = j["a"].end_pos();
j["c"] = 42;
// the newly-added value was never parsed, so it has no position
CHECK(j["c"].start_pos() == std::string::npos);
CHECK(j["c"].end_pos() == std::string::npos);
// the existing sibling's position is unaffected
CHECK(j["a"].start_pos() == a_start);
CHECK(j["a"].end_pos() == a_end);
// the parent's own recorded span is left as-is (now stale:
// it still reflects the original, shorter `{"a":1}` string)
CHECK(j.start_pos() == root_start);
CHECK(j.end_pos() == root_end);
}
SECTION("push_back on a parsed array")
{
const std::string s = R"([1,2,3])";
json j = json::parse(s);
const auto root_start = j.start_pos();
const auto root_end = j.end_pos();
const auto first_start = j[0].start_pos();
j.push_back(4);
CHECK(j.back().start_pos() == std::string::npos);
CHECK(j.back().end_pos() == std::string::npos);
CHECK(j[0].start_pos() == first_start);
CHECK(j.start_pos() == root_start);
CHECK(j.end_pos() == root_end);
}
SECTION("erase on a parsed array shifts elements but keeps their own positions")
{
const std::string s = R"([1,2,3])";
json j = json::parse(s);
const auto second_start = j[1].start_pos();
const auto third_start = j[2].start_pos();
const auto root_start = j.start_pos();
const auto root_end = j.end_pos();
j.erase(0);
// remaining elements moved down an index, but each one still
// reports the position it had *before* the erase (i.e. its
// position in the original source string, not a
// recalculated one)
CHECK(j[0].start_pos() == second_start);
CHECK(j[1].start_pos() == third_start);
// the parent's own recorded span is again left as-is
CHECK(j.start_pos() == root_start);
CHECK(j.end_pos() == root_end);
}
}
}
SECTION("input adapters")
{
SECTION("wide string input: positions count transcoded UTF-8 bytes, not wide characters")
{
// 'é' (U+00E9) is a single code unit in a wchar_t/UTF-16 string, but
// transcodes to 2 bytes in UTF-8; the lexer only ever sees the
// transcoded UTF-8 byte stream, so reported positions are byte
// offsets into that UTF-8 stream, not indices into the original
// std::wstring.
// é (rather than a literal 'é' byte sequence in this source
// file) so the wide-string literal's meaning does not depend on
// the compiler's assumed source character set (MSVC, without
// /utf-8, would otherwise decode the raw UTF-8 bytes using the
// system code page instead of as UTF-8)
const std::wstring ws = L"{\"a\":\"\u00e9\u00e9\"}";
CHECK(ws.size() == 10); // 10 wide characters
const json j = json::parse(ws);
CHECK(j.start_pos() == 0);
// the transcoded UTF-8 form is 2 bytes longer than the wide string,
// because each of the two 'é' characters becomes 2 UTF-8 bytes
CHECK(j.end_pos() == 12);
CHECK(j.end_pos() != ws.size());
const json& a = j["a"];
CHECK(a.start_pos() == 5);
CHECK(a.end_pos() == 11);
}
SECTION("BOM-prefixed input: start_pos() reflects the skipped 3-byte BOM")
{
const std::string s = "\xEF\xBB\xBF{\"a\":1}";
const json j = json::parse(s);
// the lexer silently skips the BOM before parsing the value, so
// the root value's recorded span starts right after it
CHECK(j.start_pos() == 3);
CHECK(j.end_pos() == s.size());
}
SECTION("std::istringstream: positions are consistent, not npos")
{
const std::string s = R"({"a":1,"b":2})";
std::istringstream ss(s);
const json j = json::parse(ss);
CHECK(j.start_pos() == 0);
CHECK(j.end_pos() == s.size());
CHECK(j["a"].start_pos() == 5);
}
SECTION("std::ifstream: positions are consistent, not npos")
{
const std::string s = R"({"a":1,"b":2})";
{
std::ofstream file("unit-class_parser_diagnostic_positions.tmp");
file << s;
}
{
std::ifstream f("unit-class_parser_diagnostic_positions.tmp");
const json j = json::parse(f);
CHECK(j.start_pos() == 0);
CHECK(j.end_pos() == s.size());
CHECK(j["a"].start_pos() == 5);
}
static_cast<void>(std::remove("unit-class_parser_diagnostic_positions.tmp"));
}
SECTION("iterator-pair input: positions are consistent, not npos")
{
const std::string s = R"({"a":1,"b":2})";
const json j = json::parse(s.begin(), s.end());
CHECK(j.start_pos() == 0);
CHECK(j.end_pos() == s.size());
CHECK(j["a"].start_pos() == 5);
}
SECTION("binary formats have no text positions")
{
// binary formats (CBOR, MessagePack, UBJSON, BSON, BJData) are
// parsed via detail::binary_reader, which never sets
// start_position/end_position on the values it produces (they
// have no notion of a text offset), so every value's position
// stays at its default of npos.
const json src = json::parse(R"({"a":1,"b":[1,2]})");
const json from_cbor = json::from_cbor(json::to_cbor(src));
CHECK(from_cbor.start_pos() == std::string::npos);
CHECK(from_cbor.end_pos() == std::string::npos);
CHECK(from_cbor["a"].start_pos() == std::string::npos);
CHECK(from_cbor["b"][0].start_pos() == std::string::npos);
const json from_msgpack = json::from_msgpack(json::to_msgpack(src));
CHECK(from_msgpack.start_pos() == std::string::npos);
CHECK(from_msgpack.end_pos() == std::string::npos);
const json from_ubjson = json::from_ubjson(json::to_ubjson(src));
CHECK(from_ubjson.start_pos() == std::string::npos);
CHECK(from_ubjson.end_pos() == std::string::npos);
const json from_bson_val = json::from_bson(json::to_bson(src));
CHECK(from_bson_val.start_pos() == std::string::npos);
CHECK(from_bson_val.end_pos() == std::string::npos);
}
}
SECTION("user-driven SAX consumers with no lexer report npos")
{
// json::parse() internally wires up its json_sax_dom_parser with a
// pointer to its own lexer (see parser.hpp), which is how positions
// get set at all. A user who constructs a json_sax_dom_parser
// directly (e.g. to drive it via json::sax_parse()) and does not
// supply a lexer pointer gets a consumer with m_lexer_ref == nullptr;
// every "if (m_lexer_ref)" guard in json_sax.hpp is then skipped, so
// every value it produces keeps its default, unset position (npos).
// This was previously true but silently unasserted (operator==
// ignores positions), see #5420.
json result;
nlohmann::detail::json_sax_dom_parser<json, nlohmann::detail::string_input_adapter_type> sdp(result);
const std::string s = R"({"a":1,"b":[1,2,3]})";
CHECK(json::sax_parse(s, &sdp));
CHECK(result.start_pos() == std::string::npos);
CHECK(result.end_pos() == std::string::npos);
CHECK(result["a"].start_pos() == std::string::npos);
CHECK(result["a"].end_pos() == std::string::npos);
CHECK(result["b"][0].start_pos() == std::string::npos);
CHECK(result["b"][0].end_pos() == std::string::npos);
}
}
#endif
File diff suppressed because it is too large Load Diff
+51
View File
@@ -326,6 +326,57 @@ TEST_CASE("lexicographical comparison operators")
#endif
}
SECTION("integer/float mixed comparison is exact")
{
// Widening the integer to a double loses precision past the
// mantissa, so 2^63-2 and 2^63-1 both used to compare equal to the
// double 2^63 while differing from each other. That makes equality
// intransitive and the ordering not a strict weak ordering.
const json below_two_63 = static_cast<std::int64_t>(9223372036854775806LL);
const json max_int64 = (std::numeric_limits<std::int64_t>::max)();
const json two_63 = 9223372036854775808.0;
CHECK_FALSE(below_two_63 == two_63);
CHECK_FALSE(max_int64 == two_63);
CHECK(below_two_63 != max_int64);
CHECK(below_two_63 < max_int64);
CHECK(below_two_63 < two_63);
CHECK(max_int64 < two_63);
CHECK(two_63 > max_int64);
CHECK_FALSE(two_63 < max_int64);
// the same past the unsigned range
const json max_uint64 = (std::numeric_limits<std::uint64_t>::max)();
const json two_64 = 18446744073709551616.0;
CHECK_FALSE(max_uint64 == two_64);
CHECK(max_uint64 < two_64);
CHECK(two_64 > max_uint64);
// values a double represents exactly still compare equal
CHECK(json(1) == json(1.0));
CHECK(json(1u) == json(1.0));
CHECK(json(-3) == json(-3.0));
CHECK(json(1) < json(1.5));
CHECK(json(1.5) < json(2));
CHECK(json(2) > json(1.5));
// a NaN operand stays unordered against either integer kind
CHECK_FALSE(json(1) == json(nan));
CHECK_FALSE(json(1) < json(nan));
CHECK_FALSE(json(nan) < json(1));
CHECK_FALSE(json(1u) == json(nan));
#if JSON_HAS_THREE_WAY_COMPARISON
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
CHECK((max_int64 <=> two_63) == std::partial_ordering::less); // *NOPAD*
CHECK((two_63 <=> max_int64) == std::partial_ordering::greater); // *NOPAD*
CHECK((below_two_63 <=> max_int64) == std::partial_ordering::less); // *NOPAD*
CHECK((max_uint64 <=> two_64) == std::partial_ordering::less); // *NOPAD*
CHECK((json(1) <=> json(1.0)) == std::partial_ordering::equivalent); // *NOPAD*
CHECK((json(1) <=> json(nan)) == std::partial_ordering::unordered); // *NOPAD*
#endif
}
SECTION("compares unordered")
{
std::vector<std::vector<bool>> expected =
+4 -2
View File
@@ -98,8 +98,10 @@ void check_escaped(const char* original, const char* escaped = "", bool ensure_a
void check_escaped(const char* original, const char* escaped, const bool ensure_ascii)
{
std::stringstream ss;
json::serializer s(nlohmann::detail::output_adapter<char>(ss), ' ');
s.dump_escaped(original, ensure_ascii);
nlohmann::detail::output_stream_adapter<char> adapter(ss);
json::serializer s(adapter, ' ', false, ensure_ascii);
s.dump_escaped(original);
s.flush(); // dump_escaped writes into the serializer's internal buffer
CHECK(ss.str() == escaped);
}
} // namespace
+65
View File
@@ -1389,6 +1389,37 @@ TEST_CASE("value conversion")
// CHECK(m5["one"] == "eins");
}
SECTION("reserve is called on containers that support it (#5406)")
{
// build a larger object so that a missing/incorrect reserve()
// call would be more likely to corrupt or drop elements
json j_large;
for (int i = 0; i < 100; ++i)
{
j_large[std::to_string(i)] = i;
}
SECTION("std::unordered_map (supports reserve)")
{
const auto m = j_large.get<std::unordered_map<std::string, int>>();
CHECK(m.size() == 100);
for (int i = 0; i < 100; ++i)
{
CHECK(m.at(std::to_string(i)) == i);
}
}
SECTION("std::map (no reserve, fallback path)")
{
const auto m = j_large.get<std::map<std::string, int>>();
CHECK(m.size() == 100);
for (int i = 0; i < 100; ++i)
{
CHECK(m.at(std::to_string(i)) == i);
}
}
}
SECTION("std::multimap")
{
j1.get<std::multimap<std::string, int>>();
@@ -1761,6 +1792,40 @@ TEST_CASE("std::filesystem::path")
}
#endif
// the ADL to_json overload for std::u8string only exists under the same guard
// as std::filesystem::path support (it is otherwise only reached indirectly,
// via std::filesystem::path::u8string()) -- mirror both #if conditions from
// include/nlohmann/detail/conversions/to_json.hpp exactly
#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM
#if defined(__cpp_lib_char8_t)
TEST_CASE("std::u8string")
{
SECTION("ascii")
{
const std::u8string s = u8"Path";
json const j = s;
CHECK(j.template get<std::string>() == "Path");
}
SECTION("utf-8")
{
// use \u universal-character-names (rather than raw \x byte escapes
// or literal non-ASCII source bytes) to compose the multi-byte UTF-8
// encoding -- MSVC treats \x escapes used that way inside a u8
// literal as a nonstandard extension (warning C5321), which some of
// our CI configs promote to an error; \u is portable and produces
// the exact same encoded bytes without depending on the source
// file's encoding
const std::u8string s = u8"P\u011B\u0161ina";
json const j = s;
CHECK(j.template get<std::string>() == "P\xc4\x9b\xc5\xa1ina");
}
}
#endif
#endif
TEST_CASE("std::optional")
{
SECTION("null")
+150
View File
@@ -0,0 +1,150 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
#include <deque>
#include <map>
#include <memory>
#include <string>
#include <type_traits>
#include <vector>
namespace
{
// std::deque has no capacity() member function, which the library only needs
// to detect a reallocation for JSON_DIAGNOSTICS
using deque_json = nlohmann::basic_json<std::map, std::deque>;
// a std::vector whose at() is hidden: the library performs its own bounds
// check and must not fall back to the container's checked accessor
template<class T, class Allocator = std::allocator<T>>
class vector_without_at : public std::vector<T, Allocator>
{
public:
vector_without_at() = default;
// the array of an initializer list is built from a range
template<class InputIt>
vector_without_at(InputIt first, InputIt last) : std::vector<T, Allocator>(first, last) {}
void at() = delete;
};
using no_at_json = nlohmann::basic_json<std::map, vector_without_at>;
} // namespace
TEST_CASE("array type without capacity()")
{
SECTION("the iterators take their exception specification from the container")
{
// basic_json's iterators move exactly as the container iterators do:
// their move operations are defaulted without a declared noexcept,
// because an array or object type whose iterator is not nothrow move
// constructible would otherwise have them deleted (std::deque's is not
// with libstdc++ before 11, and neither are MSVC's debug iterators)
CHECK(std::is_nothrow_move_constructible<nlohmann::json::iterator>::value ==
(std::is_nothrow_move_constructible<nlohmann::json::object_t::iterator>::value
&& std::is_nothrow_move_constructible<nlohmann::json::array_t::iterator>::value));
CHECK(std::is_nothrow_move_assignable<nlohmann::json::iterator>::value ==
(std::is_nothrow_move_assignable<nlohmann::json::object_t::iterator>::value
&& std::is_nothrow_move_assignable<nlohmann::json::array_t::iterator>::value));
CHECK(std::is_nothrow_move_constructible<nlohmann::json::const_iterator>::value ==
(std::is_nothrow_move_constructible<nlohmann::json::object_t::const_iterator>::value
&& std::is_nothrow_move_constructible<nlohmann::json::array_t::const_iterator>::value));
// and they are movable at all, which is what dropping the declared
// noexcept buys for a std::deque array
CHECK(std::is_move_constructible<deque_json::iterator>::value);
CHECK(std::is_move_assignable<deque_json::iterator>::value);
}
SECTION("adding elements")
{
deque_json j = deque_json::array();
j.push_back(1);
j.push_back("two");
j.emplace_back(3);
j += 4;
CHECK(j.size() == 4);
CHECK(j == deque_json({1, "two", 3, 4}));
CHECK(j.back() == 4);
CHECK(j.front() == 1);
}
SECTION("accessing and modifying elements")
{
auto j = deque_json::parse(R"([1,2,3])");
CHECK(j[1] == 2);
CHECK(j.at(2) == 3);
// growing through operator[] fills up with null values
j[5] = 6;
CHECK(j.size() == 6);
CHECK(j[4].is_null());
CHECK(j[5] == 6);
j.erase(0);
CHECK(j == deque_json({2, 3, nullptr, nullptr, 6}));
auto it = j.erase(j.begin());
CHECK(*it == 3);
j.insert(j.begin(), 1);
CHECK(j.front() == 1);
}
SECTION("serialization and deserialization")
{
const auto j = deque_json::parse(R"({"a":[1,[2,3]],"b":[]})");
CHECK(j.dump() == R"({"a":[1,[2,3]],"b":[]})");
CHECK(deque_json::parse(j.dump()) == j);
CHECK(deque_json::from_cbor(deque_json::to_cbor(j)) == j);
// empty containers are flattened to null and cannot be restored
const auto nested = deque_json::parse(R"({"a":[1,[2,3]]})");
CHECK(nested.flatten().unflatten() == nested);
}
SECTION("references stay valid while the array grows")
{
deque_json j = deque_json::array();
j.push_back(1);
auto& first = j[0];
for (int i = 0; i < 100; ++i)
{
j.push_back(i);
}
CHECK(&first == &j[0]);
CHECK(first == 1);
}
}
TEST_CASE("array type without at()")
{
// built in memory rather than parsed, so that the exception message does
// not gain a byte range with JSON_DIAGNOSTIC_POSITIONS
no_at_json j = {1, 2, 3};
const auto& jc = j;
CHECK(j.at(0) == 1);
CHECK(j.at(2) == 3);
CHECK(jc.at(2) == 3);
CHECK_THROWS_WITH_AS(j.at(3), "[json.exception.out_of_range.401] array index 3 is out of range", no_at_json::out_of_range);
CHECK_THROWS_WITH_AS(jc.at(3), "[json.exception.out_of_range.401] array index 3 is out of range", no_at_json::out_of_range);
CHECK(j.at(no_at_json::json_pointer("/1")) == 2);
CHECK_THROWS_AS(j.at(no_at_json::json_pointer("/3")), no_at_json::out_of_range);
}
+79
View File
@@ -0,0 +1,79 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
#include <cstdint>
#include <functional>
#include <map>
#include <memory>
#include <string>
#include <vector>
#ifdef JSON_HAS_CPP_17
#include <cstddef>
#endif
namespace
{
// a BinaryType whose value type is signed: the elements must still be
// processed as the numbers 0..255
using char_binary_json = nlohmann::basic_json <
std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t,
double, std::allocator, nlohmann::adl_serializer, std::vector<char>, void >;
#ifdef JSON_HAS_CPP_17
// a BinaryType whose value type is not an integer type at all
using byte_binary_json = nlohmann::basic_json <
std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t,
double, std::allocator, nlohmann::adl_serializer, std::vector<std::byte>, void >;
#endif
} // namespace
TEST_CASE("binary type whose value type is not std::uint8_t")
{
SECTION("a signed value type does not dump negative numbers")
{
const std::vector<char> chars{'\0', '\x01', '\xFF'};
CHECK(char_binary_json::binary(chars).dump() == R"({"bytes":[0,1,255],"subtype":null})");
CHECK(char_binary_json::binary(chars, 42).dump() == R"({"bytes":[0,1,255],"subtype":42})");
CHECK(char_binary_json::binary({}).dump() == R"({"bytes":[],"subtype":null})");
}
SECTION("the default binary type is unchanged")
{
CHECK(nlohmann::json::binary({0, 1, 255}, 42).dump() == R"({"bytes":[0,1,255],"subtype":42})");
}
#ifdef JSON_HAS_CPP_17
SECTION("dumping a value type that is not an integer")
{
const std::vector<std::byte> bytes{std::byte{0}, std::byte{1}, std::byte{0xFF}};
CHECK(byte_binary_json::binary(bytes).dump() == R"({"bytes":[0,1,255],"subtype":null})");
CHECK(byte_binary_json::binary(bytes, 42).dump() == R"({"bytes":[0,1,255],"subtype":42})");
CHECK(byte_binary_json::binary({}).dump() == R"({"bytes":[],"subtype":null})");
}
SECTION("hashing and the binary formats")
{
const std::vector<std::byte> bytes{std::byte{0}, std::byte{1}, std::byte{0xFF}};
const auto j = byte_binary_json::binary(bytes);
CHECK(std::hash<byte_binary_json> {}(j) == std::hash<byte_binary_json> {}(j));
CHECK(byte_binary_json::from_cbor(byte_binary_json::to_cbor(j)) == j);
CHECK(byte_binary_json::from_msgpack(byte_binary_json::to_msgpack(j)) == j);
// UBJSON has no binary type, so binary values are written as an array
CHECK(byte_binary_json::from_ubjson(byte_binary_json::to_ubjson(j)) == byte_binary_json({0, 1, 255}));
}
#endif
}
+323
View File
@@ -0,0 +1,323 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
#include <cstdint>
#include <map>
#include <string>
#include <type_traits>
#include <utility>
#include <vector>
namespace
{
// An ObjectType that does *not* define a key_compare member type, which is
// what every hash map looks like to the library.
//
// A hash map is deliberately not used here: object_t is probed for
// key_compare inside the definition of basic_json, that is, while basic_json
// is still an incomplete type, and whether a hash map can be instantiated
// with an incomplete mapped type depends on the standard library (libstdc++ 9
// needs the size of the mapped type for its node type and rejects it). So the
// object type wraps a std::map instead of inheriting from it: an earlier
// version derived from std::map and shadowed the inherited key_compare type
// with a same-named member function, relying on ordinary member hiding to
// make key_compare unreachable as a type. MSVC 2017 (AppVeyor, /std:c++17)
// does not honor that hiding for a typename-qualified lookup performed from
// outside the class and still resolves key_compare to the base's comparator
// type, so the library's probe incorrectly found one. Composition sidesteps
// the question entirely: with no base class, there is no key_compare to find
// under any lookup rule.
template<class Key, class T, class Compare, class Allocator>
class no_key_compare_map
{
using map_t = std::map<Key, T, Compare, Allocator>;
map_t data;
public:
using key_type = typename map_t::key_type;
using mapped_type = typename map_t::mapped_type;
using value_type = typename map_t::value_type;
using size_type = typename map_t::size_type;
using allocator_type = typename map_t::allocator_type;
using iterator = typename map_t::iterator;
using const_iterator = typename map_t::const_iterator;
// -Weffc++ asks for the member to be initialized in the member
// initialization list, which a defaulted constructor does not do; the
// exception specification a defaulted one would have carried has to be
// written out as well, or -Wnoexcept objects where the standard library
// takes noexcept(construct(...))
no_key_compare_map() noexcept(std::is_nothrow_default_constructible<map_t>::value) : data() {}
// converting between two basic_json types builds the object from a range
template<class InputIt>
no_key_compare_map(InputIt first, InputIt last) : data(first, last) {}
iterator begin() noexcept
{
return data.begin();
}
iterator end() noexcept
{
return data.end();
}
const_iterator begin() const noexcept
{
return data.begin();
}
const_iterator end() const noexcept
{
return data.end();
}
const_iterator cbegin() const noexcept
{
return data.cbegin();
}
const_iterator cend() const noexcept
{
return data.cend();
}
bool empty() const noexcept
{
return data.empty();
}
size_type size() const noexcept
{
return data.size();
}
size_type max_size() const noexcept
{
return data.max_size();
}
void clear() noexcept
{
data.clear();
}
iterator find(const key_type& key)
{
return data.find(key);
}
const_iterator find(const key_type& key) const
{
return data.find(key);
}
size_type count(const key_type& key) const
{
return data.count(key);
}
std::pair<iterator, bool> emplace(const key_type& key, const mapped_type& value)
{
return data.emplace(key, value);
}
std::pair<iterator, bool> insert(const value_type& value)
{
return data.insert(value);
}
template<class InputIt>
void insert(InputIt first, InputIt last)
{
data.insert(first, last);
}
mapped_type& operator[](const key_type& key)
{
return data[key];
}
mapped_type& at(const key_type& key)
{
return data.at(key);
}
const mapped_type& at(const key_type& key) const
{
return data.at(key);
}
iterator erase(iterator pos)
{
return data.erase(pos);
}
iterator erase(iterator first, iterator last)
{
return data.erase(first, last);
}
size_type erase(const key_type& key)
{
return data.erase(key);
}
void swap(no_key_compare_map& other) noexcept(noexcept(data.swap(other.data)))
{
data.swap(other.data);
}
friend bool operator==(const no_key_compare_map& lhs, const no_key_compare_map& rhs)
{
return lhs.data == rhs.data;
}
friend bool operator<(const no_key_compare_map& lhs, const no_key_compare_map& rhs)
{
return lhs.data < rhs.data;
}
};
using no_key_compare_json = nlohmann::basic_json<no_key_compare_map>;
// An ObjectType whose erase(iterator) returns void rather than the following
// iterator, as for instance Abseil's hash maps do
template<class Key, class T, class Compare, class Allocator>
struct void_erase_map : std::map<Key, T, Compare, Allocator>
{
using base_t = std::map<Key, T, Compare, Allocator>;
using iterator = typename base_t::iterator;
using base_t::erase;
void erase(iterator pos)
{
base_t::erase(pos);
}
};
using void_erase_json = nlohmann::basic_json<void_erase_map>;
} // namespace
TEST_CASE("object type whose erase() returns void")
{
SECTION("erasing every element through the returned iterator")
{
void_erase_json j;
for (int i = 0; i < 8; ++i)
{
j["k" + std::to_string(i)] = i;
}
std::size_t erased = 0;
for (auto it = j.begin(); it != j.end(); ++erased)
{
it = j.erase(it);
}
CHECK(erased == 8);
CHECK(j.empty());
}
SECTION("erasing in the middle returns the following element")
{
void_erase_json j;
for (int i = 0; i < 4; ++i)
{
j["k" + std::to_string(i)] = i;
}
auto it = j.begin();
++it;
const auto after = j.erase(it);
CHECK(j.size() == 3);
CHECK(after.key() == "k2");
CHECK(after.value() == 2);
CHECK(!j.contains("k1"));
}
SECTION("the other erase overloads are unaffected")
{
void_erase_json j;
j["a"] = 1;
j["b"] = 2;
j["c"] = 3;
CHECK(j.erase("a") == 1);
CHECK(j.erase("nope") == 0);
j.erase(j.begin(), j.end());
CHECK(j.empty());
}
}
TEST_CASE("object type without key_compare")
{
SECTION("object_comparator_t falls back to default_object_comparator_t")
{
CHECK(std::is_same < no_key_compare_json::object_comparator_t,
no_key_compare_json::default_object_comparator_t >::value);
}
SECTION("object types defining key_compare are unaffected")
{
CHECK(std::is_same<nlohmann::json::object_comparator_t,
nlohmann::json::object_t::key_compare>::value);
CHECK(std::is_same<nlohmann::ordered_json::object_comparator_t,
nlohmann::ordered_json::object_t::key_compare>::value);
}
SECTION("creating and accessing values")
{
no_key_compare_json j;
j["one"] = 1;
j["two"] = "zwei";
j["three"]["nested"] = true;
CHECK(j.size() == 3);
CHECK(j.at("one") == 1);
CHECK(j["two"] == "zwei");
CHECK(j["three"]["nested"] == true);
CHECK(j.contains("one"));
CHECK(!j.contains("four"));
CHECK(j.find("one") != j.end());
CHECK(j.count("one") == 1);
CHECK(j.erase("one") == 1);
CHECK(j.size() == 2);
}
SECTION("serialization and deserialization")
{
const auto j = no_key_compare_json::parse(R"({"a":[1,2,3],"b":{"c":null}})");
CHECK(j["a"].size() == 3);
CHECK(j["a"][2] == 3);
CHECK(j["b"]["c"].is_null());
CHECK(no_key_compare_json::parse(j.dump()) == j);
}
SECTION("binary formats")
{
const auto j = no_key_compare_json::parse(R"({"a":[1,2,3],"b":"x"})");
CHECK(no_key_compare_json::from_cbor(no_key_compare_json::to_cbor(j)) == j);
CHECK(no_key_compare_json::from_msgpack(no_key_compare_json::to_msgpack(j)) == j);
}
SECTION("flatten and unflatten")
{
// "o" has a key that looks like an array index, so unflatten() must
// not turn it into an array
const auto j = no_key_compare_json::parse(
R"({"c":[1,2,3],"d":{"e":"s"},"n":[[0,1],[2]],"o":{"2":"x"}})");
CHECK(j.flatten().unflatten() == j);
}
SECTION("conversion to and from nlohmann::json")
{
const auto j = no_key_compare_json::parse(R"({"a":1,"b":[true,null]})");
const nlohmann::json converted(j);
CHECK(converted.is_object());
CHECK(converted["a"] == 1);
CHECK(converted["b"][0] == true);
CHECK(converted["b"][1].is_null());
CHECK(no_key_compare_json(converted) == j);
}
}
+54 -2
View File
@@ -8,6 +8,14 @@
#include "doctest_compatibility.h"
// capture whether JSON_STRICT_NUL_HANDLING was enabled on the command line
// (e.g. -DJSON_STRICT_NUL_HANDLING=1) *before* including json.hpp, since the
// library #undefs JSON_STRICT_NUL_HANDLING itself once the header has been
// fully processed (see include/nlohmann/detail/macro_unscope.hpp)
#if defined(JSON_STRICT_NUL_HANDLING) && (JSON_STRICT_NUL_HANDLING == 1)
#define JSON_TEST_STRICT_NUL_HANDLING_ENABLED 1
#endif
#include <nlohmann/json.hpp>
using nlohmann::json;
#ifdef JSON_TEST_NO_GLOBAL_UDLS
@@ -380,6 +388,23 @@ TEST_CASE("deserialization")
CHECK(j == json({"foo", 1, 2, 3, false, {{"one", 1}}}));
}
SECTION("operator>> with a NUL byte after the value (issue #5530)")
{
// operator>> parses non-strictly (it does not require the whole
// stream to be consumed), so a NUL byte following a complete
// value is simply left unread on the stream and never reaches
// the "expected end of input" check that JSON_STRICT_NUL_HANDLING
// affects; this holds regardless of the macro (verified below for
// the opt-in state as well)
std::string data = "123";
data.push_back('\0');
std::istringstream ss(data);
json j;
ss >> j;
CHECK(j == json(123));
CHECK(ss.good());
}
SECTION("user-defined string literal")
{
CHECK("[\"foo\",1,2,3,false,{\"one\":1}]"_json == json({"foo", 1, 2, 3, false, {{"one", 1}}}));
@@ -462,6 +487,27 @@ TEST_CASE("deserialization")
CHECK_THROWS_WITH_AS(ss >> j, "[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing array - unexpected end of input; expected ']'", json::parse_error&);
}
#if defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED)
SECTION("operator>> with a NUL byte where a value is expected (JSON_STRICT_NUL_HANDLING == 1, issue #5530)")
{
// a trailing NUL byte *after* a complete value is unaffected by the
// macro (see the successful-deserialization "operator>> with a NUL
// byte after the value" section above): operator>> parses
// non-strictly and never reaches the "expected end of input" check
// that the macro changes. A NUL byte where a *value* is expected,
// however, goes through the same token dispatch as any other input
// and is affected: with the macro enabled it now raises
// parse_error.101 (like any other unrecognized byte) instead of
// being silently treated the same as an empty stream.
std::string const data(1, '\0');
std::istringstream ss(data);
json j;
CHECK_THROWS_WITH_AS(ss >> j,
"[json.exception.parse_error.101] parse error at line 1, column 1: syntax error while parsing value - invalid literal; last read: '<U+0000>'",
json::parse_error&);
}
#endif
SECTION("user-defined string literal")
{
CHECK_THROWS_WITH_AS("[\"foo\",1,2,3,false,{\"one\":1}"_json, "[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing array - unexpected end of input; expected ']'", json::parse_error&);
@@ -510,7 +556,11 @@ TEST_CASE("deserialization")
SECTION("from std::array")
{
std::array<uint8_t, 5> const v { {'t', 'r', 'u', 'e'} };
// sized to exactly the length of "true": a size of 5 would leave
// a value-initialized trailing 0x00 element that is only
// silently accepted as end-of-input by default and would fail
// under JSON_STRICT_NUL_HANDLING
std::array<uint8_t, 4> const v { {'t', 'r', 'u', 'e'} };
CHECK(json::parse(v) == json(true));
CHECK(json::accept(v));
@@ -606,7 +656,9 @@ TEST_CASE("deserialization")
SECTION("from std::array")
{
std::array<uint8_t, 5> v { {'t', 'r', 'u', 'e'} };
// sized to exactly the length of "true", see the analogous
// "from std::array" section above for why
std::array<uint8_t, 4> v { {'t', 'r', 'u', 'e'} };
CHECK(json::parse(std::begin(v), std::end(v)) == json(true));
CHECK(json::accept(std::begin(v), std::end(v)));
@@ -1,44 +0,0 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
#ifdef JSON_DIAGNOSTICS
#undef JSON_DIAGNOSTICS
#endif
#define JSON_DIAGNOSTICS 0
#define JSON_DIAGNOSTIC_POSITIONS 1
#include <nlohmann/json.hpp>
using json = nlohmann::json;
TEST_CASE("Better diagnostics with positions only")
{
SECTION("invalid type")
{
const std::string json_invalid_string = R"(
{
"address": {
"street": "Fake Street",
"housenumber": "1"
}
}
)";
json j = json::parse(json_invalid_string);
CHECK_THROWS_WITH_AS(j.at("address").at("housenumber").get<int>(),
"[json.exception.type_error.302] (bytes 108-111) type must be number, but is string", json::type_error);
}
SECTION("invalid type without positions")
{
const json j = "foo";
CHECK_THROWS_WITH_AS(j.get<int>(),
"[json.exception.type_error.302] type must be number, but is string", json::type_error);
}
}
+79 -1
View File
@@ -8,7 +8,9 @@
#include "doctest_compatibility.h"
#define JSON_DIAGNOSTICS 1
#ifndef JSON_DIAGNOSTICS
#define JSON_DIAGNOSTICS 1
#endif
#define JSON_DIAGNOSTIC_POSITIONS 1
#include <nlohmann/json.hpp>
@@ -27,8 +29,13 @@ TEST_CASE("Better diagnostics with positions")
}
)";
json j = json::parse(json_invalid_string);
#if JSON_DIAGNOSTICS
CHECK_THROWS_WITH_AS(j.at("address").at("housenumber").get<int>(),
"[json.exception.type_error.302] (/address/housenumber) (bytes 108-111) type must be number, but is string", json::type_error);
#else
CHECK_THROWS_WITH_AS(j.at("address").at("housenumber").get<int>(),
"[json.exception.type_error.302] (bytes 108-111) type must be number, but is string", json::type_error);
#endif
}
SECTION("invalid type without positions")
@@ -68,13 +75,84 @@ TEST_CASE("Better diagnostics with positions")
CHECK(j.end_pos() == root.size());
}
SECTION("copying keeps the positions of nested values (#5387)")
{
// Values nested deeper than the copy constructor's descent bound are
// copied without the call stack, on a path that has to carry the
// positions over itself; shallower ones copy their containers, which
// bring the positions along. Both sides of the bound are checked here.
const auto check_copy = [](std::size_t depth, bool objects)
{
CAPTURE(depth)
CAPTURE(objects)
const std::string opening = objects ? R"({"a":)" : "[";
const std::string closing = objects ? "}" : "]";
std::string text;
for (std::size_t i = 0; i < depth; ++i)
{
text += opening;
}
text += "12";
for (std::size_t i = 0; i < depth; ++i)
{
text += closing;
}
const json original = json::parse(text);
const json copy(original); // NOLINT(performance-unnecessary-copy-initialization)
const json* o = &original;
const json* c = &copy;
for (std::size_t level = 0; level <= depth; ++level)
{
CAPTURE(level)
REQUIRE(c->start_pos() == o->start_pos());
REQUIRE(c->end_pos() == o->end_pos());
if (level < depth)
{
o = objects ? &o->at("a") : &o->at(0);
c = objects ? &c->at("a") : &c->at(0);
}
}
};
const auto check_arrays = [&check_copy](std::size_t depth)
{
check_copy(depth, false);
};
const auto check_objects = [&check_copy](std::size_t depth)
{
check_copy(depth, true);
};
check_arrays(1);
check_arrays(127);
check_arrays(128);
check_arrays(129);
check_arrays(300);
check_objects(1);
check_objects(127);
check_objects(128);
check_objects(129);
check_objects(300);
}
SECTION("JSON patch add to primitive parent (#4292)")
{
// the JSON Patch "add" target /foo/bar/baz has a string parent
// (/foo/bar); the position of that parent is reported in the message
const json doc = json::parse(R"({"foo":{"bar":"a string"}})");
const json patch = json::parse(R"([{"op":"add","path":"/foo/bar/baz","value":1}])");
#if JSON_DIAGNOSTICS
CHECK_THROWS_WITH_AS(doc.patch(patch),
"[json.exception.out_of_range.411] (/foo/bar) (bytes 14-24) cannot add value: the JSON Patch 'add' target's parent is of type string, but must be an object or array", json::out_of_range);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch),
"[json.exception.out_of_range.411] (bytes 14-24) cannot add value: the JSON Patch 'add' target's parent is of type string, but must be an object or array", json::out_of_range);
#endif
}
}
+237
View File
@@ -273,5 +273,242 @@ TEST_CASE("Regression tests for extended diagnostics")
CHECK(j1["numbers"]["two"] == 2);
CHECK(j1["string"] == "t");
}
SECTION("Regression test for issue #5387 - copying keeps the parents of nested values")
{
// A value nested deeper than the copy constructor's descent bound is
// copied without the call stack. Every container that path creates has
// to have the parents of its children set, or the JSON Pointer in the
// diagnostic is cut short.
const std::size_t depth = 300;
SECTION("objects")
{
json j = "not a number";
std::string pointer;
for (std::size_t i = 0; i < depth; ++i)
{
j = json{{"a", j}};
pointer += "/a";
}
json const copy(j); // NOLINT(performance-unnecessary-copy-initialization)
const json* inner = &copy;
for (std::size_t i = 0; i < depth; ++i)
{
inner = &inner->at("a");
}
std::string const expected = "[json.exception.type_error.302] (" + pointer + ") type must be number, but is string";
int i = 0;
CHECK_THROWS_WITH_AS(i = inner->get<int>(), expected.c_str(), json::type_error);
CHECK(i == 0);
}
SECTION("arrays")
{
json j = "not a number";
std::string pointer;
for (std::size_t i = 0; i < depth; ++i)
{
j = json::array({j});
pointer += "/0";
}
json const copy(j); // NOLINT(performance-unnecessary-copy-initialization)
const json* inner = &copy;
for (std::size_t i = 0; i < depth; ++i)
{
inner = &inner->at(0);
}
std::string const expected = "[json.exception.type_error.302] (" + pointer + ") type must be number, but is string";
int i = 0;
CHECK_THROWS_WITH_AS(i = inner->get<int>(), expected.c_str(), json::type_error);
CHECK(i == 0);
}
}
SECTION("Regression test - swap(array_t&)/swap(object_t&) must update JSON_DIAGNOSTICS parent pointers")
{
// swap(array_t&)
{
json j = json::array();
json::array_t arr = {json::array({1})};
j.swap(arr);
// parent pointers of the moved-in elements must point into j, not
// into the now-defunct free-standing array_t
CHECK_THROWS_WITH_AS(j[0][0].get<std::string>(), "[json.exception.type_error.302] (/0/0) type must be string, but is number", json::type_error);
// must not trigger assert_invariant() in a debug/assert-enabled build
json const k = j;
CHECK(k == j);
}
// swap(object_t&)
{
json o = json::object();
json::object_t obj = {{"a", json::array({1})}};
o.swap(obj);
CHECK_THROWS_WITH_AS(o["a"][0].get<std::string>(), "[json.exception.type_error.302] (/a/0) type must be string, but is number", json::type_error);
// must not trigger assert_invariant() in a debug/assert-enabled build
json const p = o;
CHECK(p == o);
}
}
SECTION("Regression test - erase() and update() must keep JSON_DIAGNOSTICS parent pointers of ordered_json members")
{
// ordered_json keeps its members in a vector: erasing a member
// re-constructs all members after it in place, and adding a key may
// reallocate the vector; both reset the parent pointers of the members
// that were moved
using nlohmann::ordered_json;
const auto check_parents = [](const ordered_json & j)
{
// const access, so operator[] cannot repair the parent pointers
CHECK_THROWS_WITH_AS(j["z"]["x"].at(0), "[json.exception.type_error.304] (/z/x) cannot use at() with number", ordered_json::type_error);
// must not trigger assert_invariant() in a debug/assert-enabled build
ordered_json const copy = j; // NOLINT(performance-unnecessary-copy-initialization)
CHECK(copy == j);
};
// erase(key)
{
ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}};
CHECK(j.erase("a") == 1);
check_parents(j);
}
// erase(iterator)
{
ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}};
j.erase(j.begin());
check_parents(j);
}
// erase(iterator, iterator)
{
ordered_json j = {{"a", 1}, {"b", 2}, {"z", {{"x", 1}}}};
j.erase(j.begin(), j.find("z"));
check_parents(j);
}
// patch() removes via erase(iterator)
{
ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}};
j.patch_inplace(ordered_json::parse(R"([{"op": "remove", "path": "/a"}])"));
check_parents(j);
}
// update(j)
{
ordered_json j = {{"z", {{"x", 1}}}};
j.update({{"a", 1}, {"b", 2}});
check_parents(j);
}
// update(j, true), the outer and the nested vector both grow
{
ordered_json j = {{"z", {{"x", 1}}}};
j.update({{"z", {{"y", 2}}}, {"a", 1}}, true);
check_parents(j);
}
// update(j, true) around its descent bound, where the nested vectors
// grow while the objects are merged without recursing
for (const std::size_t depth :
{
nlohmann::detail::recursion_depth_limit() - 1, nlohmann::detail::recursion_depth_limit(), nlohmann::detail::recursion_depth_limit() + 2
})
{
ordered_json j = {{"z", {{"x", 1}}}};
ordered_json patch = {{"a", 1}, {"b", 2}, {"c", {{"d", 3}}}};
for (std::size_t i = 0; i < depth; ++i)
{
j = ordered_json{{"k", 0}, {"n", std::move(j)}};
patch = ordered_json{{"n", std::move(patch)}, {"l", 1}, {"m", 2}};
}
j.update(patch, true);
// must not trigger assert_invariant() on any level in a
// debug/assert-enabled build
ordered_json const copy = j; // NOLINT(performance-unnecessary-copy-initialization)
CHECK(copy == j);
}
// merge_patch() inserts "c" and removes "d" at /a/c, then inserts "e"
// at /a, which copies /a/c
{
auto j = ordered_json::parse(R"({"a": {"c": {"d": {}}}})");
j.merge_patch(ordered_json::parse(R"({"a": {"c": {"c": "s", "d": null}, "e": "s"}})"));
CHECK(j.dump() == R"({"a":{"c":{"c":"s"},"e":"s"}})");
auto const& constJ = j;
#if JSON_DIAGNOSTIC_POSITIONS
CHECK_THROWS_WITH_AS(constJ["a"]["c"]["c"].at(0), "[json.exception.type_error.304] (/a/c/c) (bytes 18-21) cannot use at() with string", ordered_json::type_error);
#else
CHECK_THROWS_WITH_AS(constJ["a"]["c"]["c"].at(0), "[json.exception.type_error.304] (/a/c/c) cannot use at() with string", ordered_json::type_error);
#endif
ordered_json const copy = j;
CHECK(copy == j);
}
}
}
TEST_CASE("Better diagnostics past the descent bound of update() and merge_patch()")
{
// Both merge objects nested more than detail::recursion_depth_limit()
// (128) levels deep without recursing; the values they add or replace
// there must still know their parents.
// The values are built rather than parsed, so that the expected messages
// carry no byte positions under JSON_DIAGNOSTIC_POSITIONS.
const std::size_t depth = 200;
json target = {{"x", 1}};
json patch = {{"y", 2}};
std::string path;
for (std::size_t i = 0; i < depth; ++i)
{
target = json{{"a", std::move(target)}};
patch = json{{"a", std::move(patch)}};
path += "/a";
}
const std::string expected_x = "[json.exception.type_error.304] (" + path + "/x) cannot use at() with number";
const std::string expected_y = "[json.exception.type_error.304] (" + path + "/y) cannot use at() with number";
SECTION("update()")
{
json j = target;
j.update(patch, true);
// walk down through const references, which leave m_parent alone
const json* p = &j;
for (std::size_t i = 0; i < depth; ++i)
{
p = &p->at("a");
}
CHECK_THROWS_WITH_AS(p->at("x").at(0), expected_x.c_str(), json::type_error);
CHECK_THROWS_WITH_AS(p->at("y").at(0), expected_y.c_str(), json::type_error);
}
SECTION("merge_patch()")
{
json j = target;
j.merge_patch(patch);
const json* p = &j;
for (std::size_t i = 0; i < depth; ++i)
{
p = &p->at("a");
}
CHECK_THROWS_WITH_AS(p->at("x").at(0), expected_x.c_str(), json::type_error);
CHECK_THROWS_WITH_AS(p->at("y").at(0), expected_y.c_str(), json::type_error);
}
}
+113
View File
@@ -13,6 +13,78 @@ using json = nlohmann::json;
using ordered_json = nlohmann::ordered_json;
#include <set>
#include <string>
namespace
{
// how detail::hash defines the hash of an array or object: the seeds of the
// elements, combined in order. Recursive, so only usable on values nested a
// few hundred levels deep - which is exactly what is needed to check that the
// iterative path taken below detail::recursion_depth_limit() computes the same.
template<typename BasicJsonType>
std::size_t reference_hash(const BasicJsonType& j)
{
using nlohmann::detail::combine;
using string_t = typename BasicJsonType::string_t;
if (!j.is_structured())
{
return std::hash<BasicJsonType> {}(j);
}
auto seed = combine(static_cast<std::size_t>(j.type()), j.size());
for (const auto& element : j.items())
{
if (j.is_object())
{
seed = combine(seed, std::hash<string_t> {}(element.key()));
}
seed = combine(seed, reference_hash(element.value()));
}
return seed;
}
// a value nested `depth` levels deep, with siblings on every level
template<typename BasicJsonType>
BasicJsonType nested(const std::size_t depth, const bool objects)
{
BasicJsonType value = "leaf";
for (std::size_t i = 0; i < depth; ++i)
{
if (objects)
{
value = BasicJsonType{{"before", i}, {"nested", std::move(value)}, {"after", {i, "x"}}};
}
else
{
value = BasicJsonType::array({i, std::move(value), BasicJsonType::object({{"k", i}})});
}
}
return value;
}
std::string nested_text(const std::size_t depth, const bool objects)
{
std::string text;
if (objects)
{
text.reserve((6 * depth) + 1);
for (std::size_t i = 0; i < depth; ++i)
{
text += "{\"a\":";
}
text += "1";
text.append(depth, '}');
}
else
{
text.assign(depth, '[');
text += "1";
text.append(depth, ']');
}
return text;
}
} // namespace
TEST_CASE("hash<nlohmann::json>")
{
@@ -111,3 +183,44 @@ TEST_CASE("hash<nlohmann::ordered_json>")
CHECK(hashes.size() == 21);
}
TEST_CASE("hash of deeply nested values")
{
SECTION("hashing past the descent bound computes the same values")
{
// every depth on either side of where the iterative path takes over
for (std::size_t depth = 0; depth <= (2 * nlohmann::detail::recursion_depth_limit()) + 10; ++depth)
{
CAPTURE(depth);
const auto arrays = nested<json>(depth, false);
const auto objects = nested<json>(depth, true);
const auto ordered = nested<ordered_json>(depth, true);
CHECK(std::hash<json> {}(arrays) == reference_hash(arrays));
CHECK(std::hash<json> {}(objects) == reference_hash(objects));
CHECK(std::hash<ordered_json> {}(ordered) == reference_hash(ordered));
}
}
SECTION("values nested too deeply for the call stack (#5545)")
{
// recursing once per level used to exhaust the call stack here; the
// values are only parsed and hashed, never copied or compared, since
// those recurse as well
const std::size_t depth = 100000;
for (const bool objects :
{
false, true
})
{
CAPTURE(objects);
const auto text = nested_text(depth, objects);
const auto a = json::parse(text);
const auto b = json::parse(text);
CHECK(std::hash<json> {}(a) == std::hash<json> {}(b));
const auto c = ordered_json::parse(text);
const auto d = ordered_json::parse(text);
CHECK(std::hash<ordered_json> {}(c) == std::hash<ordered_json> {}(d));
}
}
}
+42
View File
@@ -0,0 +1,42 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
// This file contains the C++17-only part of unit-items.cpp (structured
// bindings support for json::items()). It is kept in a separate
// translation unit so the (much larger) unit-items.cpp does not need to
// be compiled a second time just for this one SECTION.
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
using nlohmann::json;
#ifdef JSON_HAS_CPP_17
#include <map>
#include <string>
TEST_CASE("items()")
{
SECTION("object")
{
SECTION("structured bindings")
{
json j = { {"A", 1}, {"B", 2} };
std::map<std::string, int> m;
for (auto const&[key, value] : j.items())
{
m.emplace(key, value);
}
CHECK(j.get<decltype(m)>() == m);
}
}
}
#endif
-16
View File
@@ -862,22 +862,6 @@ TEST_CASE("items()")
CHECK(counter == 3);
}
#ifdef JSON_HAS_CPP_17
SECTION("structured bindings")
{
json j = { {"A", 1}, {"B", 2} };
std::map<std::string, int> m;
for (auto const&[key, value] : j.items())
{
m.emplace(key, value);
}
CHECK(j.get<decltype(m)>() == m);
}
#endif
}
SECTION("const object")
+363
View File
@@ -672,6 +672,102 @@ TEST_CASE("JSON patch")
}
}
SECTION("patch_inplace")
{
SECTION("happy path: patch_inplace mirrors patch() on success")
{
// mirrors "A.5. Replacing a Value" above, but applies the patch with
// patch_inplace() to a mutable copy instead of using patch()'s
// returned copy
json doc = R"(
{
"baz": "qux",
"foo": "bar"
}
)"_json;
json const patch = R"(
[
{ "op": "replace", "path": "/baz", "value": "boo" }
]
)"_json;
json const expected = R"(
{
"baz": "boo",
"foo": "bar"
}
)"_json;
doc.patch_inplace(patch);
CHECK(doc == expected);
}
// this test relies on the "test" operation actually throwing so the
// partial-application state can be observed right after the throw
// point; under JSON_NOEXCEPTION, JSON_THROW() calls std::abort()
// instead (there is no C++ exception to throw), and doctest's
// CHECK_THROWS_AS() is compiled out to a no-op that never even
// invokes the given expression (see doctest's "--no-throw" test
// filter, which ci_test_noexceptions passes) -- so patch()/
// patch_inplace() would never be called at all and the follow-up
// state assertions below would fail against the untouched original
#if !defined(JSON_NOEXCEPTION)
SECTION("distinguishing contract vs patch(): partial application on failure")
{
// Unlike patch(), which is all-or-nothing because it applies the
// patch to an internal copy that is simply discarded when an
// exception is thrown (leaving the original untouched no matter
// what), patch_inplace() mutates the document it is called on
// directly and immediately, operation by operation. So if a JSON
// Patch fails partway through, whatever operations already
// succeeded remain applied -- the document is left in a partially
// patched state. This is empirically verified current behavior,
// not just documented intent, and is pinned here as such.
json const original = R"(
{
"baz": "qux",
"foo": "bar"
}
)"_json;
// the first operation ("replace") succeeds; the second ("test")
// fails because the value at "/baz" no longer (and never did)
// equal "not boo"
json const patch = R"(
[
{ "op": "replace", "path": "/baz", "value": "boo" },
{ "op": "test", "path": "/baz", "value": "not boo" }
]
)"_json;
// patch() never modifies the object it is called on -- it always
// operates on (and returns) a separate copy, so the original is
// left completely untouched, regardless of success or failure.
// copy_for_patch is intentionally a real copy, not a reference
// to `original`: the whole point of this check is to catch a
// hypothetical future regression where patch() *does* mutate its
// receiver. Using a reference here would make the assertion
// below compare `original` to itself -- trivially true even if
// such a bug existed -- which is exactly what a static analyzer
// can't see when it suggests "this copy is never modified, use
// a reference instead".
json copy_for_patch = original; // NOLINT(performance-unnecessary-copy-initialization)
CHECK_THROWS_AS(copy_for_patch.patch(patch), json::other_error&);
CHECK(copy_for_patch == original);
// patch_inplace(), in contrast, already applied the successful
// "replace" operation to the document before the "test" operation
// threw -- that change is not rolled back
json doc = original;
CHECK_THROWS_AS(doc.patch_inplace(patch), json::other_error&);
CHECK(doc != original);
CHECK(doc.at("baz") == "boo");
CHECK(doc.at("foo") == "bar");
}
#endif // !defined(JSON_NOEXCEPTION)
}
SECTION("errors")
{
SECTION("unknown operation")
@@ -1388,3 +1484,270 @@ TEST_CASE("JSON patch - add to a primitive parent (regression #4292)")
CHECK_THROWS_AS(doc.patch(patch), json::out_of_range&);
}
}
TEST_CASE("JSON patch - remove with primitive or null parent (regression #5396)")
{
// Regression test for https://github.com/nlohmann/json/issues/5396
//
// RFC 6902 (§4.2) requires the target location of a "remove" operation
// to exist. When the target's parent resolves to a primitive value or
// null, the operation must fail. Previously operation_remove silently
// did nothing in this case (neither the "is_object" nor the "is_array"
// branch matched, and there was no final "else"), so the patch appeared
// to succeed without changing the document. It now throws
// out_of_range.413.
SECTION("parent is a primitive (number)")
{
json const doc = {{"a", 1}};
json const patch = {{{"op", "remove"}, {"path", "/a/b"}}};
#if JSON_DIAGNOSTICS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] (/a) cannot remove value: the JSON Patch 'remove' target's parent is of type number, but must be an object or array", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] cannot remove value: the JSON Patch 'remove' target's parent is of type number, but must be an object or array", json::out_of_range&);
#endif
}
SECTION("parent is a primitive (string)")
{
json const doc = {{"foo", {{"bar", "a string"}}}};
json const patch = {{{"op", "remove"}, {"path", "/foo/bar/baz"}}};
#if JSON_DIAGNOSTICS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] (/foo/bar) cannot remove value: the JSON Patch 'remove' target's parent is of type string, but must be an object or array", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] cannot remove value: the JSON Patch 'remove' target's parent is of type string, but must be an object or array", json::out_of_range&);
#endif
}
SECTION("top-level document is null")
{
json const doc = nullptr;
json const patch = {{{"op", "remove"}, {"path", "/a"}}};
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] cannot remove value: the JSON Patch 'remove' target's parent is of type null, but must be an object or array", json::out_of_range&);
}
SECTION("legitimate removes still work")
{
// object member
json const doc1 = {{"a", 1}, {"b", 2}};
json const patch1 = {{{"op", "remove"}, {"path", "/a"}}};
CHECK(doc1.patch(patch1) == json({{"b", 2}}));
// array element
json const doc2 = R"([1, 2, 3])"_json;
json const patch2 = {{{"op", "remove"}, {"path", "/1"}}};
CHECK(doc2.patch(patch2) == R"([1, 3])"_json);
}
}
TEST_CASE("JSON patch - move where 'from' is a proper prefix of 'path' (regression #5397)")
{
// Regression test for https://github.com/nlohmann/json/issues/5397
//
// RFC 6902 (§4.4) forbids "from" from being a proper prefix of "path"
// for a "move" operation: "a location cannot be moved into one of its
// children." "move" is implemented as remove-then-add; for an object
// target this happened to throw anyway as a side effect of the "add"
// step re-resolving through the now-removed parent, but for an array
// target the removal shifted subsequent indices, so "path" silently
// re-resolved to a different element and the operation "succeeded"
// with a corrupted result. It now throws out_of_range.414 for both
// object and array targets.
SECTION("array target (from the issue)")
{
json const doc = R"([[1,2],[3]])"_json;
json const patch = {{{"op", "move"}, {"from", "/0"}, {"path", "/0/0"}}};
#if JSON_DIAGNOSTIC_POSITIONS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-11) cannot move value: 'from' path '/0' is a proper prefix of 'path' '/0/0'", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '/0' is a proper prefix of 'path' '/0/0'", json::out_of_range&);
#endif
}
SECTION("object target")
{
json const doc = R"({"a": {"b": 1}})"_json;
json const patch = {{{"op", "move"}, {"from", "/a"}, {"path", "/a/b"}}};
#if JSON_DIAGNOSTIC_POSITIONS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-15) cannot move value: 'from' path '/a' is a proper prefix of 'path' '/a/b'", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '/a' is a proper prefix of 'path' '/a/b'", json::out_of_range&);
#endif
}
SECTION("from == path is not a proper prefix and must not be rejected")
{
// "from" equal to "path" is a no-op move; it is not a *proper*
// prefix relationship, so this new check must not reject it.
json const doc = R"({"a": 1, "b": 2})"_json;
json const patch = {{{"op", "move"}, {"from", "/a"}, {"path", "/a"}}};
CHECK(doc.patch(patch) == doc);
}
SECTION("raw string prefix that is not a pointer-token prefix must be allowed")
{
// "/ab" is a string-prefix of "/abc/x" as raw text, but "ab" and
// "abc" are different reference tokens, so this is NOT a
// pointer-token prefix relationship and the move must succeed.
// This is the key case proving the check compares tokens, not
// raw pointer text (a naive std::string prefix/rfind check on
// the undecoded pointer would wrongly reject this).
json const doc = R"({"ab": 1, "abc": {"x": 2}})"_json;
json const patch = {{{"op", "move"}, {"from", "/ab"}, {"path", "/abc/x"}}};
json const result = R"({"abc": {"x": 1}})"_json;
CHECK(doc.patch(patch) == result);
}
SECTION("escaped reference tokens are compared unescaped")
{
// "from" is the single token "a/b" (escaped as "a~1b"); "path"
// addresses member "x" of that same value, so "from" is a
// proper (token-level) prefix of "path" and must be rejected.
json const doc = R"({"a/b": {"x": 1}})"_json;
json const patch = {{{"op", "move"}, {"from", "/a~1b"}, {"path", "/a~1b/x"}}};
#if JSON_DIAGNOSTIC_POSITIONS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-17) cannot move value: 'from' path '/a~1b' is a proper prefix of 'path' '/a~1b/x'", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '/a~1b' is a proper prefix of 'path' '/a~1b/x'", json::out_of_range&);
#endif
}
SECTION("ordinary valid moves still work")
{
// unrelated top-level members
json const doc1 = R"({"a": 1, "b": 2})"_json;
json const patch1 = {{{"op", "move"}, {"from", "/a"}, {"path", "/c"}}};
CHECK(doc1.patch(patch1) == R"({"b": 2, "c": 1})"_json);
// sibling paths that share a textual prefix but are unrelated
json const doc2 = R"({"a": {"x": 1}, "b": {"y": 2}})"_json;
json const patch2 = {{{"op", "move"}, {"from", "/a/x"}, {"path", "/b/z"}}};
CHECK(doc2.patch(patch2) == R"({"a": {}, "b": {"y": 2, "z": 1}})"_json);
// "path" is a proper prefix of "from" (the reverse relationship,
// which RFC 6902 does not forbid)
json const doc3 = R"({"a": {"b": 1}})"_json;
json const patch3 = {{{"op", "move"}, {"from", "/a/b"}, {"path", "/a"}}};
CHECK(doc3.patch(patch3) == R"({"a": 1})"_json);
}
SECTION("root 'from' is a proper prefix of every non-root 'path'")
{
// the whole document is a proper prefix of any location inside it
json const doc = R"({"a": 1})"_json;
json const patch = {{{"op", "move"}, {"from", ""}, {"path", "/a"}}};
#if JSON_DIAGNOSTIC_POSITIONS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-8) cannot move value: 'from' path '' is a proper prefix of 'path' '/a'", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '' is a proper prefix of 'path' '/a'", json::out_of_range&);
#endif
}
SECTION("root 'path' is never a proper prefix violation for a non-root 'from'")
{
// the reverse of the above: moving a non-root location to the root
// is the "path is a prefix of from" relationship, which RFC 6902
// permits (already covered generally above; this pins the root
// case specifically, since root is the one path with no reference
// tokens at all)
json const doc = R"({"a": {"b": 1}})"_json;
json const patch = {{{"op", "move"}, {"from", "/a"}, {"path", ""}}};
CHECK(doc.patch(patch) == R"({"b": 1})"_json);
}
SECTION("the array-append token '-' is an ordinary child token")
{
// "-" (append-to-array) addresses a location *inside* the array,
// so "from" pointing at the array is still a proper prefix of
// "path" ending in "-" and must be rejected like any other child.
json const doc = R"({"a": [1, 2]})"_json;
json const patch = {{{"op", "move"}, {"from", "/a"}, {"path", "/a/-"}}};
#if JSON_DIAGNOSTIC_POSITIONS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-13) cannot move value: 'from' path '/a' is a proper prefix of 'path' '/a/-'", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '/a' is a proper prefix of 'path' '/a/-'", json::out_of_range&);
#endif
}
}
TEST_CASE("JSON patch - diff emits array removals in descending index order")
{
SECTION("array shrunk to empty")
{
json const source = {0, 1, 2, 3, 4};
json const target = json::array();
json const patch = json::diff(source, target);
json const expected = R"(
[
{"op": "remove", "path": "/4"},
{"op": "remove", "path": "/3"},
{"op": "remove", "path": "/2"},
{"op": "remove", "path": "/1"},
{"op": "remove", "path": "/0"}
]
)"_json;
CHECK(patch == expected);
CHECK(source.patch(patch) == target);
}
SECTION("array partially shrunk, after a replacement at a common index")
{
json const source = {0, 1, 2, 3, 4};
json const target = {0, 9};
json const patch = json::diff(source, target);
// the replacement comes first, then the removals, highest index first
json const expected = R"(
[
{"op": "replace", "path": "/1", "value": 9},
{"op": "remove", "path": "/4"},
{"op": "remove", "path": "/3"},
{"op": "remove", "path": "/2"}
]
)"_json;
CHECK(patch == expected);
CHECK(source.patch(patch) == target);
}
SECTION("nested array shrunk")
{
json const source = {{"a", {0, 1, 2}}};
json const target = {{"a", json::array()}};
json const patch = json::diff(source, target);
json const expected = R"(
[
{"op": "remove", "path": "/a/2"},
{"op": "remove", "path": "/a/1"},
{"op": "remove", "path": "/a/0"}
]
)"_json;
CHECK(patch == expected);
CHECK(source.patch(patch) == target);
}
SECTION("many removals still round-trip")
{
json source = json::array();
for (int i = 0; i < 1000; ++i)
{
source.push_back(i);
}
json const target = json::array();
json const patch = json::diff(source, target);
CHECK(patch.size() == 1000);
CHECK(patch.front().at("path") == "/999");
CHECK(patch.back().at("path") == "/0");
CHECK(source.patch(patch) == target);
}
}
+52
View File
@@ -319,6 +319,44 @@ TEST_CASE("JSON pointers")
CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&);
// #5395: contains() must not throw for a reference token that is a
// syntactically valid array index but numerically exceeds ULLONG_MAX
// (causing strtoull() to set errno to ERANGE) -- it should just report
// that the pointer does not resolve to an element
CHECK(!j.contains(jp));
CHECK(!j_const.contains(jp));
}
{
// #5395: same as above, but using the exact reproduction from the issue
json::json_pointer const jp("/99999999999999999999");
std::string const throw_msg = "[json.exception.out_of_range.404] unresolved reference token '99999999999999999999'";
CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j.at(jp) = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const.at(jp) == 1, throw_msg.c_str(), json::out_of_range&);
CHECK(!j.contains(jp));
CHECK(!j_const.contains(jp));
}
{
// #5395: a reference token that is numerically representable in
// unsigned long long but exceeds size_type's max (e.g. ULLONG_MAX
// itself on typical 64-bit platforms, where size_type's max equals
// ULLONG_MAX) must not make contains() throw either
json::json_pointer const jp("/18446744073709551615");
std::string const throw_msg = "[json.exception.out_of_range.410] array index 18446744073709551615 exceeds size_type";
CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j.at(jp) = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const.at(jp) == 1, throw_msg.c_str(), json::out_of_range&);
CHECK(!j.contains(jp));
CHECK(!j_const.contains(jp));
}
// on some machines, the check below is not constant
@@ -334,6 +372,10 @@ TEST_CASE("JSON pointers")
CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&);
// #5395: contains() must not throw for a reference token exceeding size_type's max
CHECK(!j.contains(jp));
CHECK(!j_const.contains(jp));
}
DOCTEST_MSVC_SUPPRESS_WARNING_POP
@@ -465,6 +507,16 @@ TEST_CASE("JSON pointers")
// explicit roundtrip check
CHECK(j.flatten().unflatten() == j);
// an object is only unflattened to an array if one of its keys is the
// reference token 0; this must not depend on which key is seen first
CHECK(json({{"/2", "x"}}).unflatten() == json({{"2", "x"}}));
CHECK(json({{"/10", "y"}, {"/2", "z"}}).unflatten() == json({{"10", "y"}, {"2", "z"}}));
CHECK(json({{"/0", 1}, {"/1", 2}}).unflatten() == json({1, 2}));
CHECK(json({{"/1", 2}, {"/0", 1}}).unflatten() == json({1, 2}));
CHECK(json({{"/0", 1}, {"/2", 3}}).unflatten() == json({1, nullptr, 3}));
CHECK(json({{"/a/1", 2}, {"/a/0", 1}}).unflatten() == json({{"a", {1, 2}}}));
CHECK(json({{"/a/1", 2}, {"/a/x", 1}}).unflatten() == json({{"a", {{"1", 2}, {"x", 1}}}}));
// roundtrip for primitive values
json j_null;
CHECK(j_null.flatten().unflatten() == j_null);
+151
View File
@@ -12,6 +12,7 @@
using nlohmann::json;
#include <algorithm>
#include <string>
TEST_CASE("tests on very large JSONs")
{
@@ -27,3 +28,153 @@ TEST_CASE("tests on very large JSONs")
}
}
namespace
{
// Descend a chain of single-element containers and return the value at its end,
// reporting the number of levels traversed in @a depth.
//
// The values in the test case below are nested far deeper than the call stack
// can follow, so they must not be inspected with operator== or dump(): both are
// still recursive and would overflow the stack themselves.
const json* innermost_value(const json& j, std::size_t& depth)
{
const json* current = &j;
depth = 0;
while ((current->is_array() || current->is_object()) && !current->empty())
{
current = current->is_array()
? &current->front()
: &current->begin().value();
++depth;
}
return current;
}
} // namespace
TEST_CASE("tests on deeply nested JSONs")
{
// deep enough to exhaust the call stack, but small enough to stay cheap:
// parsing is iterative, so building the values below costs little
const std::size_t depth = 100000;
SECTION("issue #5387 - stack overflow in the copy constructor")
{
SECTION("array")
{
const json j = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']'));
const json copy(j); // NOLINT(performance-unnecessary-copy-initialization): the copy is what is tested
std::size_t copy_depth = 0;
CHECK(*innermost_value(copy, copy_depth) == 0);
CHECK(copy_depth == depth);
}
SECTION("object")
{
std::string s;
s.reserve((6 * depth) + 1);
for (std::size_t i = 0; i < depth; ++i)
{
s += "{\"a\":";
}
s += '1';
s.append(depth, '}');
const json j = json::parse(s);
const json copy(j); // NOLINT(performance-unnecessary-copy-initialization): the copy is what is tested
std::size_t copy_depth = 0;
CHECK(*innermost_value(copy, copy_depth) == 1);
CHECK(copy_depth == depth);
}
SECTION("copy assignment")
{
// operator=(basic_json) takes its argument by value, so the deep
// copy happens in the copy constructor
const json j = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']'));
json target;
target = j;
std::size_t target_depth = 0;
CHECK(*innermost_value(target, target_depth) == 0);
CHECK(target_depth == depth);
}
SECTION("depths around the bound of the recursive descent")
{
// The copy constructor descends into a bounded number of levels and
// completes whatever is below that without the call stack. Cover
// every depth around that bound, so that the two ways of copying
// are known to meet cleanly - wherever the bound is set.
for (std::size_t d = 1; d <= 300; ++d)
{
CAPTURE(d);
const json array = json::parse(std::string(d, '[') + '0' + std::string(d, ']'));
const json array_copy(array); // NOLINT(performance-unnecessary-copy-initialization): the copy is what is tested
std::size_t array_depth = 0;
CHECK(*innermost_value(array_copy, array_depth) == 0);
CHECK(array_depth == d);
std::string object_text;
for (std::size_t i = 0; i < d; ++i)
{
object_text += "{\"a\":";
}
object_text += '1';
object_text.append(d, '}');
const json object = json::parse(object_text);
const json object_copy(object); // NOLINT(performance-unnecessary-copy-initialization): the copy is what is tested
std::size_t object_depth = 0;
CHECK(*innermost_value(object_copy, object_depth) == 1);
CHECK(object_depth == d);
}
}
SECTION("a value that is deep in one place only")
{
json j = json::object();
j["shallow"] = 1;
j["deep"] = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']'));
j["also_shallow"] = json::array({1, 2, 3});
const json copy(j);
CHECK(copy["shallow"] == 1);
CHECK(copy["also_shallow"] == json::array({1, 2, 3}));
std::size_t deep_depth = 0;
CHECK(*innermost_value(copy["deep"], deep_depth) == 0);
CHECK(deep_depth == depth);
}
SECTION("the copy is independent of the original")
{
const json j = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']'));
json copy(j);
// reach the innermost value without recursing and replace it
json* current = &copy;
while (current->is_array() && !current->empty())
{
current = &current->front();
}
*current = 42;
std::size_t unused = 0;
CHECK(*innermost_value(copy, unused) == 42);
CHECK(*innermost_value(j, unused) == 0);
}
}
}
+103
View File
@@ -14,6 +14,60 @@ using nlohmann::json;
using namespace nlohmann::literals; // NOLINT(google-build-using-namespace)
#endif
#include <string>
namespace
{
// RFC 7396's MergePatch, written recursively as in the RFC; only usable on
// values nested a few hundred levels deep
void reference_merge_patch(json& target, const json& patch)
{
if (!patch.is_object())
{
target = patch;
return;
}
if (!target.is_object())
{
target = json::object();
}
for (auto it = patch.begin(); it != patch.end(); ++it)
{
if (it.value().is_null())
{
target.erase(it.key());
}
else
{
reference_merge_patch(target[it.key()], it.value());
}
}
}
// objects nested `depth` levels deep under the key "a", with members that
// differ by `variant` on the way down
std::string nested_objects(const std::size_t depth, const int variant)
{
std::string text;
for (std::size_t i = 0; i < depth; ++i)
{
text += "{";
if ((i + static_cast<std::size_t>(variant)) % 3 == 0)
{
text += "\"s" + std::to_string(variant) + "\":" + std::to_string(i) + ",";
}
if (variant == 2 && i % 5 == 0)
{
text += "\"s0\":null,";
}
text += "\"a\":";
}
text += variant == 1 ? R"({"x":1,"y":null})" : "{\"y\":2}";
text.append(depth, '}');
return text;
}
} // namespace
TEST_CASE("JSON Merge Patch")
{
SECTION("examples from RFC 7396")
@@ -242,3 +296,52 @@ TEST_CASE("JSON Merge Patch")
}
}
}
TEST_CASE("JSON Merge Patch on deeply nested values")
{
SECTION("patching past the descent bound gives the same result")
{
// every depth on either side of where the iterative version takes
// over (detail::recursion_depth_limit(), 128)
for (std::size_t depth = 0; depth <= 300; ++depth)
{
CAPTURE(depth);
for (int variant = 0; variant < 3; ++variant)
{
CAPTURE(variant);
const json patch = json::parse(nested_objects(depth, variant));
json result = json::parse(nested_objects(depth, (variant + 1) % 3));
json expected = result;
result.merge_patch(patch);
reference_merge_patch(expected, patch);
CHECK(result == expected);
// a target that is not an object, and an empty one
json from_null;
from_null.merge_patch(patch);
json expected_from_null;
reference_merge_patch(expected_from_null, patch);
CHECK(from_null == expected_from_null);
}
}
}
SECTION("patches nested too deeply for the call stack (#5393)")
{
// applying a patch used to recurse once per nesting level. The result
// is only walked, never copied or compared, since those recurse too.
const std::size_t depth = 100000;
json target = json::parse(nested_objects(depth, 0));
target.merge_patch(json::parse(nested_objects(depth, 1)));
const json* p = &target;
for (std::size_t i = 0; i < depth; ++i)
{
p = &p->at("a");
}
// {"y":2} patched with {"x":1,"y":null}
CHECK(p->size() == 1);
CHECK(p->at("x") == 1);
}
}
+126
View File
@@ -11,6 +11,53 @@
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <string>
namespace
{
// update(source, true) as documented, written recursively; only usable on
// values nested a few hundred levels deep
void reference_update(json& target, const json& source)
{
for (auto it = source.begin(); it != source.end(); ++it)
{
const auto existing = target.find(it.key());
if (it.value().is_object() && existing != target.end() && existing->is_object())
{
reference_update(*existing, it.value());
}
else
{
target[it.key()] = it.value();
}
}
}
// objects nested `depth` levels deep under the key "a", with members that
// differ by `variant` on the way down
std::string nested_objects(const std::size_t depth, const int variant)
{
std::string text;
for (std::size_t i = 0; i < depth; ++i)
{
text += "{";
if ((i + static_cast<std::size_t>(variant)) % 3 == 0)
{
text += "\"s" + std::to_string(variant) + "\":" + std::to_string(i) + ",";
}
if (variant == 2 && i % 5 == 0)
{
// an object replacing a primitive, which is not merged
text += R"("s0":{"o":1},)";
}
text += "\"a\":";
}
text += variant == 1 ? "{\"x\":1}" : "{\"y\":2}";
text.append(depth, '}');
return text;
}
} // namespace
TEST_CASE("modifiers")
{
SECTION("clear()")
@@ -641,6 +688,20 @@ TEST_CASE("modifiers")
CHECK_THROWS_WITH_AS(j_array.insert(j_array.end(), j_other_array.begin(), j_other_array2.end()), "[json.exception.invalid_iterator.210] iterators do not fit",
json::invalid_iterator&);
}
SECTION("iterators not pointing into an array")
{
json j_object2 = {{"k", 1}, {"l", 2}};
json j_primitive = 5;
json j_null;
CHECK_THROWS_WITH_AS(j_array.insert(j_array.begin(), j_object2.begin(), j_object2.end()), "[json.exception.invalid_iterator.202] iterators first and last must point to arrays",
json::invalid_iterator&);
CHECK_THROWS_WITH_AS(j_array.insert(j_array.begin(), j_primitive.begin(), j_primitive.end()), "[json.exception.invalid_iterator.202] iterators first and last must point to arrays",
json::invalid_iterator&);
CHECK_THROWS_WITH_AS(j_array.insert(j_array.begin(), j_null.begin(), j_null.end()), "[json.exception.invalid_iterator.202] iterators first and last must point to arrays",
json::invalid_iterator&);
}
}
SECTION("range for object")
@@ -801,6 +862,30 @@ TEST_CASE("modifiers")
j1.update(j2, true);
CHECK(j1 == json({{"string", "t"}, {"numbers", 1}}));
}
SECTION("overwrite primitive with object")
{
json j1 = {{"k", 1}};
json const j2 = {{"k", {{"x", 2}}}};
j1.update(j2, true);
CHECK(j1 == json({{"k", {{"x", 2}}}}));
}
SECTION("overwrite array with object")
{
json j1 = {{"k", {1, 2}}};
json const j2 = {{"k", {{"x", 2}}}};
j1.update(j2, true);
CHECK(j1 == json({{"k", {{"x", 2}}}}));
}
SECTION("overwrite nested primitive with object")
{
json j1 = {{"k", {{"inner", 1}}}};
json const j2 = {{"k", {{"inner", {{"x", 2}}}}}};
j1.update(j2, true);
CHECK(j1 == json({{"k", {{"inner", {{"x", 2}}}}}}));
}
}
}
}
@@ -950,3 +1035,44 @@ TEST_CASE("modifiers")
}
}
}
TEST_CASE("update() on deeply nested values")
{
SECTION("merging past the descent bound gives the same result")
{
// every depth on either side of where the iterative version takes
// over (detail::recursion_depth_limit(), 128)
for (std::size_t depth = 0; depth <= 300; ++depth)
{
CAPTURE(depth);
for (int variant = 0; variant < 3; ++variant)
{
CAPTURE(variant);
const json source = json::parse(nested_objects(depth, variant));
json result = json::parse(nested_objects(depth, (variant + 1) % 3));
json expected = result;
result.update(source, true);
reference_update(expected, source);
CHECK(result == expected);
}
}
}
SECTION("objects nested too deeply for the call stack (#5545)")
{
// merging used to recurse once per nesting level. The result is only
// walked, never copied or compared, since those recurse too.
const std::size_t depth = 100000;
json target = json::parse(nested_objects(depth, 0));
target.update(json::parse(nested_objects(depth, 1)), true);
const json* p = &target;
for (std::size_t i = 0; i < depth; ++i)
{
p = &p->at("a");
}
CHECK(p->size() == 2);
CHECK(p->at("x") == 1);
CHECK(p->at("y") == 2);
}
}
+161
View File
@@ -1597,7 +1597,168 @@ TEST_CASE("MessagePack")
}
}
TEST_CASE("issue #5405 - array reserve for definite-length MessagePack arrays")
{
#if !defined(JSON_NOEXCEPTION)
// this SECTION relies on catching a thrown exception to distinguish
// which of two acceptable, bounded rejections a hostile header took;
// under JSON_NOEXCEPTION, JSON_THROW never produces a catchable C++
// exception (it aborts instead), so this cannot be tested that way here
SECTION("a huge claimed length with no element data must not over-allocate")
{
// 0xdd: array 32 (four-byte length); claims 0xFFFFFFFF (4294967295)
// elements but provides none. max_size() for a std::vector is far
// larger than this count, so it does not reject the header outright;
// the (capped) reservation must not attempt to allocate space for
// billions of elements before the missing data is detected.
json _;
const std::vector<uint8_t> input = {0xdd, 0xFF, 0xFF, 0xFF, 0xFF};
// On a platform where std::size_t is narrower than 64 bits (e.g.
// 32-bit), the claimed count 0xFFFFFFFF coincides with that
// platform's SIZE_MAX, which some size-narrowing checks treat the
// same as detail::unknown_size(); it may then be rejected before
// the SAX consumer's own max_size() check (out_of_range.408) rather
// than being accepted and only found short of data once the
// (capped) reservation looks for element bytes that were never
// provided (parse_error.110). Either is an acceptable, bounded
// rejection of the hostile header -- the property under test is
// that no path attempts to allocate space for billions of elements.
bool threw = false;
try
{
_ = json::from_msgpack(input);
}
catch (const json::parse_error& e)
{
threw = true;
CHECK(e.id == 110);
CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 6: syntax error while parsing MessagePack value: unexpected end of input");
}
catch (const json::out_of_range& e)
{
threw = true;
CHECK(e.id == 408);
CHECK(std::string(e.what()).find("excessive") != std::string::npos);
}
CHECK(threw);
CHECK(json::from_msgpack(input, true, false).is_discarded());
}
#endif
SECTION("arrays of various sizes decode to the same value as before the reserve optimization")
{
for (const auto size :
{
std::size_t{0}, std::size_t{1}, std::size_t{5}, // small
std::size_t{16384}, // exactly at the reserve cap
std::size_t{20000} // above the reserve cap
})
{
CAPTURE(size)
json j = json::array();
for (std::size_t i = 0; i < size; ++i)
{
j.push_back(static_cast<int>(i % 1000));
}
const auto packed = json::to_msgpack(j);
CHECK(json::from_msgpack(packed) == j);
}
}
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
{
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
// a custom SAX consumer that does not touch a DOM array sees identical events
json j = json::array();
for (int i = 0; i < 100; ++i)
{
j.push_back(i);
}
const auto packed = json::to_msgpack(j);
SaxCountdown scp(1000000); // large enough to never trigger an abort
CHECK(json::sax_parse(packed, &scp, json::input_format_t::msgpack));
}
}
TEST_CASE("regression test - MessagePack ext type rejects a subtype that doesn't fit a single byte")
{
// subtype 0-255 must still round-trip correctly (regression guard, pre-existing behavior)
CHECK(json::from_msgpack(json::to_msgpack(json::binary({1, 2}, 0))).get_binary().subtype() == 0);
CHECK(json::from_msgpack(json::to_msgpack(json::binary({1, 2}, 200))).get_binary().subtype() == 200);
CHECK(json::from_msgpack(json::to_msgpack(json::binary({1, 2}, 255))).get_binary().subtype() == 255);
// a subtype > 255 must throw instead of silently truncating
CHECK_THROWS_AS(json::to_msgpack(json::binary({1, 2}, 256)), json::out_of_range);
CHECK_THROWS_WITH_AS(json::to_msgpack(json::binary({1, 2}, 70000)), "[json.exception.out_of_range.415] subtype 70000 is too large for the MessagePack ext type (max 255)", json::out_of_range);
// a binary value with no subtype at all must be unaffected
CHECK(json::from_msgpack(json::to_msgpack(json::binary({1, 2}))).get_binary().has_subtype() == false);
}
// use this testcase outside [hide] to run it with Valgrind
TEST_CASE("MessagePack nesting does not consume the call stack")
{
// Reading a container used to call back into the value reader once per
// element, so the native call stack grew with the nesting depth of the
// input: one frame per byte for repeated 0x91 (a one-element array), which
// crashes the process long before the input is exhausted (#5104). The
// containers are kept on a heap stack now.
//
// Note that deeply nested values must not be compared, copied or dumped
// here: those operations are still recursive, and would reintroduce the
// very crash this checks for. Depth is measured by descending instead.
SECTION("an unterminated chain is reported, not crashed on")
{
json _;
const std::vector<uint8_t> input(300000, 0x91);
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(input), "[json.exception.parse_error.110] parse error at byte 300001: syntax error while parsing MessagePack value: unexpected end of input", json::parse_error&);
CHECK(json::from_msgpack(input, true, false).is_discarded());
}
SECTION("a well-formed deep value is read through the SAX interface")
{
std::vector<uint8_t> input(300000, 0x91);
input.push_back(0x01); // innermost value
SaxCountdown accept_all(600001);
CHECK(json::sax_parse(input, &accept_all, json::input_format_t::msgpack));
}
SECTION("a well-formed deep value is read into a value")
{
const std::size_t depth = 10000;
std::vector<uint8_t> input(depth, 0x91);
input.push_back(0x01);
json j = json::from_msgpack(input);
std::size_t measured = 0;
const json* p = &j;
while (p->is_array() && !p->empty())
{
p = &p->front();
++measured;
}
CHECK(measured == depth);
CHECK(p->is_number());
}
SECTION("containers are still read the same way")
{
CHECK(json::from_msgpack(std::vector<uint8_t>({0x90})) == json::array());
CHECK(json::from_msgpack(std::vector<uint8_t>({0x80})) == json::object());
CHECK(json::from_msgpack(std::vector<uint8_t>({0x92, 0x90, 0x80})) == json({json::array(), json::object()}));
CHECK(json::from_msgpack(std::vector<uint8_t>({0x91, 0x91, 0x91, 0x90})) == json({{{json::array()}}}));
CHECK(json::from_msgpack(std::vector<uint8_t>({0x81, 0xA1, 'a', 0x81, 0xA1, 'b', 0x92, 0x01, 0x02})) == json({{"a", {{"b", {1, 2}}}}}));
// array 16 and map 32, i.e. the counted forms
CHECK(json::from_msgpack(std::vector<uint8_t>({0xDC, 0x00, 0x02, 0x01, 0x02})) == json({1, 2}));
CHECK(json::from_msgpack(std::vector<uint8_t>({0xDF, 0x00, 0x00, 0x00, 0x01, 0xA1, 'k', 0xC3})) == json({{"k", true}}));
}
}
TEST_CASE("single MessagePack roundtrip")
{
SECTION("sample.json")
+34
View File
@@ -0,0 +1,34 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
// This file makes sure that none of the internal JSON_HEDLEY_* macros (vendored
// from https://nemequ.github.io/hedley/, see
// include/nlohmann/thirdparty/hedley/hedley.hpp) leak into the including
// translation unit. include/nlohmann/detail/macro_unscope.hpp is supposed to
// #undef every JSON_HEDLEY_* macro (via hedley_undef.hpp) once json.hpp has
// been fully processed. See https://github.com/nlohmann/json/issues/5408,
// where JSON_HEDLEY_PRAGMA, JSON_HEDLEY_PREDICT_TRUE, JSON_HEDLEY_PREDICT_FALSE,
// and JSON_HEDLEY_CLANG_HAS_DECLSPEC_ATTRIBUTE escaped this cleanup because
// hedley_undef.hpp had no matching #undef for them.
//
// hedley_undef_checks.inc (included below) is generated at CMake configure/
// build time by cmake/scripts/gen_hedley_undef_check.cmake, which derives the
// full list of JSON_HEDLEY_* macro names directly from hedley.hpp. That way
// this test covers every macro Hedley actually defines -- not a hardcoded
// snapshot that would silently go stale the next time `make update_hedley`
// runs -- and can never drift from the vendored header.
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
TEST_CASE("JSON_HEDLEY macros do not leak after including json.hpp")
{
#include "hedley_undef_checks.inc"
CHECK(true); // keep an assertion when nothing leaked
}
@@ -70,7 +70,7 @@ TEST_CASE("check_for_mem_leak_on_adl_to_json-2")
}
}
TEST_CASE("check_for_mem_leak_on_adl_to_json-2")
TEST_CASE("check_for_mem_leak_on_adl_to_json-3")
{
try
{
@@ -0,0 +1,91 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
// This translation unit is a dedicated, small compile-and-run check for two
// configuration macros that (per #5423) were never exercised anywhere in the
// test matrix:
// - JSON_NO_IO, which removes the library's <istream>/<ostream> support
// (operator<<, operator>>, and the stream-based overloads of dump()/parse())
// - the JSON_THROW_USER / JSON_TRY_USER / JSON_CATCH_USER trio, which lets a
// user replace the library's internal exception handling
//
// Both macros are about excluding/replacing a facility the library would
// otherwise pull in on its own, and defining one has no bearing on the other,
// so -- to keep the test matrix small -- they are exercised together in a
// single dedicated file instead of two.
//
// JSON_NO_IO requires this file itself to never rely on <iostream>/<sstream>;
// only string-based parsing/dumping is used below.
#define JSON_NO_IO 1
// The user-supplied exception macros below are a *conforming* replacement:
// they simply forward to the real throw/try/catch keywords (via a counter so
// the test can assert each macro was actually invoked, not just defined), so
// every exception-related behavior the library relies on internally --
// including rethrowing std::out_of_range as json::out_of_range in at() --
// keeps working exactly as it would with the library's own default macros.
static int json_throw_user_call_count = 0; // NOLINT(cppcoreguidelines-avoid-non-const-global-variables)
#define JSON_THROW_USER(exception) do { ++json_throw_user_call_count; throw (exception); } while (false) // NOLINT(cppcoreguidelines-macro-usage)
#define JSON_TRY_USER try // NOLINT(cppcoreguidelines-macro-usage)
#define JSON_CATCH_USER(exception) catch (exception) // NOLINT(cppcoreguidelines-macro-usage)
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
using json = nlohmann::json;
TEST_CASE("JSON_NO_IO")
{
// everything that does not touch <istream>/<ostream> must keep working:
// parsing from and dumping to std::string
const json j = json::parse(R"({"a":[1,2,3],"b":true})");
CHECK(j.dump() == R"({"a":[1,2,3],"b":true})");
CHECK(j.at("a").size() == 3);
CHECK(j.at("b").get<bool>() == true);
}
// this test relies on CHECK_THROWS_AS() actually invoking the guarded
// expression so json_throw_user_call_count gets bumped and can be observed
// afterwards; doctest's "--no-throw" test filter (which ci_test_noexceptions
// passes, together with a global -DJSON_NOEXCEPTION added to CMAKE_CXX_FLAGS
// for every translation unit in that build, this file included) compiles
// CHECK_THROWS_AS() out to a no-op that never even invokes the given
// expression -- so json::parse()/at() below would never be called at all and
// the call-count assertions would fail even though our JSON_THROW_USER
// override (which always really throws, regardless of JSON_NOEXCEPTION) would
// have worked fine on its own
#if !defined(JSON_NOEXCEPTION)
TEST_CASE("JSON_THROW_USER, JSON_TRY_USER, JSON_CATCH_USER")
{
json_throw_user_call_count = 0;
// json::parse() is [[nodiscard]] (JSON_HEDLEY_WARN_UNUSED_RESULT); under
// GCC in C++11 mode that expands to __attribute__((warn_unused_result)),
// which -- unlike a [[nodiscard]] attribute proper -- GCC does not
// consider satisfied by doctest's CHECK_THROWS_AS() wrapping the
// expression in a (void) cast, so the discarded return value would still
// be flagged under -Werror=unused-result; assign it to discard it instead,
// matching the established `json _ = json::parse(...)` pattern used
// elsewhere in the test suite (see unit-class_parser.cpp)
json _; // NOLINT(readability-identifier-naming)
// a parse error goes through JSON_THROW directly, i.e., through our
// JSON_THROW_USER override
CHECK_THROWS_AS(_ = json::parse("this is not JSON"), json::parse_error&);
CHECK(json_throw_user_call_count > 0);
// at() on an out-of-range array index internally catches std::out_of_range
// (JSON_TRY_USER/JSON_CATCH_USER) and rethrows it as json::out_of_range
// (JSON_THROW_USER again), so this exercises all three macros together
const int count_before = json_throw_user_call_count;
const json arr = json::array({1, 2, 3});
CHECK_THROWS_AS(arr.at(10), json::out_of_range&);
CHECK(json_throw_user_call_count > count_before);
}
#endif
+115
View File
@@ -81,3 +81,118 @@ TEST_CASE("regression test for issue #3732 - iteration_proxy_value<iter_impl<ord
};
static_cast<void>(fn);
}
TEST_CASE("copying an ordered_json with nested values")
{
// ordered_map is backed by a vector, so copying an object that has
// structured values takes a different route than copying a std::map-backed
// one; see https://github.com/nlohmann/json/issues/5387
ordered_json oj;
oj["z"] = 1;
oj["a"]["y"] = 2;
oj["a"]["b"]["x"] = 3;
oj["m"] = {1, 2, {{"w", 4}}};
const ordered_json copy(oj);
SECTION("the copy is equal to the original")
{
CHECK(copy == oj);
CHECK(copy.dump() == oj.dump());
}
SECTION("the key order is preserved at every level")
{
CHECK(copy.dump() == R"({"z":1,"a":{"y":2,"b":{"x":3}},"m":[1,2,{"w":4}]})");
}
SECTION("the copy is independent of the original")
{
ordered_json mutated(oj);
mutated["a"]["b"]["x"] = 99;
CHECK(oj["a"]["b"]["x"] == 3);
CHECK(mutated["a"]["b"]["x"] == 99);
}
}
TEST_CASE("regression test - diff() must account for ordered_json member order")
{
SECTION("pure reorder, no value changes")
{
ordered_json a = {{"a", 1}, {"b", 2}};
ordered_json b = {{"b", 2}, {"a", 1}};
CHECK(a != b); // order-sensitive equality
CHECK(a.patch(ordered_json::diff(a, b)) == b);
}
SECTION("new key must land at the front")
{
ordered_json c = {{"b", 2}};
ordered_json e = {{"a", 1}, {"b", 2}};
CHECK(c.patch(ordered_json::diff(c, e)) == e);
}
SECTION("reorder plus a value change on one of the reordered keys")
{
ordered_json a = {{"a", 1}, {"b", 2}};
ordered_json b = {{"b", 20}, {"a", 1}};
CHECK(a != b);
CHECK(a.patch(ordered_json::diff(a, b)) == b);
}
SECTION("reorder plus a deleted key")
{
ordered_json a = {{"a", 1}, {"b", 2}, {"c", 3}};
ordered_json b = {{"b", 2}, {"a", 1}};
CHECK(a != b);
CHECK(a.patch(ordered_json::diff(a, b)) == b);
}
SECTION("reorder plus a nested value that itself needs a recursive diff")
{
ordered_json a = {{"a", {{"x", 1}, {"y", 2}}}, {"b", 2}};
ordered_json b = {{"b", 2}, {"a", {{"x", 1}, {"y", 99}}}};
CHECK(a != b);
CHECK(a.patch(ordered_json::diff(a, b)) == b);
}
SECTION("three or more keys shuffled into a different order")
{
ordered_json a = {{"a", 1}, {"b", 2}, {"c", 3}, {"d", 4}};
ordered_json b = {{"d", 4}, {"b", 2}, {"a", 1}, {"c", 3}};
CHECK(a != b);
CHECK(a.patch(ordered_json::diff(a, b)) == b);
}
SECTION("matching order still produces a minimal patch (fast path unaffected)")
{
ordered_json a = {{"a", 1}, {"b", 2}, {"c", 3}};
ordered_json b = {{"a", 1}, {"b", 20}, {"c", 3}};
auto p = ordered_json::diff(a, b);
// only the changed value should be touched, not a wholesale remove+add
CHECK(p.size() == 1);
CHECK(p[0]["op"] == "replace");
CHECK(p[0]["path"] == "/b");
CHECK(a.patch(p) == b);
}
SECTION("plain json (std::map-backed) is unaffected by same-key-different-insertion-order")
{
json a;
a["b"] = 2;
a["a"] = 1;
json b;
b["a"] = 1;
b["b"] = 2;
// std::map iteration is always sorted by key, so a == b regardless of
// insertion order, and diff() must still produce the same minimal
// (empty) result as before this fix
CHECK(a == b);
auto p = json::diff(a, b);
CHECK(p.empty());
CHECK(a.patch(p) == b);
}
}
+489
View File
@@ -0,0 +1,489 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-FileCopyrightText: 2018 Vitaliy Manushkin <agri@akamo.info>
// SPDX-License-Identifier: MIT
// This file closes a test-coverage gap described in GitHub issue #5421:
// nlohmann::ordered_json (and other non-default basic_json specializations,
// such as the alt_string-based one from unit-alt-string.cpp) were never
// exercised through the binary formats (CBOR/MessagePack/UBJSON/BSON/BJData)
// or through flatten()/unflatten()/diff()/patch()/merge_patch().
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
#include <cstdint>
#include <string>
#include <utility>
#include <vector>
using nlohmann::json;
using nlohmann::ordered_json;
/////////////////////////////////////////////////////////////////////////////
// alt_json: a second, independent copy of the custom-string_t basic_json
// specialization defined in unit-alt-string.cpp.
//
// It is duplicated here (rather than shared via a header) because every
// unit-*.cpp file in this test suite is compiled into its own standalone
// executable (see tests/CMakeLists.txt), so there is no ODR concern in
// having the same class name defined in multiple translation units.
//
// Two members had to be added relative to the original alt_string
// (a constructor from std::string, and a find(char, pos) overload) because
// the original type was never used with the binary writers/readers before
// this file: BSON's array/document writer converts std::to_string() results
// and checks for embedded NUL characters via find(char), and the UBJSON/BSON
// high-precision-number path constructs the SAX string_t argument from a
// std::string. Neither path is exercised anywhere else in the test suite for
// this type, which is presumably why the gap was never noticed.
/////////////////////////////////////////////////////////////////////////////
class alt_string;
bool operator<(const char* op1, const alt_string& op2) noexcept; // NOLINT(misc-use-internal-linkage)
void int_to_string(alt_string& target, std::size_t value); // NOLINT(misc-use-internal-linkage)
class alt_string
{
public:
using value_type = std::string::value_type;
static constexpr auto npos = (std::numeric_limits<std::size_t>::max)();
alt_string(const char* str): str_impl(str) {}
alt_string(const char* str, std::size_t count): str_impl(str, count) {}
alt_string(std::string str): str_impl(std::move(str)) {}
alt_string(size_t count, char chr): str_impl(count, chr) {}
alt_string() = default;
alt_string& append(char ch)
{
str_impl.push_back(ch);
return *this;
}
alt_string& append(const alt_string& str)
{
str_impl.append(str.str_impl);
return *this;
}
alt_string& append(const char* s, std::size_t length)
{
str_impl.append(s, length);
return *this;
}
void push_back(char c)
{
str_impl.push_back(c);
}
template <typename op_type>
bool operator==(const op_type& op) const
{
return str_impl == op;
}
bool operator==(const alt_string& op) const
{
return str_impl == op.str_impl;
}
template <typename op_type>
bool operator!=(const op_type& op) const
{
return str_impl != op;
}
bool operator!=(const alt_string& op) const
{
return str_impl != op.str_impl;
}
std::size_t size() const noexcept
{
return str_impl.size();
}
void resize(std::size_t n)
{
str_impl.resize(n);
}
void resize(std::size_t n, char c)
{
str_impl.resize(n, c);
}
template <typename op_type>
bool operator<(const op_type& op) const noexcept
{
return str_impl < op;
}
bool operator<(const alt_string& op) const noexcept
{
return str_impl < op.str_impl;
}
const char* c_str() const
{
return str_impl.c_str();
}
char& operator[](std::size_t index)
{
return str_impl[index];
}
const char& operator[](std::size_t index) const
{
return str_impl[index];
}
char& back()
{
return str_impl.back();
}
const char& back() const
{
return str_impl.back();
}
void clear()
{
str_impl.clear();
}
const value_type* data() const
{
return str_impl.data();
}
bool empty() const
{
return str_impl.empty();
}
std::size_t find(const alt_string& str, std::size_t pos = 0) const
{
return str_impl.find(str.str_impl, pos);
}
// needed by binary_writer's BSON support, which probes string keys for
// embedded NUL characters via find(char)
std::size_t find(char c, std::size_t pos = 0) const
{
return str_impl.find(c, pos);
}
std::size_t find_first_of(char c, std::size_t pos = 0) const
{
return str_impl.find_first_of(c, pos);
}
alt_string substr(std::size_t pos = 0, std::size_t count = npos) const
{
const std::string s = str_impl.substr(pos, count);
return {s.data(), s.size()};
}
alt_string& replace(std::size_t pos, std::size_t count, const alt_string& str)
{
str_impl.replace(pos, count, str.str_impl);
return *this;
}
void reserve(std::size_t new_cap = 0)
{
str_impl.reserve(new_cap);
}
private:
std::string str_impl {}; // NOLINT(readability-redundant-member-init)
friend bool operator<(const char* /*op1*/, const alt_string& /*op2*/) noexcept;
};
void int_to_string(alt_string& target, std::size_t value)
{
target = std::to_string(value).c_str();
}
using alt_json = nlohmann::basic_json <
std::map,
std::vector,
alt_string,
bool,
std::int64_t,
std::uint64_t,
double,
std::allocator,
nlohmann::adl_serializer >;
bool operator<(const char* op1, const alt_string& op2) noexcept
{
return op1 < op2.str_impl;
}
namespace
{
// collects the object keys of j, in iteration order
std::vector<std::string> collect_keys(const ordered_json& j)
{
std::vector<std::string> result;
for (auto it = j.cbegin(); it != j.cend(); ++it)
{
result.push_back(it.key());
}
return result;
}
// a nested object/array value with keys inserted in non-alphabetical order,
// used to check both round-trip equality and (for ordered_json) that
// insertion order survives a trip through a binary format
ordered_json make_rich_ordered_json()
{
ordered_json j;
j["zebra"] = 1;
j["apple"] = ordered_json::array({1, 2, 3});
j["mango"]["z_nested"] = true;
j["mango"]["a_nested"] = nullptr;
j["banana"] = "some text";
j["cherry"] = 3.14;
return j;
}
alt_json make_rich_alt_json()
{
alt_json j;
j["zebra"] = 1;
j["apple"] = alt_json::array({1, 2, 3});
j["mango"]["z_nested"] = true;
j["mango"]["a_nested"] = nullptr;
j["banana"] = "some text";
j["cherry"] = 3.14;
return j;
}
} // namespace
TEST_CASE("ordered_json across binary formats")
{
const ordered_json original = make_rich_ordered_json();
const std::vector<std::string> original_keys = collect_keys(original);
const std::vector<std::string> original_mango_keys = collect_keys(original["mango"]);
SECTION("CBOR")
{
const auto bytes = ordered_json::to_cbor(original);
const auto restored = ordered_json::from_cbor(bytes);
CHECK(restored == original);
CHECK(collect_keys(restored) == original_keys);
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
}
SECTION("MessagePack")
{
const auto bytes = ordered_json::to_msgpack(original);
const auto restored = ordered_json::from_msgpack(bytes);
CHECK(restored == original);
CHECK(collect_keys(restored) == original_keys);
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
}
SECTION("UBJSON")
{
const auto bytes = ordered_json::to_ubjson(original);
const auto restored = ordered_json::from_ubjson(bytes);
CHECK(restored == original);
CHECK(collect_keys(restored) == original_keys);
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
}
SECTION("BSON")
{
const auto bytes = ordered_json::to_bson(original);
const auto restored = ordered_json::from_bson(bytes);
CHECK(restored == original);
CHECK(collect_keys(restored) == original_keys);
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
}
SECTION("BJData")
{
const auto bytes = ordered_json::to_bjdata(original);
const auto restored = ordered_json::from_bjdata(bytes);
CHECK(restored == original);
CHECK(collect_keys(restored) == original_keys);
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
}
}
TEST_CASE("alt_json (custom string_t) across binary formats")
{
const alt_json original = make_rich_alt_json();
SECTION("CBOR")
{
const auto bytes = alt_json::to_cbor(original);
const auto restored = alt_json::from_cbor(bytes);
CHECK(restored == original);
}
SECTION("MessagePack")
{
const auto bytes = alt_json::to_msgpack(original);
const auto restored = alt_json::from_msgpack(bytes);
CHECK(restored == original);
}
SECTION("UBJSON")
{
const auto bytes = alt_json::to_ubjson(original);
const auto restored = alt_json::from_ubjson(bytes);
CHECK(restored == original);
}
SECTION("BSON")
{
const auto bytes = alt_json::to_bson(original);
const auto restored = alt_json::from_bson(bytes);
CHECK(restored == original);
}
SECTION("BJData")
{
const auto bytes = alt_json::to_bjdata(original);
const auto restored = alt_json::from_bjdata(bytes);
CHECK(restored == original);
}
}
TEST_CASE("ordered_json operator== is sensitive to key order")
{
// Unlike nlohmann::json (whose object_t is a std::map, so equality never
// depends on insertion order), ordered_json's object_t (ordered_map) is a
// std::vector<std::pair<Key, T>> under the hood, and does not define its
// own operator==: it inherits std::vector's element-wise comparison. As a
// result, two ordered_json objects holding the very same key/value pairs
// in different insertion order compare *unequal*. This is the property
// that makes the round-trip `CHECK(restored == original)` checks above a
// meaningful order-preservation check by themselves (the explicit
// collect_keys() comparisons make that check explicit/readable, and
// guard against this operator== behavior ever changing).
ordered_json a;
a["x"] = 1;
a["y"] = 2;
ordered_json b;
b["y"] = 2;
b["x"] = 1;
CHECK(a.size() == b.size());
CHECK(a["x"] == b["x"]);
CHECK(a["y"] == b["y"]);
CHECK_FALSE(a == b);
}
TEST_CASE("duplicate keys in a binary-encoded object")
{
// CBOR encoding of a map with two entries under the same key "a": {"a": 1, "a": 2}
const std::vector<std::uint8_t> cbor_bytes
{
0xA2, 0x61, 'a', 0x01, 0x61, 'a', 0x02
};
// Both json (std::map, via operator[]) and ordered_json (ordered_map, via
// operator[]) build binary-decoded objects by looking up/creating the
// entry for each incoming key and then assigning the value into it. This
// means a repeated key does *not* produce two entries in either case;
// instead, the *first* occurrence's position is kept (relevant only for
// ordered_json) while the *last* occurrence's value wins (for both) --
// this matches operator[]'s "assign the referenced slot" semantics, and
// is worth noting because it differs from the initializer-list
// construction path (`ordered_json{{"a",1},{"a",2}}`), which builds
// through insert()/emplace() and therefore keeps the *first* value, not
// the last (see the "There are no dup keys..." case in
// unit-ordered_json.cpp).
const auto j = json::from_cbor(cbor_bytes);
const auto oj = ordered_json::from_cbor(cbor_bytes);
CHECK(j.size() == 1);
CHECK(oj.size() == 1);
CHECK(j["a"] == 2);
CHECK(oj["a"] == 2);
CHECK(j == json(oj));
}
TEST_CASE("ordered_json through flatten/unflatten")
{
const ordered_json original = make_rich_ordered_json();
const std::vector<std::string> original_keys = collect_keys(original);
const std::vector<std::string> original_mango_keys = collect_keys(original["mango"]);
const ordered_json flat = original.flatten();
const ordered_json unflattened = flat.unflatten();
CHECK(unflattened == original);
// flatten() walks the value depth-first in iteration order and
// unflatten() re-inserts each flattened key via operator[] in the flat
// object's iteration order, so for ordered_json the original key order
// (both top-level and nested) is preserved end-to-end.
CHECK(collect_keys(unflattened) == original_keys);
CHECK(collect_keys(unflattened["mango"]) == original_mango_keys);
}
TEST_CASE("ordered_json through diff/patch/patch_inplace")
{
ordered_json original;
original["one"] = 1;
original["two"] = 2;
original["three"] = 3;
ordered_json target = original;
target["one"] = 100; // replace
target.erase("two"); // remove
target["four"] = 4; // add
const ordered_json patch = ordered_json::diff(original, target);
SECTION("patch")
{
const ordered_json patched = original.patch(patch);
CHECK(patched == target);
}
SECTION("patch_inplace")
{
ordered_json copy = original;
copy.patch_inplace(patch);
CHECK(copy == target);
}
}
TEST_CASE("ordered_json through merge_patch")
{
ordered_json original;
original["a"] = 1;
original["b"] = 2;
const ordered_json patch = {{"b", nullptr}, {"c", 3}};
original.merge_patch(patch);
ordered_json expected;
expected["a"] = 1;
expected["c"] = 3;
CHECK(original == expected);
CHECK(collect_keys(original) == collect_keys(expected));
}
+3 -5
View File
@@ -29,10 +29,7 @@ using nlohmann::json;
#include <limits>
#include <cstdio>
#include "make_test_data_available.hpp"
#ifdef JSON_HAS_CPP_17
#include <variant>
#endif
#include "test_utils.hpp"
#include "fifo_map.hpp"
@@ -1373,7 +1370,8 @@ TEST_CASE("regression tests 1")
std::array<uint8_t, 28> key1 = {{ 103, 92, 117, 48, 48, 48, 55, 92, 114, 215, 126, 214, 95, 92, 34, 174, 40, 71, 38, 174, 40, 71, 38, 223, 134, 247, 127, 0 }};
std::string const key1_str(reinterpret_cast<char*>(key1.data()));
json const j = key1_str;
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 10: 0x7E", json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 10: 0x7E", json::type_error&);
}
#if JSON_USE_IMPLICIT_CONVERSIONS
+80 -766
View File
@@ -31,6 +31,8 @@ using ordered_json = nlohmann::ordered_json;
#include <type_traits>
#include <utility>
#include "test_utils.hpp"
#ifdef JSON_HAS_CPP_17
#include <any>
#include <variant>
@@ -239,209 +241,6 @@ class my_allocator : public std::allocator<T>
};
};
/////////////////////////////////////////////////////////////////////
// for #3077
/////////////////////////////////////////////////////////////////////
class FooAlloc
{};
class Foo
{
public:
explicit Foo(const FooAlloc& /* unused */ = FooAlloc()) {}
bool value = false;
};
class FooBar
{
public:
Foo foo{}; // NOLINT(readability-redundant-member-init)
};
inline void from_json(const nlohmann::json& j, FooBar& fb) // NOLINT(misc-use-internal-linkage)
{
j.at("value").get_to(fb.foo.value);
}
/////////////////////////////////////////////////////////////////////
// for #3171
/////////////////////////////////////////////////////////////////////
struct for_3171_base // NOLINT(cppcoreguidelines-special-member-functions)
{
for_3171_base(const std::string& /*unused*/ = {}) {}
virtual ~for_3171_base();
for_3171_base(const for_3171_base& other) // NOLINT(hicpp-use-equals-default,modernize-use-equals-default)
: str(other.str)
{}
for_3171_base& operator=(const for_3171_base& other)
{
if (this != &other)
{
str = other.str;
}
return *this;
}
for_3171_base(for_3171_base&& other) noexcept
: str(std::move(other.str))
{}
for_3171_base& operator=(for_3171_base&& other) noexcept
{
if (this != &other)
{
str = std::move(other.str);
}
return *this;
}
virtual void _from_json(const json& j)
{
j.at("str").get_to(str);
}
std::string str{}; // NOLINT(readability-redundant-member-init)
};
for_3171_base::~for_3171_base() = default;
struct for_3171_derived : public for_3171_base
{
for_3171_derived() = default;
~for_3171_derived() override;
explicit for_3171_derived(const std::string& /*unused*/) { }
for_3171_derived(const for_3171_derived& other) // NOLINT(hicpp-use-equals-default,modernize-use-equals-default)
: for_3171_base(other)
{}
for_3171_derived& operator=(const for_3171_derived& other)
{
if (this != &other)
{
for_3171_base::operator=(other); // Call base class assignment operator
}
return *this;
}
for_3171_derived(for_3171_derived&& other) noexcept
: for_3171_base(std::move(other))
{}
for_3171_derived& operator=(for_3171_derived&& other) noexcept
{
if (this != &other)
{
for_3171_base::operator=(std::move(other)); // Call base class move assignment operator
}
return *this;
}
};
for_3171_derived::~for_3171_derived() = default;
inline void from_json(const json& j, for_3171_base& tb) // NOLINT(misc-use-internal-linkage)
{
tb._from_json(j);
}
/////////////////////////////////////////////////////////////////////
// for #3312
/////////////////////////////////////////////////////////////////////
#ifdef JSON_HAS_CPP_20
struct for_3312
{
std::string name;
};
inline void from_json(const json& j, for_3312& obj) // NOLINT(misc-use-internal-linkage)
{
j.at("name").get_to(obj.name);
}
#endif
/////////////////////////////////////////////////////////////////////
// for #3204
/////////////////////////////////////////////////////////////////////
struct for_3204_foo
{
for_3204_foo() = default;
explicit for_3204_foo(std::string /*unused*/) {} // NOLINT(performance-unnecessary-value-param)
};
struct for_3204_bar
{
enum constructed_from_t // NOLINT(cppcoreguidelines-use-enum-class)
{
constructed_from_none = 0,
constructed_from_foo = 1,
constructed_from_json = 2
};
explicit for_3204_bar(std::function<void(for_3204_foo)> /*unused*/) noexcept // NOLINT(performance-unnecessary-value-param)
: constructed_from(constructed_from_foo) {}
explicit for_3204_bar(std::function<void(json)> /*unused*/) noexcept // NOLINT(performance-unnecessary-value-param)
: constructed_from(constructed_from_json) {}
constructed_from_t constructed_from = constructed_from_none;
};
/////////////////////////////////////////////////////////////////////
// for #3333
/////////////////////////////////////////////////////////////////////
struct for_3333 final
{
for_3333(int x_ = 0, int y_ = 0) : x(x_), y(y_) {}
template <class T>
for_3333(const T& /*unused*/)
{
CHECK(false);
}
int x = 0;
int y = 0;
};
template <>
inline for_3333::for_3333(const json& j)
: for_3333(j.value("x", 0), j.value("y", 0))
{}
/////////////////////////////////////////////////////////////////////
// for #3810
/////////////////////////////////////////////////////////////////////
struct Example_3810
{
int bla{};
Example_3810() = default;
};
NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE(Example_3810, bla) // NOLINT(misc-use-internal-linkage)
/////////////////////////////////////////////////////////////////////
// for #4740
/////////////////////////////////////////////////////////////////////
#ifdef JSON_HAS_CPP_17
struct Example_4740
{
std::optional<std::string> host = std::nullopt;
std::optional<int> port = std::nullopt;
NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(Example_4740, host, port)
};
#endif
TEST_CASE("regression tests 2")
{
SECTION("issue #1001 - Fix memory leak during parser callback")
@@ -639,7 +438,8 @@ TEST_CASE("regression tests 2")
s += static_cast<char>(i);
}
dump_test["1"] = s;
dump_test.dump(-1, ' ', true, nlohmann::json::error_handler_t::replace);
// dump() is nodiscard; this only checks that dumping does not throw/crash
utils::ignore_return_value(dump_test.dump(-1, ' ', true, nlohmann::json::error_handler_t::replace));
}
}
@@ -731,12 +531,14 @@ TEST_CASE("regression tests 2")
{
const std::array<unsigned char, 23> data = {{0x81, 0xA4, 0x64, 0x61, 0x74, 0x61, 0xC4, 0x0F, 0x33, 0x30, 0x30, 0x32, 0x33, 0x34, 0x30, 0x31, 0x30, 0x37, 0x30, 0x35, 0x30, 0x31, 0x30}};
const json j = json::from_msgpack(data.data(), data.size());
// dump() is nodiscard; this only checks that dumping does not throw
CHECK_NOTHROW(
j.dump(4, // Indent
' ', // Indent char
false, // Ensure ascii
json::error_handler_t::strict // Error
));
utils::ignore_return_value(
j.dump(4, // Indent
' ', // Indent char
false, // Ensure ascii
json::error_handler_t::strict // Error
)));
}
SECTION("PR #2181 - regression bug with lvalue")
@@ -804,7 +606,11 @@ TEST_CASE("regression tests 2")
SECTION("issue #2546 - parsing containers of std::byte")
{
const char DATA[] = R"("Hello, world!")"; // NOLINT(misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
const auto s = std::as_bytes(std::span(DATA));
// exclude the trailing '\0' that string-literal initialization adds to
// DATA: std::span(DATA) would span the full array extent (including
// that NUL), which is only silently accepted as end-of-input by default
// and would fail under JSON_STRICT_NUL_HANDLING
const auto s = std::as_bytes(std::span(DATA, sizeof(DATA) - 1));
const json j = json::parse(s);
CHECK(j.dump() == "\"Hello, world!\"");
}
@@ -959,600 +765,108 @@ TEST_CASE("regression tests 2")
CHECK(j == k);
}
#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM
// JSON_HAS_CPP_17 (do not remove; see note at top of file)
SECTION("issue #3070 - Version 3.10.3 breaks backward-compatibility with 3.10.2 ")
}
TEST_CASE("regression test - parser callback must not lose a duplicate key's prior value")
{
// a callback that rejects only the scalar value 2
const json::parser_callback_t drop_value_2 = [](int /*depth*/, json::parse_event_t ev, json & v) noexcept
{
nlohmann::detail::std_fs::path text_path("/tmp/text.txt");
const json j(text_path);
return !(ev == json::parse_event_t::value && v == 2);
};
const auto j_path = j.get<nlohmann::detail::std_fs::path>();
CHECK(j_path == text_path);
#if DOCTEST_CLANG || DOCTEST_GCC >= DOCTEST_COMPILER(8, 4, 0)
// only known to work on Clang and GCC >=8.4
CHECK_THROWS_WITH_AS(nlohmann::detail::std_fs::path(json(1)), "[json.exception.type_error.302] type must be string, but is number", json::type_error);
#endif
}
#endif
SECTION("issue #3077 - explicit constructor with default does not compile")
SECTION("duplicate key, second (scalar) value rejected - prior value is restored")
{
json j;
j[0]["value"] = true;
std::vector<FooBar> foo;
j.get_to(foo);
const json j = json::parse(R"({"a":1,"a":2})", drop_value_2);
CHECK(j.dump() == "{\"a\":1}");
}
SECTION("issue #3108 - ordered_json doesn't support range based erase")
SECTION("duplicate key, second value is an object rejected at object_end - prior value is restored")
{
ordered_json j = {1, 2, 2, 4};
auto last = std::unique(j.begin(), j.end());
j.erase(last, j.end());
CHECK(j.dump() == "[1,2,4]");
j.erase(std::remove_if(j.begin(), j.end(), [](const ordered_json & val)
const json j = json::parse(R"({"a":1,"a":{"x":2}})",
[](int depth, json::parse_event_t ev, json& /*parsed*/) noexcept
{
return val == 2;
}), j.end());
CHECK(j.dump() == "[1,4]");
return !(ev == json::parse_event_t::object_end && depth == 1);
});
CHECK(j.dump() == "{\"a\":1}");
}
SECTION("issue #3343 - json and ordered_json are not interchangeable")
SECTION("duplicate key, second value is an array rejected at array_end - prior value is restored")
{
json::object_t jobj({ { "product", "one" } });
ordered_json::object_t ojobj({{"product", "one"}});
auto jit = jobj.begin();
auto ojit = ojobj.begin();
CHECK(jit->first == ojit->first);
CHECK(jit->second.get<std::string>() == ojit->second.get<std::string>());
}
SECTION("issue #3171 - if class is_constructible from std::string wrong from_json overload is being selected, compilation failed")
{
const json j{{ "str", "value"}};
// failed with: error: no match for ‘operator=’ (operand types are ‘for_3171_derived’ and ‘const nlohmann::basic_json<>::string_t’
// {aka ‘const std::__cxx11::basic_string<char>’})
// s = *j.template get_ptr<const typename BasicJsonType::string_t*>();
auto td = j.get<for_3171_derived>();
CHECK(td.str == "value");
}
#ifdef JSON_HAS_CPP_20
SECTION("issue #3312 - Parse to custom class from unordered_json breaks on G++11.2.0 with C++20")
{
// see test for #3171
const ordered_json j = {{"name", "class"}};
for_3312 obj{};
j.get_to(obj);
CHECK(obj.name == "class");
}
#endif
#if defined(JSON_HAS_CPP_17) && JSON_USE_IMPLICIT_CONVERSIONS
SECTION("issue #3428 - Error occurred when converting nlohmann::json to std::any")
{
const json j;
const std::any a1 = j;
std::any&& a2 = j;
CHECK(a1.type() == typeid(j));
CHECK(a2.type() == typeid(j));
}
#endif
SECTION("issue #3204 - ambiguous regression")
{
const for_3204_bar bar_from_foo([](for_3204_foo) noexcept {}); // NOLINT(performance-unnecessary-value-param)
const for_3204_bar bar_from_json([](json) noexcept {}); // NOLINT(performance-unnecessary-value-param)
CHECK(bar_from_foo.constructed_from == for_3204_bar::constructed_from_foo);
CHECK(bar_from_json.constructed_from == for_3204_bar::constructed_from_json);
}
SECTION("issue #3333 - Ambiguous conversion from nlohmann::basic_json<> to custom class")
{
const json j
const json j = json::parse(R"({"a":1,"a":[9,9]})",
[](int depth, json::parse_event_t ev, json& /*parsed*/) noexcept
{
{"x", 1},
{"y", 2}
};
const for_3333 p = j;
CHECK(p.x == 1);
CHECK(p.y == 2);
return !(ev == json::parse_event_t::array_end && depth == 1);
});
CHECK(j.dump() == "{\"a\":1}");
}
SECTION("issue #3810 - ordered_json doesn't support construction from C array of custom type")
SECTION("duplicate key, second value accepted (scalar) - last value wins")
{
Example_3810 states[45]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
// fix "not used" warning
states[0].bla = 1;
const auto* const expected = R"([{"bla":1},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0}])";
// This works:
nlohmann::json j;
j["test"] = states;
CHECK(j["test"].dump() == expected);
// This doesn't compile:
nlohmann::ordered_json oj;
oj["test"] = states;
CHECK(oj["test"].dump() == expected);
}
#ifdef JSON_HAS_CPP_17
SECTION("issue #4740 - build issue with std::optional")
{
const auto t1 = Example_4740();
const auto j1 = nlohmann::json(t1);
CHECK(j1.dump() == "{\"host\":null,\"port\":null}");
const auto t2 = j1.get<Example_4740>();
CHECK(!t2.host.has_value());
CHECK(!t2.port.has_value());
// improve coverage
auto t3 = Example_4740();
t3.port = 80;
t3.host = "example.com";
const auto j2 = nlohmann::json(t3);
CHECK(j2.dump() == "{\"host\":\"example.com\",\"port\":80}");
const auto t4 = j2.get<Example_4740>();
CHECK(t4.host.has_value());
CHECK(t4.port.has_value());
}
#endif
#if !defined(_MSVC_LANG)
// MSVC returns garbage on invalid enum values, so this test is excluded
// there.
SECTION("issue #4762 - json exception 302 with unhelpful explanation : type must be number, but is number")
{
// In #4762, the main issue was that a json object with an invalid type
// returned "number" as type_name(), because this was the default case.
// This test makes sure we now return "invalid" instead.
json j;
j.m_data.m_type = static_cast<json::value_t>(100); // NOLINT(clang-analyzer-optin.core.EnumCastOutOfRange)
CHECK(j.type_name() == "invalid");
}
#endif
#ifdef JSON_HAS_CPP_17
SECTION("issue #4804: from_cbor incompatible with std::vector<std::byte> as binary_t")
{
const std::vector<std::uint8_t> data = {0x80};
const auto decoded = json_4804::from_cbor(data);
CHECK((decoded == json_4804::array()));
}
SECTION("discussion #4209 - custom BinaryType direct assignment and round-tripping")
{
// Test that assigning a custom BinaryType directly creates a binary value, not an array
const std::vector<std::byte> original{std::byte{1}, std::byte{2}, std::byte{3}};
const json_4804 j = original;
CHECK(j.is_binary());
CHECK(!j.is_array());
// Test round-tripping: extracting the binary value back as the custom container type
const auto extracted = j.get<std::vector<std::byte>>();
CHECK(extracted == original);
// Test that the default json alias behavior is unchanged: std::vector<uint8_t> -> array
const json default_json = std::vector<std::uint8_t> {1, 2, 3};
CHECK(default_json.is_array());
CHECK(!default_json.is_binary());
}
SECTION("discussion #4209 - custom BinaryType extraction from parsed array")
{
// Test that extracting a custom BinaryType from a parsed JSON array still works
// (not just from a binary-typed node)
const auto j = json_4804::parse("[1,2,3]");
CHECK(j.is_array());
CHECK(!j.is_binary());
// Extracting as custom BinaryType should work from arrays
const auto extracted = j.get<std::vector<std::byte>>();
CHECK(extracted.size() == 3);
CHECK(extracted[0] == std::byte{1});
CHECK(extracted[1] == std::byte{2});
CHECK(extracted[2] == std::byte{3});
}
SECTION("issue #5046 - implicit conversion of return json to std::optional no longer implicit")
{
const json jval{};
auto GetValue = [](const json & valRoot) -> std::optional<json>
{
if (valRoot.contains("default"))
{
return valRoot.at("default");
}
return std::nullopt;
};
auto result = GetValue(jval);
CHECK(!result.has_value());
}
#endif
#if JSON_HAS_RANGES == 1
SECTION("issue #4440 - assert when using std::views::filter and GCC 10")
{
auto noOpFilter = std::views::filter([](auto&&) noexcept
const json j = json::parse(R"({"a":1,"a":2})", [](int, json::parse_event_t, json&) noexcept
{
return true;
});
json j = {1, 2, 3};
auto filtered = j | noOpFilter;
CHECK(*filtered.begin() == 1);
CHECK(j.dump() == "{\"a\":2}");
}
#endif
#if JSON_HAS_RANGES && !defined(__MINGW32__)
SECTION("issue #4916 - constructing array from C++20 ranges view does not work")
SECTION("duplicate key, second value accepted (object) - last value wins")
{
std::vector<int> nums{1, 2, 37, 42, 21};
auto filteredNums = nums | std::views::filter([](int i)
const json j = json::parse(R"({"a":1,"a":{"x":2}})", [](int, json::parse_event_t, json&) noexcept
{
return i > 10;
return true;
});
json const j(filteredNums);
CHECK(j.type() == json::value_t::array);
CHECK(j == json({37, 42, 21}));
CHECK(j.dump() == "{\"a\":{\"x\":2}}");
}
#endif
// owning_view is not available in libstdc++ < 12
#if JSON_HAS_RANGES && !defined(__MINGW32__) && !(defined(__GLIBCXX__) && _GLIBCXX_RELEASE < 12)
SECTION("issue #4916 - constructing array from prvalue C++20 ranges view (owning_view)")
SECTION("brand new (non-duplicate) key, value rejected - member is fully absent")
{
json const j(std::vector<int> {1, 2, 37, 42, 21} | std::views::filter([](int i)
{
return i > 10;
}));
CHECK(j.type() == json::value_t::array);
CHECK(j == json({37, 42, 21}));
const json j = json::parse(R"({"a":1,"b":2})", drop_value_2);
CHECK(j.dump() == "{\"a\":1}");
}
#endif
#if JSON_HAS_RANGES && !defined(__MINGW32__)
SECTION("issue #4916 - constructing array from C++20 transform view (prvalue elements)")
SECTION("duplicate key nested two levels deep")
{
std::vector<int> nums{1, 2, 3};
auto t = nums | std::views::transform([](int i) noexcept
{
return i * 2;
});
json const j(t);
CHECK(j.type() == json::value_t::array);
CHECK(j == json({2, 4, 6}));
const json j = json::parse(R"({"outer":{"a":1,"a":2}})", drop_value_2);
CHECK(j.dump() == "{\"outer\":{\"a\":1}}");
}
#endif
}
TEST_CASE_TEMPLATE("issue #4798 - nlohmann::json::to_msgpack() encode float NaN as double", T, double, float) // NOLINT(readability-math-missing-parentheses, bugprone-throwing-static-initialization)
{
// With issue #4798, we encode NaN, infinity, and -infinity as float instead
// of double to allow for smaller encodings.
const json jx = std::numeric_limits<T>::quiet_NaN();
const json jy = std::numeric_limits<T>::infinity();
const json jz = -std::numeric_limits<T>::infinity();
/////////////////////////////////////////////////////////////////////////
// MessagePack
/////////////////////////////////////////////////////////////////////////
// expected MessagePack values
const std::vector<std::uint8_t> msgpack_x = {{0xCA, 0x7F, 0xC0, 0x00, 0x00}};
const std::vector<std::uint8_t> msgpack_y = {{0xCA, 0x7F, 0x80, 0x00, 0x00}};
const std::vector<std::uint8_t> msgpack_z = {{0xCA, 0xFF, 0x80, 0x00, 0x00}};
CHECK(json::to_msgpack(jx) == msgpack_x);
CHECK(json::to_msgpack(jy) == msgpack_y);
CHECK(json::to_msgpack(jz) == msgpack_z);
CHECK(std::isnan(json::from_msgpack(msgpack_x).get<T>()));
CHECK(json::from_msgpack(msgpack_y).get<T>() == std::numeric_limits<T>::infinity());
CHECK(json::from_msgpack(msgpack_z).get<T>() == -std::numeric_limits<T>::infinity());
// Make sure the other MessagePakc encodings for NaN, infinity, and
// -infinity are still supported.
const std::vector<std::uint8_t> msgpack_x_2 = {{0xCB, 0x7F, 0xF8, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}};
const std::vector<std::uint8_t> msgpack_y_2 = {{0xCB, 0x7F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}};
const std::vector<std::uint8_t> msgpack_z_2 = {{0xCB, 0xFF, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}};
CHECK(std::isnan(json::from_msgpack(msgpack_x_2).get<T>()));
CHECK(json::from_msgpack(msgpack_y_2).get<T>() == std::numeric_limits<T>::infinity());
CHECK(json::from_msgpack(msgpack_z_2).get<T>() == -std::numeric_limits<T>::infinity());
/////////////////////////////////////////////////////////////////////////
// CBOR
/////////////////////////////////////////////////////////////////////////
// expected CBOR values
const std::vector<std::uint8_t> cbor_x = {{0xF9, 0x7E, 0x00}};
const std::vector<std::uint8_t> cbor_y = {{0xF9, 0x7C, 0x00}};
const std::vector<std::uint8_t> cbor_z = {{0xF9, 0xfC, 0x00}};
CHECK(json::to_cbor(jx) == cbor_x);
CHECK(json::to_cbor(jy) == cbor_y);
CHECK(json::to_cbor(jz) == cbor_z);
CHECK(std::isnan(json::from_cbor(cbor_x).get<T>()));
CHECK(json::from_cbor(cbor_y).get<T>() == std::numeric_limits<T>::infinity());
CHECK(json::from_cbor(cbor_z).get<T>() == -std::numeric_limits<T>::infinity());
// Make sure the other CBOR encodings for NaN, infinity, and -infinity are
// still supported.
const std::vector<std::uint8_t> cbor_x_2 = {{0xFA, 0x7F, 0xC0, 0x00, 0x00}};
const std::vector<std::uint8_t> cbor_y_2 = {{0xFA, 0x7F, 0x80, 0x00, 0x00}};
const std::vector<std::uint8_t> cbor_z_2 = {{0xFA, 0xFF, 0x80, 0x00, 0x00}};
const std::vector<std::uint8_t> cbor_x_3 = {{0xFB, 0x7F, 0xF8, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}};
const std::vector<std::uint8_t> cbor_y_3 = {{0xFB, 0x7F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}};
const std::vector<std::uint8_t> cbor_z_3 = {{0xFB, 0xFF, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}};
CHECK(std::isnan(json::from_cbor(cbor_x_2).get<T>()));
CHECK(json::from_cbor(cbor_y_2).get<T>() == std::numeric_limits<T>::infinity());
CHECK(json::from_cbor(cbor_z_2).get<T>() == -std::numeric_limits<T>::infinity());
CHECK(std::isnan(json::from_cbor(cbor_x_3).get<T>()));
CHECK(json::from_cbor(cbor_y_3).get<T>() == std::numeric_limits<T>::infinity());
CHECK(json::from_cbor(cbor_z_3).get<T>() == -std::numeric_limits<T>::infinity());
}
TEST_CASE("regression test #5074 - portable workaround for single-element brace init")
{
json const j_obj = {{"key", "value"}};
json const j = json::array({j_obj});
CHECK(j.is_array());
CHECK(j.size() == 1);
CHECK(j[0] == j_obj);
}
#if defined(JSON_BRACE_INIT_COPY_SEMANTICS) && (JSON_BRACE_INIT_COPY_SEMANTICS == 1)
TEST_CASE("regression test #5074 - single-element brace init with JSON_BRACE_INIT_COPY_SEMANTICS")
{
// with JSON_BRACE_INIT_COPY_SEMANTICS: single-element brace init copies/moves
json const j_obj = {{"key", "value"}, {"num", 42}};
json const j_arr = {1, 2, 3};
// object: brace init copies instead of wrapping
json const j1{j_obj};
CHECK(j1.is_object());
CHECK(j1 == j_obj);
// array: brace init copies instead of wrapping
json const j2{j_arr};
CHECK(j2.is_array());
CHECK(j2.size() == 3);
CHECK(j2 == j_arr);
// primitives still work as initializer lists
json const j3{true};
CHECK(j3.is_boolean());
json const j4{42};
CHECK(j4.is_number_integer());
}
#endif
struct Example_5122
{
float b = 2;
nlohmann::ordered_map<std::string, std::string> c{}; // NOLINT(readability-redundant-member-init): needed for GCC -Weffc++
int a = 1;
NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(Example_5122, b, c, a)
};
TEST_CASE("regression test #5122 - from_json into types holding nlohmann::ordered_map")
{
Example_5122 src;
src.c.emplace("first", "1");
src.c.emplace("second", "2");
ordered_json const j = src;
Example_5122 const dst = j.get<Example_5122>();
CHECK(dst.b == src.b);
CHECK(dst.a == src.a);
REQUIRE(dst.c.size() == src.c.size());
auto src_it = src.c.begin();
auto dst_it = dst.c.begin();
for (; src_it != src.c.end(); ++src_it, ++dst_it)
SECTION("three occurrences of the same key - middle rejected, last accepted")
{
CHECK(dst_it->first == src_it->first);
CHECK(dst_it->second == src_it->second);
const json j = json::parse(R"({"k":1,"k":2,"k":3})", drop_value_2);
CHECK(j.dump() == "{\"k\":3}");
}
}
// -Wself-assign-overloaded was introduced in Clang 7. Gate the pragma on
// __has_warning so older Clang versions do not error with "unknown warning
// group". The __has_warning check has to stay inside the __clang__ branch
// because GCC does not provide it and would tokenize-error on the argument.
#if defined(__clang__) && defined(__has_warning)
#if __has_warning("-Wself-assign-overloaded")
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
DOCTEST_CLANG_SUPPRESS_WARNING("-Wself-assign-overloaded")
#endif
#endif
TEST_CASE("regression test #5122 - nlohmann::ordered_map copy-assignment is self-assignment safe")
TEST_CASE("regression test - excessive binary container size honors allow_exceptions=false")
{
nlohmann::ordered_map<std::string, std::string> m;
m.emplace("first", "1");
m.emplace("second", "2");
// CBOR array with declared length 2^63
const std::vector<std::uint8_t> cbor = {0x9b, 0x80, 0, 0, 0, 0, 0, 0, 0};
// CBOR map with declared length 2^63
const std::vector<std::uint8_t> cbor_m = {0xbb, 0x80, 0, 0, 0, 0, 0, 0, 0};
// UBJSON array with declared length 2^63-1
const std::vector<std::uint8_t> ubj = {'[', '#', 'L', 0x7f, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff};
// BJData array with declared length 2^63-1 (little endian)
const std::vector<std::uint8_t> bjd = {'[', '#', 'L', 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x7f};
// Insertion order is preserved by ordered_map, so we can check it directly.
m = m;
// allow_exceptions=false must report failure instead of throwing/aborting
CHECK(json::from_cbor(cbor, true, false).is_discarded());
CHECK(json::from_cbor(cbor_m, true, false).is_discarded());
CHECK(json::from_ubjson(ubj, true, false).is_discarded());
CHECK(json::from_bjdata(bjd, true, false).is_discarded());
REQUIRE(m.size() == 2);
auto it = m.begin();
CHECK(it->first == "first");
CHECK(it->second == "1");
++it;
CHECK(it->first == "second");
CHECK(it->second == "2");
}
// allow_exceptions=true (the default) must still throw exactly as before.
// The exact message text is not checked here: on platforms where
// std::size_t is 32-bit, the CBOR reader's own length-narrowing check
// (get_cbor_container_size(), unrelated to this fix) intercepts a
// declared length of 2^63 before it ever reaches the check this test
// targets, with different (but equally valid, and already correct)
// wording -- see unit-cbor.cpp for coverage of that message.
json _;
CHECK_THROWS_AS(_ = json::from_cbor(cbor), json::out_of_range);
#if defined(__clang__) && defined(__has_warning)
#if __has_warning("-Wself-assign-overloaded")
DOCTEST_CLANG_SUPPRESS_WARNING_POP
#endif
#endif
TEST_CASE("regression test #5122 - nlohmann::ordered_map move-assignment transfers contents")
{
nlohmann::ordered_map<std::string, std::string> src;
src.emplace("first", "1");
src.emplace("second", "2");
nlohmann::ordered_map<std::string, std::string> dst;
dst.emplace("stale", "x");
dst = std::move(src);
REQUIRE(dst.size() == 2);
auto it = dst.begin();
CHECK(it->first == "first");
CHECK(it->second == "1");
++it;
CHECK(it->first == "second");
CHECK(it->second == "2");
// Re-assigning into the moved-from object must leave it in a usable state.
src = nlohmann::ordered_map<std::string, std::string> {};
src.emplace("after-move", "3");
REQUIRE(src.size() == 1);
CHECK(src.begin()->first == "after-move");
}
// Stand-in for a third-party library (e.g., Eigen as of 3.4, which added
// STL-compatible begin()/end() to its vector types), living in its own
// namespace with its own to_json overload for its vector type.
namespace issue_4320_eigen
{
// "array-compatible" from the library's point of view (it has begin()/end()),
// but for which this (fake) third-party namespace provides its own to_json.
struct vector3
{
double v[3]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays,cppcoreguidelines-use-default-member-init,modernize-use-default-member-init)
vector3(double x, double y, double z) : v{x, y, z} {} // NOLINT(hicpp-member-init,cppcoreguidelines-pro-type-member-init)
double x() const
{
return v[0];
}
double y() const
{
return v[1];
}
double z() const
{
return v[2];
}
double* begin()
{
return v;
}
double* end()
{
return v + 3;
}
const double* begin() const
{
return v;
}
const double* end() const
{
return v + 3;
}
};
inline void to_json(json& j, const vector3& v) // NOLINT(misc-use-internal-linkage)
{
j = {{"x", v.x()}, {"y", v.y()}, {"z", v.z()}};
}
} // namespace issue_4320_eigen
// The user's own namespace, using the (fake) Eigen type as an implementation
// detail behind a payload type that has nothing to do with vectors/arrays.
namespace issue_4320
{
// Publicly derives from issue_4320_eigen::vector3 but does *not* define its
// own to_json - it is only ever used as a temporary to reach the base
// class's to_json via ADL.
struct vector3_wrapper : issue_4320_eigen::vector3
{
using issue_4320_eigen::vector3::vector3;
};
struct payload
{
double x, y, z;
};
inline vector3_wrapper to_eigen(const payload& p) // NOLINT(misc-use-internal-linkage)
{
return {p.x, p.y, p.z};
}
inline void to_json(json& j, const payload& p) // NOLINT(misc-use-internal-linkage)
{
// Unqualified call, passing a *derived* vector3_wrapper: relies on ADL
// finding issue_4320_eigen::to_json(json&, const vector3&) through the
// vector3 base class, via a derived-to-base conversion. Must NOT resolve
// to the library's own generic array-compatible to_json (an exact-match
// template for vector3_wrapper, since it also has begin()/end()), which
// would serialize this as [x, y, z] instead of {"x":x, "y":y, "z":z}.
to_json(j, to_eigen(p));
}
} // namespace issue_4320
TEST_CASE("issue #4320 - custom base class must not leak nlohmann::detail into ADL")
{
// Before the fix, basic_json unconditionally derived from a type living in
// nlohmann::detail (json_default_base), which made nlohmann::detail an
// associated namespace of every basic_json for ADL purposes. That leaked
// the library's internal generic-array to_json overload into unqualified
// to_json() calls made from user code, silently bypassing user-defined
// to_json overloads reached via a derived-to-base conversion.
const issue_4320::payload p{1.0, 2.0, 3.0};
json j;
to_json(j, p);
CHECK(j == json({{"x", 1.0}, {"y", 2.0}, {"z", 3.0}}));
}
TEST_CASE("issue #5338 - truncated CBOR tagged binary subtype is rejected")
{
const std::vector<std::vector<std::uint8_t>> truncated_tags =
{
{0xD8},
{0xD9, 0x00},
{0xDA, 0x00, 0x00, 0x00},
{0xDB, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}
};
for (const auto& data : truncated_tags)
{
CAPTURE(data);
for (const auto tag_handler :
{
json::cbor_tag_handler_t::ignore, json::cbor_tag_handler_t::store
})
{
CAPTURE(tag_handler);
const auto result = json::from_cbor(data, true, false, tag_handler);
CHECK(result.is_discarded());
}
}
// regression guard: a genuinely truncated CBOR input must remain discarded
CHECK(json::from_cbor(std::vector<std::uint8_t> {0x9b, 0, 0, 0, 0, 0, 0, 0, 0x02}, true, false).is_discarded());
}
DOCTEST_CLANG_SUPPRESS_WARNING_POP
+928
View File
@@ -0,0 +1,928 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
// cmake/test.cmake selects the C++ standard versions with which to build a
// unit test based on the presence of JSON_HAS_CPP_<VERSION> macros.
// When using macros that are only defined for particular versions of the standard
// (e.g., JSON_HAS_FILESYSTEM for C++17 and up), please mention the corresponding
// version macro in a comment close by, like this:
// JSON_HAS_CPP_<VERSION> (do not remove; see note at top of file)
#include "doctest_compatibility.h"
// for some reason including this after the json header leads to linker errors with VS 2017...
#include <locale>
// skip tests if JSON_DisableEnumSerialization=ON (#4384): std::byte is a
// scoped enum, so get<std::byte>() (needed below to get<std::vector<std::byte>>()
// from a plain JSON array, not just from an already-binary value) relies on
// enum serialization being enabled
#if defined(JSON_DISABLE_ENUM_SERIALIZATION) && (JSON_DISABLE_ENUM_SERIALIZATION == 1)
#define SKIP_TESTS_FOR_ENUM_SERIALIZATION
#endif
#define JSON_TESTS_PRIVATE
#include <nlohmann/json.hpp>
using json = nlohmann::json;
using ordered_json = nlohmann::ordered_json;
#ifdef JSON_TEST_NO_GLOBAL_UDLS
using namespace nlohmann::literals; // NOLINT(google-build-using-namespace)
#endif
#include <cstdio>
#include <deque>
#include <list>
#include <type_traits>
#include <utility>
#ifdef JSON_HAS_CPP_17
#include <any>
#include <variant>
#endif
#ifdef JSON_HAS_CPP_17
#if __has_include(<optional>)
#include <optional>
#elif __has_include(<experimental/optional>)
#endif
/////////////////////////////////////////////////////////////////////
// for #4804
/////////////////////////////////////////////////////////////////////
using json_4804 = nlohmann::basic_json<std::map, // ObjectType
std::vector, // ArrayType
std::string, // StringType
bool, // BooleanType
std::int64_t, // NumberIntegerType
std::uint64_t, // NumberUnsignedType
double, // NumberFloatType
std::allocator, // AllocatorType
nlohmann::adl_serializer, // JSONSerializer
std::vector<std::byte>, // BinaryType
void // CustomBaseClass
>;
#endif
#ifdef JSON_HAS_CPP_20
#if __has_include(<span>)
#include <span>
#endif
#endif
/////////////////////////////////////////////////////////////////////
// for #4825 - explicitly instantiating basic_json must compile; this
// forces instantiation of binary_writer::write_bjdata_ndarray, whose
// static_cast<string_t> was ambiguous under explicit instantiation on
// C++17. Merely compiling this translation unit is the regression test.
/////////////////////////////////////////////////////////////////////
template class nlohmann::basic_json<>;
/////////////////////////////////////////////////////////////////////
// for #4440
/////////////////////////////////////////////////////////////////////
#if JSON_HAS_RANGES == 1
#include <ranges>
#endif
// NLOHMANN_JSON_SERIALIZE_ENUM uses a static std::pair
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
DOCTEST_CLANG_SUPPRESS_WARNING("-Wexit-time-destructors")
/////////////////////////////////////////////////////////////////////
// for #3077
/////////////////////////////////////////////////////////////////////
class FooAlloc
{};
class Foo
{
public:
explicit Foo(const FooAlloc& /* unused */ = FooAlloc()) {}
bool value = false;
};
class FooBar
{
public:
Foo foo{}; // NOLINT(readability-redundant-member-init)
};
inline void from_json(const nlohmann::json& j, FooBar& fb) // NOLINT(misc-use-internal-linkage)
{
j.at("value").get_to(fb.foo.value);
}
/////////////////////////////////////////////////////////////////////
// for #3171
/////////////////////////////////////////////////////////////////////
struct for_3171_base // NOLINT(cppcoreguidelines-special-member-functions)
{
for_3171_base(const std::string& /*unused*/ = {}) {}
virtual ~for_3171_base();
for_3171_base(const for_3171_base& other) // NOLINT(hicpp-use-equals-default,modernize-use-equals-default)
: str(other.str)
{}
for_3171_base& operator=(const for_3171_base& other)
{
if (this != &other)
{
str = other.str;
}
return *this;
}
for_3171_base(for_3171_base&& other) noexcept
: str(std::move(other.str))
{}
for_3171_base& operator=(for_3171_base&& other) noexcept
{
if (this != &other)
{
str = std::move(other.str);
}
return *this;
}
virtual void _from_json(const json& j)
{
j.at("str").get_to(str);
}
std::string str{}; // NOLINT(readability-redundant-member-init)
};
for_3171_base::~for_3171_base() = default;
struct for_3171_derived : public for_3171_base
{
for_3171_derived() = default;
~for_3171_derived() override;
explicit for_3171_derived(const std::string& /*unused*/) { }
for_3171_derived(const for_3171_derived& other) // NOLINT(hicpp-use-equals-default,modernize-use-equals-default)
: for_3171_base(other)
{}
for_3171_derived& operator=(const for_3171_derived& other)
{
if (this != &other)
{
for_3171_base::operator=(other); // Call base class assignment operator
}
return *this;
}
for_3171_derived(for_3171_derived&& other) noexcept
: for_3171_base(std::move(other))
{}
for_3171_derived& operator=(for_3171_derived&& other) noexcept
{
if (this != &other)
{
for_3171_base::operator=(std::move(other)); // Call base class move assignment operator
}
return *this;
}
};
for_3171_derived::~for_3171_derived() = default;
inline void from_json(const json& j, for_3171_base& tb) // NOLINT(misc-use-internal-linkage)
{
tb._from_json(j);
}
/////////////////////////////////////////////////////////////////////
// for #3312
/////////////////////////////////////////////////////////////////////
#ifdef JSON_HAS_CPP_20
struct for_3312
{
std::string name;
};
inline void from_json(const json& j, for_3312& obj) // NOLINT(misc-use-internal-linkage)
{
j.at("name").get_to(obj.name);
}
#endif
/////////////////////////////////////////////////////////////////////
// for #3204
/////////////////////////////////////////////////////////////////////
struct for_3204_foo
{
for_3204_foo() = default;
explicit for_3204_foo(std::string /*unused*/) {} // NOLINT(performance-unnecessary-value-param)
};
struct for_3204_bar
{
enum constructed_from_t // NOLINT(cppcoreguidelines-use-enum-class)
{
constructed_from_none = 0,
constructed_from_foo = 1,
constructed_from_json = 2
};
explicit for_3204_bar(std::function<void(for_3204_foo)> /*unused*/) noexcept // NOLINT(performance-unnecessary-value-param)
: constructed_from(constructed_from_foo) {}
explicit for_3204_bar(std::function<void(json)> /*unused*/) noexcept // NOLINT(performance-unnecessary-value-param)
: constructed_from(constructed_from_json) {}
constructed_from_t constructed_from = constructed_from_none;
};
/////////////////////////////////////////////////////////////////////
// for #3333
/////////////////////////////////////////////////////////////////////
struct for_3333 final
{
for_3333(int x_ = 0, int y_ = 0) : x(x_), y(y_) {}
template <class T>
for_3333(const T& /*unused*/)
{
CHECK(false);
}
int x = 0;
int y = 0;
};
template <>
inline for_3333::for_3333(const json& j)
: for_3333(j.value("x", 0), j.value("y", 0))
{}
/////////////////////////////////////////////////////////////////////
// for #3810
/////////////////////////////////////////////////////////////////////
struct Example_3810
{
int bla{};
Example_3810() = default;
};
NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE(Example_3810, bla) // NOLINT(misc-use-internal-linkage)
/////////////////////////////////////////////////////////////////////
// for #4740
/////////////////////////////////////////////////////////////////////
#ifdef JSON_HAS_CPP_17
struct Example_4740
{
std::optional<std::string> host = std::nullopt;
std::optional<int> port = std::nullopt;
NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(Example_4740, host, port)
};
#endif
TEST_CASE("regression tests 3")
{
#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM
// JSON_HAS_CPP_17 (do not remove; see note at top of file)
SECTION("issue #3070 - Version 3.10.3 breaks backward-compatibility with 3.10.2 ")
{
nlohmann::detail::std_fs::path text_path("/tmp/text.txt");
const json j(text_path);
const auto j_path = j.get<nlohmann::detail::std_fs::path>();
CHECK(j_path == text_path);
#if DOCTEST_CLANG || DOCTEST_GCC >= DOCTEST_COMPILER(8, 4, 0)
// only known to work on Clang and GCC >=8.4
CHECK_THROWS_WITH_AS(nlohmann::detail::std_fs::path(json(1)), "[json.exception.type_error.302] type must be string, but is number", json::type_error);
#endif
}
#endif
SECTION("issue #3077 - explicit constructor with default does not compile")
{
json j;
j[0]["value"] = true;
std::vector<FooBar> foo;
j.get_to(foo);
}
SECTION("issue #3108 - ordered_json doesn't support range based erase")
{
ordered_json j = {1, 2, 2, 4};
auto last = std::unique(j.begin(), j.end());
j.erase(last, j.end());
CHECK(j.dump() == "[1,2,4]");
j.erase(std::remove_if(j.begin(), j.end(), [](const ordered_json & val)
{
return val == 2;
}), j.end());
CHECK(j.dump() == "[1,4]");
}
SECTION("issue #3343 - json and ordered_json are not interchangeable")
{
json::object_t jobj({ { "product", "one" } });
ordered_json::object_t ojobj({{"product", "one"}});
auto jit = jobj.begin();
auto ojit = ojobj.begin();
CHECK(jit->first == ojit->first);
CHECK(jit->second.get<std::string>() == ojit->second.get<std::string>());
}
SECTION("issue #3171 - if class is_constructible from std::string wrong from_json overload is being selected, compilation failed")
{
const json j{{ "str", "value"}};
// failed with: error: no match for ‘operator=’ (operand types are ‘for_3171_derived’ and ‘const nlohmann::basic_json<>::string_t’
// {aka ‘const std::__cxx11::basic_string<char>’})
// s = *j.template get_ptr<const typename BasicJsonType::string_t*>();
auto td = j.get<for_3171_derived>();
CHECK(td.str == "value");
}
#ifdef JSON_HAS_CPP_20
SECTION("issue #3312 - Parse to custom class from unordered_json breaks on G++11.2.0 with C++20")
{
// see test for #3171
const ordered_json j = {{"name", "class"}};
for_3312 obj{};
j.get_to(obj);
CHECK(obj.name == "class");
}
#endif
#if defined(JSON_HAS_CPP_17) && JSON_USE_IMPLICIT_CONVERSIONS
SECTION("issue #3428 - Error occurred when converting nlohmann::json to std::any")
{
const json j;
const std::any a1 = j;
std::any&& a2 = j;
CHECK(a1.type() == typeid(j));
CHECK(a2.type() == typeid(j));
}
#endif
SECTION("issue #3204 - ambiguous regression")
{
const for_3204_bar bar_from_foo([](for_3204_foo) noexcept {}); // NOLINT(performance-unnecessary-value-param)
const for_3204_bar bar_from_json([](json) noexcept {}); // NOLINT(performance-unnecessary-value-param)
CHECK(bar_from_foo.constructed_from == for_3204_bar::constructed_from_foo);
CHECK(bar_from_json.constructed_from == for_3204_bar::constructed_from_json);
}
SECTION("issue #3333 - Ambiguous conversion from nlohmann::basic_json<> to custom class")
{
const json j
{
{"x", 1},
{"y", 2}
};
const for_3333 p = j;
CHECK(p.x == 1);
CHECK(p.y == 2);
}
SECTION("issue #3810 - ordered_json doesn't support construction from C array of custom type")
{
Example_3810 states[45]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
// fix "not used" warning
states[0].bla = 1;
const auto* const expected = R"([{"bla":1},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0}])";
// This works:
nlohmann::json j;
j["test"] = states;
CHECK(j["test"].dump() == expected);
// This doesn't compile:
nlohmann::ordered_json oj;
oj["test"] = states;
CHECK(oj["test"].dump() == expected);
}
#ifdef JSON_HAS_CPP_17
SECTION("issue #4740 - build issue with std::optional")
{
const auto t1 = Example_4740();
const auto j1 = nlohmann::json(t1);
CHECK(j1.dump() == "{\"host\":null,\"port\":null}");
const auto t2 = j1.get<Example_4740>();
CHECK(!t2.host.has_value());
CHECK(!t2.port.has_value());
// improve coverage
auto t3 = Example_4740();
t3.port = 80;
t3.host = "example.com";
const auto j2 = nlohmann::json(t3);
CHECK(j2.dump() == "{\"host\":\"example.com\",\"port\":80}");
const auto t4 = j2.get<Example_4740>();
CHECK(t4.host.has_value());
CHECK(t4.port.has_value());
}
#endif
#if !defined(_MSVC_LANG)
// MSVC returns garbage on invalid enum values, so this test is excluded
// there.
SECTION("issue #4762 - json exception 302 with unhelpful explanation : type must be number, but is number")
{
// In #4762, the main issue was that a json object with an invalid type
// returned "number" as type_name(), because this was the default case.
// This test makes sure we now return "invalid" instead.
json j;
j.m_data.m_type = static_cast<json::value_t>(100); // NOLINT(clang-analyzer-optin.core.EnumCastOutOfRange)
CHECK(j.type_name() == "invalid");
}
#endif
#ifdef JSON_HAS_CPP_17
SECTION("issue #4804: from_cbor incompatible with std::vector<std::byte> as binary_t")
{
const std::vector<std::uint8_t> data = {0x80};
const auto decoded = json_4804::from_cbor(data);
CHECK((decoded == json_4804::array()));
}
#ifndef SKIP_TESTS_FOR_ENUM_SERIALIZATION
SECTION("discussion #4209 - custom BinaryType direct assignment and round-tripping")
{
// Test that assigning a custom BinaryType directly creates a binary value, not an array
const std::vector<std::byte> original{std::byte{1}, std::byte{2}, std::byte{3}};
const json_4804 j = original;
CHECK(j.is_binary());
CHECK(!j.is_array());
// Test round-tripping: extracting the binary value back as the custom container type
const auto extracted = j.get<std::vector<std::byte>>();
CHECK(extracted == original);
// Test that the default json alias behavior is unchanged: std::vector<uint8_t> -> array
const json default_json = std::vector<std::uint8_t> {1, 2, 3};
CHECK(default_json.is_array());
CHECK(!default_json.is_binary());
}
SECTION("discussion #4209 - custom BinaryType extraction from parsed array")
{
// Test that extracting a custom BinaryType from a parsed JSON array still works
// (not just from a binary-typed node)
const auto j = json_4804::parse("[1,2,3]");
CHECK(j.is_array());
CHECK(!j.is_binary());
// Extracting as custom BinaryType should work from arrays
const auto extracted = j.get<std::vector<std::byte>>();
CHECK(extracted.size() == 3);
CHECK(extracted[0] == std::byte{1});
CHECK(extracted[1] == std::byte{2});
CHECK(extracted[2] == std::byte{3});
}
#endif
SECTION("issue #5046 - implicit conversion of return json to std::optional no longer implicit")
{
const json jval{};
auto GetValue = [](const json & valRoot) -> std::optional<json>
{
if (valRoot.contains("default"))
{
return valRoot.at("default");
}
return std::nullopt;
};
auto result = GetValue(jval);
CHECK(!result.has_value());
}
#endif
#if JSON_HAS_RANGES == 1
SECTION("issue #4440 - assert when using std::views::filter and GCC 10")
{
auto noOpFilter = std::views::filter([](auto&&) noexcept
{
return true;
});
json j = {1, 2, 3};
auto filtered = j | noOpFilter;
CHECK(*filtered.begin() == 1);
}
#endif
#if JSON_HAS_RANGES && !defined(__MINGW32__)
SECTION("issue #4916 - constructing array from C++20 ranges view does not work")
{
std::vector<int> nums{1, 2, 37, 42, 21};
auto filteredNums = nums | std::views::filter([](int i)
{
return i > 10;
});
json const j(filteredNums);
CHECK(j.type() == json::value_t::array);
CHECK(j == json({37, 42, 21}));
}
#endif
// owning_view is not available in libstdc++ < 12
#if JSON_HAS_RANGES && !defined(__MINGW32__) && !(defined(__GLIBCXX__) && _GLIBCXX_RELEASE < 12)
SECTION("issue #4916 - constructing array from prvalue C++20 ranges view (owning_view)")
{
json const j(std::vector<int> {1, 2, 37, 42, 21} | std::views::filter([](int i)
{
return i > 10;
}));
CHECK(j.type() == json::value_t::array);
CHECK(j == json({37, 42, 21}));
}
#endif
#if JSON_HAS_RANGES && !defined(__MINGW32__)
SECTION("issue #4916 - constructing array from C++20 transform view (prvalue elements)")
{
std::vector<int> nums{1, 2, 3};
auto t = nums | std::views::transform([](int i) noexcept
{
return i * 2;
});
json const j(t);
CHECK(j.type() == json::value_t::array);
CHECK(j == json({2, 4, 6}));
}
#endif
}
TEST_CASE_TEMPLATE("issue #4798 - nlohmann::json::to_msgpack() encode float NaN as double", T, double, float) // NOLINT(readability-math-missing-parentheses, bugprone-throwing-static-initialization)
{
// With issue #4798, we encode NaN, infinity, and -infinity as float instead
// of double to allow for smaller encodings.
const json jx = std::numeric_limits<T>::quiet_NaN();
const json jy = std::numeric_limits<T>::infinity();
const json jz = -std::numeric_limits<T>::infinity();
/////////////////////////////////////////////////////////////////////////
// MessagePack
/////////////////////////////////////////////////////////////////////////
// expected MessagePack values
const std::vector<std::uint8_t> msgpack_x = {{0xCA, 0x7F, 0xC0, 0x00, 0x00}};
const std::vector<std::uint8_t> msgpack_y = {{0xCA, 0x7F, 0x80, 0x00, 0x00}};
const std::vector<std::uint8_t> msgpack_z = {{0xCA, 0xFF, 0x80, 0x00, 0x00}};
CHECK(json::to_msgpack(jx) == msgpack_x);
CHECK(json::to_msgpack(jy) == msgpack_y);
CHECK(json::to_msgpack(jz) == msgpack_z);
CHECK(std::isnan(json::from_msgpack(msgpack_x).get<T>()));
CHECK(json::from_msgpack(msgpack_y).get<T>() == std::numeric_limits<T>::infinity());
CHECK(json::from_msgpack(msgpack_z).get<T>() == -std::numeric_limits<T>::infinity());
// Make sure the other MessagePakc encodings for NaN, infinity, and
// -infinity are still supported.
const std::vector<std::uint8_t> msgpack_x_2 = {{0xCB, 0x7F, 0xF8, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}};
const std::vector<std::uint8_t> msgpack_y_2 = {{0xCB, 0x7F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}};
const std::vector<std::uint8_t> msgpack_z_2 = {{0xCB, 0xFF, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}};
CHECK(std::isnan(json::from_msgpack(msgpack_x_2).get<T>()));
CHECK(json::from_msgpack(msgpack_y_2).get<T>() == std::numeric_limits<T>::infinity());
CHECK(json::from_msgpack(msgpack_z_2).get<T>() == -std::numeric_limits<T>::infinity());
/////////////////////////////////////////////////////////////////////////
// CBOR
/////////////////////////////////////////////////////////////////////////
// expected CBOR values
const std::vector<std::uint8_t> cbor_x = {{0xF9, 0x7E, 0x00}};
const std::vector<std::uint8_t> cbor_y = {{0xF9, 0x7C, 0x00}};
const std::vector<std::uint8_t> cbor_z = {{0xF9, 0xfC, 0x00}};
CHECK(json::to_cbor(jx) == cbor_x);
CHECK(json::to_cbor(jy) == cbor_y);
CHECK(json::to_cbor(jz) == cbor_z);
CHECK(std::isnan(json::from_cbor(cbor_x).get<T>()));
CHECK(json::from_cbor(cbor_y).get<T>() == std::numeric_limits<T>::infinity());
CHECK(json::from_cbor(cbor_z).get<T>() == -std::numeric_limits<T>::infinity());
// Make sure the other CBOR encodings for NaN, infinity, and -infinity are
// still supported.
const std::vector<std::uint8_t> cbor_x_2 = {{0xFA, 0x7F, 0xC0, 0x00, 0x00}};
const std::vector<std::uint8_t> cbor_y_2 = {{0xFA, 0x7F, 0x80, 0x00, 0x00}};
const std::vector<std::uint8_t> cbor_z_2 = {{0xFA, 0xFF, 0x80, 0x00, 0x00}};
const std::vector<std::uint8_t> cbor_x_3 = {{0xFB, 0x7F, 0xF8, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}};
const std::vector<std::uint8_t> cbor_y_3 = {{0xFB, 0x7F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}};
const std::vector<std::uint8_t> cbor_z_3 = {{0xFB, 0xFF, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}};
CHECK(std::isnan(json::from_cbor(cbor_x_2).get<T>()));
CHECK(json::from_cbor(cbor_y_2).get<T>() == std::numeric_limits<T>::infinity());
CHECK(json::from_cbor(cbor_z_2).get<T>() == -std::numeric_limits<T>::infinity());
CHECK(std::isnan(json::from_cbor(cbor_x_3).get<T>()));
CHECK(json::from_cbor(cbor_y_3).get<T>() == std::numeric_limits<T>::infinity());
CHECK(json::from_cbor(cbor_z_3).get<T>() == -std::numeric_limits<T>::infinity());
}
TEST_CASE("regression test #5074 - portable workaround for single-element brace init")
{
json const j_obj = {{"key", "value"}};
json const j = json::array({j_obj});
CHECK(j.is_array());
CHECK(j.size() == 1);
CHECK(j[0] == j_obj);
}
struct Example_5122
{
float b = 2;
nlohmann::ordered_map<std::string, std::string> c{}; // NOLINT(readability-redundant-member-init): needed for GCC -Weffc++
int a = 1;
NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(Example_5122, b, c, a)
};
TEST_CASE("regression test #5122 - from_json into types holding nlohmann::ordered_map")
{
Example_5122 src;
src.c.emplace("first", "1");
src.c.emplace("second", "2");
ordered_json const j = src;
Example_5122 const dst = j.get<Example_5122>();
CHECK(dst.b == src.b);
CHECK(dst.a == src.a);
REQUIRE(dst.c.size() == src.c.size());
auto src_it = src.c.begin();
auto dst_it = dst.c.begin();
for (; src_it != src.c.end(); ++src_it, ++dst_it)
{
CHECK(dst_it->first == src_it->first);
CHECK(dst_it->second == src_it->second);
}
}
// -Wself-assign-overloaded was introduced in Clang 7. Gate the pragma on
// __has_warning so older Clang versions do not error with "unknown warning
// group". The __has_warning check has to stay inside the __clang__ branch
// because GCC does not provide it and would tokenize-error on the argument.
#if defined(__clang__) && defined(__has_warning)
#if __has_warning("-Wself-assign-overloaded")
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
DOCTEST_CLANG_SUPPRESS_WARNING("-Wself-assign-overloaded")
#endif
#endif
TEST_CASE("regression test #5122 - nlohmann::ordered_map copy-assignment is self-assignment safe")
{
nlohmann::ordered_map<std::string, std::string> m;
m.emplace("first", "1");
m.emplace("second", "2");
// Insertion order is preserved by ordered_map, so we can check it directly.
m = m;
REQUIRE(m.size() == 2);
auto it = m.begin();
CHECK(it->first == "first");
CHECK(it->second == "1");
++it;
CHECK(it->first == "second");
CHECK(it->second == "2");
}
#if defined(__clang__) && defined(__has_warning)
#if __has_warning("-Wself-assign-overloaded")
DOCTEST_CLANG_SUPPRESS_WARNING_POP
#endif
#endif
TEST_CASE("regression test #5122 - nlohmann::ordered_map move-assignment transfers contents")
{
nlohmann::ordered_map<std::string, std::string> src;
src.emplace("first", "1");
src.emplace("second", "2");
nlohmann::ordered_map<std::string, std::string> dst;
dst.emplace("stale", "x");
dst = std::move(src);
REQUIRE(dst.size() == 2);
auto it = dst.begin();
CHECK(it->first == "first");
CHECK(it->second == "1");
++it;
CHECK(it->first == "second");
CHECK(it->second == "2");
// Re-assigning into the moved-from object must leave it in a usable state.
src = nlohmann::ordered_map<std::string, std::string> {};
src.emplace("after-move", "3");
REQUIRE(src.size() == 1);
CHECK(src.begin()->first == "after-move");
}
// Stand-in for a third-party library (e.g., Eigen as of 3.4, which added
// STL-compatible begin()/end() to its vector types), living in its own
// namespace with its own to_json overload for its vector type.
namespace issue_4320_eigen
{
// "array-compatible" from the library's point of view (it has begin()/end()),
// but for which this (fake) third-party namespace provides its own to_json.
struct vector3
{
double v[3]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays,cppcoreguidelines-use-default-member-init,modernize-use-default-member-init)
vector3(double x, double y, double z) : v{x, y, z} {} // NOLINT(hicpp-member-init,cppcoreguidelines-pro-type-member-init)
double x() const
{
return v[0];
}
double y() const
{
return v[1];
}
double z() const
{
return v[2];
}
double* begin()
{
return v;
}
double* end()
{
return v + 3;
}
const double* begin() const
{
return v;
}
const double* end() const
{
return v + 3;
}
};
inline void to_json(json& j, const vector3& v) // NOLINT(misc-use-internal-linkage)
{
j = {{"x", v.x()}, {"y", v.y()}, {"z", v.z()}};
}
} // namespace issue_4320_eigen
// The user's own namespace, using the (fake) Eigen type as an implementation
// detail behind a payload type that has nothing to do with vectors/arrays.
namespace issue_4320
{
// Publicly derives from issue_4320_eigen::vector3 but does *not* define its
// own to_json - it is only ever used as a temporary to reach the base
// class's to_json via ADL.
struct vector3_wrapper : issue_4320_eigen::vector3
{
using issue_4320_eigen::vector3::vector3;
};
struct payload
{
double x, y, z;
};
inline vector3_wrapper to_eigen(const payload& p) // NOLINT(misc-use-internal-linkage)
{
return {p.x, p.y, p.z};
}
inline void to_json(json& j, const payload& p) // NOLINT(misc-use-internal-linkage)
{
// Unqualified call, passing a *derived* vector3_wrapper: relies on ADL
// finding issue_4320_eigen::to_json(json&, const vector3&) through the
// vector3 base class, via a derived-to-base conversion. Must NOT resolve
// to the library's own generic array-compatible to_json (an exact-match
// template for vector3_wrapper, since it also has begin()/end()), which
// would serialize this as [x, y, z] instead of {"x":x, "y":y, "z":z}.
to_json(j, to_eigen(p));
}
} // namespace issue_4320
TEST_CASE("issue #4320 - custom base class must not leak nlohmann::detail into ADL")
{
// Before the fix, basic_json unconditionally derived from a type living in
// nlohmann::detail (json_default_base), which made nlohmann::detail an
// associated namespace of every basic_json for ADL purposes. That leaked
// the library's internal generic-array to_json overload into unqualified
// to_json() calls made from user code, silently bypassing user-defined
// to_json overloads reached via a derived-to-base conversion.
const issue_4320::payload p{1.0, 2.0, 3.0};
json j;
to_json(j, p);
CHECK(j == json({{"x", 1.0}, {"y", 2.0}, {"z", 3.0}}));
}
TEST_CASE("issue #5338 - truncated CBOR tagged binary subtype is rejected")
{
const std::vector<std::vector<std::uint8_t>> truncated_tags =
{
{0xD8},
{0xD9, 0x00},
{0xDA, 0x00, 0x00, 0x00},
{0xDB, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}
};
for (const auto& data : truncated_tags)
{
CAPTURE(data);
for (const auto tag_handler :
{
json::cbor_tag_handler_t::ignore, json::cbor_tag_handler_t::store
})
{
CAPTURE(tag_handler);
const auto result = json::from_cbor(data, true, false, tag_handler);
CHECK(result.is_discarded());
}
}
}
TEST_CASE("issue #5402 - update(merge_objects=true) overwrites a primitive with an object")
{
json t = {{"k", 1}};
t.update(json{{"k", {{"x", 2}}}}, true);
CHECK(t == json({{"k", {{"x", 2}}}}));
json mixed = {{"keep", {{"a", 1}}}, {"replace", 1}};
mixed.update(json{{"keep", {{"b", 2}}}, {"replace", {{"x", 2}}}}, true);
CHECK(mixed == json({{"keep", {{"a", 1}, {"b", 2}}}, {"replace", {{"x", 2}}}}));
}
TEST_CASE("regression test #5476 - array type without reserve()")
{
// the capacity reserved for definite-length arrays must not require the
// array type to have a reserve() member function
using deque_json = nlohmann::basic_json<std::map, std::deque>;
SECTION("std::deque")
{
const auto j = deque_json::parse(R"({"a":[1,[2,3]],"b":[]})");
CHECK(j.dump() == R"({"a":[1,[2,3]],"b":[]})");
// the binary formats pass a definite length to start_array()
CHECK(deque_json::from_cbor(deque_json::to_cbor(j)) == j);
CHECK(deque_json::from_msgpack(deque_json::to_msgpack(j)) == j);
// parse() instantiates the callback parser as well, which reserves too
const auto with_callback = deque_json::parse(R"([1,2,3])", [](int /*depth*/, deque_json::parse_event_t /*event*/, deque_json& /*parsed*/) noexcept
{
return true;
});
CHECK(with_callback == deque_json({1, 2, 3}));
}
SECTION("std::vector still reserves")
{
json array = json::array();
for (int i = 0; i < 100; ++i)
{
array.push_back(i);
}
const auto j = json::from_cbor(json::to_cbor(array));
CHECK(j == array);
CHECK(j.get_ref<const json::array_t&>().capacity() >= 100);
}
SECTION("the reservation stays capped")
{
// CBOR array announcing 2^32-1 elements, but truncated right after the
// header: the input must be rejected without reserving that capacity
const std::vector<std::uint8_t> truncated = {0x9A, 0xFF, 0xFF, 0xFF, 0xFF};
CHECK(json::from_cbor(truncated, true, false).is_discarded());
}
}
DOCTEST_CLANG_SUPPRESS_WARNING_POP
+253 -6
View File
@@ -15,6 +15,8 @@ using nlohmann::json;
#include <sstream>
#include <iomanip>
#include "test_utils.hpp"
TEST_CASE("serialization")
{
SECTION("operator<<")
@@ -84,8 +86,9 @@ TEST_CASE("serialization")
{
const json j = "ä\xA9ü";
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&);
CHECK_THROWS_WITH_AS(j.dump(1, ' ', false, json::error_handler_t::strict), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&);
CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump(1, ' ', false, json::error_handler_t::strict)), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&);
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"äü\"");
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"ä\xEF\xBF\xBDü\"");
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"\\u00e4\\ufffd\\u00fc\"");
@@ -95,8 +98,9 @@ TEST_CASE("serialization")
{
const json j = "123\xC2";
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] incomplete UTF-8 string; last byte: 0xC2", json::type_error&);
CHECK_THROWS_AS(j.dump(1, ' ', false, json::error_handler_t::strict), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] incomplete UTF-8 string; last byte: 0xC2", json::type_error&);
CHECK_THROWS_AS(utils::ignore_return_value(j.dump(1, ' ', false, json::error_handler_t::strict)), json::type_error&);
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123\"");
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\"");
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd\"");
@@ -106,8 +110,9 @@ TEST_CASE("serialization")
{
const json j = "123\xF1\xB0\x34\x35\x36";
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0x34", json::type_error&);
CHECK_THROWS_AS(j.dump(1, ' ', false, json::error_handler_t::strict), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0x34", json::type_error&);
CHECK_THROWS_AS(utils::ignore_return_value(j.dump(1, ' ', false, json::error_handler_t::strict)), json::type_error&);
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123456\"");
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\x34\x35\x36\"");
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd456\"");
@@ -382,3 +387,245 @@ TEST_CASE("dump for basic_json with long double number_float_t")
check_same(100.0L, 100.0);
}
}
TEST_CASE("serialization of strings (bulk fast path)")
{
// These cases exercise the SWAR bulk-copy fast path in dump_escaped and the
// internal write buffer: long runs, escapes interrupting runs, 0x7F/DEL,
// multibyte UTF-8 under both ensure_ascii settings, and payloads larger than
// the write buffer.
SECTION("long unescaped ASCII exceeds the write buffer")
{
const std::string big(3000, 'a');
const json j = big;
CHECK(j.dump() == '"' + big + '"');
CHECK(j.dump(-1, ' ', true) == '"' + big + '"');
// round-trips
CHECK(json::parse(j.dump()) == j);
}
SECTION("runs interrupted by escapes")
{
const json j = std::string(500, 'x') + "\n\"\\" + std::string(500, 'y');
const std::string out = j.dump();
CHECK(out == '"' + std::string(500, 'x') + "\\n\\\"\\\\" + std::string(500, 'y') + '"');
CHECK(json::parse(out) == j);
}
SECTION("DEL (0x7F) depends on ensure_ascii")
{
const json j = std::string("a\x7f" "b");
CHECK(j.dump(-1, ' ', false) == "\"a\x7f" "b\""); // copied verbatim
CHECK(j.dump(-1, ' ', true) == "\"a\\u007fb\""); // escaped
}
SECTION("multibyte UTF-8 under both ensure_ascii settings")
{
const json j = std::string("A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z"); // A é 你 😀 Z
// not escaping non-ASCII: bytes are copied through the bulk validator
CHECK(j.dump(-1, ' ', false) == "\"A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z\"");
// ensure_ascii: escaped (with a surrogate pair for the emoji)
CHECK(j.dump(-1, ' ', true) == "\"A\\u00e9\\u4f60\\ud83d\\ude00Z\"");
CHECK(json::parse(j.dump(-1, ' ', true)) == j);
}
SECTION("many small structural writes exceed the write buffer")
{
json arr = json::array();
for (int i = 0; i < 2000; ++i)
{
arr.push_back(i);
}
const std::string out = arr.dump();
CHECK(out.front() == '[');
CHECK(out.back() == ']');
CHECK(json::parse(out) == arr);
json obj = json::object();
for (int i = 0; i < 500; ++i)
{
obj["key" + std::to_string(i)] = i;
}
CHECK(json::parse(obj.dump()) == obj);
CHECK(json::parse(obj.dump(2)) == obj);
// an array of many empty strings emits a long run of single-character
// writes ('"', '"', ',') at shallow nesting depth, so the write buffer
// fills and flushes mid-run without the deep recursion that would
// overflow the stack on some debug builds
json many_empty = json::array();
for (int i = 0; i < 500; ++i)
{
many_empty.push_back("");
}
const std::string out2 = many_empty.dump();
CHECK(out2.size() > 1024); // spans multiple write-buffer flushes
CHECK(out2.front() == '[');
CHECK(out2.back() == ']');
CHECK(json::parse(out2) == many_empty);
}
SECTION("invalid UTF-8 handling is unaffected by the fast path")
{
const json j = std::string("valid\xff" "more");
CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0xFF", json::type_error&);
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"valid\xef\xbf\xbd" "more\"");
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"valid\\ufffdmore\"");
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"validmore\"");
}
}
TEST_CASE("indentation is written straight into the write buffer")
{
// put_indent() memsets the indentation into the write buffer instead of
// copying it out of a pre-grown indentation string. These cases cover an
// indentation wider than the buffer, a non-space indentation character, and
// nesting deep enough that the accumulated indentation spans several
// buffer-fulls - the situations the old grow-a-string approach got wrong.
SECTION("indent_step wider than the write buffer")
{
const json j = {{"a", 1}};
// 2000 > the 1024-byte write buffer, and > the 512 the indentation
// string used to start at
CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"a\": 1\n}");
// several whole buffer-fulls, so the buffer is refilled once and then
// flushed repeatedly
CHECK(j.dump(5000) == "{\n" + std::string(5000, ' ') + "\"a\": 1\n}");
CHECK(j.dump(5000, '\t') == "{\n" + std::string(5000, '\t') + "\"a\": 1\n}");
// an exact multiple of the buffer size
CHECK(j.dump(4096) == "{\n" + std::string(4096, ' ') + "\"a\": 1\n}");
}
SECTION("a non-space indentation character is used throughout")
{
const json j = {{"a", 1}};
// 600 is past the point where the indentation used to be grown, which
// is where a hard-coded space would have shown up
CHECK(j.dump(600, '\t') == "{\n" + std::string(600, '\t') + "\"a\": 1\n}");
CHECK(j.dump(3, '.') == "{\n...\"a\": 1\n}");
}
SECTION("accumulated indentation spans several buffer-fulls")
{
// five levels deep at 400 per level: the innermost value is indented by
// 2000 characters, reached in steps that each straddle the buffer end
json j = json::array({1});
for (int i = 0; i < 4; ++i)
{
j = json::array({j});
}
const std::string out = j.dump(400);
CHECK(out.find(std::string("\n") + std::string(2000, ' ') + "1\n") != std::string::npos);
CHECK(json::parse(out) == j);
}
SECTION("binary values are indented the same way")
{
// a binary value is serialized as an object with "bytes" and
// "subtype" keys; the byte array itself is always written compactly
// (see dump_byte()), so only the surrounding object's indentation
// goes through put_indent()
const json j = json::binary({1, 2, 3}, 128);
CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"bytes\": [1, 2, 3],\n"
+ std::string(2000, ' ') + "\"subtype\": 128\n}");
CHECK(j.dump(2000, '\t') == "{\n" + std::string(2000, '\t') + "\"bytes\": [1, 2, 3],\n"
+ std::string(2000, '\t') + "\"subtype\": 128\n}");
}
SECTION("indentation is unchanged for ordinary widths")
{
const json j = {{"a", {1, 2}}, {"b", nullptr}};
CHECK(j.dump(2) == "{\n \"a\": [\n 1,\n 2\n ],\n \"b\": null\n}");
CHECK(j.dump(0) == "{\n\"a\": [\n1,\n2\n],\n\"b\": null\n}");
}
}
TEST_CASE("serialization of deeply nested values")
{
// dump() descends into a bounded number of levels and writes out whatever
// is nested deeper than that without the call stack; see
// https://github.com/nlohmann/json/issues/5387
SECTION("nested deeper than the call stack could follow")
{
// parsing is iterative, so building these costs little
const std::size_t depth = 100000;
const std::string array_text = std::string(depth, '[') + '0' + std::string(depth, ']');
CHECK(json::parse(array_text).dump() == array_text);
std::string object_text;
object_text.reserve((6 * depth) + 1);
for (std::size_t i = 0; i < depth; ++i)
{
object_text += "{\"a\":";
}
object_text += '1';
object_text.append(depth, '}');
CHECK(json::parse(object_text).dump() == object_text);
}
SECTION("depths around the bound of the descent")
{
// Cover every depth around the bound, so that the two ways of writing a
// value are known to meet cleanly - wherever the bound is set.
for (std::size_t d = 1; d <= 300; ++d)
{
CAPTURE(d);
const std::string array_text = std::string(d, '[') + '7' + std::string(d, ']');
CHECK(json::parse(array_text).dump() == array_text);
std::string object_text;
for (std::size_t i = 0; i < d; ++i)
{
object_text += "{\"k\":";
}
object_text += '7';
object_text.append(d, '}');
CHECK(json::parse(object_text).dump() == object_text);
}
}
SECTION("pretty-printing across the bound")
{
for (std::size_t d = 120; d <= 140; ++d)
{
CAPTURE(d);
const json j = json::parse(std::string(d, '[') + '7' + std::string(d, ']'));
std::string expected;
for (std::size_t i = 0; i < d; ++i)
{
expected += std::string(2 * i, ' ') + "[\n";
}
expected += std::string(2 * d, ' ') + '7';
for (std::size_t i = d; i > 0; --i)
{
expected += '\n' + std::string(2 * (i - 1), ' ') + ']';
}
CHECK(j.dump(2) == expected);
}
}
SECTION("an empty container below the bound")
{
// an empty container is written out in full and never descended into,
// so it must not gain a newline when it is reached iteratively
for (std::size_t d = 125; d <= 135; ++d)
{
CAPTURE(d);
const std::string compact = std::string(d, '[') + "[]" + std::string(d, ']');
CHECK(json::parse(compact).dump() == compact);
const std::string with_object = std::string(d, '[') + "{}" + std::string(d, ']');
CHECK(json::parse(with_object).dump() == with_object);
}
}
}
+30
View File
@@ -17,6 +17,7 @@
#include <nlohmann/json.hpp>
using json = nlohmann::json;
using ordered_json = nlohmann::ordered_json;
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
#if JSON_HAS_STD_FORMAT
@@ -52,6 +53,23 @@ TEST_CASE("std::formatter<nlohmann::json>")
CHECK(std::format("{:2}", j) == j.dump(2));
CHECK(std::format("{:#2}", j) == j.dump(2));
CHECK(std::format("{:8}", j) == j.dump(8));
// multi-digit widths must accumulate every digit, not just the first
CHECK(std::format("{:12}", j) == j.dump(12));
CHECK(std::format("{:#12}", j) == j.dump(12));
CHECK(std::format("{:10}", j) == j.dump(10));
}
SECTION("bare alignment with no fill character defaults to a space indent character")
{
const json j = {{"foo", 1}, {"bar", {1, 2, 3}}};
// without a preceding fill character, the alignment character itself must not
// be mistaken for the indent character -- the default space is kept
CHECK(std::format("{:<}", j) == j.dump());
CHECK(std::format("{:>}", j) == j.dump());
CHECK(std::format("{:^}", j) == j.dump());
CHECK(std::format("{:<3}", j) == j.dump(3, ' '));
CHECK(std::format("{:>3}", j) == j.dump(3, ' '));
CHECK(std::format("{:^3}", j) == j.dump(3, ' '));
}
SECTION("fill-and-align sets the indent character, like dump(indent, indent_char)")
@@ -93,4 +111,16 @@ TEST_CASE("std::formatter<nlohmann::json>")
}
}
TEST_CASE("std::formatter<nlohmann::ordered_json>")
{
// spot-check a non-default basic_json instantiation, since the formatter
// is written against the generic NLOHMANN_BASIC_JSON_TPL_DECLARATION
// template and must actually instantiate (and behave correctly) for
// template arguments other than nlohmann::json
const ordered_json j = {{"foo", 1}, {"bar", {1, 2, 3}}};
CHECK(std::format("{}", j) == j.dump());
CHECK(std::format("{:#}", j) == j.dump(4));
CHECK(std::format("{:2}", j) == j.dump(2));
}
#endif
+362
View File
@@ -15,6 +15,7 @@ using nlohmann::json;
#include <fstream>
#include <set>
#include "make_test_data_available.hpp"
#include "round_trip_corpus.hpp"
#include "test_utils.hpp"
namespace
@@ -2149,6 +2150,280 @@ TEST_CASE("UBJSON")
}
}
TEST_CASE("UBJSON nesting does not consume the call stack")
{
// Containers used to be read by calling back into the value reader once
// per element, so the native call stack grew with the nesting depth of the
// input. '[' alone opens a container, so a payload of repeated '[' crashed
// the process (#5104), as did the optimized forms, which reach the same
// path through a type or size annotation. The containers are kept on a
// heap stack now.
//
// Deeply nested values must not be compared, copied or dumped here: those
// operations are still recursive and would reintroduce the crash.
json _;
SECTION("containers that end at a marker")
{
const std::vector<uint8_t> input(500000, '[');
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(input), "[json.exception.parse_error.110] parse error at byte 500001: syntax error while parsing UBJSON value: unexpected end of input", json::parse_error&);
CHECK(json::from_ubjson(input, true, false).is_discarded());
}
SECTION("containers with a size")
{
std::vector<uint8_t> input;
for (std::size_t i = 0; i < 100000; ++i)
{
input.push_back('[');
input.push_back('#');
input.push_back('i');
input.push_back(1);
}
CHECK_THROWS_AS(_ = json::from_ubjson(input), json::parse_error&);
CHECK(json::from_ubjson(input, true, false).is_discarded());
}
SECTION("containers with a type and a size")
{
// '[' is a permitted optimized type in UBJSON, so each element of such
// a container is itself a container, read without a marker of its own
std::vector<uint8_t> input;
for (std::size_t i = 0; i < 100000; ++i)
{
const std::vector<uint8_t> level = {'[', '$', '[', '#', 'i', 1};
input.insert(input.end(), level.begin(), level.end());
}
CHECK_THROWS_AS(_ = json::from_ubjson(input), json::parse_error&);
CHECK(json::from_ubjson(input, true, false).is_discarded());
}
SECTION("a well-formed deep value is read through the SAX interface")
{
std::vector<uint8_t> input(100000, '[');
input.insert(input.end(), 100000, ']');
SaxCountdown accept_all(1000000);
CHECK(json::sax_parse(input, &accept_all, json::input_format_t::ubjson));
}
SECTION("a well-formed deep value is read into a value")
{
const std::size_t depth = 10000;
std::vector<uint8_t> input(depth, '[');
input.insert(input.end(), depth, ']');
json j = json::from_ubjson(input);
std::size_t measured = 0;
const json* p = &j;
while (p->is_array() && !p->empty())
{
p = &p->front();
++measured;
}
// the innermost array is empty, so the descent stops one level short
CHECK(measured == depth - 1);
}
SECTION("containers are still read the same way")
{
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', ']'})) == json::array());
CHECK(json::from_ubjson(std::vector<uint8_t>({'{', '}'})) == json::object());
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '#', 'i', 0})) == json::array());
CHECK(json::from_ubjson(std::vector<uint8_t>({'{', '#', 'i', 0})) == json::object());
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '$', 'i', '#', 'i', 2, 1, 2})) == json({1, 2}));
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '#', 'i', 2, 'i', 1, 'i', 2})) == json({1, 2}));
CHECK(json::from_ubjson(std::vector<uint8_t>({'{', '$', 'i', '#', 'i', 1, 'i', 1, 'a', 1})) == json({{"a", 1}}));
// a no-op is not a value, so a container of them holds none
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '$', 'N', '#', 'i', 2})) == json::array());
// sized and unsized forms nested inside one another
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '[', '#', 'i', 2, 'i', 1, 'i', 2, ']'})) == json({{1, 2}}));
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '#', 'i', 1, '[', 'i', 1, ']'})) == json({{1}}));
// an optimized container of containers
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '$', '[', '#', 'i', 2, 'i', 1, ']', 'i', 2, ']'})) == json({{1}, {2}}));
}
SECTION("BJData containers are still read the same way")
{
// the ND-array wrapper and the binary shortcut are complete values,
// not containers the reader descends into
CHECK(json::from_bjdata(std::vector<uint8_t>({'[', '$', 'U', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6})) ==
json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}));
CHECK(json::from_bjdata(std::vector<uint8_t>({'[', '$', 'i', '#', 'i', 2, 1, 2})) == json({1, 2}));
CHECK(json::from_bjdata(std::vector<uint8_t>({'[', '[', 'i', 1, ']', ']'})) == json({{1}}));
}
}
TEST_CASE("UBJSON optimized arrays of a valueless type are bounded")
{
// An element of type 'Z', 'T' or 'F' is encoded by its marker alone, so an
// optimized array of one of those has no payload and the declared count is
// the only thing deciding how much is allocated. Ten bytes used to produce
// billions of values (#2793); every other type costs at least one byte per
// element and is bounded by the end of the input.
json _;
SECTION("an excessive count is rejected")
{
// 'l' is a big-endian int32: 0x7FFFFFFF elements, about 34 GB of value;
// OSS-Fuzz reported this shape as a parse_ubjson_fuzzer timeout
// (testcase 6347769435193344, no issue filed)
for (const auto marker :
{'Z', 'T', 'F'
})
{
const std::vector<uint8_t> input = {'[', '$', static_cast<uint8_t>(marker), '#', 'l', 0x7F, 0xFF, 0xFF, 0xFF};
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(input), "[json.exception.out_of_range.408] syntax error while parsing UBJSON size: excessive array size", json::out_of_range&);
CHECK(json::from_ubjson(input, true, false).is_discarded());
}
}
SECTION("ordinary counts are unaffected")
{
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '$', 'Z', '#', 'i', 3})) == json({nullptr, nullptr, nullptr}));
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '$', 'T', '#', 'i', 2})) == json({true, true}));
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '$', 'F', '#', 'i', 2})) == json({false, false}));
// 'N' is a no-op rather than a value, and still yields an empty array
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', '$', 'N', '#', 'i', 2})) == json::array());
}
SECTION("a type with a payload is unaffected")
{
// A count past the limit is not rejected for 'U', which costs a byte
// per element and is bounded by the end of the input instead. The
// count is kept just past the limit rather than made huge, because a
// count that also exceeds the array's max_size() is reported as
// out_of_range before the input runs out, and max_size() depends on
// the width of std::size_t.
const std::vector<uint8_t> input = {'[', '$', 'U', '#', 'l', 0x00, 0x10, 0x00, 0x01};
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(input), "[json.exception.parse_error.110] parse error at byte 10: syntax error while parsing UBJSON number: unexpected end of input", json::parse_error&);
CHECK(json::from_ubjson(input, true, false).is_discarded());
}
SECTION("the writer stays within what the reader accepts")
{
// below the limit the optimized form is used and is tiny; above it the
// writer falls back so that the result can still be read back
json const at_limit(1048576, nullptr);
const auto v_at_limit = json::to_ubjson(at_limit, true, true);
CHECK(v_at_limit.size() == 9);
CHECK(v_at_limit.at(1) == '$');
CHECK(json::from_ubjson(v_at_limit) == at_limit);
json const above_limit(1048577, nullptr);
const auto v_above_limit = json::to_ubjson(above_limit, true, true);
CHECK(v_above_limit.at(1) != '$');
CHECK(json::from_ubjson(v_above_limit) == above_limit);
}
}
TEST_CASE("issue #5405 - array reserve for definite-length UBJSON arrays")
{
#if !defined(JSON_NOEXCEPTION)
// this SECTION relies on catching a thrown exception to distinguish
// which of two acceptable, bounded rejections a hostile header took;
// under JSON_NOEXCEPTION, JSON_THROW never produces a catchable C++
// exception (it aborts instead), so this cannot be tested that way here
SECTION("a huge claimed length with no element data must not over-allocate")
{
// optimized form [$type#count: type 'i' (int8), count as a four-byte
// 'l' (int32) of 0x7FFFFFFF (2147483647), but no element data at all.
// max_size() for a std::vector is far larger than this count, so it
// does not reject the header outright; the (capped) reservation must
// not attempt to allocate space for billions of elements before the
// missing data is detected.
json _;
const std::vector<uint8_t> input = {'[', '$', 'i', '#', 'l', 0x7F, 0xFF, 0xFF, 0xFF};
// On a platform where std::vector<json>::max_size() is smaller than
// the claimed count (e.g. 32-bit, where max_size() is bounded by a
// 32-bit SIZE_MAX divided by sizeof(json)), the SAX consumer's own
// check rejects the header outright (out_of_range.408, with the
// claimed count in the message) instead of accepting it and only
// finding it short of data once the (capped) reservation looks for
// element bytes that were never provided (parse_error.110). Either
// is an acceptable, bounded rejection of the hostile header -- the
// property under test is that no path attempts to allocate space
// for billions of elements.
bool threw = false;
try
{
_ = json::from_ubjson(input);
}
catch (const json::parse_error& e)
{
threw = true;
CHECK(e.id == 110);
CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 10: syntax error while parsing UBJSON number: unexpected end of input");
}
catch (const json::out_of_range& e)
{
threw = true;
CHECK(e.id == 408);
CHECK(std::string(e.what()).find("excessive array size") != std::string::npos);
}
CHECK(threw);
// json_sax_dom_parser::start_array()'s max_size() check (unlike the
// scanner's own parse_error path) throws unconditionally via
// JSON_THROW rather than going through sax->parse_error(), so it is
// not gated by allow_exceptions=false on a platform where this
// header hits that check (e.g. 32-bit, see above) -- allow either
// a discarded result or the same out_of_range it throws with
// exceptions enabled.
try
{
CHECK(json::from_ubjson(input, true, false).is_discarded());
}
catch (const json::out_of_range& e)
{
CHECK(e.id == 408);
}
}
#endif
SECTION("arrays of various sizes decode to the same value as before the reserve optimization")
{
for (const auto size :
{
std::size_t{0}, std::size_t{1}, std::size_t{5}, // small
std::size_t{16384}, // exactly at the reserve cap
std::size_t{20000} // above the reserve cap
})
{
CAPTURE(size)
json j = json::array();
for (std::size_t i = 0; i < size; ++i)
{
j.push_back(static_cast<int>(i % 1000));
}
// exercise both the plain and the optimized [$type#count encoding
const auto packed_plain = json::to_ubjson(j);
CHECK(json::from_ubjson(packed_plain) == j);
const auto packed_optimized = json::to_ubjson(j, true, true);
CHECK(json::from_ubjson(packed_optimized) == j);
}
}
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
{
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
// a custom SAX consumer that does not touch a DOM array sees identical events
json j = json::array();
for (int i = 0; i < 100; ++i)
{
j.push_back(i);
}
const auto packed = json::to_ubjson(j, true, true);
SaxCountdown scp(1000000); // large enough to never trigger an abort
CHECK(json::sax_parse(packed, &scp, json::input_format_t::ubjson));
}
}
TEST_CASE("Universal Binary JSON Specification Examples 1")
{
SECTION("Null Value")
@@ -2503,6 +2778,93 @@ TEST_CASE("all UBJSON first bytes")
}
#endif
TEST_CASE("UBJSON use_type requires use_size")
{
SECTION("non-empty array throws other_error.502")
{
const json j = {1, 2, 3};
CHECK_THROWS_WITH_AS(json::to_ubjson(j, false, true),
"[json.exception.other_error.502] use_type requires use_size = true",
json::other_error&);
}
SECTION("non-empty object throws other_error.502")
{
const json j = {{"a", 1}, {"b", 2}};
CHECK_THROWS_WITH_AS(json::to_ubjson(j, false, true),
"[json.exception.other_error.502] use_type requires use_size = true",
json::other_error&);
}
SECTION("scalars do not throw with use_type=true, use_count=false")
{
CHECK_NOTHROW(json::to_ubjson(42, false, true));
CHECK_NOTHROW(json::to_ubjson(3.14, false, true));
CHECK_NOTHROW(json::to_ubjson("hello", false, true));
CHECK_NOTHROW(json::to_ubjson(true, false, true));
CHECK_NOTHROW(json::to_ubjson(nullptr, false, true));
}
SECTION("empty containers do not throw with use_type=true, use_count=false")
{
CHECK_NOTHROW(json::to_ubjson(json::array(), false, true));
CHECK_NOTHROW(json::to_ubjson(json::object(), false, true));
}
SECTION("valid combinations on non-empty containers")
{
const json j = {1, 2, 3};
CHECK_NOTHROW(json::to_ubjson(j, false, false));
CHECK_NOTHROW(json::to_ubjson(j, true, false));
CHECK_NOTHROW(json::to_ubjson(j, true, true));
}
}
TEST_CASE("UBJSON round-trip invariants")
{
// This checks what the parse_ubjson_fuzzer driver checks (see
// tests/src/fuzzer-parse_ubjson.cpp), so that a regression shows up in CI
// rather than as an OSS-Fuzz report: every value from_ubjson() returns
// (j1) can be serialized with any combination of options, the result can
// be parsed back (j2), and serializing j2 again with the same options
// reproduces the exact bytes. Beyond the driver, this also checks that j2
// equals j1. Values are compared with dump() rather than operator==,
// because a NaN never compares equal to itself.
struct options
{
bool use_size;
bool use_type;
};
const std::vector<options> all_options =
{
{false, false},
{true, false},
{true, true},
};
for (const auto& j0 : utils::round_trip_corpus::values())
{
// turn the corpus value into a value as from_ubjson() returns it; this
// has no binary values, as UBJSON writes them as arrays of integers
for (const auto& initial : all_options)
{
const json j1 = json::from_ubjson(json::to_ubjson(j0, initial.use_size, initial.use_type));
for (const auto& o : all_options)
{
INFO("j1 = " << j1.dump() << ", use_size = " << o.use_size << ", use_type = " << o.use_type);
const std::vector<std::uint8_t> vec = json::to_ubjson(j1, o.use_size, o.use_type);
json j2;
// anything the library writes must be parsable by the library
REQUIRE_NOTHROW(j2 = json::from_ubjson(vec));
CHECK(j2.dump() == j1.dump());
CHECK(json::to_ubjson(j2, o.use_size, o.use_type) == vec);
}
}
}
}
TEST_CASE("UBJSON roundtrips" * doctest::skip())
{
SECTION("input from self-generated UBJSON files")
+5 -2
View File
@@ -17,6 +17,7 @@ using nlohmann::json;
#include <sstream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
TEST_CASE("Unicode (1/5)" * doctest::skip())
{
@@ -240,7 +241,8 @@ void roundtrip(bool success_expected, const std::string& s)
if (success_expected)
{
// serialization succeeds
CHECK_NOTHROW(j.dump());
// dump() is nodiscard; this only checks that dumping does not throw
CHECK_NOTHROW(utils::ignore_return_value(j.dump()));
// exclude parse test for U+0000
if (s[0] != '\0')
@@ -259,7 +261,8 @@ void roundtrip(bool success_expected, const std::string& s)
else
{
// serialization fails
CHECK_THROWS_AS(j.dump(), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// parsing JSON text fails
CHECK_THROWS_AS(_ = json::parse(ps), json::parse_error&);
+3 -1
View File
@@ -19,6 +19,7 @@ using nlohmann::json;
#include <iostream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
// this test suite uses static variables with non-trivial destructors
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
@@ -97,7 +98,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
else
{
// strict mode must throw if success is not expected
CHECK_THROWS_AS(j.dump(), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// ignore and replace must create different dumps
CHECK(s_ignored != s_replaced);
+5 -3
View File
@@ -19,6 +19,7 @@ using nlohmann::json;
#include <iostream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
// this test suite uses static variables with non-trivial destructors
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
@@ -97,7 +98,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
else
{
// strict mode must throw if success is not expected
CHECK_THROWS_AS(j.dump(), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// ignore and replace must create different dumps
CHECK(s_ignored != s_replaced);
@@ -304,8 +306,8 @@ TEST_CASE("Unicode (3/5)" * doctest::skip())
{
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip fourth second byte
if (0x80 <= byte3 && byte3 <= 0xBF)
// skip correct fourth byte
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
+4 -2
View File
@@ -19,6 +19,7 @@ using nlohmann::json;
#include <iostream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
// this test suite uses static variables with non-trivial destructors
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
@@ -97,7 +98,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
else
{
// strict mode must throw if success is not expected
CHECK_THROWS_AS(j.dump(), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// ignore and replace must create different dumps
CHECK(s_ignored != s_replaced);
@@ -305,7 +307,7 @@ TEST_CASE("Unicode (4/5)" * doctest::skip())
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip correct fourth byte
if (0x80 <= byte3 && byte3 <= 0xBF)
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
+4 -2
View File
@@ -19,6 +19,7 @@ using nlohmann::json;
#include <iostream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
// this test suite uses static variables with non-trivial destructors
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
@@ -97,7 +98,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
else
{
// strict mode must throw if success is not expected
CHECK_THROWS_AS(j.dump(), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// ignore and replace must create different dumps
CHECK(s_ignored != s_replaced);
@@ -305,7 +307,7 @@ TEST_CASE("Unicode (5/5)" * doctest::skip())
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip correct fourth byte
if (0x80 <= byte3 && byte3 <= 0xBF)
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
+239
View File
@@ -18,7 +18,12 @@
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <array> // array
#include <cstddef> // size_t
#include <cstdint> // uint8_t
#include <list>
#include <string> // string
#include <vector> // vector
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
#include <iterator>
@@ -212,6 +217,66 @@ TEST_CASE("Parse with heterogeneous iterator and sentinel types")
CHECK(j2.at(0) == 1);
}
// A type whose data() hands out raw bytes but whose size() counts something
// else - here fixed-size records. Reading [data(), data() + size()) as bytes
// would silently truncate the input, so data() and size() alone must not be
// taken as evidence of contiguous byte storage.
struct record_buffer
{
using value_type = std::array<char, 4>;
std::string bytes;
const char* data() const noexcept
{
return bytes.data();
}
std::size_t size() const noexcept
{
return bytes.size() / sizeof(value_type);
}
const char* begin() const noexcept
{
return bytes.data();
}
const char* end() const noexcept
{
return bytes.data() + bytes.size();
}
};
TEST_CASE("Contiguous byte containers take the pointer adapter")
{
// Containers with contiguous single-byte storage are routed through the
// pointer-based adapter so the bulk fast paths apply in every standard, not
// only in C++20 where the library iterators model std::contiguous_iterator.
CHECK(nlohmann::detail::is_contiguous_byte_container<std::string>::value);
CHECK(nlohmann::detail::is_contiguous_byte_container<std::vector<char>>::value);
CHECK(nlohmann::detail::is_contiguous_byte_container<std::vector<std::uint8_t>>::value);
CHECK(nlohmann::detail::is_contiguous_byte_container<std::array<char, 4>>::value);
// input_adapter() takes its container by forwarding reference, so the trait
// is also asked about reference types
CHECK(nlohmann::detail::is_contiguous_byte_container<std::string&>::value);
CHECK(nlohmann::detail::is_contiguous_byte_container<const std::string&>::value);
// everything else keeps the iterator-based adapter
CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container<std::list<char>>::value);
CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container<std::vector<int>>::value);
CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container<const char*>::value);
// including a type that has data() and size() but whose size() does not
// count the units data() points at: its value_type says so
CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container<record_buffer>::value);
// and such a container still parses through its iterators, in full - taking
// it for a byte container would stop after data() + size() bytes
const record_buffer buffer{"[1,2,3,4,5]"};
CHECK(buffer.data() == buffer.bytes.data());
CHECK(buffer.size() * sizeof(record_buffer::value_type) < buffer.bytes.size());
CHECK(json::parse(buffer) == json({1, 2, 3, 4, 5}));
}
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
TEST_CASE("Parse with std::counted_iterator and std::default_sentinel_t")
@@ -228,6 +293,180 @@ TEST_CASE("Parse with std::counted_iterator and std::default_sentinel_t")
const std::counted_iterator<iterator_type> first2(json_str.begin(), len);
CHECK(json::accept(first2, std::default_sentinel));
}
TEST_CASE("std::counted_iterator reaches the contiguous fast paths")
{
// A sized sentinel makes the remaining element count computable in O(1), so
// std::counted_iterator over a contiguous iterator must reach the same bulk
// string/number scanners as a plain pointer - not just the byte-at-a-time
// fallback (see #5268 for the equivalent memcpy fast path).
#if JSON_HAS_RANGES
// JSON_HAS_RANGES is 0 on standard libraries with an incomplete <ranges>
// (libstdc++ < 11, libc++ < 16), where the adapter deliberately falls back
// to the byte-at-a-time scanner; everything below still has to work there.
using adapter_type = nlohmann::detail::iterator_input_adapter<std::counted_iterator<const char*>, std::default_sentinel_t>;
CHECK(adapter_type::supports_bulk_scan);
CHECK(adapter_type::supports_seek);
#endif
// exercise every fast path: long ASCII run, multibyte UTF-8, escapes, and
// integer/floating-point numbers
const std::string json_str =
R"({"ascii":"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",)"
"\"utf8\":\"\xe4\xb8\xad\xe6\x96\x87\xf0\x9f\x98\x80\xc3\xa9\","
R"("escaped":"aéb\n\\","ints":[0,-1,18446744073709551615,-9223372036854775808],)"
R"("floats":[1.5,-2.25e3,0.30000000000000004]})";
const auto len = static_cast<std::iter_difference_t<const char*>>(json_str.size());
const std::counted_iterator<const char*> first(json_str.data(), len);
const json j = json::parse(first, std::default_sentinel);
// parsing through the pointer adapter must give exactly the same result
CHECK(j == json::parse(json_str));
#if !defined(JSON_NOEXCEPTION)
// Diagnostics that quote the offending token are reconstructed from the
// already-consumed input (supports_seek), a path a sized sentinel only
// reaches now; check a few that include the "last read" text. Parsing
// invalid input aborts when exceptions are off, hence the guard.
// Raw strings and explicit bytes: an escaped literal and two literals
// written next to each other both read as mistakes to static analysis.
const auto byte = [](int value)
{
return std::string(1, static_cast<char>(value));
};
const std::vector<std::string> diagnostic_docs =
{
"1\nx",
"truX",
"[tru]",
R"("abc)",
R"(["\ud834"])",
R"(["a)" + byte(0x01) + R"(b"])",
R"([")" + byte(0xC3) + byte(0x28) + R"("])",
"[1e]",
R"(["aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaX)"
};
for (const auto& text : diagnostic_docs)
{
CAPTURE(text);
const std::counted_iterator<const char*> it(text.data(), static_cast<std::iter_difference_t<const char*>>(text.size()));
std::string counted_message;
std::string string_message;
try
{
const json counted_result = json::parse(it, std::default_sentinel);
static_cast<void>(counted_result);
}
catch (const json::parse_error& e)
{
counted_message = e.what();
}
try
{
const json string_result = json::parse(text);
static_cast<void>(string_result);
}
catch (const json::parse_error& e)
{
string_message = e.what();
}
CHECK_FALSE(counted_message.empty());
CHECK(counted_message == string_message);
}
// and errors must still be reported identically
const std::string bad = "[01\n]";
const std::counted_iterator<const char*> bad_first(bad.data(), static_cast<std::iter_difference_t<const char*>>(bad.size()));
std::string counted_what;
std::string string_what;
try
{
const json counted_result = json::parse(bad_first, std::default_sentinel);
static_cast<void>(counted_result);
}
catch (const json::parse_error& e)
{
counted_what = e.what();
}
try
{
const json string_result = json::parse(bad);
static_cast<void>(string_result);
}
catch (const json::parse_error& e)
{
string_what = e.what();
}
CHECK_FALSE(counted_what.empty());
CHECK(counted_what == string_what);
#endif
}
#if !defined(JSON_NOEXCEPTION)
// several cases below are truncated on purpose, and parsing invalid input
// aborts when exceptions are off
TEST_CASE("std::counted_iterator bulk scanning stops at the counted end")
{
// The count, not the size of the underlying buffer, is the end of the
// input: the bulk scanners must never look at the bytes behind it, even
// though they are readable. Each case is compared against parsing the
// equivalent prefix as a std::string.
const auto via_counted = [](const std::string & buf, std::size_t n) -> std::string
{
const std::counted_iterator<const char*> first(buf.data(), static_cast<std::iter_difference_t<const char*>>(n));
try
{
const json j = json::parse(first, std::default_sentinel);
return "OK|" + j.dump();
}
catch (const json::parse_error& e)
{
return {e.what()};
}
};
const auto via_prefix = [](const std::string & buf, std::size_t n) -> std::string
{
try
{
const json j = json::parse(buf.substr(0, n));
return "OK|" + j.dump();
}
catch (const json::parse_error& e)
{
return {e.what()};
}
};
struct testcase // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init)
{
const char* buffer;
std::size_t count;
};
const std::vector<testcase> cases =
{
{"[\"abc\"]____TRAILING____", 7}, // exact fit, tail hidden
{"[\"abcdefghijklmnop\"]____", 8}, // cut inside a string
{"[\"abc\"]____", 6}, // cut just before the closing quote
{"[12345]xxxxx", 4}, // cut inside a number
{"[123]999999", 5}, // number ends exactly at the count
{"[\"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\"]", 12}, // closing quote only behind the count
{"[\"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\"]", 19}, // cut inside an 8-byte SWAR stride
{"[\"\xe4\xb8\xad\xe6\x96\x87\"]", 5}, // cut inside a UTF-8 sequence
{"[\"\xe4\xb8\xad\xe6\x96\x87\"]____", 10}, // complete UTF-8, tail hidden
{"[1.25e3]TRAILINGDIGITS999", 7}, // number token reaches the count
};
for (const auto& tc : cases)
{
CAPTURE(tc.buffer);
CAPTURE(tc.count);
const std::string buffer = tc.buffer;
CHECK(via_counted(buffer, tc.count) == via_prefix(buffer, tc.count));
}
}
#endif
#endif
} // namespace