diff --git a/.github/CONTRIBUTING.md b/.github/CONTRIBUTING.md index b0f65230c..4125b066a 100644 --- a/.github/CONTRIBUTING.md +++ b/.github/CONTRIBUTING.md @@ -108,7 +108,9 @@ The tests are located in [`tests/src/unit-*.cpp`](https://github.com/nlohmann/js are structured along the features of the library or the nature of the tests. Usually, it should be clear from the context which existing file needs to be extended, and only very few cases require creating new test files. -When fixing a bug, edit `unit-regression2.cpp` and add a section referencing the fixed issue. +When fixing a bug, edit `unit-regression3.cpp` and add a section referencing the fixed issue. +`unit-regression2.cpp` holds the older tests; the two files exist because a single one grew large enough for the +MinGW linker to fail relocating it, so please keep adding to the smaller file rather than growing the larger one. #### Exceptions @@ -156,6 +158,15 @@ make amalgamate Running `make amalgamate` will also apply automatic formatting to the source files using [`Artistic Style`](https://astyle.sourceforge.net/). This formatting may modify your source files in-place. Be certain to review and commit any changes to avoid unintended formatting diffs in commits. +If you add, rename, or remove a header in `include/nlohmann`, also regenerate the header list in +[`BUILD.bazel`](https://github.com/nlohmann/json/blob/develop/BUILD.bazel) (requires CMake) by executing: + +```shell +make BUILD.bazel +``` + +The amalgamation check in CI fails if any of these generated files is out of date. + ## Recommended documentation - The library’s [README file](https://github.com/nlohmann/json/blob/master/README.md) is an excellent starting point to diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md index 537095324..16d808485 100644 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -2,6 +2,7 @@ - [ ] The changes are described in detail, both the what and why. - [ ] If applicable, an [existing issue](https://github.com/nlohmann/json/issues) is referenced. +- [ ] If applicable, a fixed [OSS-Fuzz](https://issues.oss-fuzz.com) issue is referenced as `OSS-Fuzz: ` (see [fuzz testing](https://github.com/nlohmann/json/blob/develop/tests/fuzzing.md#handling-oss-fuzz-reports)). - [ ] The [Code coverage](https://coveralls.io/github/nlohmann/json) remained at 100%. A test case for every new line of code. - [ ] If applicable, the [documentation](https://json.nlohmann.me) is updated. - [ ] The source code is amalgamated by running `make amalgamate`. diff --git a/.github/labeler.yml b/.github/labeler.yml index 828660daf..b4c176960 100644 --- a/.github/labeler.yml +++ b/.github/labeler.yml @@ -29,6 +29,27 @@ labels: files: - ".github/external_ci/.*" +- label: "CI" + files: + - ".github/(dependabot|labeler)\\.yml" + +- label: "aspect: binary formats" + files: + - "include/nlohmann/detail/input/binary_reader\\.hpp" + - "include/nlohmann/detail/output/binary_writer\\.hpp" + - "tests/src/unit-(bson|cbor|msgpack|ubjson|bjdata|binary_formats)" + - "tests/src/fuzzer-parse_(bson|cbor|msgpack|ubjson|bjdata)" + - "docs/mkdocs/docs/features/binary_formats/" + - "docs/mkdocs/docs/(api/basic_json|examples)/(to|from)_(bson|cbor|msgpack|ubjson|bjdata)" + +- label: "aspect: binary formats" + title: "(?i)(bson|cbor|msgpack|messagepack|ubjson|bjdata|binary format)" + +- label: "python" + files: + - "\\.py$" + - "requirements[^/]*\\.txt$" + - label: "S" size-below: 10 - label: "M" diff --git a/.github/workflows/check_amalgamation.yml b/.github/workflows/check_amalgamation.yml index 60a3f8240..f692e434a 100644 --- a/.github/workflows/check_amalgamation.yml +++ b/.github/workflows/check_amalgamation.yml @@ -11,7 +11,7 @@ jobs: runs-on: ubuntu-latest steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit @@ -34,7 +34,7 @@ jobs: steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit @@ -57,18 +57,31 @@ jobs: python3 -mvenv venv venv/bin/pip3 install -r $MAIN_DIR/tools/astyle/requirements.txt - - name: Regenerate amalgamation and formatting + - name: Regenerate amalgamation, formatting, and BUILD.bazel run: | cd $MAIN_DIR python3 $TOOL_DIR/amalgamate.py -c $TOOL_DIR/config_json.json -s . python3 $TOOL_DIR/amalgamate.py -c $TOOL_DIR/config_json_fwd.json -s . + # the header list of the Bazel "json" target must match the files in include/ + cmake -P cmake/scripts/gen_bazel_build_file.cmake + ${{ github.workspace }}/venv/bin/astyle --project=tools/astyle/.astylerc --suffix=none --quiet \ $INCLUDE_DIR/json.hpp $INCLUDE_DIR/json_fwd.hpp + # fail loudly if a directory is renamed or removed: find would only warn + # about the missing path and silently drop its files from the check + SOURCE_DIRS="docs/mkdocs/docs/examples include tests" + for DIR in $SOURCE_DIRS; do + if [ ! -d "$DIR" ]; then + echo "::error::source directory '$DIR' does not exist" + exit 1 + fi + done + ${{ github.workspace }}/venv/bin/astyle --project=tools/astyle/.astylerc --suffix=none --quiet \ - $(find docs/examples include tests -type f \( -name '*.hpp' -o -name '*.cpp' -o -name '*.cu' \) -not -path 'tests/thirdparty/*' -not -path 'tests/abi/include/nlohmann/*' | sort) + $(find $SOURCE_DIRS -type f \( -name '*.hpp' -o -name '*.cpp' -o -name '*.cu' \) -not -path 'tests/thirdparty/*' -not -path 'tests/abi/include/nlohmann/*' | sort) - name: Build patch and check for differences id: diff @@ -77,7 +90,7 @@ jobs: mkdir -p ${{ github.workspace }}/patch git diff --patch --no-color > ${{ github.workspace }}/patch/amalgamation.patch if [ -s ${{ github.workspace }}/patch/amalgamation.patch ]; then - echo "The source code has not been amalgamated/formatted correctly. Diff:" + echo "The source code has not been amalgamated/formatted correctly or BUILD.bazel is out of date. Diff:" cat ${{ github.workspace }}/patch/amalgamation.patch echo "has_diff=true" >> "$GITHUB_OUTPUT" else diff --git a/.github/workflows/cifuzz.yml b/.github/workflows/cifuzz.yml index da3df626f..2e50fee9c 100644 --- a/.github/workflows/cifuzz.yml +++ b/.github/workflows/cifuzz.yml @@ -9,7 +9,7 @@ jobs: runs-on: ubuntu-22.04 steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit diff --git a/.github/workflows/codeql-analysis.yml b/.github/workflows/codeql-analysis.yml index c21a3cd24..1c2e73cd1 100644 --- a/.github/workflows/codeql-analysis.yml +++ b/.github/workflows/codeql-analysis.yml @@ -27,7 +27,7 @@ jobs: steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit @@ -38,14 +38,14 @@ jobs: # Initializes the CodeQL tools for scanning. - name: Initialize CodeQL - uses: github/codeql-action/init@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 + uses: github/codeql-action/init@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: languages: c-cpp # Autobuild attempts to build any compiled languages (C/C++, C#, or Java). # If this step fails, then you should remove it and run the build manually (see below) - name: Autobuild - uses: github/codeql-action/autobuild@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 + uses: github/codeql-action/autobuild@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 - name: Perform CodeQL Analysis - uses: github/codeql-action/analyze@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 + uses: github/codeql-action/analyze@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 diff --git a/.github/workflows/comment_check_amalgamation.yml b/.github/workflows/comment_check_amalgamation.yml index 87e1b7b58..4667329d2 100644 --- a/.github/workflows/comment_check_amalgamation.yml +++ b/.github/workflows/comment_check_amalgamation.yml @@ -19,7 +19,7 @@ jobs: pull-requests: write steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit @@ -95,13 +95,13 @@ jobs: issue_number: issue_number, owner: context.repo.owner, repo: context.repo.repo, - body: '## 🔴 Amalgamation check failed! 🔴\nThe source code has not been amalgamated and/or formatted correctly.' + body: '## 🔴 Amalgamation check failed! 🔴\nThe source code has not been amalgamated and/or formatted correctly, or `BUILD.bazel` is out of date.' + (hasPatch ? '\n\n📎 A ready-to-apply patch is attached to the [failed workflow run](' + runUrl + ') as the `amalgamation-patch` artifact.' + ' Download it, then apply it locally from the repository root with:' + '\n\n```shell\ngit apply amalgamation.patch\n```\n\n' + 'This does not require installing astyle yourself.' : '') + (first ? '\n\n@' + author + ' Please read and follow the [Contribution Guidelines]' - + '(https://github.com/nlohmann/json/blob/develop/.github/CONTRIBUTING.md#files-to-change).' + + '(https://github.com/nlohmann/json/blob/develop/.github/CONTRIBUTING.md#amalgamate-the-source-code).' : '') }) diff --git a/.github/workflows/dependency-review.yml b/.github/workflows/dependency-review.yml index 2c6300252..aa09008a3 100644 --- a/.github/workflows/dependency-review.yml +++ b/.github/workflows/dependency-review.yml @@ -17,7 +17,7 @@ jobs: runs-on: ubuntu-latest steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit diff --git a/.github/workflows/flawfinder.yml b/.github/workflows/flawfinder.yml index 5477195d8..2c8befd40 100644 --- a/.github/workflows/flawfinder.yml +++ b/.github/workflows/flawfinder.yml @@ -27,7 +27,7 @@ jobs: security-events: write steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit @@ -43,6 +43,6 @@ jobs: output: 'flawfinder_results.sarif' - name: Upload analysis results to GitHub Security tab - uses: github/codeql-action/upload-sarif@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: sarif_file: ${{github.workspace}}/flawfinder_results.sarif diff --git a/.github/workflows/labeler.yml b/.github/workflows/labeler.yml index 2222b77f2..3b4571b95 100644 --- a/.github/workflows/labeler.yml +++ b/.github/workflows/labeler.yml @@ -17,7 +17,7 @@ jobs: steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit diff --git a/.github/workflows/publish_documentation.yml b/.github/workflows/publish_documentation.yml index d0066885e..189d95419 100644 --- a/.github/workflows/publish_documentation.yml +++ b/.github/workflows/publish_documentation.yml @@ -7,7 +7,6 @@ on: - develop paths: - docs/mkdocs/** - - docs/examples/** workflow_dispatch: # we don't want to have concurrent jobs, and we don't want to cancel running jobs to avoid broken publications @@ -27,7 +26,7 @@ jobs: runs-on: ubuntu-22.04 steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit diff --git a/.github/workflows/scorecards.yml b/.github/workflows/scorecards.yml index 63ca0b90d..113da079a 100644 --- a/.github/workflows/scorecards.yml +++ b/.github/workflows/scorecards.yml @@ -36,7 +36,7 @@ jobs: steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit @@ -76,6 +76,6 @@ jobs: # Upload the results to GitHub's code scanning dashboard. - name: "Upload to code-scanning" - uses: github/codeql-action/upload-sarif@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: sarif_file: results.sarif diff --git a/.github/workflows/semgrep.yml b/.github/workflows/semgrep.yml index 812e4a0d7..dc326db55 100644 --- a/.github/workflows/semgrep.yml +++ b/.github/workflows/semgrep.yml @@ -32,7 +32,7 @@ jobs: runs-on: ubuntu-latest steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit @@ -61,7 +61,7 @@ jobs: # Upload SARIF file generated in previous step - name: Upload SARIF file - uses: github/codeql-action/upload-sarif@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: sarif_file: semgrep.sarif if: always() diff --git a/.github/workflows/stale.yml b/.github/workflows/stale.yml index fd9cbb80f..1e690ccdd 100644 --- a/.github/workflows/stale.yml +++ b/.github/workflows/stale.yml @@ -16,7 +16,7 @@ jobs: steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index 2ef56a80b..d8c27c621 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -35,7 +35,7 @@ jobs: runs-on: ubuntu-latest steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit @@ -60,7 +60,7 @@ jobs: target: [ci_test_amalgamation, ci_test_single_header, ci_cppcheck, ci_cpplint, ci_reproducible_tests, ci_non_git_tests, ci_offline_testdata, ci_reuse_compliance, ci_test_valgrind] steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit @@ -100,7 +100,7 @@ jobs: container: ubuntu:focal strategy: matrix: - target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls] + target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_skiplibraryversioncheck, ci_test_simdutf, ci_test_strict_nul_handling, ci_test_no_thread_local] steps: - name: Install build-essential run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev @@ -118,7 +118,7 @@ jobs: runs-on: ubuntu-latest steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit @@ -369,7 +369,7 @@ jobs: runs-on: ubuntu-latest steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit @@ -392,7 +392,7 @@ jobs: target: [ci_test_examples, ci_test_build_documentation] steps: - name: Harden Runner - uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 with: egress-policy: audit diff --git a/.github/workflows/windows.yml b/.github/workflows/windows.yml index 068c5a0f1..ea742773f 100644 --- a/.github/workflows/windows.yml +++ b/.github/workflows/windows.yml @@ -124,11 +124,11 @@ jobs: steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - name: Run CMake (Release) - run: cmake -S . -B build -G "Visual Studio 17 2022" -A ARM64 -DJSON_BuildTests=On -DCMAKE_CXX_FLAGS="/W4 /WX" + run: cmake -S . -B build -G "Visual Studio 18 2026" -A ARM64 -DJSON_BuildTests=On -DCMAKE_CXX_FLAGS="/W4 /WX" if: matrix.build_type == 'Release' shell: pwsh - name: Run CMake (Debug) - run: cmake -S . -B build -G "Visual Studio 17 2022" -A ARM64 -DJSON_BuildTests=On -DJSON_FastTests=ON -DCMAKE_CXX_FLAGS="/W4 /WX" + run: cmake -S . -B build -G "Visual Studio 18 2026" -A ARM64 -DJSON_BuildTests=On -DJSON_FastTests=ON -DCMAKE_CXX_FLAGS="/W4 /WX" if: matrix.build_type == 'Debug' shell: pwsh - name: Build @@ -158,6 +158,10 @@ jobs: # to fit: IMAGE_REL_AMD64_SECREL against `.debug_line'" because the # MinGW linker cannot relocate the debug sections this test produces. # The tests are only built and run here, so the debug info is not used. + # Do not add -O1 here to shrink the objects further: it does make them + # link, but the binaries clang 11.0.1 and clang 18.1.8 then produce crash + # before doctest prints its first line - 39 of 102 tests on clang 18. + # Keep the objects small by splitting the test files instead. - name: Run CMake run: cmake -S . -B build ^ -DCMAKE_CXX_COMPILER="C:/Program Files/LLVM/bin/clang++.exe" ^ diff --git a/BUILD.bazel b/BUILD.bazel index de0ff7145..b13e62c22 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -30,8 +30,10 @@ cc_library( "include/nlohmann/detail/input/input_adapters.hpp", "include/nlohmann/detail/input/json_sax.hpp", "include/nlohmann/detail/input/lexer.hpp", + "include/nlohmann/detail/input/number_parse.hpp", "include/nlohmann/detail/input/parser.hpp", "include/nlohmann/detail/input/position_t.hpp", + "include/nlohmann/detail/input/string_scan.hpp", "include/nlohmann/detail/iterators/internal_iterator.hpp", "include/nlohmann/detail/iterators/iter_impl.hpp", "include/nlohmann/detail/iterators/iteration_proxy.hpp", @@ -49,12 +51,14 @@ cc_library( "include/nlohmann/detail/meta/detected.hpp", "include/nlohmann/detail/meta/identity_tag.hpp", "include/nlohmann/detail/meta/is_sax.hpp", + "include/nlohmann/detail/meta/logic.hpp", "include/nlohmann/detail/meta/std_fs.hpp", "include/nlohmann/detail/meta/type_traits.hpp", "include/nlohmann/detail/meta/void_t.hpp", "include/nlohmann/detail/output/binary_writer.hpp", "include/nlohmann/detail/output/output_adapters.hpp", "include/nlohmann/detail/output/serializer.hpp", + "include/nlohmann/detail/recursion_depth_limit.hpp", "include/nlohmann/detail/string_concat.hpp", "include/nlohmann/detail/string_escape.hpp", "include/nlohmann/detail/string_utils.hpp", diff --git a/CMakeLists.txt b/CMakeLists.txt index 9669946a5..4c43c23ff 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -59,6 +59,7 @@ option(JSON_LegacyDiscardedValueComparison "Enable legacy discarded value compar option(JSON_Install "Install CMake targets during install step." ${MAIN_PROJECT}) option(JSON_MultipleHeaders "Use non-amalgamated version of the library." ON) option(JSON_SystemInclude "Include as system headers (skip for clang-tidy)." OFF) +option(JSON_StrictNulHandling "Build with strict NUL-byte handling enabled." OFF) if (JSON_CI) include(ci) @@ -108,6 +109,10 @@ if (JSON_Diagnostics) message(STATUS "Diagnostics enabled (JSON_DIAGNOSTICS=1)") endif() +if (JSON_StrictNulHandling) + message(STATUS "Strict NUL-byte handling enabled (JSON_STRICT_NUL_HANDLING=1)") +endif() + if (JSON_Diagnostic_Positions) message(STATUS "Diagnostic positions enabled (JSON_DIAGNOSTIC_POSITIONS=1)") endif() @@ -141,6 +146,7 @@ target_compile_definitions( $<$:JSON_DIAGNOSTICS=1> $<$:JSON_DIAGNOSTIC_POSITIONS=1> $<$:JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1> + $<$:JSON_STRICT_NUL_HANDLING=1> ) target_include_directories( diff --git a/FILES.md b/FILES.md index b68167336..263647146 100644 --- a/FILES.md +++ b/FILES.md @@ -250,12 +250,16 @@ Further documentation: ### `BUILD.bazel` -The file can be updated by calling +The build definition for [Bazel](https://bazel.build). The file is generated by +`cmake/scripts/gen_bazel_build_file.cmake`, which derives the header list from the files in `include`; change the +script rather than editing the file by hand. The file can be updated by calling ```shell make BUILD.bazel ``` +The "Check amalgamation" workflow fails if the file is out of date. + ### `meson.build` The build definition for the [Meson](https://mesonbuild.com) build system. diff --git a/Makefile b/Makefile index d99d6f5f3..871ea7995 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: pretty clean ChangeLog.md release +.PHONY: pretty clean ChangeLog.md release update_hedley update_hedley_undef BUILD.bazel ########################################################################## # configuration @@ -30,8 +30,9 @@ AMALGAMATED_FWD_FILE=single_include/nlohmann/json_fwd.hpp # main target all: @echo "amalgamate - amalgamate files single_include/nlohmann/json{,_fwd}.hpp from the include/nlohmann sources" + @echo "BUILD.bazel - regenerate the Bazel BUILD file from the include/nlohmann sources" @echo "ChangeLog.md - generate ChangeLog file" - @echo "check-amalgamation - check whether sources have been amalgamated" + @echo "check-amalgamation - check whether sources have been amalgamated and BUILD.bazel is up to date" @echo "clean - remove built files" @echo "doctest - compile example files and check their output" @echo "fuzz_testing - prepare fuzz testing of the JSON parser" @@ -41,6 +42,8 @@ all: @echo "fuzz_testing_ubjson - prepare fuzz testing of the UBJSON parser" @echo "pretty - beautify code with Artistic Style" @echo "run_benchmarks - build and run benchmarks" + @echo "update_hedley - download Hedley and regenerate hedley.hpp / hedley_undef.hpp" + @echo "update_hedley_undef - rebuild hedley_undef.hpp from the JSON_HEDLEY_* #define names in hedley.hpp" ########################################################################## @@ -170,8 +173,13 @@ check-amalgamation: @diff $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_FWD_FILE)~ || (echo "===================================================================\n Amalgamation required! Please read the contribution guidelines\n in file .github/CONTRIBUTING.md.\n===================================================================" ; mv $(AMALGAMATED_FWD_FILE)~ $(AMALGAMATED_FWD_FILE) ; false) @mv $(AMALGAMATED_FILE)~ $(AMALGAMATED_FILE) @mv $(AMALGAMATED_FWD_FILE)~ $(AMALGAMATED_FWD_FILE) + @mv BUILD.bazel BUILD.bazel~ + @$(MAKE) BUILD.bazel + @diff BUILD.bazel BUILD.bazel~ || (echo "===================================================================\n BUILD.bazel is out of date! Please run 'make BUILD.bazel'.\n===================================================================" ; mv BUILD.bazel~ BUILD.bazel ; false) + @mv BUILD.bazel~ BUILD.bazel -BUILD.bazel: $(SRCS) +# generate the Bazel BUILD file; phony, because a removed header would not trigger a rebuild +BUILD.bazel: cmake -P cmake/scripts/gen_bazel_build_file.cmake ########################################################################## @@ -241,11 +249,24 @@ update_hedley: rm -f include/nlohmann/thirdparty/hedley/hedley.hpp include/nlohmann/thirdparty/hedley/hedley_undef.hpp curl https://raw.githubusercontent.com/nemequ/hedley/master/hedley.h -o include/nlohmann/thirdparty/hedley/hedley.hpp $(SED) -i 's/HEDLEY_/JSON_HEDLEY_/g' include/nlohmann/thirdparty/hedley/hedley.hpp - grep "[[:blank:]]*#[[:blank:]]*undef" include/nlohmann/thirdparty/hedley/hedley.hpp | grep -v "__" | sort | uniq | $(SED) 's/ //g' | $(SED) 's/undef/undef /g' > include/nlohmann/thirdparty/hedley/hedley_undef.hpp $(SED) -i '1s/^/#pragma once\n\n/' include/nlohmann/thirdparty/hedley/hedley.hpp - $(SED) -i '1s/^/#pragma once\n\n/' include/nlohmann/thirdparty/hedley/hedley_undef.hpp + $(MAKE) update_hedley_undef $(MAKE) amalgamate +# Rebuild hedley_undef.hpp from every JSON_HEDLEY_* name that hedley.hpp +# #defines. Hedley does not #undef all of its public macros internally (see +# #5408), so grepping those #undef lines misses names such as +# JSON_HEDLEY_PRAGMA. cmake/scripts/gen_hedley_undef_check.cmake is the +# single source of truth for this extraction (tests/CMakeLists.txt uses the +# same script, in MODE=checks, to generate the matching leak-check test), so +# the vendored header, the generated #undef list, and the regression test +# cannot drift apart. +update_hedley_undef: + cmake -DHEDLEY_HPP=include/nlohmann/thirdparty/hedley/hedley.hpp \ + -DOUTPUT=include/nlohmann/thirdparty/hedley/hedley_undef.hpp \ + -DMODE=undef \ + -P cmake/scripts/gen_hedley_undef_check.cmake + ########################################################################## # serve_header.py ########################################################################## diff --git a/README.md b/README.md index 4b4236bb6..becce71a1 100644 --- a/README.md +++ b/README.md @@ -1421,7 +1421,7 @@ I deeply appreciate the help of the following people. 6. [Joshua C. Randall](https://github.com/jrandall) fixed a bug in the floating-point serialization. 7. [Aaron Burghardt](https://github.com/aburgh) implemented code to parse streams incrementally. Furthermore, he greatly improved the parser class by allowing the definition of a filter function to discard undesired elements while parsing. 8. [Daniel Kopeček](https://github.com/dkopecek) fixed a bug in the compilation with GCC 5.0. -9. [Florian Weber](https://github.com/Florianjw) fixed a bug in and improved the performance of the comparison operators. +9. [Fiona Johanna Weber](https://github.com/Fiona-J-W) fixed a bug in and improved the performance of the comparison operators. 10. [Eric Cornelius](https://github.com/EricMCornelius) pointed out a bug in the handling with NaN and infinity values. He also improved the performance of the string escaping. 11. [易思龙](https://github.com/likebeta) implemented a conversion from anonymous enums. 12. [kepkin](https://github.com/kepkin) patiently pushed forward the support for Microsoft Visual Studio. @@ -1523,14 +1523,14 @@ I deeply appreciate the help of the following people. 108. [Kevin Tonon](https://github.com/ktonon) overworked the C++11 compiler checks in CMake. 109. [Axel Huebl](https://github.com/ax3l) simplified a CMake check and added support for the [Spack package manager](https://spack.io). 110. [Carlos O'Ryan](https://github.com/coryan) fixed a typo. -111. [James Upjohn](https://github.com/jammehcow) fixed a version number in the compilers section. +111. [James Upjohn](https://github.com/jupjohn) fixed a version number in the compilers section. 112. [Chuck Atkins](https://github.com/chuckatkins) adjusted the CMake files to the CMake packaging guidelines and provided documentation for the CMake integration. 113. [Jan Schöppach](https://github.com/dns13) fixed a typo. 114. [martin-mfg](https://github.com/martin-mfg) fixed a typo. 115. [Matthias Möller](https://github.com/TinyTinni) removed the dependency from `std::stringstream`. 116. [agrianius](https://github.com/agrianius) added code to use alternative string implementations. 117. [Daniel599](https://github.com/Daniel599) allowed to use more algorithms with the `items()` function. -118. [Julius Rakow](https://github.com/jrakow) fixed the Meson include directory and fixed the links to [cppreference.com](https://cppreference.com). +118. [Julius Rakow](https://github.com/juliusrakow) fixed the Meson include directory and fixed the links to [cppreference.com](https://cppreference.com). 119. [Sonu Lohani](https://github.com/sonulohani) fixed the compilation with MSVC 2015 in debug mode. 120. [grembo](https://github.com/grembo) fixed the test suite and re-enabled several test cases. 121. [Hyeon Kim](https://github.com/simnalamburt) introduced the macro `JSON_INTERNAL_CATCH` to control the exception handling inside the library. @@ -1581,7 +1581,7 @@ I deeply appreciate the help of the following people. 166. [Mark Beckwith](https://github.com/wythe) fixed a typo. 167. [yann-morin-1998](https://github.com/yann-morin-1998) helped to reduce the CMake requirement to version 3.1. 168. [Konstantin Podsvirov](https://github.com/podsvirov) maintains a package for the MSYS2 software distro. -169. [remyabel](https://github.com/remyabel) added GNUInstallDirs to the CMake files. +169. [remyabel](https://github.com/remyabel2) added GNUInstallDirs to the CMake files. 170. [Taylor Howard](https://github.com/taylorhoward92) fixed a unit test. 171. [Gabe Ron](https://github.com/Macr0Nerd) implemented the `to_string` method. 172. [Watal M. Iwasaki](https://github.com/heavywatal) fixed a Clang warning. @@ -1608,7 +1608,7 @@ I deeply appreciate the help of the following people. 193. [Hubert Chathi](https://github.com/uhoreg) made CMake's version config file architecture-independent. 194. [OmnipotentEntity](https://github.com/OmnipotentEntity) implemented the binary values for CBOR, MessagePack, BSON, and UBJSON. 195. [ArtemSarmini](https://github.com/ArtemSarmini) fixed a compilation issue with GCC 10 and fixed a leak. -196. [Evgenii Sopov](https://github.com/sea-kg) integrated the library to the wsjcpp package manager. +196. [Evgenii Sopov](https://github.com/sea5kg) integrated the library to the wsjcpp package manager. 197. [Sergey Linev](https://github.com/linev) fixed a compiler warning. 198. [Miguel Magalhães](https://github.com/magamig) fixed the year in the copyright. 199. [Gareth Sylvester-Bradley](https://github.com/garethsb-sony) fixed a compilation issue with MSVC. @@ -1702,7 +1702,7 @@ I deeply appreciate the help of the following people. 287. [NN](https://github.com/NN---) added the Visual Studio output directory to `.gitignore`. 288. [Romain Reignier](https://github.com/romainreignier) improved the performance of the vector output adapter. 289. [Mike](https://github.com/Mike-Leo-Smith) fixed the `std::iterator_traits`. -290. [Richard Hozák](https://github.com/zxey) added macro `JSON_NO_ENUM` to disable default enum conversions. +290. [Richard Hozák](https://github.com/richardhozak) added macro `JSON_NO_ENUM` to disable default enum conversions. 291. [vakokako](https://github.com/vakokako) fixed tests when compiling with C++20. 292. [Alexander “weej” Jones](https://github.com/alexweej) fixed an example in the README. 293. [Eli Schwartz](https://github.com/eli-schwartz) added more files to the `include.zip` archive. @@ -1727,7 +1727,7 @@ I deeply appreciate the help of the following people. 312. [Gareth Sylvester-Bradley](https://github.com/garethsb) added `operator/=` and `operator/` to construct JSON pointers. 313. [Michael Macnair](https://github.com/mykter) added support for afl-fuzz testing. 314. [Berkus Decker](https://github.com/berkus) fixed a typo in the README. -315. [Illia Polishchuk](https://github.com/effolkronium) improved the CMake testing. +315. [Illia Polishchuk](https://github.com/ilqvya) improved the CMake testing. 316. [Ikko Ashimine](https://github.com/eltociear) fixed a typo. 317. [Raphael Grimm](https://github.com/barcode) added the possibility to define a custom base class. 318. [tocic](https://github.com/tocic) fixed typos in the documentation. @@ -1797,6 +1797,66 @@ I deeply appreciate the help of the following people. 382. [bitFiedler](https://github.com/bitFiedler) made GDB pretty printer work with Python 3.8. 383. [Gianfranco Costamagna](https://github.com/LocutusOfBorg) fixed a compiler warning. 384. [risa2000](https://github.com/risa2000) made `std::filesystem::path` conversion to/from UTF-8 encoded string explicit. +385. [AM](https://github.com/maqnouch) fixed typos in the README. +386. [dmenendez-gruposantander](https://github.com/dmenendez-gruposantander) fixed typos in the comments of the examples. +387. [Mihai Stan](https://github.com/mstan-xx) fixed comparisons against the literal `0`. +388. [Matt Gumbel](https://github.com/intelmatt) fixed some `-Weffc++` warnings. +389. [vimpunk](https://github.com/vimpunk) moved a lambda out of an unevaluated context to support older compilers. +390. [Chris Harris](https://github.com/cjh1) fixed the compilation with GCC 4.8. +391. [Palmer Dabbelt](https://github.com/palmer-dabbelt) generated and installed a pkg-config file. +392. [Gus Pozuelo](https://github.com/ap-viavi) made `ordered_map` compatible with GCC 5.5, Clang 3.6, and Xcode 9. +393. [AK](https://github.com/Lioncky) fixed an MSVC build error caused by the `min`/`max` macros from `windows.h`. +394. [Sergiu Deitsch](https://github.com/sergiud) provided a fallback for missing `char8_t` support. +395. [Xiaochuan Ye](https://github.com/XueSongTap) fixed `from_msgpack` for `std::byte` input by specializing `std::char_traits`. +396. [Ville Vesilehto](https://github.com/thevilledev) fixed an overflow in the BJData size calculation and rejected overflowing negative integers in CBOR. +397. [NmPassTHFan](https://github.com/nmpassthf) replaced the deprecated `std::is_trivial` for C++26. +398. [Chris Ever](https://github.com/chirsz-ever) added the `ignore_trailing_commas` parser option. +399. [Kuan-Fu Wu](https://github.com/kfwu1999) fixed the example code for `json_pointer` initialization. +400. [David Kilzer](https://github.com/ddkilzer) added a missing header to the input adapters. +401. [Miko](https://github.com/mikomikotaishi) added proper C++20 module support, simplified the module API, and fixed missing exports. +402. [hitgirl](https://github.com/hitgil) fixed the CMake configuration when cross-compiling. +403. [Devon Thomas](https://github.com/ThomaDevOSU) mentioned the Artistic Style formatting in the contribution guidelines. +404. [Erik Hu](https://github.com/Erikhu1) made Coveralls upload errors non-fatal in the CI. +405. [co63oc](https://github.com/co63oc) fixed typos. +406. [DmitriBogdanov](https://github.com/DmitriBogdanov) fixed broken package manager links in the documentation. +407. [Bander](https://github.com/banderzhm) improved the MSVC compatibility of the C++ modules. +408. [Andy Choi](https://github.com/ccpong) removed an unnecessary `template` keyword before `get` in the README and the documentation. +409. [SamareshSingh](https://github.com/ssam18) fixed single-element brace initialization to copy/move instead of wrapping in an array, fixed the `WITH_DEFAULT` macros for `ordered_map`, and handled moved events in `serve_header.py`. +410. [Aditya](https://github.com/Lumowhisp) improved the documentation of the documentation generation. +411. [cheese1](https://github.com/cheese1) clarified the README. +412. [KhloodElhossiny](https://github.com/khloodelhossiny) enabled `std::string_view` keys in `operator[]`. +413. [Charles Cabergs](https://github.com/cacharle) fixed a `-Wtautological-constant-out-of-range-compare` warning. +414. [EALePain](https://github.com/EALePain) made the `std::tuple` conversion work with reference types such as `std::tie`. +415. [koala_oishi](https://github.com/chibi-dogs) fixed grammatical wording in the README. +416. [riccardoori11](https://github.com/riccardoori11) fixed a typo in the documentation. +417. [Swastik Bose](https://github.com/VasuBhakt) fixed the parent pointers after `update()` with `JSON_DIAGNOSTICS` and fixed the Doxygen autolinking of requirements. +418. [trdesilva](https://github.com/trdesilva) added `front`, `pop_front`, and `push_front` to `json_pointer`. +419. [Akhilesh Arora](https://github.com/akhilesharora) fixed an incomplete-type error with `ordered_json`. +420. [Hariom Phulre](https://github.com/hariomphulre) fixed the C++20 modules compilation with GCC. +421. [Kirill Lokotkov](https://github.com/RUSLoker) fixed printing `long double` values. +422. [George Sedov](https://github.com/radistmorse) added the `NLOHMANN_DEFINE_TYPE_*_WITH_NAMES` macros. +423. [Caillin Nugent](https://github.com/nugentcaillin) added the `NLOHMANN_JSON_SERIALIZE_ENUM_STRICT` macro. +424. [Cosmin D.](https://github.com/drcosmin) fixed `std::filesystem::path` conversions and added an MSVC workaround for `std::unique_ptr`. +425. [Paul Dreik](https://github.com/pauldreik) fixed a test relying on implementation-specific behavior. +426. [Daniel Falk](https://github.com/daniel-falk) added missing copyright notices to the SBOM. +427. [Federico Sfriso](https://github.com/federicosfriso05-dotcom) added support for constructing JSON values from C++20 range views. +428. [Luke Banicevic](https://github.com/banaboi) fixed corrupt BSON output for lengths exceeding `INT32_MAX`, cleaned up the BSON writer, and improved the documentation. +429. [Patrick Armstrong](https://github.com/Patrick10199) updated the CBOR references and the half-precision float assertions. +430. [Yash Bavadiya](https://github.com/xevrion) added checks to all BSON reads. +431. [hum4nBeing](https://github.com/hum4nBeing) fixed the overflow handling of high-precision numbers in UBJSON. +432. [tomatotomata](https://github.com/tomatotomata) added checks for reading CBOR tagged subtypes. +433. [YingqiDuan](https://github.com/YingqiDuan) documented the BSON interoperability. +434. [KBS](https://github.com/youdie006) documented the standards compliance and the strictness of `parse()` and `operator>>`. +435. [Petr Bělohlávek](https://github.com/petrbel) added Clang 21 and 22 to the CI. +436. [Dmitry Rantovov](https://github.com/darkdi) fixed the placement of a CBOR documentation block. +437. [ljcjclljc](https://github.com/ljcjclljc) fixed the comparison of large unsigned integers with signed integers. +438. [Sahil Kamate](https://github.com/sahilkamate03) fixed the handling of CBOR tags 0-5 and 21-23. +439. [Krishnanand G](https://github.com/Krishnanand-G) made the UBJSON writer reject `use_type` without `use_size`. +440. [whn](https://github.com/Whning0513) documented the lenient BSON input handling and corrected the complexity of `to_bson`. +441. [elix3r](https://github.com/22elix3r) fixed `update()` with `merge_objects` when merging a primitive into an object. +442. [Avionic Harshit](https://github.com/avionicharshit-byte) made `diff()` linear when an array shrinks. +443. [Qatadaha Bin Matloob](https://github.com/qatcod) fixed comparisons between integers and floats and fixed unparsable BJData output. +444. [Wu Shuwen](https://github.com/dajiaohuang) removed an unused include. Thanks a lot for helping out! Please [let me know](mailto:mail@nlohmann.me) if I forgot someone. diff --git a/cmake/ci.cmake b/cmake/ci.cmake index 9cc850564..a99788633 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -213,18 +213,38 @@ add_custom_target(ci_test_legacycomparison ) ############################################################################### -# Enable brace-init copy semantics. +# Validate UTF-8 with simdutf. ############################################################################### -add_custom_target(ci_test_brace_init_copy_semantics +add_custom_target(ci_test_simdutf COMMAND ${CMAKE_COMMAND} -DCMAKE_BUILD_TYPE=Debug -GNinja - -DJSON_BuildTests=ON -DJSON_FastTests=ON - -DCMAKE_CXX_FLAGS=-DJSON_BRACE_INIT_COPY_SEMANTICS=1 - -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_brace_init_copy_semantics - COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_brace_init_copy_semantics - COMMAND cd ${PROJECT_BINARY_DIR}/build_brace_init_copy_semantics && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure - COMMENT "Compile and test with brace-init copy semantics enabled" + -DJSON_BuildTests=ON -DJSON_TestSimdutf=ON + # simdutf needs C++17, so the library falls back to its scalar validator + # below that: build the suite at C++11 to cover the fallback with the macro + # defined, and at C++17 to run every test against simdutf itself + "-DJSON_TestStandards=11\;17" + -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_simdutf + COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_simdutf + COMMAND cd ${PROJECT_BINARY_DIR}/build_simdutf && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure + COMMENT "Compile and test with simdutf UTF-8 validation enabled" +) + +############################################################################### +# Enable strict NUL-byte handling. +############################################################################### + +add_custom_target(ci_test_strict_nul_handling + COMMAND ${CMAKE_COMMAND} + -DCMAKE_BUILD_TYPE=Debug -GNinja + -DJSON_BuildTests=ON -DJSON_FastTests=ON -DJSON_StrictNulHandling=ON + -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_strict_nul_handling + COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_strict_nul_handling + # unit-testsuites contains a fixture (a "1e308" test value) that relies on the + # legacy NUL-as-end-of-input behavior this macro disables; exclude it here, as + # it is expected to fail under strict NUL handling and is out of scope for it + COMMAND cd ${PROJECT_BINARY_DIR}/build_strict_nul_handling && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure -E "test-testsuites" + COMMENT "Compile and test with strict NUL-byte handling enabled" ) ############################################################################### @@ -242,6 +262,59 @@ add_custom_target(ci_test_noglobaludls COMMENT "Compile and test with global UDLs disabled" ) +############################################################################### +# Disable enum serialization. +############################################################################### + +add_custom_target(ci_test_disableenumserialization + COMMAND ${CMAKE_COMMAND} + -DCMAKE_BUILD_TYPE=Debug -GNinja + -DJSON_BuildTests=ON -DJSON_FastTests=ON -DJSON_DisableEnumSerialization=ON + -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_disableenumserialization + COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_disableenumserialization + COMMAND cd ${PROJECT_BINARY_DIR}/build_disableenumserialization && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure + COMMENT "Compile and test with enum serialization disabled" +) + +############################################################################### +# Skip the multiple-inclusion library version check. +############################################################################### + +# tests/src/skip_library_version_check.cpp deliberately simulates a scenario +# (mixing two differently-versioned inclusions of the library in one +# translation unit) that unavoidably triggers the compiler's own "macro +# redefined" warning, so -- unlike the ci_test_* targets above -- it is +# compiled directly here, with a modest warning set, instead of being folded +# into the library's own -Weverything/-Werror unit test matrix. +add_custom_target(ci_test_skiplibraryversioncheck + COMMAND ${CMAKE_COMMAND} -E make_directory ${PROJECT_BINARY_DIR}/skip_library_version_check + COMMAND ${CMAKE_CXX_COMPILER} -std=c++11 -Wall -Wextra + -I${PROJECT_SOURCE_DIR}/include + ${PROJECT_SOURCE_DIR}/tests/src/skip_library_version_check.cpp + -o ${PROJECT_BINARY_DIR}/skip_library_version_check/skip_library_version_check + COMMAND ${PROJECT_BINARY_DIR}/skip_library_version_check/skip_library_version_check + COMMENT "Compile and run a translation unit simulating a mismatched library version, with JSON_SKIP_LIBRARY_VERSION_CHECK defined" +) + +############################################################################### +# Disable thread-local storage. +############################################################################### + +# Without thread-local storage, the copy constructor cannot bound its descent +# and copies every object and array without the call stack. That path is +# otherwise only reached by values nested deeper than the bound, so this target +# is what runs the whole test suite through it. +add_custom_target(ci_test_no_thread_local + COMMAND ${CMAKE_COMMAND} + -DCMAKE_BUILD_TYPE=Debug -GNinja + -DJSON_BuildTests=ON + -DCMAKE_CXX_FLAGS=-DJSON_NO_THREAD_LOCAL + -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_no_thread_local + COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_no_thread_local + COMMAND cd ${PROJECT_BINARY_DIR}/build_no_thread_local && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure + COMMENT "Compile and test without thread-local storage" +) + ############################################################################### # Coverage. ############################################################################### @@ -294,7 +367,7 @@ file(GLOB_RECURSE INDENT_FILES ${PROJECT_SOURCE_DIR}/tests/src/*.cpp ${PROJECT_SOURCE_DIR}/tests/src/*.hpp ${PROJECT_SOURCE_DIR}/tests/benchmarks/src/benchmarks.cpp - ${PROJECT_SOURCE_DIR}/docs/examples/*.cpp + ${PROJECT_SOURCE_DIR}/docs/mkdocs/docs/examples/*.cpp ) set(include_dir ${PROJECT_SOURCE_DIR}/single_include/nlohmann) diff --git a/cmake/scripts/gen_bazel_build_file.cmake b/cmake/scripts/gen_bazel_build_file.cmake index e754d387d..3c7db9493 100644 --- a/cmake/scripts/gen_bazel_build_file.cmake +++ b/cmake/scripts/gen_bazel_build_file.cmake @@ -1,24 +1,58 @@ # generate Bazel BUILD file +# +# usage: cmake -P cmake/scripts/gen_bazel_build_file.cmake (or: make BUILD.bazel) +# +# The header list of the "json" target is derived from the files in include/. Everything else is fixed text below, +# so edit this script rather than BUILD.bazel. -set(PROJECT_ROOT "${CMAKE_CURRENT_LIST_DIR}/../..") +get_filename_component(PROJECT_ROOT "${CMAKE_CURRENT_LIST_DIR}/../.." ABSOLUTE) set(BUILD_FILE "${PROJECT_ROOT}/BUILD.bazel") -file(GLOB_RECURSE HEADERS LIST_DIRECTORIES false RELATIVE "${PROJECT_ROOT}" "include/*.hpp") +file(GLOB_RECURSE HEADERS LIST_DIRECTORIES false RELATIVE "${PROJECT_ROOT}" "${PROJECT_ROOT}/include/*.hpp") +list(SORT HEADERS) + +set(CONTENT [=[ +load("@rules_cc//cc:cc_library.bzl", "cc_library") +load("@rules_license//rules:license.bzl", "license") + +package( + default_applicable_licenses = [":license"], +) + +exports_files([ + "LICENSE.MIT", +]) + +license( + name = "license", + license_kinds = ["@rules_license//licenses/spdx:MIT"], + license_text = "LICENSE.MIT", +) -file(WRITE "${BUILD_FILE}" [=[ cc_library( name = "json", hdrs = [ ]=]) foreach(header ${HEADERS}) - file(APPEND "${BUILD_FILE}" " \"${header}\",\n") + string(APPEND CONTENT " \"${header}\",\n") endforeach() -file(APPEND "${BUILD_FILE}" [=[ +string(APPEND CONTENT [=[ ], includes = ["include"], visibility = ["//visibility:public"], alwayslink = True, ) + +cc_library( + name = "singleheader-json", + hdrs = [ + "single_include/nlohmann/json.hpp", + ], + includes = ["single_include"], + visibility = ["//visibility:public"], +) ]=]) + +file(WRITE "${BUILD_FILE}" "${CONTENT}") diff --git a/cmake/scripts/gen_hedley_undef_check.cmake b/cmake/scripts/gen_hedley_undef_check.cmake new file mode 100644 index 000000000..fc8cfec2c --- /dev/null +++ b/cmake/scripts/gen_hedley_undef_check.cmake @@ -0,0 +1,112 @@ +# Shared extractor for the JSON_HEDLEY_* macro names defined in hedley.hpp. +# +# Every macro that hedley.hpp #defines must be #undef-ed again once json.hpp +# has been fully processed (see include/nlohmann/detail/macro_unscope.hpp +# and https://github.com/nlohmann/json/issues/5408). Deriving the macro list +# straight from hedley.hpp here -- instead of hand-maintaining it in two +# places -- means hedley_undef.hpp and the regression test that checks for +# leaked macros can never drift apart, even after a future `make +# update_hedley` pulls in new macros from upstream Hedley. +# +# MODE=undef (default): write hedley_undef.hpp (SPDX header, #pragma once, +# one #undef per macro name) -- used by `make update_hedley_undef` +# MODE=checks: write one #ifdef/FAIL_CHECK/#endif per macro name, +# meant to be #include-d inside a TEST_CASE -- used by +# tests/CMakeLists.txt to (re)generate the include for +# tests/src/unit-no-macro-leak.cpp +# +# Required variables: +# HEDLEY_HPP path to include/nlohmann/thirdparty/hedley/hedley.hpp +# OUTPUT path of the file to (over)write +# Optional: +# MODE "undef" (default) or "checks" + +if(NOT DEFINED HEDLEY_HPP OR NOT DEFINED OUTPUT) + message(FATAL_ERROR "HEDLEY_HPP and OUTPUT must be set") +endif() + +if(NOT EXISTS "${HEDLEY_HPP}") + message(FATAL_ERROR "Hedley header not found: ${HEDLEY_HPP}") +endif() + +if(NOT DEFINED MODE) + set(MODE undef) +endif() + +if(NOT MODE STREQUAL "undef" AND NOT MODE STREQUAL "checks") + message(FATAL_ERROR "MODE must be undef or checks, got: ${MODE}") +endif() + +# Line-anchored, like `grep -oE "^[[:blank:]]*#[[:blank:]]*define[[:blank:]]+JSON_HEDLEY_[A-Za-z0-9_]+"`. +# Unanchored matching would also pick up JSON_HEDLEY_* mentions inside +# comments or string literals elsewhere in the file, which must not turn +# into #undef lines. +file(STRINGS "${HEDLEY_HPP}" hedley_lines) +set(macro_names) +foreach(line IN LISTS hedley_lines) + if("${line}" MATCHES "^[ \t]*#[ \t]*define[ \t]+(JSON_HEDLEY_[A-Za-z0-9_]+)") + list(APPEND macro_names "${CMAKE_MATCH_1}") + endif() +endforeach() + +if(NOT macro_names) + message(FATAL_ERROR "No JSON_HEDLEY_* macros found in ${HEDLEY_HPP}") +endif() + +list(REMOVE_DUPLICATES macro_names) +# Lexicographic, locale-independent (ASCII-only names) -- matches `LC_ALL=C sort`. +list(SORT macro_names COMPARE STRING) +list(LENGTH macro_names macro_count) + +set(generated "") +if(MODE STREQUAL "undef") + # Same banner `make update_hedley_undef` would stamp by hand, so the + # recipe is self-contained and its output is byte-stable across reruns. + # The embedded SPDX tags below are part of the *generated* file's + # content, not a REUSE header for this .cmake script itself (which is + # already covered by the blanket "Files: *" rule in .reuse/dep5) -- keep + # them wrapped in REUSE-IgnoreStart/End so `reuse lint` does not try to + # parse "MIT\n")" as this file's own SPDX-License-Identifier value. + # REUSE-IgnoreStart + string(APPEND generated "// __ _____ _____ _____\n") + string(APPEND generated "// __| | __| | | | JSON for Modern C++\n") + string(APPEND generated "// | | |__ | | | | | | version 3.12.0\n") + string(APPEND generated "// |_____|_____|_____|_|___| https://github.com/nlohmann/json\n") + string(APPEND generated "//\n") + string(APPEND generated "// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann \n") + string(APPEND generated "// SPDX-License-Identifier: MIT\n") + # REUSE-IgnoreEnd + string(APPEND generated "\n") + string(APPEND generated "#pragma once\n") + string(APPEND generated "\n") + foreach(name IN LISTS macro_names) + string(APPEND generated "#undef ${name}\n") + endforeach() +else() + string(APPEND generated "// This file is generated by cmake/scripts/gen_hedley_undef_check.cmake\n") + string(APPEND generated "// from include/nlohmann/thirdparty/hedley/hedley.hpp. Do not edit it by\n") + string(APPEND generated "// hand -- it is regenerated on every build. ${macro_count} macros checked.\n\n") + foreach(name IN LISTS macro_names) + string(APPEND generated "#ifdef ${name}\n") + string(APPEND generated " FAIL_CHECK(\"${name} leaked after including nlohmann/json.hpp\");\n") + string(APPEND generated "#endif\n") + endforeach() +endif() + +get_filename_component(output_dir "${OUTPUT}" DIRECTORY) +if(output_dir) + file(MAKE_DIRECTORY "${output_dir}") +endif() + +# Avoid rewriting the file (and busting downstream incremental rebuilds) +# when the content has not actually changed. +set(write_output TRUE) +if(EXISTS "${OUTPUT}") + file(READ "${OUTPUT}" existing_content) + if(existing_content STREQUAL generated) + set(write_output FALSE) + endif() +endif() +if(write_output) + file(WRITE "${OUTPUT}" "${generated}") +endif() diff --git a/docs/mkdocs/docs/api/basic_json/accept.md b/docs/mkdocs/docs/api/basic_json/accept.md index 5ba009c3e..0cdcae3a8 100644 --- a/docs/mkdocs/docs/api/basic_json/accept.md +++ b/docs/mkdocs/docs/api/basic_json/accept.md @@ -90,6 +90,10 @@ Linear in the length of the input. The parser is a predictive LL(1) parser. A UTF-8 byte order mark is silently ignored. +By default, a `'\0'` (NUL) byte anywhere in the input is treated as end of input, rather than as an ordinary (and, +outside of a string, invalid) byte; see the [FAQ entry](../../home/faq.md#nul-bytes-in-the-input) for details and the +[`JSON_STRICT_NUL_HANDLING`](../macros/json_strict_nul_handling.md) macro to opt into rejecting it instead. + ## Examples ??? example @@ -111,6 +115,8 @@ A UTF-8 byte order mark is silently ignored. - [parse](parse.md) - deserialize from a compatible input - [sax_parse](sax_parse.md) - parse input using the SAX interface - [operator>>](../operator_gtgt.md) - deserialize from stream +- [`JSON_STRICT_NUL_HANDLING`](../macros/json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input + instead of treating it as end of input ## Version history @@ -120,6 +126,8 @@ A UTF-8 byte order mark is silently ignored. - Added `ignore_trailing_commas` in version 3.13.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- `JSON_STRICT_NUL_HANDLING` added in version 3.13.0 to optionally reject a NUL byte in the input instead of treating + it as end of input; planned to become the default in version 4.0.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/array_t.md b/docs/mkdocs/docs/api/basic_json/array_t.md index dd2b901d5..b6e41a6ad 100644 --- a/docs/mkdocs/docs/api/basic_json/array_t.md +++ b/docs/mkdocs/docs/api/basic_json/array_t.md @@ -14,7 +14,11 @@ To store objects in C++, a type is defined by the template parameters explained ## Template parameters `ArrayType` -: container type to store arrays (e.g., `std::vector` or `std::list`) +: container type to store arrays. It must be a vector-like container: the library uses `operator[]`, `at()`, and + `resize()`, and requires random-access iterators. `#!cpp std::vector` and `#!cpp std::deque` qualify; + `#!cpp std::list` does not. See + [Template Parameter Requirements](../../features/types/template_parameters.md#arraytype) for the full list of + requirements. `AllocatorType` : the allocator to use for objects (e.g., `std::allocator`) @@ -66,3 +70,4 @@ Arrays are stored as pointers in a `basic_json` type. That is, for any access to ## Version history - Added in version 1.0.0. +- Made `capacity()` optional, so that array types such as `#!cpp std::deque` can be used, in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/binary_t.md b/docs/mkdocs/docs/api/basic_json/binary_t.md index 64506b25f..36600748a 100644 --- a/docs/mkdocs/docs/api/basic_json/binary_t.md +++ b/docs/mkdocs/docs/api/basic_json/binary_t.md @@ -42,7 +42,9 @@ represent a byte array in modern C++. `value_type` must additionally be exactly one byte wide (e.g., `std::uint8_t`/`char`/`std::byte`): the binary serializers (CBOR, MessagePack, BSON, UBJSON) read and write the container's raw bytes via `reinterpret_cast`, which is only correct for byte-sized elements -- a container like - `#!cpp std::vector` will not work as `BinaryType`. + `#!cpp std::vector` will not work as `BinaryType`. The elements must be stored contiguously, and + the binary readers additionally require `resize()` and `operator[]`. See + [Template Parameter Requirements](../../features/types/template_parameters.md#binarytype) for the full list. ## Notes @@ -50,6 +52,11 @@ represent a byte array in modern C++. The default values for `BinaryType` is `#!cpp std::vector`. +#### Supported byte types + +`#!cpp std::vector`, `#!cpp std::vector`, and `#!cpp std::vector` are supported. +Regardless of which of them is configured, [`dump`](dump.md) writes the bytes as the numbers 0..255. + #### Custom BinaryType behavior When a custom `BinaryType` is configured (other than the default `#!cpp std::vector`), you can assign @@ -126,3 +133,6 @@ type `#!cpp binary_t*` must be dereferenced. ## Version history - Added in version 3.8.0. Changed the type of subtype to `std::uint64_t` in version 3.10.0. +- Fixed [`dump`](dump.md), [`std::hash`](std_hash.md), and [`to_ubjson`](to_ubjson.md) for byte types that are not + integers (e.g., `#!cpp std::byte`) in version 3.13.0. `dump` now writes the bytes of a signed byte type (e.g., + `#!cpp char`) as 0..255 rather than as negative numbers. diff --git a/docs/mkdocs/docs/api/basic_json/boolean_t.md b/docs/mkdocs/docs/api/basic_json/boolean_t.md index c30afefdc..bfb8c3426 100644 --- a/docs/mkdocs/docs/api/basic_json/boolean_t.md +++ b/docs/mkdocs/docs/api/basic_json/boolean_t.md @@ -11,6 +11,14 @@ literals `#!json true` and `#!json false`. To store boolean values in C++, a type is defined by the template parameter `BooleanType` which chooses the type to use. +## Template parameters + +`BooleanType` +: the type to store booleans. As it is stored directly inside a `basic_json` value (in a union), it must be a + trivially default-constructible, trivially copyable, and trivially destructible type that is convertible to and + from `#!cpp bool`. See + [Template Parameter Requirements](../../features/types/template_parameters.md#booleantype). + ## Notes #### Default type diff --git a/docs/mkdocs/docs/api/basic_json/index.md b/docs/mkdocs/docs/api/basic_json/index.md index bd33f31ac..c5556ea3b 100644 --- a/docs/mkdocs/docs/api/basic_json/index.md +++ b/docs/mkdocs/docs/api/basic_json/index.md @@ -35,6 +35,10 @@ class basic_json; | `BinaryType` | type for binary arrays | [`binary_t`](binary_t.md) | | `CustomBaseClass` | extension point for user code | [`json_base_class_t`](json_base_class_t.md) | +The library imposes a number of requirements on these types that are not expressed as C++ concepts, such as the +container operations `object_t` and `array_t` must provide, or the fact that `StringType` must be `char`-based. They +are collected in [Template Parameter Requirements](../../features/types/template_parameters.md). + ## Specializations - [**json**](../json.md) - default specialization diff --git a/docs/mkdocs/docs/api/basic_json/insert.md b/docs/mkdocs/docs/api/basic_json/insert.md index 14d5823c1..fcb1e6e44 100644 --- a/docs/mkdocs/docs/api/basic_json/insert.md +++ b/docs/mkdocs/docs/api/basic_json/insert.md @@ -88,6 +88,8 @@ Strong exception safety: if an exception occurs, the original value stays intact do not belong to the same JSON value; example: `"iterators do not fit"` - Throws [`invalid_iterator.211`](../../home/exceptions.md#jsonexceptioninvalid_iterator211) if `first` or `last` are iterators into container for which insert is called; example: `"passed iterators may not belong to container"` + - Throws [`invalid_iterator.202`](../../home/exceptions.md#jsonexceptioninvalid_iterator202) if `first` or `last` + do not point to an array; example: `"iterators first and last must point to arrays"` 4. The function can throw the following exceptions: - Throws [`type_error.309`](../../home/exceptions.md#jsonexceptiontype_error309) if called on JSON values other than arrays; example: `"cannot use insert() with string"` diff --git a/docs/mkdocs/docs/api/basic_json/json_base_class_t.md b/docs/mkdocs/docs/api/basic_json/json_base_class_t.md index dfe5f1cb7..7add54098 100644 --- a/docs/mkdocs/docs/api/basic_json/json_base_class_t.md +++ b/docs/mkdocs/docs/api/basic_json/json_base_class_t.md @@ -21,8 +21,11 @@ The default value for `CustomBaseClass` is `void`. In this case, an #### Limitations -The type `CustomBaseClass` has to be a default-constructible class. +The type `CustomBaseClass` has to be a default-constructible, non-`final` class. `basic_json` only supports copy/move construction/assignment if `CustomBaseClass` does so as well. +A `CustomBaseClass` with non-static data members forfeits `basic_json`'s +[standard layout](https://en.cppreference.com/w/cpp/named_req/StandardLayoutType) guarantee. See +[Template Parameter Requirements](../../features/types/template_parameters.md#custombaseclass). ## Examples diff --git a/docs/mkdocs/docs/api/basic_json/json_serializer.md b/docs/mkdocs/docs/api/basic_json/json_serializer.md index 24a37735c..b92cf3d17 100644 --- a/docs/mkdocs/docs/api/basic_json/json_serializer.md +++ b/docs/mkdocs/docs/api/basic_json/json_serializer.md @@ -19,6 +19,12 @@ using json_serializer = JSONSerializer; The default values for `json_serializer` is [`adl_serializer`](../adl_serializer/index.md). +#### Requirements + +A custom serializer must provide `#!cpp static void to_json(basic_json&, T)` for every type it serializes, and either +`#!cpp static void from_json(const basic_json&, T&)` or `#!cpp static T from_json(const basic_json&)` for every type it +deserializes. See [Template Parameter Requirements](../../features/types/template_parameters.md#jsonserializer). + ## Examples ??? example diff --git a/docs/mkdocs/docs/api/basic_json/number_float_t.md b/docs/mkdocs/docs/api/basic_json/number_float_t.md index 3e8933da6..83c7011c5 100644 --- a/docs/mkdocs/docs/api/basic_json/number_float_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_float_t.md @@ -20,6 +20,16 @@ used. To store floating-point numbers in C++, a type is defined by the template parameter `NumberFloatType` which chooses the type to use. +## Template parameters + +`NumberFloatType` +: the type to store floating-point numbers. Parsing and serialization are implemented in terms of + `#!cpp std::strtof`/`#!cpp std::strtod`/`#!cpp std::strtold` and `#!cpp std::snprintf`, so the type must be + `#!cpp float`, `#!cpp double`, or `#!cpp long double`. The + [binary formats](../../features/binary_formats/index.md) additionally require `#!cpp float` or `#!cpp double`, + because they have no encoding for `#!cpp long double`. See + [Template Parameter Requirements](../../features/types/template_parameters.md#numberfloattype). + ## Notes #### Default type diff --git a/docs/mkdocs/docs/api/basic_json/number_integer_t.md b/docs/mkdocs/docs/api/basic_json/number_integer_t.md index 79cbdf8ca..9a2ffab7f 100644 --- a/docs/mkdocs/docs/api/basic_json/number_integer_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_integer_t.md @@ -20,6 +20,13 @@ used. To store integer numbers in C++, a type is defined by the template parameter `NumberIntegerType` which chooses the type to use. +## Template parameters + +`NumberIntegerType` +: the type to store signed integers. It must be a **signed integral** type (`#!cpp std::is_integral`) with a + `#!cpp std::numeric_limits` specialization, and it is stored directly inside a `basic_json` value. See + [Template Parameter Requirements](../../features/types/template_parameters.md#numberintegertype-and-numberunsignedtype). + ## Notes #### Default type diff --git a/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md b/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md index f1010f2a6..674f7711d 100644 --- a/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md @@ -20,6 +20,14 @@ used. To store unsigned integer numbers in C++, a type is defined by the template parameter `NumberUnsignedType` which chooses the type to use. +## Template parameters + +`NumberUnsignedType` +: the type to store unsigned integers. It must be an **unsigned integral** type (`#!cpp std::is_integral`) with a + `#!cpp std::numeric_limits` specialization, and it must be able to represent the absolute value of every + [`number_integer_t`](number_integer_t.md) value. See + [Template Parameter Requirements](../../features/types/template_parameters.md#numberintegertype-and-numberunsignedtype). + ## Notes #### Default type diff --git a/docs/mkdocs/docs/api/basic_json/object_comparator_t.md b/docs/mkdocs/docs/api/basic_json/object_comparator_t.md index d41b98229..bbda0a0e7 100644 --- a/docs/mkdocs/docs/api/basic_json/object_comparator_t.md +++ b/docs/mkdocs/docs/api/basic_json/object_comparator_t.md @@ -30,3 +30,5 @@ and [`default_object_comparator_t`](default_object_comparator_t.md) otherwise. - Added in version 3.0.0. - Changed to be conditionally defined as `#!cpp typename object_t::key_compare` or `default_object_comparator_t` in version 3.11.0. +- Fixed the fallback to `default_object_comparator_t`, which previously failed to compile for object types without a + `key_compare` member type, in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/object_t.md b/docs/mkdocs/docs/api/basic_json/object_t.md index de41b86e4..6ce393a1d 100644 --- a/docs/mkdocs/docs/api/basic_json/object_t.md +++ b/docs/mkdocs/docs/api/basic_json/object_t.md @@ -18,7 +18,11 @@ To store objects in C++, a type is defined by the template parameters described ## Template parameters `ObjectType` -: the container to store objects (e.g., `std::map` or `std::unordered_map`) +: the container to store objects. Its template parameters must have the same order and meaning as those of + `std::map`; in particular, the third parameter is a comparator. `#!cpp std::unordered_map`, whose third parameter + is a hash function, therefore needs an adapter -- see + [Template Parameter Requirements](../../features/types/template_parameters.md#objecttype) for the full list of + requirements, an adapter example, and the containers that are known to work. `StringType` : the type of the keys or names (e.g., `std::string`). The comparison function `std::less` is used to @@ -122,3 +126,4 @@ the object is silently converted as an array of key-value pairs, which is incorr ## Version history - Added in version 1.0.0. +- Allowed object types whose `erase(iterator)` returns `#!cpp void` in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/parse.md b/docs/mkdocs/docs/api/basic_json/parse.md index ef22e9873..20bb1c708 100644 --- a/docs/mkdocs/docs/api/basic_json/parse.md +++ b/docs/mkdocs/docs/api/basic_json/parse.md @@ -103,6 +103,10 @@ A UTF-8 byte order mark is silently ignored. Invalid Unicode escapes and unpaired surrogates in the input are reported as [`parse_error.101`](../../home/exceptions.md#jsonexceptionparse_error101) with a detailed message. +By default, a `'\0'` (NUL) byte anywhere in the input is treated as end of input, rather than as an ordinary (and, +outside of a string, invalid) byte; see the [FAQ entry](../../home/faq.md#nul-bytes-in-the-input) for details and the +[`JSON_STRICT_NUL_HANDLING`](../macros/json_strict_nul_handling.md) macro to opt into rejecting it instead. + ## Examples ??? example "Parsing from a character array" @@ -236,6 +240,8 @@ Invalid Unicode escapes and unpaired surrogates in the input are reported as - [accept](accept.md) - check if the input is valid JSON - [sax_parse](sax_parse.md) - parse input using the SAX interface - [operator>>](../operator_gtgt.md) - deserialize from stream +- [`JSON_STRICT_NUL_HANDLING`](../macros/json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input + instead of treating it as end of input ## Version history @@ -246,6 +252,8 @@ Invalid Unicode escapes and unpaired surrogates in the input are reported as - Added `ignore_trailing_commas` in version 3.13.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- `JSON_STRICT_NUL_HANDLING` added in version 3.13.0 to optionally reject a NUL byte in the input instead of treating + it as end of input; planned to become the default in version 4.0.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/patch.md b/docs/mkdocs/docs/api/basic_json/patch.md index 432e7c128..fa25b2699 100644 --- a/docs/mkdocs/docs/api/basic_json/patch.md +++ b/docs/mkdocs/docs/api/basic_json/patch.md @@ -34,6 +34,10 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va ("add", "remove", "move") - Throws [`out_of_range.411`](../../home/exceptions.md#jsonexceptionout_of_range411) if an "add" operation's target location has a parent that is neither an object nor an array. +- Throws [`out_of_range.413`](../../home/exceptions.md#jsonexceptionout_of_range413) if a "remove" operation's target + location has a parent that is neither an object nor an array. +- Throws [`out_of_range.414`](../../home/exceptions.md#jsonexceptionout_of_range414) if a "move" operation's "from" + location is a proper prefix of its "path" location. - Throws [`other_error.501`](../../home/exceptions.md#jsonexceptionother_error501) if "test" operation was unsuccessful. @@ -75,3 +79,7 @@ is thrown. In any case, the original value is not changed: the patch is applied - Added in version 2.0.0. - Added [`out_of_range.411`](../../home/exceptions.md#jsonexceptionout_of_range411) and stopped relying on an internal assertion when an "add" operation's target location has a non-object/non-array parent in version 3.13.0. +- Added [`out_of_range.413`](../../home/exceptions.md#jsonexceptionout_of_range413) and stopped silently ignoring a "remove" operation whose target + location has a non-object/non-array parent in version 3.13.0. +- Added [`out_of_range.414`](../../home/exceptions.md#jsonexceptionout_of_range414) and rejected a "move" operation whose "from" location is a proper + prefix of its "path" location instead of silently producing a corrupted result in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/patch_inplace.md b/docs/mkdocs/docs/api/basic_json/patch_inplace.md index ae2b5e6a9..7ae85aaaa 100644 --- a/docs/mkdocs/docs/api/basic_json/patch_inplace.md +++ b/docs/mkdocs/docs/api/basic_json/patch_inplace.md @@ -30,6 +30,10 @@ No guarantees, value may be corrupted by an unsuccessful patch operation. ("add", "remove", "move") - Throws [`out_of_range.411`](../../home/exceptions.md#jsonexceptionout_of_range411) if an "add" operation's target location has a parent that is neither an object nor an array. +- Throws [`out_of_range.413`](../../home/exceptions.md#jsonexceptionout_of_range413) if a "remove" operation's target + location has a parent that is neither an object nor an array. +- Throws [`out_of_range.414`](../../home/exceptions.md#jsonexceptionout_of_range414) if a "move" operation's "from" + location is a proper prefix of its "path" location. - Throws [`other_error.501`](../../home/exceptions.md#jsonexceptionother_error501) if "test" operation was unsuccessful. @@ -72,3 +76,7 @@ function throws an exception. - Added in version 3.11.0. - Added [`out_of_range.411`](../../home/exceptions.md#jsonexceptionout_of_range411) and stopped relying on an internal assertion when an "add" operation's target location has a non-object/non-array parent in version 3.13.0. +- Added [`out_of_range.413`](../../home/exceptions.md#jsonexceptionout_of_range413) and stopped silently ignoring a "remove" operation whose target + location has a non-object/non-array parent in version 3.13.0. +- Added [`out_of_range.414`](../../home/exceptions.md#jsonexceptionout_of_range414) and rejected a "move" operation whose "from" location is a proper + prefix of its "path" location instead of silently producing a corrupted result in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/string_t.md b/docs/mkdocs/docs/api/basic_json/string_t.md index 97c586f28..e8be0fe0e 100644 --- a/docs/mkdocs/docs/api/basic_json/string_t.md +++ b/docs/mkdocs/docs/api/basic_json/string_t.md @@ -23,6 +23,11 @@ JSON class into byte-sized characters during deserialization. `StringType`. To work with wide-character data, convert it to/from UTF-8 at the boundary instead -- see the FAQ's [wide string handling](../../home/faq.md#wide-string-handling) section for a conversion recipe. + Beyond the character type, the library expects a substantial part of the `#!cpp std::string` interface (contiguous + null-terminated `data()`, `substr()`, `find()`, `append()`, ...). See + [Template Parameter Requirements](../../features/types/template_parameters.md#stringtype) for the full list and + for the string types that are known to work. + ## Notes #### Default type @@ -78,3 +83,5 @@ and an example. ## Version history - Added in version 1.0.0. +- Removed the requirement that `string_t` be implicitly convertible from `#!cpp std::string`, which the BSON writer and + the UBJSON reader relied on, in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/swap.md b/docs/mkdocs/docs/api/basic_json/swap.md index 3a3d288fb..aa5aa6c4c 100644 --- a/docs/mkdocs/docs/api/basic_json/swap.md +++ b/docs/mkdocs/docs/api/basic_json/swap.md @@ -34,10 +34,14 @@ void swap(typename binary_t::container_type& other); ``` 1. Exchanges the contents of the JSON value with those of `other`. Does not invoke any move, copy, or swap operations on - individual elements. All iterators and references remain valid. The past-the-end iterator is invalidated. + individual elements. All iterators and references remain valid. The past-the-end iterator is invalidated. If macro + [`JSON_DIAGNOSTIC_POSITIONS`](../macros/json_diagnostic_positions.md) is defined to `#!cpp 1`, the + [`start_pos()`](start_pos.md)/[`end_pos()`](end_pos.md) diagnostic positions are exchanged along with the value. 2. Exchanges the contents of the JSON value from `left` with those of `right`. Does not invoke any move, copy, or swap operations on individual elements. All iterators and references remain valid. The past-the-end iterator is - invalidated. Implemented as a friend function callable via ADL. + invalidated. Implemented as a friend function callable via ADL. If macro + [`JSON_DIAGNOSTIC_POSITIONS`](../macros/json_diagnostic_positions.md) is defined to `#!cpp 1`, the + [`start_pos()`](start_pos.md)/[`end_pos()`](end_pos.md) diagnostic positions are exchanged along with the value. 3. Exchanges the contents of a JSON array with those of `other`. Does not invoke any move, copy, or swap operations on individual elements. All iterators and references remain valid. The past-the-end iterator is invalidated. 4. Exchanges the contents of a JSON object with those of `other`. Does not invoke any move, copy, or swap operations on diff --git a/docs/mkdocs/docs/api/basic_json/to_bjdata.md b/docs/mkdocs/docs/api/basic_json/to_bjdata.md index 9c1efb183..2cf936a51 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/to_bjdata.md @@ -52,6 +52,11 @@ optional, `#!cpp bjdata_version_t::draft2` by default. Strong guarantee: if an exception is thrown, there are no changes in the JSON value. +## Exceptions + +- Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` + is false. + ## Complexity Linear in the size of the JSON value `j`. diff --git a/docs/mkdocs/docs/api/basic_json/to_bson.md b/docs/mkdocs/docs/api/basic_json/to_bson.md index 6374d4c86..786fbc9e0 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bson.md +++ b/docs/mkdocs/docs/api/basic_json/to_bson.md @@ -46,7 +46,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va ## Complexity -Linear in the size of the JSON value `j`. +Proportional to the size of the JSON value `j` multiplied by its maximum nesting +depth, `O(n × d)`. BSON length prefixes are computed recursively before nested +values are written. ## Examples diff --git a/docs/mkdocs/docs/api/basic_json/to_ubjson.md b/docs/mkdocs/docs/api/basic_json/to_ubjson.md index 39b184e74..694fbbba6 100644 --- a/docs/mkdocs/docs/api/basic_json/to_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/to_ubjson.md @@ -45,6 +45,11 @@ The exact mapping and its limitations are described on a [dedicated page](../../ Strong guarantee: if an exception is thrown, there are no changes in the JSON value. +## Exceptions + +- Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` + is false. + ## Complexity Linear in the size of the JSON value `j`. diff --git a/docs/mkdocs/docs/api/basic_json/unflatten.md b/docs/mkdocs/docs/api/basic_json/unflatten.md index 1af243b58..ac88cd73e 100644 --- a/docs/mkdocs/docs/api/basic_json/unflatten.md +++ b/docs/mkdocs/docs/api/basic_json/unflatten.md @@ -37,7 +37,14 @@ Linear in the size of the JSON value. ## Notes Empty objects and arrays are flattened by [`flatten()`](flatten.md) to `#!json null` values and cannot unflattened to -their original type. Apart from this example, for a JSON value `j`, the following is always true: +their original type. + +A flattened array and a flattened object whose keys are array indices are indistinguishable, because both are +described by the same JSON pointers. A value is therefore restored as an array if and only if one of its keys is the +reference token `0`, and as an object otherwise: `#!json {"2": 1}` is restored unchanged, whereas `#!json {"0": 1}` is +restored as `#!json [1]`. This decision does not depend on the order in which the flattened object is iterated. + +Apart from these two cases, for a JSON value `j`, the following is always true: `#!cpp j == j.flatten().unflatten()`. ## Examples @@ -63,3 +70,4 @@ their original type. Apart from this example, for a JSON value `j`, the followin ## Version history - Added in version 2.0.0. +- Made the array/object decision independent of the object's iteration order in version 3.13.0. diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index 507c04932..70f02a7a6 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -14,6 +14,11 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_DIAGNOSTIC_POSITIONS**](json_diagnostic_positions.md) - access positions of elements - [**JSON_NOEXCEPTION**](json_noexception.md) - switch off exceptions +## Parsing + +- [**JSON_STRICT_NUL_HANDLING**](json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of + treating it as end of input + ## Language support - [**JSON_HAS_CPP_11**
**JSON_HAS_CPP_14**
**JSON_HAS_CPP_17**
**JSON_HAS_CPP_20**](json_has_cpp_11.md) - set supported C++ standard @@ -22,8 +27,10 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_HAS_STD_FORMAT**](json_has_std_format.md) - control `std::format`/`std::formatter` support - [**JSON_HAS_THREE_WAY_COMPARISON**](json_has_three_way_comparison.md) - control 3-way comparison support - [**JSON_NO_IO**](json_no_io.md) - switch off functions relying on certain C++ I/O headers +- [**JSON_NO_THREAD_LOCAL**](json_no_thread_local.md) - switch off the use of `thread_local` storage - [**JSON_SKIP_UNSUPPORTED_COMPILER_CHECK**](json_skip_unsupported_compiler_check.md) - do not warn about unsupported compilers - [**JSON_USE_GLOBAL_UDLS**](json_use_global_udls.md) - place user-defined string literals (UDLs) into the global namespace +- [**JSON_USE_SIMDUTF**](json_use_simdutf.md) - use the simdutf library to accelerate UTF-8 validation ## Library version diff --git a/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md b/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md index 970c20537..2301a0486 100644 --- a/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md +++ b/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md @@ -38,6 +38,28 @@ The default value is `0` (disabled — existing behavior is preserved). This macro must be defined **before** including ``. Defining it after the include has no effect. +!!! warning "Applies to every single-element list" + + The macro does not only affect a single JSON value in braces. **Any** single-element braced list is treated as its + element, so it no longer creates a one-element array: + + ```cpp + json j1 = {1}; // 1, not [1] + json j2 = {"text"}; // "text", not ["text"] + json j3 = {{1, 2}}; // [1,2], not [[1,2]] + ``` + + Code that relies on these producing arrays must use `json::array()` instead (see below). Lists with more than one + element, and a single `[string, value]` pair such as `{{"key", "value"}}`, which still creates an object, are not + affected. The library's own conversions are not affected either: for example, `std::tuple{5}` still becomes + `[5]`. + +!!! note "ABI compatibility" + + The value of this macro is encoded in the [namespace](../../features/namespace.md) (tag `_bics`), resulting in + distinct symbol names. Translation units compiled with and without it can therefore be linked into the same program + without One Definition Rule (ODR) violations, but they cannot exchange instances of library types. + !!! tip "Workaround without the macro" To explicitly create a single-element array without enabling this macro, use `json::array()`: diff --git a/docs/mkdocs/docs/api/macros/json_no_thread_local.md b/docs/mkdocs/docs/api/macros/json_no_thread_local.md new file mode 100644 index 000000000..126116ec3 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_no_thread_local.md @@ -0,0 +1,47 @@ +# JSON_NO_THREAD_LOCAL + +```cpp +#define JSON_NO_THREAD_LOCAL +``` + +When defined, the library does not use `#!cpp thread_local` storage. This is relevant for the few environments whose +toolchain does not support it. + +The copy constructor copies the first levels of a value by copying the containers, which copy their elements, and +completes whatever is nested deeper than that without the call stack, so that copying a value cannot exhaust the stack +however deeply it is nested. It counts the levels it has descended into in a `#!cpp thread_local` variable, as a counter +shared between threads would be raced. + +Without that counter, no descent can be bounded safely, so objects and arrays are copied without the call stack right +away. Copying keeps working exactly as it does otherwise - the same values come out, and deeply nested values are copied +just as safely - but copying is slower, because the containers no longer copy themselves. Copying the benchmark +documents takes 9% (`canada.json`) to 34% (`twitter.json`) longer; values built mostly from objects are affected the +most. + +## Default definition + +By default, `#!cpp JSON_NO_THREAD_LOCAL` is not defined. + +```cpp +#undef JSON_NO_THREAD_LOCAL +``` + +The library defines it by itself for Clang targeting MinGW, which does not survive the `#!cpp thread_local` storage: +copying a value segfaults there, with both old and current Clang versions, while GCC targeting MinGW is unaffected. + +## Examples + +??? example + + The code below forces the library not to use `#!cpp thread_local` storage. + + ```cpp + #define JSON_NO_THREAD_LOCAL 1 + #include + + ... + ``` + +## Version history + +- Added in version 3.12.1. diff --git a/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md b/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md new file mode 100644 index 000000000..1832f2353 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md @@ -0,0 +1,126 @@ +# JSON_STRICT_NUL_HANDLING + +```cpp +#define JSON_STRICT_NUL_HANDLING /* value */ +``` + +When defined to `1`, a `'\0'` (NUL) byte in JSON text input is rejected with `parse_error.101`, like any other +unexpected byte, instead of being silently treated as end of input. + +The macro only affects the JSON text parser ([`parse`](../basic_json/parse.md), [`accept`](../basic_json/accept.md), +[`sax_parse`](../basic_json/sax_parse.md), and [`operator>>`](../operator_gtgt.md)). There are three cases where a NUL +byte is still not rejected: + +- The binary formats ([`from_bjdata`](../basic_json/from_bjdata.md), [`from_bson`](../basic_json/from_bson.md), + [`from_cbor`](../basic_json/from_cbor.md), [`from_msgpack`](../basic_json/from_msgpack.md), + [`from_ubjson`](../basic_json/from_ubjson.md)) are never affected: there, `0x00` is ordinary data. +- A bare `const char*` pointer has no length of its own, so its length is still determined with `strlen()`. The first + NUL byte therefore still marks the end of the input, and nothing after it is read. +- One trailing `'\0'` at the end of a `char` array (e.g., a string literal) is trimmed; see the warning below. + +## Default definition + +The default value is `0` (disabled — existing behavior is preserved). + +```cpp +#define JSON_STRICT_NUL_HANDLING 0 +``` + +## Notes + +!!! note "Background" + + By default, a `'\0'` byte anywhere in the input is treated the same as the real end of the input, rather than as + an ordinary (and, outside of a string, invalid) byte. Everything from that byte onward is silently ignored, + without a parse error - including further, otherwise well-formed JSON: + + ```cpp + json::parse(std::string("123") + '\0'); // == 123, no error + json::parse(std::string("123") + '\0' + "true"); // == 123, the "true" is silently ignored too + ``` + + This falls out of the same convention used when no explicit input length is given at all: parsing from a + `const char*` already stops at the first NUL byte via `strlen()`, since a bare pointer has no length of its own. + The library applies that same NUL-terminated-C-string convention uniformly, rather than only when a length is + genuinely unavailable - so a `std::string`, iterator range, or container whose content happens to include a NUL + byte is affected the same way a raw `const char*` would be (see the + [FAQ entry](../../home/faq.md#nul-bytes-in-the-input) for a fuller explanation). + + This was not fixed unconditionally, because doing so is backwards-incompatible for any caller who happens to + depend on the current behavior - even unknowingly, for instance because their input already contains trailing + padding they never noticed was being discarded (see [#5530](https://github.com/nlohmann/json/issues/5530)). + This macro instead offers an opt-in path to the corrected behavior ahead of version 4.0.0, where it is planned to + become the default. + +!!! warning "Opt-in only" + + This macro must be defined **before** including ``. Defining it after the include has no + effect. + + Enabling it also changes how a `char` array (including a string literal, e.g. `json::parse("123")`) is read: such + an array normally carries a trailing `'\0'` contributed by the compiler, not by the source text. With this macro + enabled, that one trailing byte is trimmed if present so that parsing a string literal keeps working; every other + byte in the array - including any `'\0'` that is not the very last element - is read as real data and rejected + like any other unexpected byte. Arrays of any other element type (`unsigned char`, `std::uint8_t`, ...), as used + for CBOR or MessagePack, are never affected by this trimming; their full extent - including a genuine trailing + `0x00` - is always preserved, in both states of this macro. + +!!! tip "Workaround without the macro" + + To reject a NUL byte without enabling this macro, trim your input yourself before calling `parse()`: + + ```cpp + s.resize(s.find('\0')); // drop everything from the first NUL onward, if any + json::parse(s); + ``` + +## Examples + +??? example "Default behavior (macro not defined)" + + Without the macro, a NUL byte silently ends parsing at that point: + + ```cpp + #include + + using json = nlohmann::json; + + int main() + { + json j = json::parse(std::string("123") + '\0' + "true"); + // j is 123 -- the '\0' and everything after it is silently ignored + } + ``` + +??? example "Opt-in strict handling (macro defined to 1)" + + With the macro, a NUL byte is rejected like any other unexpected byte: + + ```cpp + #define JSON_STRICT_NUL_HANDLING 1 + #include + + using json = nlohmann::json; + + int main() + { + json j = json::parse(std::string("123") + '\0' + "true"); + // throws parse_error.101 -- the NUL byte is now invalid input, + // exactly like any other unexpected trailing byte + + json ok = json::parse("123"); + // ok is 123 -- parsing from a string literal still works + } + ``` + +## See also + +- [FAQ: NUL bytes in the input](../../home/faq.md#nul-bytes-in-the-input) +- [**parse**](../basic_json/parse.md) - deserialize from a compatible input +- [**accept**](../basic_json/accept.md) - check if the input is valid JSON +- [**operator>>**](../operator_gtgt.md) - deserialize from stream + +## Version history + +- Added in version 3.13.0. +- Planned to become the default (with the macro removed) in version 4.0.0. diff --git a/docs/mkdocs/docs/api/macros/json_throw_user.md b/docs/mkdocs/docs/api/macros/json_throw_user.md index b02918cf8..7f4fb6a8d 100644 --- a/docs/mkdocs/docs/api/macros/json_throw_user.md +++ b/docs/mkdocs/docs/api/macros/json_throw_user.md @@ -12,9 +12,11 @@ Controls how exceptions are handled by the library. 1. This macro overrides [`#!cpp catch`](https://en.cppreference.com/w/cpp/language/try_catch) calls inside the library. - The argument is the type of the exception to catch. As of version 3.8.0, the library only catches `std::out_of_range` - exceptions internally to rethrow them as [`json::out_of_range`](../../home/exceptions.md#out-of-range) exceptions. - The macro is always followed by a scope. + The argument is the type of the exception to catch. The library uses it in a single place: to swallow any exception + escaping the parent-pointer check that [`JSON_DIAGNOSTICS`](json_diagnostics.md) adds to the class invariant. The + places where the library catches its own [`json::out_of_range`](../../home/exceptions.md#out-of-range) exceptions + use `JSON_INTERNAL_CATCH` instead, which `JSON_CATCH_USER` also overrides unless `JSON_INTERNAL_CATCH_USER` is + defined. The macro is always followed by a scope. 2. This macro overrides `#!cpp throw` calls inside the library. The argument is the exception to be thrown. Note that `JSON_THROW_USER` should leave the current scope (e.g., by throwing or aborting), as continuing after it may yield undefined behavior. diff --git a/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md b/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md index 22f6d0072..e3d5fb29d 100644 --- a/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md +++ b/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md @@ -24,6 +24,14 @@ By default, implicit conversions are enabled. You can prepare existing code by already defining `JSON_USE_IMPLICIT_CONVERSIONS` to `0` and replace any implicit conversions with calls to [`get`](../basic_json/get.md). +!!! tip "Automatic migration" + + The community-maintained clang-tidy check `modernize-nlohmann-json-explicit-conversions` rewrites implicit + conversions into explicit calls to [`get`](../basic_json/get.md); for example, `#!cpp int i = j;` becomes + `#!cpp int i = j.get();`. The check is not part of clang-tidy itself, and it does not catch every case (for + example, constructing a `std::optional` from a JSON value), so review the result. See + [discussion #4610](https://github.com/nlohmann/json/discussions/4610) for how to build and use it. + !!! hint "CMake option" Implicit conversions can also be controlled with the CMake option diff --git a/docs/mkdocs/docs/api/macros/json_use_simdutf.md b/docs/mkdocs/docs/api/macros/json_use_simdutf.md new file mode 100644 index 000000000..611c1e9f1 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_use_simdutf.md @@ -0,0 +1,71 @@ +# JSON_USE_SIMDUTF + +```cpp +#define JSON_USE_SIMDUTF +``` + +When defined, the parser validates the UTF-8 content of JSON strings that come from a **contiguous byte input** +(`std::string`, `std::vector`/``, string literals, `const char*` ranges, …) using the +[simdutf](https://github.com/simdutf/simdutf) library instead of the built-in scalar validator. On text with many +non-ASCII characters (e.g. CJK or emoji) this can validate several times faster. + +This is an **opt-in external dependency**. The library itself remains header-only and its behavior is unchanged: the +same input is accepted or rejected either way, and every parse error is reported at the same position with the same +message (simdutf is only used to fast-path *valid* runs; anything it flags falls back to the scalar path so the exact +diagnostic is preserved). Streaming inputs (files, `std::istream`, wide strings, user-defined adapters) always use the +scalar path. + +When `JSON_USE_SIMDUTF` is defined you must make the `simdutf.h` header available on the include path and link the +simdutf library. When it is not defined, no simdutf header is included and there is no dependency. + +!!! note "Requires C++17" + + simdutf requires C++17 and its header rejects older standards with an `#!cpp #error`. The backend is therefore only + compiled in from C++17 on. In C++11 and C++14 the macro has no effect and the scalar validator is used, which + accepts and rejects exactly the same input -- only throughput differs. Setting the macro project-wide is therefore + safe even when some translation units are built with an older standard. + +!!! warning "Define consistently" + + The macro selects between two definitions of the same inline validation function. It must therefore be defined + identically for **every** translation unit that includes the library; mixing translation units that define it with + ones that do not is an ODR violation. Prefer setting it as a compile definition on the target rather than with + `#!cpp #define` in individual source files. + +## Default definition + +By default, `#!cpp JSON_USE_SIMDUTF` is not defined and the portable C++11 scalar validator is used. + +```cpp +#undef JSON_USE_SIMDUTF +``` + +## Examples + +??? example + + The code below enables the simdutf backend for UTF-8 validation. + + ```cpp + #define JSON_USE_SIMDUTF 1 + #include + + ... + ``` + + The project must also link against simdutf, e.g. with CMake: + + ```cmake + target_compile_definitions(your_target PRIVATE JSON_USE_SIMDUTF) + target_link_libraries(your_target PRIVATE simdutf::simdutf) + ``` + +!!! hint "Testing this configuration" + + The unit tests can be built against the simdutf backend with the CMake option `JSON_TestSimdutf` (`OFF` by + default), which fetches simdutf and defines `JSON_USE_SIMDUTF` for every test target. The `ci_test_simdutf` target + runs the whole test suite in that configuration. + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/operator_gtgt.md b/docs/mkdocs/docs/api/operator_gtgt.md index c418e256d..4cb6bc6ca 100644 --- a/docs/mkdocs/docs/api/operator_gtgt.md +++ b/docs/mkdocs/docs/api/operator_gtgt.md @@ -57,6 +57,13 @@ input >> j3; // j3 == [2] Note that reading concatenated values does **not** work for [JSON Lines](../features/parsing/json_lines.md) (newline-delimited JSON) input -- see that page for why and for the recommended alternative. +By default, a `'\0'` (NUL) byte encountered while reading a value is treated as end of input, rather than as an +ordinary (and, outside of a string, invalid) byte; see the [FAQ entry](../home/faq.md#nul-bytes-in-the-input) for +details and the [`JSON_STRICT_NUL_HANDLING`](macros/json_strict_nul_handling.md) macro to opt into rejecting it +instead. Because `operator>>` only parses a single value and does not require the rest of the stream to be consumed, +a NUL byte *after* a complete value has no effect on `operator>>` either way; it only matters while a value is still +being read. + !!! warning "Deprecation" This function replaces function `#!cpp std::istream& operator<<(basic_json& j, std::istream& i)` which has @@ -83,9 +90,13 @@ Note that reading concatenated values does **not** work for [JSON Lines](../feat - [accept](basic_json/accept.md) - check if the input is valid JSON - [parse](basic_json/parse.md) - deserialize from a compatible input +- [`JSON_STRICT_NUL_HANDLING`](macros/json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input + instead of treating it as end of input ## Version history - Added in version 1.0.0. - Changed in version 4.0.0 to leave the character that terminates a number in the stream, so that the stream is positioned right after the parsed value for every value type. +- `JSON_STRICT_NUL_HANDLING` added in version 3.13.0 to optionally reject a NUL byte in the input instead of treating + it as end of input; planned to become the default in version 4.0.0. diff --git a/docs/mkdocs/docs/community/assurance_case.md b/docs/mkdocs/docs/community/assurance_case.md new file mode 100644 index 000000000..87d8d6f14 --- /dev/null +++ b/docs/mkdocs/docs/community/assurance_case.md @@ -0,0 +1,70 @@ +# Assurance case + +This page argues why the library meets its security requirements. It describes the threats the library faces, where the +trust boundaries lie, and how the library's design and the [quality assurance](quality_assurance.md) counter these +threats. To report a vulnerability, see the [security policy](security_policy.md). + +## Threat model + +The library parses, stores, and serializes JSON values in memory. It does not open network connections, does not open +files (it only reads from streams or `std::FILE*` handles that the caller has already opened), does not read environment +variables, and does not implement cryptography or handle credentials. + +The primary threat is therefore **untrusted input**: JSON text or binary data (BJData, BSON, CBOR, MessagePack, UBJSON) +that an attacker controls, passed to [`parse`](../api/basic_json/parse.md), [`accept`](../api/basic_json/accept.md), +[`sax_parse`](../api/basic_json/sax_parse.md), or one of the `from_*` functions such as +[`from_cbor`](../api/basic_json/from_cbor.md). Such input may try to + +- make the library read or write out of bounds (malformed lengths, truncated input, invalid UTF-8), +- trigger undefined behavior (integer overflow in sizes or numbers, invalid casts), +- exhaust memory (huge announced sizes), or +- exhaust the call stack (deeply nested arrays and objects). + +## Trust boundaries + +- **Untrusted:** all serialized input read by the parser, the SAX interface, and the binary readers. The library must + handle every possible input by either producing a value or throwing a [`parse_error`](../home/exceptions.md#parse-errors) + (or returning `false` when exceptions are disabled for the call). +- **Trusted:** the C++ code that calls the library. Calling a function with violated preconditions, for instance + accessing an array with [`operator[]`](../api/basic_json/operator%5B%5D.md) out of range, is a programming error and + not a security boundary. Such preconditions are checked with [runtime assertions](../features/assertions.md) in debug + builds; functions such as [`at`](../api/basic_json/at.md) offer checked access with exceptions. + +## Secure design + +- **Strict parsing.** The parser accepts exactly the JSON grammar of [RFC 8259](https://datatracker.ietf.org/doc/html/rfc8259). + Extensions such as [comments](../features/comments.md) and [trailing commas](../features/trailing_commas.md) must be + enabled explicitly. Invalid UTF-8 is rejected. +- **Errors are reported, not ignored.** Malformed input results in a [`parse_error`](../home/exceptions.md#parse-errors) + with the byte position of the error. Binary readers do not trust announced sizes: strings and binary values grow + only as bytes are actually read, arrays reserve at most a fixed number of elements up front, and sizes that no + container can hold are rejected. +- **Memory is owned by values.** Each `basic_json` value owns its content, and there is no manual memory management in + user code. The destructor does not recurse, so destroying a deeply nested value does not exhaust the stack. +- **Bounded recursion.** The JSON parser and the binary readers keep their state in explicit stacks instead of + recursing per nesting level. Operations that walk a value, such as [`dump`](../api/basic_json/dump.md), copying, + hashing, and [`merge_patch`](../api/basic_json/merge_patch.md), recurse only up to a fixed depth and continue with an + explicit stack below it. Some operations, such as comparison, [`diff`](../api/basic_json/diff.md), + [`flatten`](../api/basic_json/flatten.md), and the binary writers, still recurse once per nesting level; work on them + is in progress. Applications that process untrusted input can limit its nesting depth with a + [parser callback](../features/parsing/parser_callbacks.md). +- **Invariants are checked.** The class invariant (for instance, that the pointer for the stored type is never null) is + checked with runtime assertions throughout the test suite. + +## Common weaknesses + +The following table maps the relevant classes of the [Common Weakness Enumeration](https://cwe.mitre.org) to the +measures that counter them. The measures are described in detail in [Quality assurance](quality_assurance.md). + +| Weakness | Countermeasures | +|---------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------| +| Out-of-bounds read/write ([CWE-125](https://cwe.mitre.org/data/definitions/125.html), [CWE-787](https://cwe.mitre.org/data/definitions/787.html)) | bounds checks on all reads from the input; AddressSanitizer and Valgrind on the test suite; OSS-Fuzz | +| Integer overflow ([CWE-190](https://cwe.mitre.org/data/definitions/190.html)) | UndefinedBehaviorSanitizer with integer overflow detection; Clang-Tidy; Cppcheck | +| Use after free, double free ([CWE-416](https://cwe.mitre.org/data/definitions/416.html), [CWE-415](https://cwe.mitre.org/data/definitions/415.html)) | ownership of all memory by values; AddressSanitizer and Valgrind; Clang Static Analyzer | +| Memory leaks ([CWE-401](https://cwe.mitre.org/data/definitions/401.html)) | Valgrind (Memcheck) on the test suite | +| Uncontrolled recursion ([CWE-674](https://cwe.mitre.org/data/definitions/674.html)) | iterative parser, binary readers, and destructor; bounded recursion in value operations; tests with deeply nested inputs | +| Uncontrolled resource consumption ([CWE-400](https://cwe.mitre.org/data/definitions/400.html)) | allocations based on announced sizes are capped; OSS-Fuzz with memory limits | +| Undefined behavior in general ([CWE-758](https://cwe.mitre.org/data/definitions/758.html)) | UndefinedBehaviorSanitizer; runtime assertions; Clang-Tidy, Cppcheck, Clang Static Analyzer, Infer | + +In addition, every line of the library is covered by the unit tests, and all parsers are fuzz-tested around the clock +by [OSS-Fuzz](https://github.com/google/oss-fuzz/tree/master/projects/json). diff --git a/docs/mkdocs/docs/community/index.md b/docs/mkdocs/docs/community/index.md index 50baeab25..7b7f5c07a 100644 --- a/docs/mkdocs/docs/community/index.md +++ b/docs/mkdocs/docs/community/index.md @@ -5,4 +5,6 @@ - [Contribution Guidelines](contribution_guidelines.md) - guidelines how to contribute to this project - [Governance](governance.md) - the governance model of this project - [Quality Assurance](quality_assurance.md) - how the quality of this project is assured +- [Roadmap](roadmap.md) - what the project will and will not do - [Security Policy](security_policy.md) - the security policy of the project +- [Assurance Case](assurance_case.md) - why the library meets its security requirements diff --git a/docs/mkdocs/docs/community/quality_assurance.md b/docs/mkdocs/docs/community/quality_assurance.md index 4196f3532..bd35516b8 100644 --- a/docs/mkdocs/docs/community/quality_assurance.md +++ b/docs/mkdocs/docs/community/quality_assurance.md @@ -164,6 +164,9 @@ Note: Some modern features (like C++20 ranges or filesystem support) may be disa - [x] The parser is tested against extensive correctness suites for JSON compliance. - [x] In addition, the library is continuously fuzz-tested at [OSS-Fuzz](https://google.github.io/oss-fuzz/) where the library is checked against billions of inputs. +- [x] Every crash reported by OSS-Fuzz is fixed together with a unit test that reproduces it, and the fix references + the OSS-Fuzz issue. The round-trip checks of the fuzzer drivers are also part of the unit tests. See the + [fuzz testing documentation](https://github.com/nlohmann/json/blob/develop/tests/fuzzing.md#handling-oss-fuzz-reports). ## Static analysis diff --git a/docs/mkdocs/docs/community/roadmap.md b/docs/mkdocs/docs/community/roadmap.md new file mode 100644 index 000000000..e8c407d3f --- /dev/null +++ b/docs/mkdocs/docs/community/roadmap.md @@ -0,0 +1,43 @@ +# Roadmap + +This page describes what the project intends to do, and what it does not intend to do, over the next year. Concrete +work items are tracked in the [GitHub milestones](https://github.com/nlohmann/json/milestones) and the +[issue tracker](https://github.com/nlohmann/json/issues). + +## What the project will do + +- **Keep the C++11 baseline.** The library will continue to compile with every + [supported C++11 compiler](https://github.com/nlohmann/json/blob/develop/README.md#supported-compilers). Features of + later standards are only used when they are guarded by the `JSON_HAS_CPP_*` macros. +- **Stay conformant to JSON.** The parser and serializer follow [RFC 8259](https://datatracker.ietf.org/doc/html/rfc8259). + Extensions such as [comments](../features/comments.md) or [trailing commas](../features/trailing_commas.md) remain + opt-in. +- **Keep the 3.x public API stable.** Releases follow [semantic versioning](https://semver.org). Changes that would + break existing code are only added behind a feature macro, so users can opt in and test their code before a next + major release. +- **Support a broad range of compilers and platforms.** The [CI](quality_assurance.md) keeps testing old and new + versions of GCC, Clang, MSVC, and other compilers on Linux, macOS, and Windows. +- **Keep the quality assurance up.** Every change keeps the test coverage at 100%, passes the static and dynamic + analysis, and is fuzz-tested by OSS-Fuzz, see [Quality assurance](quality_assurance.md). +- **Harden the library against hostile input.** Handling deeply nested values without exhausting the call stack is + ongoing work. +- **Fix bugs and security issues** reported through the issue tracker and the [security policy](security_policy.md). + +## What the project will not do + +- **Break the public API of version 3.x.** See the + [contribution guidelines](https://github.com/nlohmann/json/blob/develop/.github/CONTRIBUTING.md#break-the-public-api) + for what counts as a breaking change. +- **Require a newer C++ standard than C++11.** +- **Break JSON conformance** or enable non-standard extensions by default. +- **Add dependencies** or require a build step. The library remains header-only, and the single header + `json.hpp` remains a complete distribution. +- **Trade simplicity for speed or memory efficiency.** Performance improvements are welcome, but the library is not + meant to compete with the fastest JSON libraries, see [Design goals](../home/design_goals.md). + +## Version 4.0 + +There is no decision yet on whether or when a version 4.0 with breaking changes will be released. Proposals that need +a major version, for instance stricter type conversions, are collected in issue +[#3453](https://github.com/nlohmann/json/issues/3453). Until then, such changes are only added as opt-in behavior +behind feature macros. diff --git a/docs/mkdocs/docs/examples/custom_array_type.cpp b/docs/mkdocs/docs/examples/custom_array_type.cpp new file mode 100644 index 000000000..63651cadf --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_array_type.cpp @@ -0,0 +1,19 @@ +#include +#include + +#include + +#include "custom_array_type.hpp" + +using custom_json = nlohmann::basic_json; + +int main() +{ + custom_json j = custom_json::array(); + j.push_back(1); + j.push_back(2); + j.push_back(3); + + std::cout << j.dump() << std::endl; + std::cout << std::boolalpha << (custom_json::parse(j.dump()) == j) << std::endl; +} diff --git a/docs/mkdocs/docs/examples/custom_array_type.hpp b/docs/mkdocs/docs/examples/custom_array_type.hpp new file mode 100644 index 000000000..750d52a81 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_array_type.hpp @@ -0,0 +1,152 @@ +#pragma once + +#include +#include +#include + +// A minimal, self-contained ArrayType built around a private std::vector. +// See https://json.nlohmann.me/features/types/template_parameters/#arraytype +template> +class custom_array_type +{ + using vector_t = std::vector; + vector_t data_; + + public: + using value_type = typename vector_t::value_type; + using size_type = typename vector_t::size_type; + using iterator = typename vector_t::iterator; + using const_iterator = typename vector_t::const_iterator; + + custom_array_type() = default; + custom_array_type(const custom_array_type&) = default; + custom_array_type(custom_array_type&&) = default; + custom_array_type& operator=(const custom_array_type&) = default; + custom_array_type& operator=(custom_array_type&&) = default; + + template + custom_array_type(InputIt first, InputIt last) : data_(first, last) {} + + custom_array_type(size_type count, const T& value) : data_(count, value) {} + + iterator begin() + { + return data_.begin(); + } + iterator end() + { + return data_.end(); + } + const_iterator begin() const + { + return data_.begin(); + } + const_iterator end() const + { + return data_.end(); + } + const_iterator cbegin() const + { + return data_.cbegin(); + } + const_iterator cend() const + { + return data_.cend(); + } + + bool empty() const + { + return data_.empty(); + } + size_type size() const + { + return data_.size(); + } + size_type max_size() const + { + return data_.max_size(); + } + void clear() + { + data_.clear(); + } + void resize(size_type n) + { + data_.resize(n); + } + + T& operator[](size_type pos) + { + return data_[pos]; + } + const T& operator[](size_type pos) const + { + return data_[pos]; + } + + T& back() + { + return data_.back(); + } + const T& back() const + { + return data_.back(); + } + + void push_back(const T& value) + { + data_.push_back(value); + } + void push_back(T&& value) + { + data_.push_back(std::move(value)); + } + + template + void emplace_back(Args&& ... args) + { + data_.emplace_back(std::forward(args)...); + } + + void pop_back() + { + data_.pop_back(); + } + + iterator insert(const_iterator pos, const T& value) + { + return data_.insert(pos, value); + } + iterator insert(const_iterator pos, size_type count, const T& value) + { + return data_.insert(pos, count, value); + } + template + iterator insert(const_iterator pos, InputIt first, InputIt last) + { + return data_.insert(pos, first, last); + } + + iterator erase(const_iterator pos) + { + return data_.erase(pos); + } + iterator erase(const_iterator first, const_iterator last) + { + return data_.erase(first, last); + } + + void swap(custom_array_type& other) + { + data_.swap(other.data_); + } + + friend bool operator==(const custom_array_type& lhs, const custom_array_type& rhs) + { + return lhs.data_ == rhs.data_; + } + friend bool operator<(const custom_array_type& lhs, const custom_array_type& rhs) + { + return lhs.data_ < rhs.data_; + } +}; diff --git a/docs/mkdocs/docs/examples/custom_array_type.output b/docs/mkdocs/docs/examples/custom_array_type.output new file mode 100644 index 000000000..f5ceaa555 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_array_type.output @@ -0,0 +1,2 @@ +[1,2,3] +true diff --git a/docs/mkdocs/docs/examples/custom_binary_type.cpp b/docs/mkdocs/docs/examples/custom_binary_type.cpp new file mode 100644 index 000000000..cea34ea38 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_binary_type.cpp @@ -0,0 +1,21 @@ +#include +#include +#include +#include +#include + +#include + +#include "custom_binary_type.hpp" + +using custom_json = nlohmann::basic_json; + +int main() +{ + const auto j = custom_json::binary({0x01, 0x02, 0x03}); + + std::cout << j.dump() << std::endl; + std::cout << std::boolalpha << (custom_json::from_cbor(custom_json::to_cbor(j)) == j) << std::endl; +} diff --git a/docs/mkdocs/docs/examples/custom_binary_type.hpp b/docs/mkdocs/docs/examples/custom_binary_type.hpp new file mode 100644 index 000000000..77b4ad49a --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_binary_type.hpp @@ -0,0 +1,112 @@ +#pragma once + +#include +#include +#include + +// A minimal, self-contained BinaryType built around a private std::vector. +// See https://json.nlohmann.me/features/types/template_parameters/#binarytype +class custom_binary_type +{ + using vector_t = std::vector; + vector_t data_; + + public: + using value_type = vector_t::value_type; + using size_type = vector_t::size_type; + using iterator = vector_t::iterator; + using const_iterator = vector_t::const_iterator; + + custom_binary_type() = default; + custom_binary_type(const custom_binary_type&) = default; + custom_binary_type(custom_binary_type&&) = default; + custom_binary_type& operator=(const custom_binary_type&) = default; + custom_binary_type& operator=(custom_binary_type&&) = default; + + template + custom_binary_type(InputIt first, InputIt last) : data_(first, last) {} + + // so basic_json::binary({0x01, 0x02}) can build one directly + custom_binary_type(std::initializer_list init) : data_(init) {} + + size_type size() const + { + return data_.size(); + } + bool empty() const + { + return data_.empty(); + } + void clear() + { + data_.clear(); + } + void resize(size_type n) + { + data_.resize(n); + } + + // read-only is enough: the writers only ever read from a binary value + const std::uint8_t* data() const + { + return data_.data(); + } + + std::uint8_t& operator[](size_type pos) + { + return data_[pos]; + } + std::uint8_t operator[](size_type pos) const + { + return data_[pos]; + } + + std::uint8_t& back() + { + return data_.back(); + } + std::uint8_t back() const + { + return data_.back(); + } + + iterator begin() + { + return data_.begin(); + } + iterator end() + { + return data_.end(); + } + const_iterator begin() const + { + return data_.begin(); + } + const_iterator end() const + { + return data_.end(); + } + const_iterator cbegin() const + { + return data_.cbegin(); + } + const_iterator cend() const + { + return data_.cend(); + } + + template + iterator insert(const_iterator pos, InputIt first, InputIt last) + { + return data_.insert(pos, first, last); + } + + friend bool operator==(const custom_binary_type& lhs, const custom_binary_type& rhs) + { + return lhs.data_ == rhs.data_; + } + friend bool operator<(const custom_binary_type& lhs, const custom_binary_type& rhs) + { + return lhs.data_ < rhs.data_; + } +}; diff --git a/docs/mkdocs/docs/examples/custom_binary_type.output b/docs/mkdocs/docs/examples/custom_binary_type.output new file mode 100644 index 000000000..b4814d6ed --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_binary_type.output @@ -0,0 +1,2 @@ +{"bytes":[1,2,3],"subtype":null} +true diff --git a/docs/mkdocs/docs/examples/custom_object_type.cpp b/docs/mkdocs/docs/examples/custom_object_type.cpp new file mode 100644 index 000000000..d4f99ccf0 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_object_type.cpp @@ -0,0 +1,26 @@ +#include +#include +#include + +#include + +#include "custom_object_type.hpp" + +using custom_json = nlohmann::basic_json; + +int main() +{ + custom_json j; + j["pi"] = 3.141; + j["happy"] = true; + j["list"] = {1, 2, 3}; + + std::cout << j.dump(2) << std::endl; + std::cout << std::boolalpha << (custom_json::parse(j.dump()) == j) << std::endl; + + // custom_object_type has no key_compare member, so object_comparator_t + // falls back to its default + std::cout << std::boolalpha + << std::is_same::value + << std::endl; +} diff --git a/docs/mkdocs/docs/examples/custom_object_type.hpp b/docs/mkdocs/docs/examples/custom_object_type.hpp new file mode 100644 index 000000000..da715ed4e --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_object_type.hpp @@ -0,0 +1,144 @@ +#pragma once + +#include +#include + +// A minimal, self-contained ObjectType built around a private std::map. +// key_compare is deliberately not exposed: when an ObjectType has no +// key_compare member, the library falls back to its own default comparator. +// See https://json.nlohmann.me/features/types/template_parameters/#objecttype +template +class custom_object_type +{ + using map_t = std::map; + map_t data_; + + public: + using key_type = typename map_t::key_type; + using mapped_type = typename map_t::mapped_type; + using value_type = typename map_t::value_type; + using size_type = typename map_t::size_type; + using iterator = typename map_t::iterator; + using const_iterator = typename map_t::const_iterator; + + custom_object_type() = default; + custom_object_type(const custom_object_type&) = default; + custom_object_type(custom_object_type&&) = default; + custom_object_type& operator=(const custom_object_type&) = default; + custom_object_type& operator=(custom_object_type&&) = default; + + template + custom_object_type(InputIt first, InputIt last) : data_(first, last) {} + + iterator begin() + { + return data_.begin(); + } + iterator end() + { + return data_.end(); + } + const_iterator begin() const + { + return data_.begin(); + } + const_iterator end() const + { + return data_.end(); + } + const_iterator cbegin() const + { + return data_.cbegin(); + } + const_iterator cend() const + { + return data_.cend(); + } + + bool empty() const + { + return data_.empty(); + } + size_type size() const + { + return data_.size(); + } + size_type max_size() const + { + return data_.max_size(); + } + void clear() + { + data_.clear(); + } + + iterator find(const key_type& key) + { + return data_.find(key); + } + const_iterator find(const key_type& key) const + { + return data_.find(key); + } + size_type count(const key_type& key) const + { + return data_.count(key); + } + + std::pair emplace(const key_type& key, const mapped_type& value) + { + return data_.emplace(key, value); + } + + std::pair insert(const value_type& value) + { + return data_.insert(value); + } + + template + void insert(InputIt first, InputIt last) + { + data_.insert(first, last); + } + + mapped_type& operator[](const key_type& key) + { + return data_[key]; + } + + mapped_type& at(const key_type& key) + { + return data_.at(key); + } + const mapped_type& at(const key_type& key) const + { + return data_.at(key); + } + + iterator erase(iterator pos) + { + return data_.erase(pos); + } + iterator erase(iterator first, iterator last) + { + return data_.erase(first, last); + } + size_type erase(const key_type& key) + { + return data_.erase(key); + } + + void swap(custom_object_type& other) + { + data_.swap(other.data_); + } + + friend bool operator==(const custom_object_type& lhs, const custom_object_type& rhs) + { + return lhs.data_ == rhs.data_; + } + friend bool operator<(const custom_object_type& lhs, const custom_object_type& rhs) + { + return lhs.data_ < rhs.data_; + } +}; diff --git a/docs/mkdocs/docs/examples/custom_object_type.output b/docs/mkdocs/docs/examples/custom_object_type.output new file mode 100644 index 000000000..48b3e9630 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_object_type.output @@ -0,0 +1,11 @@ +{ + "happy": true, + "list": [ + 1, + 2, + 3 + ], + "pi": 3.141 +} +true +true diff --git a/docs/mkdocs/docs/examples/custom_string_type.cpp b/docs/mkdocs/docs/examples/custom_string_type.cpp new file mode 100644 index 000000000..63b798fc6 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_string_type.cpp @@ -0,0 +1,20 @@ +#include +#include +#include + +#include + +#include "custom_string_type.hpp" + +using custom_json = nlohmann::basic_json; + +int main() +{ + custom_json j; + j["pi"] = 3.141; + j["happy"] = true; + j["list"] = {1, 2, 3}; + + std::cout << j.dump(2) << std::endl; + std::cout << std::boolalpha << (custom_json::parse(j.dump()) == j) << std::endl; +} diff --git a/docs/mkdocs/docs/examples/custom_string_type.hpp b/docs/mkdocs/docs/examples/custom_string_type.hpp new file mode 100644 index 000000000..ec48501fc --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_string_type.hpp @@ -0,0 +1,134 @@ +#pragma once + +#include +#include + +// A minimal, self-contained StringType built around a private std::string. +// Wraps rather than inherits, so it exposes exactly what the library needs +// and nothing more of std::string's interface. +// +// Covers the "Always required" members, the extras needed for the binary +// formats, and the extras needed for JSON Pointer / flatten / unflatten / +// diff. Extending it further (e.g. for std::hash or to_bson) is +// a matter of adding the extra members listed in the "Required for other +// functionality" table. +// +// See https://json.nlohmann.me/features/types/template_parameters/#stringtype +class custom_string_type +{ + std::string data_; + + public: + using value_type = char; + using size_type = std::string::size_type; + using iterator = std::string::iterator; + using const_iterator = std::string::const_iterator; + + static constexpr size_type npos = std::string::npos; + + custom_string_type() = default; + custom_string_type(const custom_string_type&) = default; + custom_string_type(custom_string_type&&) = default; + custom_string_type& operator=(const custom_string_type&) = default; + custom_string_type& operator=(custom_string_type&&) = default; + + // not explicit: the library relies on being able to hand it a string literal + custom_string_type(const char* s) : data_(s) {} + custom_string_type(const char* s, size_type count) : data_(s, count) {} + custom_string_type(size_type count, char ch) : data_(count, ch) {} + + size_type size() const + { + return data_.size(); + } + bool empty() const + { + return data_.empty(); + } + void clear() + { + data_.clear(); + } + void resize(size_type n) + { + data_.resize(n); + } + void resize(size_type n, char c) + { + data_.resize(n, c); + } + void reserve(size_type n) + { + data_.reserve(n); + } + + // must stay null-terminated -- the parser hands this to std::strtoull & + // friends; std::string::data() has guaranteed that since C++11 + const char* data() const + { + return data_.data(); + } + + void push_back(char c) + { + data_.push_back(c); + } + + char& operator[](size_type pos) + { + return data_[pos]; + } + char operator[](size_type pos) const + { + return data_[pos]; + } + + custom_string_type& append(const char* s, size_type count) + { + data_.append(s, count); + return *this; + } + custom_string_type& append(const custom_string_type& other) + { + data_.append(other.data_); + return *this; + } + + size_type find_first_of(char c, size_type pos = 0) const + { + return data_.find_first_of(c, pos); + } + + iterator begin() + { + return data_.begin(); + } + iterator end() + { + return data_.end(); + } + const_iterator begin() const + { + return data_.begin(); + } + const_iterator end() const + { + return data_.end(); + } + + friend bool operator==(const custom_string_type& lhs, const custom_string_type& rhs) + { + return lhs.data_ == rhs.data_; + } + friend bool operator<(const custom_string_type& lhs, const custom_string_type& rhs) + { + return lhs.data_ < rhs.data_; + } + + // not required by the library itself, but dump() returns a custom_string_type + // and this makes `std::cout << j.dump()` work as expected + friend std::ostream& operator<<(std::ostream& os, const custom_string_type& s) + { + return os << s.data_; + } +}; diff --git a/docs/mkdocs/docs/examples/custom_string_type.output b/docs/mkdocs/docs/examples/custom_string_type.output new file mode 100644 index 000000000..d792e2a72 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_string_type.output @@ -0,0 +1,10 @@ +{ + "happy": true, + "list": [ + 1, + 2, + 3 + ], + "pi": 3.141 +} +true diff --git a/docs/mkdocs/docs/examples/get__ValueType_const.output b/docs/mkdocs/docs/examples/get__ValueType_const.output index 5cd9cd3aa..e7e9b5d59 100644 --- a/docs/mkdocs/docs/examples/get__ValueType_const.output +++ b/docs/mkdocs/docs/examples/get__ValueType_const.output @@ -4,8 +4,8 @@ Hello, world! 1 2 3 4 5 -string: "Hello, world!" number: {"floating-point":17.23,"integer":42} null: null +string: "Hello, world!" boolean: true array: [1,2,3,4,5] diff --git a/docs/mkdocs/docs/examples/get_to.output b/docs/mkdocs/docs/examples/get_to.output index 5cd9cd3aa..e7e9b5d59 100644 --- a/docs/mkdocs/docs/examples/get_to.output +++ b/docs/mkdocs/docs/examples/get_to.output @@ -4,8 +4,8 @@ Hello, world! 1 2 3 4 5 -string: "Hello, world!" number: {"floating-point":17.23,"integer":42} null: null +string: "Hello, world!" boolean: true array: [1,2,3,4,5] diff --git a/docs/mkdocs/docs/examples/operator__ValueType.output b/docs/mkdocs/docs/examples/operator__ValueType.output index a3bd9fff4..de471ec02 100644 --- a/docs/mkdocs/docs/examples/operator__ValueType.output +++ b/docs/mkdocs/docs/examples/operator__ValueType.output @@ -4,9 +4,9 @@ Hello, world! 1 2 3 4 5 -string: "Hello, world!" number: {"floating-point":17.23,"integer":42} null: null +string: "Hello, world!" boolean: true array: [1,2,3,4,5] [json.exception.type_error.302] type must be boolean, but is string diff --git a/docs/mkdocs/docs/examples/parser_callback_t.cpp b/docs/mkdocs/docs/examples/parser_callback_t.cpp index 45794cca6..99c4298e8 100644 --- a/docs/mkdocs/docs/examples/parser_callback_t.cpp +++ b/docs/mkdocs/docs/examples/parser_callback_t.cpp @@ -9,13 +9,13 @@ int main() auto text = R"({"IDs": [116, 943], "Width": 800})"; // discard the array when the parser reads its opening bracket - json j_array_start = json::parse(text, [](int /*depth*/, json::parse_event_t event, json & /*parsed*/) + json j_array_start = json::parse(text, [](int /*depth*/, json::parse_event_t event, json& /*parsed*/) { return event != json::parse_event_t::array_start; }); // discard the same array when the parser reads its closing bracket - json j_array_end = json::parse(text, [](int /*depth*/, json::parse_event_t event, json & /*parsed*/) + json j_array_end = json::parse(text, [](int /*depth*/, json::parse_event_t event, json& /*parsed*/) { return event != json::parse_event_t::array_end; }); @@ -33,7 +33,7 @@ int main() }); // discard the top-level object - json j_root = json::parse(text, [](int /*depth*/, json::parse_event_t event, json & /*parsed*/) + json j_root = json::parse(text, [](int /*depth*/, json::parse_event_t event, json& /*parsed*/) { return event != json::parse_event_t::object_end; }); diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index faff64f16..a0c84edaf 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -116,17 +116,22 @@ The library uses the following mapping from JSON values types to BJData types ac ``` Likewise, when a JSON object in the above form is serialized using - [`to_bjdata`](../../api/basic_json/to_bjdata.md), it is automatically converted into a compact BJData ND-array. When - the 1-dimensional vector stored in `"_ArraySize_"` contains a single integer or two integers with one being 1, a - regular 1-D optimized array is generated instead. + [`to_bjdata`](../../api/basic_json/to_bjdata.md), it is automatically converted into a compact BJData ND-array. - An object is only converted if the annotation actually describes a packed array; otherwise it is serialized as a - regular JSON object. This requires all of the following: + When parsing, an ND-array whose dimension vector is empty, contains a single integer, contains two integers with the + first being 1, or contains a 0 is returned as a regular (possibly empty) array rather than an annotated object. + + An object is only converted if the annotation describes a packed array that is parsed back into the same annotated + object; otherwise it is serialized as a regular JSON object, so the annotation is never lost in a round trip. This requires + all of the following: - `"_ArrayType_"` is one of `uint8`, `int8`, `uint16`, `int16`, `uint32`, `int32`, `uint64`, `int64`, `single`, `double`, `char`, or `byte`, - - every entry of `"_ArraySize_"` is a non-negative integer, and their product is representable as a `std::size_t`, - - `"_ArrayData_"` holds exactly that many elements, and + - `"_ArraySize_"` is an array, since the dimensions are written as the ND-array header's length, + - `"_ArraySize_"` has at least two entries and is not a 1×N row vector (first entry 1), since other shapes are + parsed back as a regular array, + - every entry of `"_ArraySize_"` is a positive integer, and their product is representable as a `std::size_t`, + - `"_ArrayData_"` is an array holding exactly that many elements, and - every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"` (a floating-point number for `single` and `double`, an integer otherwise). @@ -203,6 +208,16 @@ The library maps BJData types to JSON value types as follows: The mapping is **complete** in the sense that any BJData value can be converted to a JSON value. +!!! info "Round trips" + + A value returned by [`from_bjdata`](../../api/basic_json/from_bjdata.md) can be serialized with + [`to_bjdata`](../../api/basic_json/to_bjdata.md) using any combination of options and parsed back into an equal + value, and serializing that value again with the same options produces the same bytes. The exception is binary + values: they are only written as an optimized binary array (`[$B`) if Draft 3 is enabled and both `use_size` and + `use_type` are set. Otherwise, they are written as arrays of integers and parsed back as such (see the notes on + binary values above), and serializing such an array again may choose different, but equally valid, type markers. + The bytes can then differ, but parsing them again yields the same value. + ??? example ```cpp diff --git a/docs/mkdocs/docs/features/binary_formats/bson.md b/docs/mkdocs/docs/features/binary_formats/bson.md index 8c1112e84..95c82e873 100644 --- a/docs/mkdocs/docs/features/binary_formats/bson.md +++ b/docs/mkdocs/docs/features/binary_formats/bson.md @@ -98,6 +98,17 @@ The library maps BSON record types to JSON value types as follows: This library deserializes BSON type `0x11` (Timestamp) as a `number_unsigned` value. The 64-bit value is preserved, but the Timestamp type information is not. +!!! warning "Lenient BSON input handling" + + The BSON reader is lenient in a few areas where the BSON specification is more restrictive: + + - array element keys are not checked against the required decimal sequence (`0`, `1`, `2`, ...), + - any non-zero byte is accepted as `true` for the boolean type, and + - the payload for binary subtype `0x02` is returned as-is, including its inner length prefix. + + If BSON input must be validated for strict specification compliance, validate it separately before passing it to + `from_bson()`. + ??? example ```cpp diff --git a/docs/mkdocs/docs/features/binary_formats/ubjson.md b/docs/mkdocs/docs/features/binary_formats/ubjson.md index 76956d60a..be545b9fe 100644 --- a/docs/mkdocs/docs/features/binary_formats/ubjson.md +++ b/docs/mkdocs/docs/features/binary_formats/ubjson.md @@ -69,6 +69,13 @@ The library uses the following mapping from JSON values types to UBJSON types ac Note that `use_size = true` alone may result in larger representations - the benefit of this parameter is that the receiving side is immediately informed on the number of elements of the container. + An array whose type marker is `Z` (null), `T` (true) or `F` (false) stores no payload at all, because the marker + already is the value. Its declared count is therefore the only thing that decides how much memory the receiving side + allocates, and a handful of bytes can describe billions of elements. `from_ubjson` rejects such an array with + [`out_of_range.408`](../../home/exceptions.md#jsonexceptionout_of_range408) when the count exceeds 1,048,576 + (`1 << 20`), and `to_ubjson` writes longer arrays of these types without the annotation, so any value it produces + can be read back. + !!! info "Binary values" If the JSON data contains the binary type, the value stored is a list of integers, as suggested by the UBJSON diff --git a/docs/mkdocs/docs/features/conversions.md b/docs/mkdocs/docs/features/conversions.md index 765c610ab..94a660ec8 100644 --- a/docs/mkdocs/docs/features/conversions.md +++ b/docs/mkdocs/docs/features/conversions.md @@ -54,6 +54,30 @@ json j = {1.0, "hello", 42}; auto t = j.get>(); // {1.0, "hello", 42} ``` +!!! warning "Serializing a `std::pair`/`std::tuple` whose every element is a string-keyed pair" + + When *every* element of a `#!cpp std::pair` or `#!cpp std::tuple` is itself a two-element array whose first + element is a string (for example `#!cpp std::pair`), serializing it produces a JSON **object** + instead of the expected array: + + ```cpp + using kv = std::pair; + json j = std::pair{{"a", 1}, {"b", 2}}; // {"a":1,"b":2}, not [["a",1],["b",2]] + ``` + + This is a consequence of the [brace-initializer object-detection rule](creating_values.md): the same rule that + lets `#!cpp json{{"a", 1}, {"b", 2}}` create an object also fires here. The resulting object cannot be read back + into the original type (`#!cpp get>()` throws [`type_error.302`](../home/exceptions.md#jsonexceptiontype_error302)), + and duplicate keys collapse into one, losing elements. This only affects `#!cpp std::pair`/`#!cpp std::tuple` + themselves; a `#!cpp std::vector>`, or a pair/tuple with at least one element that is + not a string-keyed pair, serializes to an array as expected. To force an array, build one explicitly from the + elements with [`array`](../api/basic_json/array.md): + + ```cpp + std::pair p{{"a", 1}, {"b", 2}}; + json a = json::array({p.first, p.second}); // [["a",1],["b",2]] + ``` + !!! info "Extracting references into a tuple" A tuple type may also hold references (e.g. `#!cpp std::tuple`) to avoid copying: `get` diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index 1d169fdeb..e7baba0ae 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -91,6 +91,13 @@ security reasons (e.g., Intel Software Guard Extensions (SGX)). See [full documentation of `JSON_NO_IO`](../api/macros/json_no_io.md). +## `JSON_NO_THREAD_LOCAL` + +When defined, the library does not use `#!cpp thread_local` storage. Copying a value then always avoids the call stack +rather than descending into a bounded number of levels first, which is slower but yields the same values. + +See [full documentation of `JSON_NO_THREAD_LOCAL`](../api/macros/json_no_thread_local.md). + ## `JSON_SKIP_LIBRARY_VERSION_CHECK` When defined, the library will not create a compiler warning when a different version of the library was already @@ -105,6 +112,19 @@ using the library with compilers that do not fully support C++11 and may only wo See [full documentation of `JSON_SKIP_UNSUPPORTED_COMPILER_CHECK`](../api/macros/json_skip_unsupported_compiler_check.md). +## `JSON_STRICT_NUL_HANDLING` + +When defined to `1`, a `'\0'` (NUL) byte anywhere in the input is rejected with `parse_error.101`, like any other +unexpected byte, instead of being silently treated as end of input (see the +[FAQ entry](../home/faq.md#nul-bytes-in-the-input) for background). The default value is `0`, which preserves the +existing behavior; this is planned to become the default in version 4.0.0. + +The strict handling can also be enabled with the CMake option +[`JSON_StrictNulHandling`](../integration/cmake.md#json_strictnulhandling) (`OFF` by default) which sets +`JSON_STRICT_NUL_HANDLING` accordingly. + +See [full documentation of `JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md). + ## `JSON_THROW_USER(exception)` This macro overrides `#!cpp throw` calls inside the library. The argument is the exception to be thrown. @@ -137,6 +157,14 @@ behavior is deprecated and switched off (`0`) by default. See [full documentation of `JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md). +## `JSON_USE_SIMDUTF` + +When defined, UTF-8 validation of JSON strings read from contiguous byte input is delegated to the +[simdutf](https://github.com/simdutf/simdutf) library instead of the built-in scalar validator. This is an opt-in +external dependency and is not defined by default. + +See [full documentation of `JSON_USE_SIMDUTF`](../api/macros/json_use_simdutf.md). + ## `NLOHMANN_DEFINE_TYPE_*(...)`, `NLOHMANN_DEFINE_DERIVED_TYPE_*(...)` The library defines 12 macros to simplify the serialization/deserialization of types. See the page on diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index 5542c1f88..09e53f3a2 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -15,6 +15,9 @@ The complete default namespace name is derived as follows: - [`JSON_DIAGNOSTICS`](../api/macros/json_diagnostics.md) defined non-zero appends `_diag`. - [`JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md) defined non-zero appends `_ldvcmp`. + - [`JSON_DIAGNOSTIC_POSITIONS`](../api/macros/json_diagnostic_positions.md) defined non-zero appends `_dp`. + - [`JSON_BRACE_INIT_COPY_SEMANTICS`](../api/macros/json_brace_init_copy_semantics.md) defined non-zero appends + `_bics`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/docs/mkdocs/docs/features/object_order.md b/docs/mkdocs/docs/features/object_order.md index f62474efd..200913fd2 100644 --- a/docs/mkdocs/docs/features/object_order.md +++ b/docs/mkdocs/docs/features/object_order.md @@ -51,7 +51,11 @@ If you do want to preserve the **insertion order**, you can use the type [`nlohm --8<-- "examples/ordered_json.output" ``` -Alternatively, you can use a more sophisticated ordered map like [`tsl::ordered_map`](https://github.com/Tessil/ordered-map) ([integration](https://github.com/nlohmann/json/issues/546#issuecomment-304447518)) or [`nlohmann::fifo_map`](https://github.com/nlohmann/fifo_map) ([integration](https://github.com/nlohmann/json/issues/485#issuecomment-333652309)). +Alternatively, [`nlohmann::fifo_map`](https://github.com/nlohmann/fifo_map) also preserves the insertion order and, unlike [`ordered_map`](../api/ordered_map.md), keeps a lookup index, so it does not have the quadratic cost described below. It is used through a small adapter ([integration](https://github.com/nlohmann/json/issues/485#issuecomment-333652309)). + +If the order does not matter and you only want faster lookup, `boost::unordered_flat_map`, `absl::flat_hash_map`, `absl::node_hash_map`, and several other hash maps work through an adapter that restores the template argument order `basic_json` expects; see [Template Parameter Requirements](types/template_parameters.md#objecttype). Note these are *unordered*, not insertion-ordered. + +[`tsl::ordered_map`](https://github.com/Tessil/ordered-map) cannot be used: its iterators expose the mapped value as `const`, while `basic_json` needs to modify it in place. The [`ordered_map`](../api/ordered_map.md) behind `nlohmann::ordered_json` is deliberately minimal and has no lookup index, so every key access is a linear scan and building an object of `n` keys costs O(n²). This is unnoticeable at diff --git a/docs/mkdocs/docs/features/types/index.md b/docs/mkdocs/docs/features/types/index.md index 5990ac708..e6078b825 100644 --- a/docs/mkdocs/docs/features/types/index.md +++ b/docs/mkdocs/docs/features/types/index.md @@ -79,7 +79,8 @@ template< class NumberFloatType = double, template class AllocatorType = std::allocator, template class JSONSerializer = adl_serializer, - class BinaryType = std::vector + class BinaryType = std::vector, + class CustomBaseClass = void > class basic_json; ``` @@ -106,6 +107,10 @@ using number_float_t = NumberFloatType; using binary_t = nlohmann::byte_container_with_subtype; ``` +Not every type can be passed for these template arguments: the library uses the resulting types in ways that imply a +number of requirements, for instance that `StringType` is `char`-based or that `ArrayType` is vector-like. These +requirements are collected in [Template Parameter Requirements](template_parameters.md). + ## Objects diff --git a/docs/mkdocs/docs/features/types/template_parameters.md b/docs/mkdocs/docs/features/types/template_parameters.md new file mode 100644 index 000000000..760972dff --- /dev/null +++ b/docs/mkdocs/docs/features/types/template_parameters.md @@ -0,0 +1,747 @@ +# Template Parameter Requirements + +Class [`basic_json`](../../api/basic_json/index.md) is configurable through eleven template parameters. The library +never formally states what a type passed for one of these parameters has to provide -- the requirements are implied by +the way the library uses the resulting [`object_t`](../../api/basic_json/object_t.md), +[`array_t`](../../api/basic_json/array_t.md), [`string_t`](../../api/basic_json/string_t.md), etc. This page collects +these requirements so they do not have to be discovered by trial and error. Each section lists the concrete types +that are known to work for that parameter and the ones that do not, checked against Boost 1.83, Abseil 20250127.0, +Folly, EASTL 3.21, `ankerl::unordered_dense`, `phmap`, `gtl`, `robin_hood`, `tsl::ordered_map`, and Qt 6. + +## How to read this page + +Requirements are split into two groups: + +- **Always required** -- needed to instantiate `basic_json` at all, or needed by functions that virtually every program + uses (construction, element access, [`dump`](../../api/basic_json/dump.md)). +- **Required for ...** -- only needed when a particular part of the API is instantiated. Member function templates are + only instantiated when they are used, so a type may be perfectly usable even though it does not satisfy these + requirements, as long as the corresponding functions are never called. + +!!! warning "Requirements are not checked" + + Three requirements are checked with a `#!cpp static_assert`: the array iterator category, the width of + [`BinaryType`](#binarytype)'s `value_type`, and [`NumberUnsignedType`](#numberintegertype-and-numberunsignedtype) + being at least as wide as [`NumberIntegerType`](#numberintegertype-and-numberunsignedtype). The rest are not + diagnosed with dedicated error messages, and violating most of them results in a compiler error somewhere inside + the library. Four violations are not caught at compile time at all: + + - A [`StringType`](#stringtype) whose `data()` is not null-terminated compiles and silently misparses numbers, + because the lexer hands the buffer to `#!cpp std::strtoull`/`#!cpp std::strtoll`/`#!cpp std::strtod`. + - A stateful [`AllocatorType`](#allocatortype) compiles and silently ignores its state: allocation, deallocation, + and [`get_allocator()`](../../api/basic_json/get_allocator.md) each use a different default-constructed instance. + - The two [cross-specialization conversions](#cross-specialization-conversions) below. These abort on an assertion + in a normal build, and only fail silently under `#!cpp NDEBUG`. + +## Overview + +| Template parameter | Default | Notable substitutes | +|-------------------------------------------------------------------|-----------------------------------|-----------------------------------------------------------------------| +| [`ObjectType`](#objecttype) | `std::map` | [`nlohmann::ordered_map`](../../api/ordered_map.md), Abseil hash maps | +| [`ArrayType`](#arraytype) | `std::vector` | `#!cpp std::deque` | +| [`StringType`](#stringtype) | `std::string` | `std::string`-like types over `char` | +| [`BooleanType`](#booleantype) | `bool` | none worth using | +| [`NumberIntegerType`](#numberintegertype-and-numberunsignedtype) | `std::int64_t` | any signed integer type | +| [`NumberUnsignedType`](#numberintegertype-and-numberunsignedtype) | `std::uint64_t` | any unsigned integer type at least as wide as `NumberIntegerType` | +| [`NumberFloatType`](#numberfloattype) | `double` | `float` (`long double`: no binary formats) | +| [`AllocatorType`](#allocatortype) | `std::allocator` | stateless allocators | +| [`JSONSerializer`](#jsonserializer) | `adl_serializer` | serializers with the same interface | +| [`BinaryType`](#binarytype) | `#!cpp std::vector` | `#!cpp std::vector` | +| [`CustomBaseClass`](#custombaseclass) | `void` | any default-constructible class | + +!!! warning "Third-party containers and incomplete types" + + `object_t` is instantiated inside the definition of `basic_json` -- it is probed for a `key_compare` member to + form [`object_comparator_t`](../../api/basic_json/object_comparator_t.md) -- i.e. while `basic_json` is still an + incomplete type. `#!cpp std::map` is required by the standard to support incomplete mapped types; most + third-party maps are not, and inspecting the mapped type at class scope (for instance with + `#!cpp std::is_trivially_move_assignable`) makes them unusable as `ObjectType`, no matter how their template + arguments are adapted. This rules out `absl::btree_map`, `phmap::btree_map`, `gtl::btree_map`, + `robin_hood::unordered_node_map`, `folly::F14FastMap`, and `eastl::hash_map`. + + `array_t` is only *named* in the class definition and is not instantiated until `basic_json` is complete, so an + `ArrayType` that inspects its value type at class scope is generally fine -- `boost::container::small_vector` and + `static_vector` both reject incomplete value types yet work here. `absl::InlinedVector` is the exception: the + `#!cpp std::is_trivially_move_assignable` it evaluates while instantiating itself re-enters the + library's own trait machinery mid-instantiation. + +!!! note "Folly requires C++20" + + Folly's headers use `#!cpp consteval` and `#!cpp std::type_identity`, so any `basic_json` specialization that + names a Folly type has to be compiled as C++20 or later, whatever the rest of the library supports. + +## `ObjectType` + +`ObjectType` is instantiated as + +```cpp +using object_t = ObjectType>>; // allocator_type +``` + +i.e., the template arguments follow the order and meaning of `std::map`. + +### Always required + +- The template must be usable with **four** type arguments in the order shown above. The third argument is a + **comparator**; containers that expect something else in this position (e.g., a hash function) need an alias template + or wrapper -- see [Notes](#notes). +- An optional member type `key_compare`. If it is present it becomes + [`object_comparator_t`](../../api/basic_json/object_comparator_t.md); otherwise + [`default_object_comparator_t`](../../api/basic_json/default_object_comparator_t.md) is used. +- Member types `key_type`, `mapped_type`, `value_type`, and `iterator`. +- `value_type` must behave like `#!cpp std::pair`; the library accesses `.first` and + `.second` on it. +- `iterator` must be default-constructible and satisfy + [LegacyBidirectionalIterator](https://en.cppreference.com/w/cpp/named_req/BidirectionalIterator). The type returned + by `cbegin()`/`cend()` must satisfy the same requirements. +- Constructors: default, copy, move, and from an iterator range `(first, last)`. +- Member functions `begin()`, `end()`, `cbegin()`, `cend()`, `empty()`, `size()`, `max_size()`, `clear()`, + `find(key)`, `count(key)`, `emplace(key, value)`, `insert(value_type)`, `insert(first, last)`, `operator[](key)`, + `erase(iterator)`, and `erase(first, last)`. `erase(iterator)` may return the following iterator or `#!cpp void`; + in the latter case the library computes the successor itself, before erasing. +- `erase(key)` is **optional**: if the container does not provide one, the library falls back to `find(key)` followed + by `erase(iterator)`. +- `at(key)` is required only by [`to_ubjson`](../../api/basic_json/to_ubjson.md) and + [`to_bjdata`](../../api/basic_json/to_bjdata.md), but every container tried here provides it. +- `emplace` and `insert(value_type)` must return `#!cpp std::pair` and must have **unique-key** + semantics; multimaps cannot be used. +- The type must be swappable (via `std::swap` or an ADL `swap`). +- The comparison operators `==` and `<`; `!=`, `<=`, `>`, and `>=` are derived from them. Where the library uses + three-way comparison (C++20), `==` and `<=>` are required **instead** -- the six two-way operators do not satisfy + it. They implement [`basic_json`'s comparison operators](../../api/basic_json/operator_eq.md). + +### Required for heterogeneous key lookup + +The overloads of [`at`](../../api/basic_json/at.md), [`operator[]`](../../api/basic_json/operator%5B%5D.md), +[`find`](../../api/basic_json/find.md), [`contains`](../../api/basic_json/contains.md), +[`count`](../../api/basic_json/count.md), [`erase`](../../api/basic_json/erase.md), and +[`value`](../../api/basic_json/value.md) that accept a key type other than `object_t::key_type` require + +- a **transparent** comparator, i.e. [`object_comparator_t`](../../api/basic_json/object_comparator_t.md) has a member + type `is_transparent` (this is why the default comparator is `#!cpp std::less<>` since C++14), and +- corresponding heterogeneous `find`, `count`, `erase`, and `operator[]` overloads on the container. + +### Notes + +#### `std::unordered_map` needs an adapter + +`#!cpp std::unordered_map` cannot be passed directly: its third template parameter is a hash function, but +`basic_json` passes a comparator in that position. An alias template or wrapper that restores the expected argument +order makes it usable: + +```cpp +template +struct unordered_map_object + : std::unordered_map, std::equal_to, Allocator> +{ + using base_t = std::unordered_map, std::equal_to, Allocator>; + using base_t::base_t; +}; + +using unordered_json = nlohmann::basic_json; +``` + +Whether `#!cpp std::unordered_map` can be instantiated at all depends on the standard library: `object_t` is formed +while `basic_json` is still incomplete (see the warning above), and libstdc++ 9 needs the size of the mapped type to +instantiate the hash map's node type, so the adapter does not compile there. Newer libstdc++ versions, and the hash +maps listed below, do not have that problem. + +The adapter above works verbatim for Abseil's, Boost's, `phmap`'s and `gtl`'s hash maps, which all place the hash +function third and take a `#!cpp std::pair` allocator fifth. Two need a different adapter: + +- `ankerl::unordered_dense` expects an allocator over `#!cpp std::pair` (non-const key), so the allocator has + to be rebound to that or dropped. +- `robin_hood`'s fifth parameter is the non-type `MaxLoadFactor100`, so its adapter must drop the allocator entirely. + +None of these hash maps defines `key_compare`, so all of them additionally rely on `object_comparator_t` falling back +to [`default_object_comparator_t`](../../api/basic_json/default_object_comparator_t.md); see +[`object_comparator_t`](../../api/basic_json/object_comparator_t.md). + +#### Abseil hash maps + +`absl::flat_hash_map` and `absl::node_hash_map` tolerate an incomplete value type, but they take a hash function as +their third template argument. The same adapter as for `#!cpp std::unordered_map` makes them usable: + +```cpp +template +struct flat_hash_object + : absl::flat_hash_map, std::equal_to, Allocator> +{ + using base_t = absl::flat_hash_map, std::equal_to, Allocator>; + using base_t::base_t; +}; + +using flat_hash_json = nlohmann::basic_json; +``` + +`absl::node_hash_map` keeps references to the mapped values valid across insertions; `absl::flat_hash_map` does not, +which makes it behave like [`ordered_json`](../../api/ordered_json.md) with respect to +[iterator invalidation](../../api/basic_json/index.md#iterator-invalidation). Both expose a `capacity()` member +function, so [`JSON_DIAGNOSTICS`](../../api/macros/json_diagnostics.md) treats them conservatively and keeps the +parent pointers correct either way. + +#### Iteration order + +The library never relies on the container's iteration order for correctness; it does determine the order in which +object keys are serialized by [`dump`](../../api/basic_json/dump.md) and visited by +[`items`](../../api/basic_json/items.md). See [Object Order](../object_order.md). + +#### `capacity()` marks a container as insertion-ordered + +With [`JSON_DIAGNOSTICS`](../../api/macros/json_diagnostics.md) enabled, the library detects insertion-ordered maps by +probing for a `capacity()` member function (`nlohmann::ordered_map` inherits it from `std::vector`) and refreshes all +parent pointers after every insertion. An `ObjectType` that happens to have a `capacity()` member is therefore treated +conservatively -- this is correct, but slower. + +#### Key order and duplicate keys + +The library does not sort or de-duplicate keys itself; the behavior described in +[`object_t`](../../api/basic_json/object_t.md) is entirely the behavior of the chosen container. + +!!! tip "Reference implementation" + + `docs/mkdocs/docs/examples/custom_object_type.hpp` wraps a private `#!cpp std::map` and satisfies every + requirement above. It does not define `key_compare`, so `object_comparator_t` falls back to + [`default_object_comparator_t`](../../api/basic_json/default_object_comparator_t.md) -- a good starting point for + a custom `ObjectType`. + + ```cpp + --8<-- "examples/custom_object_type.hpp" + ``` + +??? example "Compiling and using it" + + ```cpp + --8<-- "examples/custom_object_type.cpp" + ``` + + Output: + + ```json + --8<-- "examples/custom_object_type.output" + ``` + +### Compatible containers + +| Container | Notes | +|----------------------------------------------------------------------------------|-------------------------------------------------------------------------------| +| `#!cpp std::map` (default) | | +| [`nlohmann::ordered_map`](../../api/ordered_map.md) | used by [`ordered_json`](../../api/ordered_json.md); keeps insertion order | +| [`nlohmann::fifo_map`](https://github.com/nlohmann/fifo_map) | keeps insertion order; adapter puts `fifo_map_compare` in the comparator slot | +| `boost::container::map`, `boost::container::flat_map` | no adapter needed | +| `#!cpp std::unordered_map` | through the adapter above; not with libstdc++ 9, see the note | +| `boost::unordered_map`, `boost::unordered_flat_map`, `boost::unordered_node_map` | through the adapter above | +| `absl::flat_hash_map`, `absl::node_hash_map` | through the adapter above; `flat_hash_map` moves mapped values on rehash | +| `phmap::flat_hash_map`, `phmap::node_hash_map`, `gtl::flat_hash_map` | through the adapter above | +| `ankerl::unordered_dense::map` and `segmented_map` | adapter must rebind or drop the allocator | +| `robin_hood::unordered_flat_map` | adapter must drop the allocator | +| `folly::F14NodeMap` | through the adapter above; requires C++20, see the note above | +| `folly::sorted_vector_map` | alias must drop the allocator, whose value type it disagrees on | + +### Containers that cannot be used + +| Container | Reason | +|--------------------------------------------------------------------------|---------------------------------------------------------------------------------------------------------------------| +| `absl::btree_map`, `phmap::btree_map`, `gtl::btree_map` | require a complete mapped type | +| `robin_hood::unordered_node_map`, `folly::F14FastMap`, `eastl::hash_map` | require a complete mapped type | +| `eastl::map` | EASTL iterators do not work with `#!cpp std::iterator_traits` | +| `tsl::ordered_map` | its iterators expose the mapped value as `#!cpp const` | +| `QMap` | no `value_type` member type | +| `QHash` | its `value_type` is the mapped type rather than a key/value pair, and its iterators dereference to the mapped value | +| `#!cpp std::multimap`, `#!cpp std::unordered_multimap` | `emplace` does not return `#!cpp std::pair` | + +## `ArrayType` + +`ArrayType` is instantiated as + +```cpp +using array_t = ArrayType>; +``` + +### Always required + +- The template must be usable with **two** type arguments (value type and allocator). +- Member types `value_type` and `iterator`. +- Constructors: default, copy, and move; and from an iterator range `(first, last)`. +- Member functions `begin()`, `end()`, `cbegin()`, `cend()`, `empty()`, `size()`, `max_size()`, `clear()`, + `operator[](size_type)`, `back()`, `push_back()`, `emplace_back()`, `pop_back()`, `resize()`, + `insert()` (single element, count, and range), `erase(pos)`, and `erase(first, last)`. + `basic_json::insert(pos, initializer_list)` goes through the range overload, so no initializer-list `insert` is + needed. `at(size_type)` is **not** required: [`basic_json::at(size_type)`](../../api/basic_json/at.md) checks the + index itself and then uses `operator[]`. +- `iterator` must be default-constructible, and it as well as the type returned by `cbegin()`/`cend()` must satisfy + [LegacyRandomAccessIterator](https://en.cppreference.com/w/cpp/named_req/RandomAccessIterator). + A `#!cpp static_assert` only checks for + [LegacyBidirectionalIterator](https://en.cppreference.com/w/cpp/named_req/BidirectionalIterator), but + [`dump`](../../api/basic_json/dump.md) (`cend() - 1`), + [`erase(idx)`](../../api/basic_json/erase.md) (`begin() + idx`), and the random-access operations of + [`basic_json::iterator`](../../api/basic_json/begin.md) require random access. +- The comparison operators, as for [`ObjectType`](#objecttype): `==` and `<`, or `==` and `<=>` under C++20. + +### Required for individual functions + +- A member type `value_type`, for [`to_bson`](../../api/basic_json/to_bson.md) of an array. +- A constructor from `(count, value)`, for + [`basic_json(size_type, const basic_json&)`](../../api/basic_json/basic_json.md). +- Swappability, via `#!cpp std::swap` or an ADL `swap`, for [`swap(array_t&)`](../../api/basic_json/swap.md). + +!!! note "`capacity()` is optional" + + With [`JSON_DIAGNOSTICS`](../../api/macros/json_diagnostics.md) enabled, the library reads `array_t::capacity()` + to find out whether adding an element reallocated the array and moved its elements, which would invalidate the + parent pointers. An array type without a `capacity()` member function is handled conservatively: the parent + pointers of all elements are refreshed after every insertion, which makes adding *n* elements cost O(*n*²). Only + diagnostics builds pay this; without them `capacity()` is never called. + +!!! tip "Reference implementation" + + `docs/mkdocs/docs/examples/custom_array_type.hpp` wraps a private `#!cpp std::vector` and satisfies every + requirement above -- a good starting point for a custom `ArrayType`. + + ```cpp + --8<-- "examples/custom_array_type.hpp" + ``` + +??? example "Compiling and using it" + + ```cpp + --8<-- "examples/custom_array_type.cpp" + ``` + + Output: + + ```json + --8<-- "examples/custom_array_type.output" + ``` + +### Compatible containers + +| Container | Notes | +|---------------------------------------------------------|-------------------------------------------------------------------------------------------| +| `#!cpp std::vector` (default) | | +| `#!cpp std::deque` | references survive appends, but not insertions elsewhere; see the `capacity()` note above | +| `#!cpp std::pmr::vector` | through an alias, as the allocator comes from `AllocatorType` instead | +| `boost::container::vector`, `deque`, `devector` | | +| `boost::container::stable_vector` | the only one tried that keeps references valid across *every* insertion | +| `boost::container::small_vector`, `folly::small_vector` | through an alias that fixes the inline capacity | +| `boost::container::static_vector` | through the same kind of alias, for arrays that stay within the fixed capacity | +| `folly::fbvector` | requires C++20, see the note above | + +### Containers that cannot be used + +| Container | Reason | +|-------------------------------------|-----------------------------------------------------------------------------------------------| +| `#!cpp std::list` | no `operator[]`, and no random-access iterators | +| `eastl::vector`, `QList`, `QVector` | no `max_size()`; they handle the incomplete value type fine | +| `absl::InlinedVector` | requires a complete value type, see the note above | +| `absl::FixedArray` | the size is fixed at construction, so `resize`, `push_back`, `insert` and `erase` are missing | + +## `StringType` + +`StringType` is used **both** for JSON string values and for the keys of JSON objects +(`string_t` and `object_t::key_type`). + +### Always required + +- A member type `value_type` that is one byte wide and `char`-compatible. The library stores and processes UTF-8 + encoded `char` data and hands `data()` to `#!cpp std::strtoull`/`#!cpp std::strtoll`. + `#!cpp std::wstring`, `#!cpp std::u16string`, and `#!cpp std::u32string` are **not** valid choices; see the FAQ on + [wide string handling](../../home/faq.md#wide-string-handling). +- Constructors: default, copy, move, from `#!cpp const char*` (which must not be `#!cpp explicit`), from + `#!cpp (const char*, size_type)`, and from `#!cpp (size_type, char)`; and copy or move assignment. +- Member functions `size()`, `clear()`, `resize(n, c)`, `data()`, `push_back(char)`, and `operator[]` + (const and non-const, returning references). `c_str()` and `back()` are **not** required. +- `data()` must return a pointer to a contiguous, **null-terminated** buffer -- the parser hands it to + `#!cpp std::strtoull`. A type whose `data()` is not null-terminated does not fail to compile; it silently + misparses numbers. +- `append(const char*, size_type)`, used by [`dump`](../../api/basic_json/dump.md), and `append(const StringType&)`, + used by the CBOR reader for indefinite-length strings. The library's internal string concatenation additionally has + to append a `#!cpp char` and a `#!cpp const char*`; for each it selects between `append(arg)`, `#!cpp operator+=`, + `append(first, last)`, and `append(data, size)`. +- The comparison operator `==` against another `StringType`, and `<` for use as a key of the chosen + [`ObjectType`](#objecttype) (with the default comparator, `#!cpp std::less<>` must be able to compare two + `StringType` values, and a `StringType` with the key types used for lookup). `!=` is never applied to a + `StringType`, and `==` against `#!cpp const char*` is resolved by the implicit `#!cpp const char*` constructor. + +### Required for the binary formats + +- `resize(n)`, used by the readers to make room for a block of bytes. +- Non-const `operator[]`, into which the readers `#!cpp std::memcpy` those bytes. A non-`#!cpp const` `data()` would + serve just as well, but `#!cpp std::string` has only had one since C++17, and the library still supports C++11. + +### Required for JSON Pointer, `flatten`, and `diff` + +- A static member `npos` and the member function `find_first_of(char, size_type)` -- together with `data()`, + `reserve(n)`, and `append(const char*, size_type)` they implement the escaping and unescaping of reference tokens + described in RFC 6901. Neither `find(const StringType&, size_type)`, nor `substr(pos, count)`, nor + `replace(pos, count, const StringType&)` is required. +- `empty()`. +- `begin()` and `end()` -- used by + [`operator[](const json_pointer&)`](../../api/basic_json/operator%5B%5D.md) to decide whether a reference token + denotes an array index. + +### Required for other functionality + +| Functionality | Additional requirement | +|-----------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| [`diff`](../../api/basic_json/diff.md), [`items`](../../api/basic_json/items.md), [`std::hash`](../../api/basic_json/std_hash.md) | conversion of a `#!cpp std::size_t` to `StringType`: either assignability from the result of `#!cpp std::to_string`, or an ADL overload `#!cpp void int_to_string(StringType&, std::size_t)` | +| [`std::hash`](../../api/basic_json/std_hash.md) | additionally a specialization of `#!cpp std::hash` | +| [`to_bson`](../../api/basic_json/to_bson.md) | `find(value_type)` and `npos` | +| [`parse`](../../api/basic_json/parse.md) from a `string_t` | the input adapters must accept it; otherwise pass a character range | +| `#!cpp operator<<(std::ostream&, const json_pointer&)` | streamability to `#!cpp std::ostream` | +| exception messages | `data()` and `size()`, or `begin()` and `end()` | + +### Compatible types + +| Type | Notes | +|-----------------------------------------------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `#!cpp std::string` (default) | | +| `#!cpp std::basic_string` with a custom **stateless** allocator | | +| `#!cpp std::pmr::string` | see the warning below before relying on the memory resource | +| `boost::container::string` | needs a user-supplied `#!cpp std::hash` specialization (Boost provides `boost::hash` instead) | +| `folly::fbstring` | requires C++20, see the note above | +| `eastl::string` | needs a user-supplied `#!cpp std::hash` and an ADL `int_to_string` (it is not assignable from a `#!cpp std::string`); [`parse`](../../api/basic_json/parse.md) does not accept it directly -- pass a character range or a `#!cpp std::string` | +| a custom string class in a user-defined namespace | if the requirements above are met | + +### Types that cannot be used + +| Type | Reason | +|----------------------------------------------------------------------|---------------------------------------------------------------------------------------------------------| +| `#!cpp std::wstring`, `#!cpp std::u16string`, `#!cpp std::u32string` | the character type is not one byte wide | +| `#!cpp std::u8string` | one byte wide, but `#!cpp char8_t` is not `#!cpp char`-compatible | +| `absl::Cord` | no `value_type`, and the storage is not contiguous | +| `QString` | no `append(const char*, size_type)`; its `QChar` is also two bytes wide, though that is never diagnosed | + +!!! warning "A `std::pmr::string` mostly does not use the memory resource you choose" + + `basic_json` cannot be given an allocator or a memory resource. `AllocatorType` is default-constructed at every + allocation and has to be stateless (see [`AllocatorType`](#allocatortype)), and string values the library creates + are constructed with their own default allocator. So: + + - Every string the library itself produces -- from [`parse`](../../api/basic_json/parse.md), from + [`dump`](../../api/basic_json/dump.md), or by default construction -- allocates from + `#!cpp std::pmr::get_default_resource()`. + - **Copying** an arena-backed string into a value silently drops its memory resource: the copy lands on the + default resource, because `#!cpp std::pmr::polymorphic_allocator` does not propagate on copy construction. + Nothing warns about this. + - **Moving** one in does keep it, and later growth still allocates from that arena -- but it does not survive a + copy of the enclosing `basic_json`. + - Passing `#!cpp std::pmr::polymorphic_allocator` as `AllocatorType` does not work around any of this; it does + not compile. + + Apart from moving a string in, the only way to redirect these allocations is the process-global + `#!cpp std::pmr::set_default_resource()`. + +!!! tip "Reference implementation" + + `docs/mkdocs/docs/examples/custom_string_type.hpp` wraps a private `#!cpp std::string` and satisfies every + requirement above -- a good starting point for a custom `StringType`. The unit test + `tests/src/unit-alt-string.cpp` contains a more thorough variant, `alt_string`, exercised against a larger part + of the API. + + ```cpp + --8<-- "examples/custom_string_type.hpp" + ``` + +??? example "Compiling and using it" + + ```cpp + --8<-- "examples/custom_string_type.cpp" + ``` + + Output: + + ```json + --8<-- "examples/custom_string_type.output" + ``` + +## `BooleanType` + +`boolean_t` is stored **directly** inside `basic_json`, as a member of an anonymous union. + +### Always required + +- A literal type that is trivially default-constructible, trivially copyable, and trivially destructible; otherwise the + union's special member functions are deleted. +- **Implicitly** convertible from `#!cpp bool` -- an `#!cpp explicit` constructor is not enough, because the + `to_json` overload for a custom `BooleanType` is constrained on `#!cpp std::is_convertible` -- and contextually + convertible to `#!cpp bool` (here an `#!cpp explicit operator bool` is fine). +- Comparison operators `==`, `!=`, `<`, `<=`, `>`, `>=` (or `<=>`). +- Convertible from and to `#!cpp bool` through the serializer, because + [`get()`](../../api/basic_json/get.md) is used internally. + +There is little reason to use anything other than `#!cpp bool` here. + +### Compatible types + +`#!cpp bool` is the only usable choice. Another trivially copyable type that is implicitly convertible to and from +`#!cpp bool` -- `#!cpp std::uint8_t`, say -- does compile, and JSON booleans still round-trip, but the type then +serves as both `boolean_t` and an ordinary integer: `basic_json` can no longer be constructed or assigned from a +`#!cpp std::uint8_t` at all (the boolean and unsigned-integer `to_json` overloads become ambiguous), and +[`get()`](../../api/basic_json/get.md) on a number throws +[`type_error.302`](../../home/exceptions.md#jsonexceptiontype_error302) instead of returning the value. + +## `NumberIntegerType` and `NumberUnsignedType` + +Both types are stored **directly** inside `basic_json`'s union. + +### Always required + +- `#!cpp std::is_integral` must be satisfied: `NumberIntegerType` must be a **signed** integer type, + `NumberUnsignedType` an **unsigned** integer type. Class types are not supported -- among others, the constructors + taking integer values are constrained on `#!cpp std::is_integral`. +- Trivially default-constructible, trivially copyable, and trivially destructible (union member). +- `#!cpp std::numeric_limits` must be specialized for both types. +- `NumberUnsignedType` must be able to represent the absolute value of every `NumberIntegerType` value; serialization + of negative numbers converts the value to `NumberUnsignedType`. A `#!cpp static_assert` requires it to be at least as + wide as `NumberIntegerType`, which is what that amounts to for the standard integer types. +- Both types must fit into the internal 64-character number buffer used by + [`dump`](../../api/basic_json/dump.md), which is the case for all standard integer types. +- [`std::hash`](../../api/basic_json/std_hash.md) additionally requires `#!cpp std::hash` specializations. + +### Notes + +The number types influence what the parser accepts: an integer literal that does not round-trip through the chosen type +is stored as [`number_float_t`](../../api/basic_json/number_float_t.md) instead. Choosing types narrower than 64 bits +therefore silently changes parse results rather than raising an error. See +[Number Handling](number_handling.md) for details. + +### Compatible types + +| Type pair | Support | +|----------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `#!cpp std::int64_t` / `#!cpp std::uint64_t` (default) | full | +| `#!cpp std::int32_t` / `#!cpp std::uint32_t`, `#!cpp long long` / `#!cpp unsigned long long` | full; narrower types change which literals the parser can represent | +| any other pair of standard signed/unsigned integer types | full | +| class types, enumerations | not usable; `#!cpp std::is_integral` must hold | +| `#!cpp bool`, or a type already used for another member of the union | not usable; `#!cpp std::is_integral` is in fact `#!cpp true`, but the `get_impl_ptr` overloads for `boolean_t`, `number_integer_t`, `number_unsigned_t` and `number_float_t` would collide | + +## `NumberFloatType` + +`number_float_t` is stored **directly** inside `basic_json`'s union. + +### Always required + +- Trivially default-constructible, trivially copyable, and trivially destructible (union member). +- `#!cpp std::numeric_limits` must be specialized; `max_digits10` is used to size the conversion. +- `#!cpp std::isfinite` must be applicable to the type. + +### Required for parsing and serialization + +`NumberFloatType` must be one of `#!cpp float`, `#!cpp double`, or `#!cpp long double`: + +- The [parser](../parsing/index.md) converts number literals with `#!cpp std::strtof`, `#!cpp std::strtod`, or + `#!cpp std::strtold`; the library provides overloads for exactly these three types. +- [`dump`](../../api/basic_json/dump.md) falls back to `#!cpp std::snprintf` with the `%g` and `%Lg` conversion + specifiers, for which the library likewise provides only `#!cpp double` and `#!cpp long double` overloads + (`#!cpp float` is promoted to `#!cpp double`). + +If `#!cpp std::numeric_limits` describes an IEEE 754 binary32 or binary64 number, `dump` uses the +Grisu2 algorithm, which produces the shortest representation that round-trips. Otherwise the `snprintf` fallback with +`max_digits10` digits is used. + +### Required for the binary formats + +`NumberFloatType` must be `#!cpp float` or `#!cpp double`. The writers for +[CBOR, MessagePack, UBJSON, BJData, and BSON](../binary_formats/index.md) map a floating-point value onto an IEEE 754 +binary32 or binary64 field and have no encoding for `#!cpp long double`. + +### Compatible types + +| Type | Support | +|--------------------------|-----------------------------------------------------------------------------------------------------------------------| +| `#!cpp double` (default) | full; short round-trip output through Grisu2 | +| `#!cpp float` | full; short round-trip output through Grisu2 | +| `#!cpp long double` | `dump` and `parse` only; the binary format writers do not compile, as they only handle IEEE 754 binary32 and binary64 | +| any other type | not usable | + +## `AllocatorType` + +`AllocatorType` is instantiated with **one** argument, for each of `object_t`, `array_t`, `string_t`, `binary_t`, +`basic_json`, and `#!cpp std::pair`. + +### Always required + +- The template must be usable with exactly one type argument. The library instantiates `AllocatorType` directly and + never uses `#!cpp std::allocator_traits<...>::rebind_alloc`. +- It must satisfy the [Allocator](https://en.cppreference.com/w/cpp/named_req/Allocator) named requirement so that + `#!cpp std::allocator_traits` can be used with it. +- It must be **default-constructible and stateless**. Objects are allocated with a default-constructed allocator and + deallocated with a *different* default-constructed allocator, and + [`get_allocator()`](../../api/basic_json/get_allocator.md) returns a default-constructed instance. Allocators + carrying state are not supported, so there is no way to tell a `basic_json` where to allocate from; see the note + under [`StringType`](#stringtype) for what that means in practice. A stateful allocator is **not diagnosed**: it + compiles and silently ignores the state. +- It must support **incomplete types**: `AllocatorType` is instantiated inside the definition of + `basic_json` itself. +- `#!cpp std::allocator_traits>::pointer` becomes + [`basic_json::pointer`](../../api/basic_json/index.md#container-types), and iterators are constructed from raw + `#!cpp basic_json*` values. The `pointer` type must therefore be a plain pointer; fancy pointers are not supported. + +### Compatible types + +| Type | Support | +|-------------------------------------------------------------------|----------------------------------------| +| `#!cpp std::allocator` (default) | full | +| a custom stateless allocator template | full | +| stateful allocators, e.g. `#!cpp std::pmr::polymorphic_allocator` | not usable; see the requirements above | + +## `JSONSerializer` + +`JSONSerializer` is instantiated as `JSONSerializer` and defaults to +[`adl_serializer`](../../api/adl_serializer/index.md). + +### Always required + +- The template must accept **two** type arguments. It does not have to give the second one a default -- `basic_json` + declares the parameter as `#!cpp template class JSONSerializer`, so uses such as + `#!cpp JSONSerializer` inside the library supply `#!cpp void` themselves. The second parameter exists so that + partial specializations can be constrained by SFINAE. +- For every type `T` that is converted **to** a JSON value, a static member function + `#!cpp static void to_json(basic_json&, T)` must exist. +- For every type `T` that is converted **from** a JSON value, either + `#!cpp static void from_json(const basic_json&, T&)` or `#!cpp static T from_json(const basic_json&)` must exist. + The latter form is required for types that are not default-constructible; see + [Arbitrary Types Conversions](../arbitrary_types.md). +- To support the [converting constructor](../../api/basic_json/basic_json.md) between different `basic_json` + specializations, `to_json` must be available for `boolean_t`, `number_integer_t`, `number_unsigned_t`, + `number_float_t`, `string_t`, `object_t`, `array_t`, and `binary_t` of the *source* specialization. + +### Compatible types + +| Type | Support | +|---------------------------------------------------------------------------|-------------------------------------------------------------------| +| [`nlohmann::adl_serializer`](../../api/adl_serializer/index.md) (default) | full | +| a class template deriving from `adl_serializer` | full; the usual way to change behavior while keeping the defaults | +| an unrelated template with the same interface | full, but it has to handle every type the library converts | + +## `BinaryType` + +`BinaryType` is not a JSON type; it is used for the byte strings of the +[binary formats](../binary_formats/index.md). It is wrapped as + +```cpp +using binary_t = nlohmann::byte_container_with_subtype; +``` + +### Always required + +- A non-`final` class type -- [`byte_container_with_subtype`](../../api/byte_container_with_subtype/index.md) derives + from it publicly. +- A member type `value_type` that is **exactly one byte** wide (e.g., `#!cpp std::uint8_t`, `#!cpp char`, or + `#!cpp std::byte`). Readers and writers reinterpret the container's storage as raw bytes, so a wider `value_type` is + rejected with a `#!cpp static_assert`. +- Contiguous storage: the binary readers `#!cpp std::memcpy` into `#!cpp &binary[n]`, the writers `reinterpret_cast` + `data()`. `#!cpp data() + n` would do for the readers too, but they share one helper with + [`StringType`](#stringtype), whose non-`#!cpp const` `data()` is C++17 and later only. +- Default-constructible, copy-constructible, and move-constructible. +- Member functions `size()`, `empty()`, `data()`, `resize()`, `operator[]`, `back()`, `begin()`, `end()`, `cbegin()`, + and `cend()` with random-access iterators, and `insert(pos, first, last)`, which the CBOR reader uses to join the + chunks of an indefinite-length byte string. `push_back()` is **not** required. +- Comparison operators: `==` is used by + [`byte_container_with_subtype`](../../api/byte_container_with_subtype/index.md), the relational operators by + [`basic_json`'s comparison operators](../../api/basic_json/operator_le.md). + +### Required for individual functions + +- `clear()`, for [`basic_json::clear()`](../../api/basic_json/clear.md). + +`max_size()`, `at()`, `reserve()`, `erase()`, `pop_back()`, and `emplace_back()` are **not** used at all. + +See [`binary_t`](../../api/basic_json/binary_t.md) for how a non-default `BinaryType` changes the meaning of assigning +such a container to a `basic_json` value. + +!!! tip "Reference implementation" + + `docs/mkdocs/docs/examples/custom_binary_type.hpp` wraps a private `#!cpp std::vector` and satisfies + every requirement above -- a good starting point for a custom `BinaryType`. + + ```cpp + --8<-- "examples/custom_binary_type.hpp" + ``` + +??? example "Compiling and using it" + + ```cpp + --8<-- "examples/custom_binary_type.cpp" + ``` + + Output: + + ```json + --8<-- "examples/custom_binary_type.output" + ``` + +### Compatible containers + +| Container | Notes | +|---------------------------------------------------------------------------------------------|---------------------------------------------------------------------------| +| `#!cpp std::vector` (default) | | +| `#!cpp std::vector`, `#!cpp std::vector` | `dump()` writes the bytes as 0..255 whichever is used | +| `boost::container::vector`, `boost::container::small_vector` | | +| `absl::InlinedVector` | usable here, unlike as an `ArrayType`, because the value type is complete | +| `eastl::vector` | usable here, unlike as an `ArrayType`, because `max_size()` is not needed | +| `folly::fbvector` | requires C++20, see the note above | + +### Containers that cannot be used + +| Container | Reason | +|------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `QByteArray` | no `empty()` (it spells that `isEmpty()`); its `insert` takes an index rather than an iterator; and it converts to `string_t`, which makes `to_json` ambiguous between a string and a binary value | +| `#!cpp std::string` | `binary_t::container_type` and `string_t` would be the same type, so the two [`swap`](../../api/basic_json/swap.md) overloads collide and `basic_json` cannot be instantiated at all | +| `#!cpp std::deque` | storage is not contiguous, so there is no `data()` | +| containers whose `value_type` is wider than one byte | see above -- accepted by the compiler, wrong at runtime | + +## `CustomBaseClass` + +`CustomBaseClass` is an extension point: unless it is `#!cpp void` (the default, which selects the empty +`nlohmann::json_default_base`), `basic_json` publicly derives from it. + +### Always required + +- A non-`final`, default-constructible class type. +- `basic_json` is copy-/move-constructible and copy-/move-assignable only if `CustomBaseClass` is. + +### Notes + +`basic_json` is documented to be a +[StandardLayoutType](https://en.cppreference.com/w/cpp/named_req/StandardLayoutType). Because `basic_json` has +non-static data members of its own, a `CustomBaseClass` with non-static data members forfeits this guarantee. + +Note the namespace of `CustomBaseClass` becomes an associated namespace of `basic_json` for the purpose of +argument-dependent lookup. + +See [`json_base_class_t`](../../api/basic_json/json_base_class_t.md) for an example. + +### Compatible types + +| Type | Support | +|----------------------------------------------|----------------------------------------------------------------------------| +| `#!cpp void` (default) | an empty base class is used; no effect on `basic_json` | +| any default-constructible, non-`final` class | full; see [`json_base_class_t`](../../api/basic_json/json_base_class_t.md) | + +## Cross-specialization conversions + +Converting a value from one `basic_json` specialization into another (see the +[converting constructor](../../api/basic_json/basic_json.md)) imposes two additional requirements that are not +diagnosed at compile time. With assertions enabled they abort on the `#!cpp JSON_ASSERT` at the end of the converting +constructor; under `#!cpp NDEBUG` they fail **silently** at runtime: + +- The target `string_t` must be directly constructible from the source `string_t`. Otherwise the string is converted to + an array of character codes. +- The target `object_t::key_type` must be directly constructible from the source object's key type. Otherwise the + object is converted to an array of key/value pairs. + +See [issue #3425](https://github.com/nlohmann/json/issues/3425), [`string_t`](../../api/basic_json/string_t.md), and +[`object_t`](../../api/basic_json/object_t.md). + +## See also + +- [Types](index.md) -- overview of how JSON values are stored +- [Number Handling](number_handling.md) -- how the number types affect parsing and serialization +- [Object Order](../object_order.md) -- using an insertion-ordered `ObjectType` +- [`basic_json`](../../api/basic_json/index.md) -- API documentation of the class template diff --git a/docs/mkdocs/docs/home/architecture.md b/docs/mkdocs/docs/home/architecture.md index aba2be580..6c0892f11 100644 --- a/docs/mkdocs/docs/home/architecture.md +++ b/docs/mkdocs/docs/home/architecture.md @@ -1,34 +1,125 @@ # Architecture -!!! info - - This page is still under construction. Its goal is to provide a high-level overview of the library's architecture. - This should help new contributors to get an idea of the used concepts and where to make changes. +This page gives a high-level overview of the library's architecture. It should help new contributors to get an idea of +the used concepts and where to make changes. ## Overview -The main structure is class [nlohmann::basic_json](../api/basic_json/index.md). +The library is built around a single class template, [`nlohmann::basic_json`](../api/basic_json/index.md). A +`basic_json` value is a node in a tree of JSON values. All other components either create such a tree from an input +(parsing), write a tree to an output (serialization), or give access to it (iterators, JSON Pointer, conversions). -- public API -- container interface -- iterators +```mermaid +flowchart LR + input[/"input
(string, stream,
iterator range, file)"/] + ia["input adapter"] + lexer["lexer"] + parser["parser"] + breader["binary_reader"] + sax["SAX interface"] + value[("basic_json
value tree")] + serializer["serializer"] + bwriter["binary_writer"] + oa["output adapter"] + output[/"output
(string, stream,
vector)"/] -## Template specializations + input --> ia + ia --> lexer --> parser --> sax + ia --> breader --> sax + sax --> value + value --> serializer --> oa + value --> bwriter --> oa + oa --> output +``` -- describe template parameters of `basic_json` -- [`json`](../api/json.md) -- [`ordered_json`](../api/ordered_json.md) via [`ordered_map`](../api/ordered_map.md) +- **JSON text** is read by an [input adapter](#input-adapters), tokenized by the lexer, and turned into SAX events by + the parser. +- **Binary formats** (BJData, BSON, CBOR, MessagePack, UBJSON) are read by an input adapter and turned into the same SAX + events by the `binary_reader`. +- A [SAX consumer](#sax-interface) receives the events. The one used by [`parse`](../api/basic_json/parse.md) builds a + `basic_json` value tree. +- The `serializer` (JSON text) or the `binary_writer` (binary formats) writes a value tree to an + [output adapter](#output-adapters). + +## Source layout + +The public headers are in [`include/nlohmann`](https://github.com/nlohmann/json/tree/develop/include/nlohmann): + +- [`json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/json.hpp) defines class [`basic_json`](../api/basic_json/index.md). +- [`json_fwd.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/json_fwd.hpp) contains forward declarations. +- [`adl_serializer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/adl_serializer.hpp), [`byte_container_with_subtype.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/byte_container_with_subtype.hpp), and [`ordered_map.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/ordered_map.hpp) define + [`adl_serializer`](../api/adl_serializer/index.md), + [`byte_container_with_subtype`](../api/byte_container_with_subtype/index.md), and + [`ordered_map`](../api/ordered_map.md). + +Everything else lives in [`detail/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail) and namespace `nlohmann::detail`, which is not part of the public API. Paths +below are relative to `include/nlohmann`. + +| Component | Location | +|-----------|----------| +| Value type enumeration | [`detail/value_t.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/value_t.hpp) | +| Input adapters | [`detail/input/input_adapters.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/input_adapters.hpp) | +| Lexer | [`detail/input/lexer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/lexer.hpp), [`detail/input/number_parse.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/number_parse.hpp), [`detail/input/string_scan.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/string_scan.hpp) | +| Parser | [`detail/input/parser.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/parser.hpp) | +| SAX interface and DOM builders | [`detail/input/json_sax.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/json_sax.hpp) | +| Binary format readers | [`detail/input/binary_reader.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/binary_reader.hpp) | +| JSON serializer | [`detail/output/serializer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/serializer.hpp), [`detail/conversions/to_chars.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/to_chars.hpp) | +| Binary format writers | [`detail/output/binary_writer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/binary_writer.hpp) | +| Output adapters | [`detail/output/output_adapters.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/output_adapters.hpp) | +| Iterators | [`detail/iterators/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail/iterators) | +| Conversions from/to arbitrary types | [`detail/conversions/from_json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/from_json.hpp), [`detail/conversions/to_json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/to_json.hpp) | +| JSON Pointer | [`detail/json_pointer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/json_pointer.hpp) | +| Exceptions | [`detail/exceptions.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/exceptions.hpp) | +| Type traits and C++ feature backports | [`detail/meta/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail/meta) | +| Macros | [`detail/macro_scope.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/macro_scope.hpp), [`detail/macro_unscope.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/macro_unscope.hpp), [`detail/abi_macros.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/abi_macros.hpp) | + +The single-header version [`single_include/nlohmann/json.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json.hpp) +is generated from these files with `make amalgamate` and must not be edited by hand. + +## Template parameters + +[`basic_json`](../api/basic_json/index.md) is parameterized by the types it uses to store values and to convert from and to other types: + +| Template parameter | Default | Used for | +|----------------------|-----------------------------|-------------------------------------------------------------------| +| `ObjectType` | `std::map` | objects, see [`object_t`](../api/basic_json/object_t.md) | +| `ArrayType` | `std::vector` | arrays, see [`array_t`](../api/basic_json/array_t.md) | +| `StringType` | `std::string` | strings and object keys, see [`string_t`](../api/basic_json/string_t.md) | +| `BooleanType` | `bool` | Booleans, see [`boolean_t`](../api/basic_json/boolean_t.md) | +| `NumberIntegerType` | `std::int64_t` | signed integers, see [`number_integer_t`](../api/basic_json/number_integer_t.md) | +| `NumberUnsignedType` | `std::uint64_t` | unsigned integers, see [`number_unsigned_t`](../api/basic_json/number_unsigned_t.md) | +| `NumberFloatType` | `double` | floating-point numbers, see [`number_float_t`](../api/basic_json/number_float_t.md) | +| `AllocatorType` | `std::allocator` | allocating objects, arrays, strings, and binary values | +| `JSONSerializer` | `adl_serializer` | conversions from/to other types, see [`adl_serializer`](../api/adl_serializer/index.md) | +| `BinaryType` | `std::vector` | binary values, see [`binary_t`](../api/basic_json/binary_t.md) | +| `CustomBaseClass` | `void` | an optional base class, see [`json_base_class_t`](../api/basic_json/json_base_class_t.md) | + +The library provides two specializations: + +- [`json`](../api/json.md) uses all default template arguments. +- [`ordered_json`](../api/ordered_json.md) uses [`ordered_map`](../api/ordered_map.md) as `ObjectType` to keep the + insertion order of object keys. + +The requirements on the template arguments are listed in +[Template Parameter Requirements](../features/types/template_parameters.md). ## Value storage -Values are stored as a tagged union of [value_t](../api/basic_json/value_t.md) and json_value. +Each [`basic_json`](../api/basic_json/index.md) value stores its content as a tagged union: an enumeration [`value_t`](../api/basic_json/value_t.md) +names the type of the value, and a union `json_value` holds the value itself. Both are members of the nested struct +`data`, which is the only data member `m_data` of `basic_json`: ```cpp -/// the type of the current element -value_t m_type = value_t::null; +struct data +{ + /// the type of the current element + value_t m_type = value_t::null; -/// the value of the current element -json_value m_value = {}; + /// the value of the current element + json_value m_value = {}; +}; + +data m_data = {}; ``` with @@ -68,42 +159,83 @@ union json_value { }; ``` -## Parsing inputs (deserialization) +Objects, arrays, strings, and binary values are allocated on the heap with `AllocatorType`, and the union only stores a +pointer to them. This keeps a `basic_json` value small: one pointer-sized union and one byte for the type. The class +maintains the invariant that the pointer matching `m_type` is never null; `assert_invariant()` checks it with +[runtime assertions](../features/assertions.md). -Input is read via **input adapters** that abstract a source with a common interface: +## Input adapters + +Input is read via **input adapters** that abstract a source. Every input adapter provides this interface: ```cpp -/// read a single character -std::char_traits::int_type get_character() noexcept; +/// the type of the characters in the input +using char_type = ...; -/// read multiple characters to a destination buffer and -/// returns the number of characters successfully read +/// read a single character; returns std::char_traits::eof() at the end of the input +typename std::char_traits::int_type get_character(); + +/// read up to count * sizeof(T) bytes into dest and return the number of bytes read +/// (used by the binary readers) template std::size_t get_elements(T* dest, std::size_t count = 1); ``` -List examples of input adapters. +The lexer detects two optional extensions at compile time. Only `iterator_input_adapter` provides them, and only for +random-access input of single-byte characters: -## SAX Interface +- `supports_seek`, `get_consumed_count()`, and `copy_consumed_range()` let the lexer reconstruct already consumed input + for error messages instead of copying every character it reads. +- `supports_bulk_scan`, `bulk_data()`, `bulk_remaining()`, and `bulk_skip()` let the lexer scan strings directly in + contiguous memory, several bytes at a time. -TODO +The function `input_adapter` picks the right adapter for the argument passed to `parse`, `accept`, `sax_parse`, or the +`from_*` functions: -## Writing outputs (serialization) +- `iterator_input_adapter` reads from an iterator range, which also covers strings, containers, and pointers. +- `wide_string_input_adapter` reads from ranges of `wchar_t`, `char16_t`, or `char32_t` and converts them to UTF-8. + It cannot be used for binary formats; its `get_elements()` throws. +- `input_stream_adapter` reads from a `std::istream`. +- `file_input_adapter` reads from a `std::FILE*`. + +## SAX interface + +The parser does not build values itself. It reports what it reads as events to a [SAX](../features/parsing/sax_interface.md) +consumer, which implements the interface [`json_sax`](../api/json_sax/index.md): `null`, `boolean`, `number_integer`, +`number_unsigned`, `number_float`, `string`, `binary`, `start_object`, `key`, `end_object`, `start_array`, `end_array`, +and `parse_error`. + +The library comes with two consumers in `detail/input/json_sax.hpp`: + +- `json_sax_dom_parser` builds a [`basic_json`](../api/basic_json/index.md) value tree. [`parse`](../api/basic_json/parse.md) uses it. +- `json_sax_dom_callback_parser` does the same, but calls a [parser callback](../features/parsing/parser_callbacks.md) + for each event, which can skip values. `parse` uses it when a callback is given. + +The `binary_reader` emits the same events for binary formats, so [`sax_parse`](../api/basic_json/sax_parse.md) works +with a user-defined consumer for JSON and for all binary formats alike. + +## Output adapters Output is written via **output adapters**: ```cpp -template void write_character(CharType c); -template void write_characters(const CharType* s, std::size_t length); ``` -List examples of output adapters. +The `serializer` (used by [`dump`](../api/basic_json/dump.md) and [`operator<<`](../api/operator_ltlt.md)) and the +`binary_writer` (used by the `to_*` functions) write to one of these adapters: + +- `output_vector_adapter` appends to a `std::vector`. +- `output_stream_adapter` writes to a `std::ostream`. +- `output_string_adapter` appends to a string. ## Value conversion +Values are converted from and to other types with the `JSONSerializer` template parameter. The default, +[`adl_serializer`](../api/adl_serializer/index.md), calls the free functions + ```cpp template void to_json(basic_json& j, const T& t); @@ -112,13 +244,23 @@ template void from_json(const basic_json& j, T& t); ``` +found by argument-dependent lookup. The library defines them for standard types in `detail/conversions`; users add them +for their own types, see [Arbitrary Type Conversions](../features/arbitrary_types.md). The +[serialization macros](../features/macros.md) generate these functions. + ## Additional features -- JSON Pointers -- Binary formats -- Custom base class -- Conversion macros +- [JSON Pointer](../features/json_pointer.md) (class `json_pointer`) addresses values inside a tree. It is also the + basis of [JSON Patch](../features/json_patch.md). +- [Binary formats](../features/binary_formats/index.md) are read by `binary_reader` and written by `binary_writer`. +- A [custom base class](../api/basic_json/json_base_class_t.md) can add members to every [`basic_json`](../api/basic_json/index.md) value. +- [Serialization macros](../features/macros.md) generate `to_json` and `from_json` functions for user-defined types. ## Details namespace -- C++ feature backports +Namespace `nlohmann::detail` contains all implementation details. It is not part of the public API and may change in any +release. Besides the components above, it contains: + +- type traits to detect the capabilities of user-defined types (`detail/meta/type_traits.hpp`), +- backports of C++14/17 features to C++11 (`detail/meta/cpp_future.hpp`), and +- helpers such as `string_concat` and `string_escape`. diff --git a/docs/mkdocs/docs/home/customers.md b/docs/mkdocs/docs/home/customers.md index 73ac83ee8..72802ff17 100644 --- a/docs/mkdocs/docs/home/customers.md +++ b/docs/mkdocs/docs/home/customers.md @@ -8,41 +8,72 @@ the result of an internet search. If you know further customers of the library, ## Space Exploration - [**Peregrine Lunar Lander Flight 01**](https://en.wikipedia.org/wiki/Peregrine_Mission_One) - The library was used for payload management in the **Peregrine Moon Lander**, developed by **Astrobotic Technology** and launched as part of NASA's **Commercial Lunar Payload Services (CLPS)** program. After six days in orbit, the spacecraft was intentionally redirected into Earth's atmosphere, where it burned up over the Pacific Ocean on **January 18, 2024**. +- [**NASA Unsteady Pressure-Sensitive Paint Processing**](https://github.com/nasa/upsp-processing), NASA software for processing high-speed video recordings of wind tunnel tests on launch vehicle and aircraft models +- [**Terma TEMU**](https://temu.terma.com/docs/public/temu-release-notes/latest/copying/json-for-modern-cpp.html), an emulator of spacecraft on-board computers used to develop and validate flight software for European space missions ## Automotive - [**Alexa Auto SDK**](https://github.com/alexa/alexa-auto-sdk), a software development kit enabling the integration of Alexa into automotive systems - [**Apollo**](https://github.com/ApolloAuto/apollo), a framework for building autonomous driving systems - [**Automotive Grade Linux (AGL)**](https://download.automotivelinux.org/AGL/release/jellyfish/latest/qemux86-64/deploy/licenses/nlohmann-json/), a collaborative open-source platform for automotive software development +- [**Autoware**](https://github.com/autowarefoundation/autoware_universe), an open-source software stack for autonomous driving built on ROS 2 +- [**Eclipse S-CORE**](https://github.com/eclipse-score/nlohmann_json), an open-source software platform for the software-defined vehicle backed by major automotive manufacturers and suppliers - [**Genesis Motor** (infotainment)](http://webmanual.genesis.com/ccIC/AVNT/JW/KOR/English/reference010.html), a luxury automotive brand - [**Hyundai** (infotainment)](https://www.hyundai.com/wsvc/ww/download.file.do?id=/content/hyundai/ww/data/opensource/data/GN7-2022/licenseCode/info), a global automotive brand - [**Kia** (infotainment)](http://webmanual.kia.com/PREM_GEN6/AVNT/RJPE/KOR/Korean/reference010.html), a global automotive brand - [**Mercedes-Benz Operating System (MB.OS)**](https://group.mercedes-benz.com/careers/about-us/mercedes-benz-operating-system/), a core component of the vehicle software ecosystem from Mercedes-Benz +- [**NVIDIA DRIVE OS**](https://developer.nvidia.com/docs/drive/drive-os/6.0.5/public/driveworks-nvcgf/dwx_open_source_attribution.html), the operating system and DriveWorks SDK powering NVIDIA's platform for autonomous vehicles - [**Rivian** (infotainment)](https://assets.ctfassets.net/2md5qhoeajym/3cwyo4eoufk4yingUwusFt/ded2c47da620fdfc99c88c7156d2c1d8/In-Vehicle_OSS_Attribution_2024__11-24_.pdf), an electric vehicle manufacturer - [**Suzuki** (infotainment)](https://www.globalsuzuki.com/motorcycle/ipc/oss/oss_48KA_00.pdf), a global automotive and motorcycle manufacturer ## Gaming and Entertainment +- [**Anno 117: Pax Romana**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a city-building strategy game set in the Roman Empire - [**Assassin's Creed: Mirage**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a stealth-action game set in the Middle East, focusing on the journey of a young assassin with classic parkour and stealth mechanics +- [**Battlefield 6**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a military first-person shooter known for its large-scale multiplayer battles +- [**Battlefield: REDSEC**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a free-to-play battle royale experience set in the Battlefield universe +- [**BioMenace: Remastered**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a remaster of the classic side-scrolling platform shooter - [**Chasm: The Rift**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a first-person shooter blending horror and adventure, where players navigate dark realms and battle monsters - [**College Football 25**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a college football simulation game featuring gameplay that mimics real-life college teams and competitions +- [**College Football 26**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a college football simulation game featuring licensed teams and stadiums +- [**College Football 27**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), the latest installment of the college football simulation series - [**Concepts**](https://concepts.app/en/licenses), a digital sketching app designed for creative professionals, offering flexible drawing tools for illustration, design, and brainstorming - [**Depthkit**](https://www.depthkit.tv/third-party-licenses), a tool for creating and capturing volumetric video, enabling immersive 3D experiences and interactive content +- [**Dune: Awakening**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), an open-world survival MMO set on the desert planet Arrakis +- [**EA Sports FC 25**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), an association football simulation with club, career, and online modes +- [**EA Sports FC 26**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), the latest installment of the association football simulation series +- [**EA Sports UFC 6**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a mixed martial arts fighting simulation +- [**FiveM**](https://github.com/citizenfx/fivem), a modification framework for Grand Theft Auto V that powers custom multiplayer servers +- [**FLUX:: Immersive**](https://doc.flux.audio/syrah/Credits.html), a suite of professional audio processing and immersive mixing plugins used in music and post-production - [**IMG.LY**](https://img.ly/acknowledgements), a platform offering creative tools and SDKs for integrating advanced image and video editing in applications +- [**immersivetech**](https://immersitech.io/open-source-third-party-software/), a technology company focused on immersive experiences, providing tools and solutions for virtual and augmented reality applications +- [**Kodi**](https://github.com/xbmc/xbmc/blob/master/xbmc/utils/JSONVariantWriter.cpp), a home theater and media center application - [**LOOT**](https://loot.readthedocs.io/_/downloads/en/0.13.0/pdf/), a tool for optimizing the load order of game plugins, commonly used in The Elder Scrolls and Fallout series +- [**LunaTranslator**](https://github.com/HIllya51/LunaTranslator/blob/main/src/NativeImpl/LunaSubprocess/aspatch.cpp), a real-time translation tool for visual novels +- [**MaaAssistantArknights**](https://github.com/MaaAssistantArknights/MaaAssistantArknights/blob/dev-v2/src/MaaCore/Vision/Roguelike/BlackFlow/BlackFlowMapAnalyzer.cpp), an automation assistant for the mobile game Arknights - [**Madden NFL 25**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a sports simulation game capturing the excitement of American football with realistic gameplay and team management features +- [**Madden NFL 26**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), an American football simulation with franchise and team management modes +- [**Madden NFL 27**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), the latest installment of the American football simulation series - [**Marne**](https://marne.io/licenses), an unofficial private server platform for hosting custom Battlefield 1 game experiences - [**Minecraft**](https://www.minecraft.net/zh-hant/attribution), a popular sandbox video game +- [**Mumble**](https://github.com/mumble-voip/mumble), a low-latency, open-source voice chat application widely used by gaming communities - [**NHL 22**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a hockey simulation game offering realistic gameplay, team management, and various modes to enhance the hockey experience +- [**OBS Studio**](https://github.com/obsproject/obs-studio), a free and open-source suite for video recording and live streaming +- [**OpenRCT2**](https://github.com/OpenRCT2/OpenRCT2/blob/develop/src/openrct2/core/JsonFwd.hpp), an open source re-implementation of RollerCoaster Tycoon 2 - [**Pixelpart**](https://pixelpart.net/documentation/book/third-party.html), a 2D animation and video compositing software that allows users to create animated graphics and visual effects with a focus on simplicity and ease of use - [**Razer Cortex**](https://mysupport.razer.com/app/answers/detail/a_id/14146/~/open-source-software-for-razer-software), a gaming performance optimizer and system booster designed to enhance the gaming experience - [**Red Dead Redemption II**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), an open-world action-adventure game following an outlaw's story in the late 1800s, emphasizing deep storytelling and immersive gameplay +- [**RetroArch**](https://github.com/libretro/RetroArch), a frontend for emulators, game engines, and media players built on the libretro API +- [**shadPS4**](https://github.com/shadps4-emu/shadPS4/blob/main/src/core/user_manager.h), a PlayStation 4 emulator for Windows, Linux and macOS +- [**skate.**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a free-to-play skateboarding game set in an open world - [**Snapchat**](https://www.snap.com/terms/license-android), a multimedia messaging and augmented reality app for communication and entertainment +- [**Steel Century Groove**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), an action game released in 2026 +- [**Sunshine**](https://github.com/LizardByte/Sunshine/blob/master/src/confighttp.h), a self-hosted game streaming host compatible with Moonlight clients - [**Tactics Ogre: Reborn**](https://www.square-enix-games.com/en_US/documents/tactics-ogre-reborn-pc-installer-software-and-associated-plug-ins-disclosure), a tactical role-playing game featuring strategic battles and deep storytelling elements - [**Throne and Liberty**](https://www.amazon.com/gp/help/customer/display.html?nodeId=T7fLNw5oAevCMtJFPj&pop-up=1), an MMORPG that offers an expansive fantasy world with dynamic gameplay and immersive storytelling - [**Unity Vivox**](https://docs.unity3d.com/Packages/com.unity.services.vivox@15.1/license/Third%20Party%20Notices.html), a communication service that enables voice and text chat functionality in multiplayer games developed with Unity +- [**xemu**](https://github.com/xemu-project/xemu), an emulator of the original Xbox console - [**Zool: Redimensioned**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a modern reimagining of the classic platformer featuring fast-paced gameplay and vibrant environments -- [**immersivetech**](https://immersitech.io/open-source-third-party-software/), a technology company focused on immersive experiences, providing tools and solutions for virtual and augmented reality applications ## Consumer Electronics @@ -50,109 +81,195 @@ the result of an internet search. If you know further customers of the library, - [**Canon CanoScan LIDE**](https://carolburo.com/wp-content/uploads/2024/06/LiDE400_OnlineManual_Win_FR_V02.pdf), a series of flatbed scanners offering high-resolution image scanning for home and office use - [**Canon PIXMA Printers**](https://www.mediaexpert.pl/products/files/73/7338196/Instrukcja-obslugi-CANON-Pixma-TS7450i.pdf), a line of all-in-one inkjet printers known for high-quality printing and wireless connectivity - [**Cisco Webex Desk Camera**](https://www.cisco.com/c/dam/en_us/about/doing_business/open_source/docs/CiscoWebexDeskCamera-23-1622100417.pdf), a video camera designed for professional-quality video conferencing and remote collaboration +- [**DJI Edge SDK**](https://github.com/dji-sdk/Edge-SDK-V2-Demo), the reference applications for DJI's Edge SDK, used to build edge computing services on DJI drone docks +- [**Elgato Stream Deck**](https://github.com/elgatosf/streamdeck-obs-plugin2), a family of programmable control surfaces for content creators and their plugin ecosystem +- [**Instagrid**](https://instagrid.co/intellectual-property/foss), a manufacturer of portable, high-performance battery systems for professional mobile power supply +- [**iRobot**](https://iot-content.irobot.com/iw/sfsites/c/cms/delivery/media/MCKRLTPDJSSJBNJKDA5SG5UVVIIQ), a manufacturer of autonomous home robots including the Roomba vacuum cleaner range +- [**Logitech Logi Bolt**](https://opensource.logitech.com/wiki/Logi_BoltApp/), the management application for Logitech's secure wireless connectivity technology +- [**Novitus**](https://novitus.pl/licencjepensource), a manufacturer of fiscal cash registers and point-of-sale devices - [**Philips Hue Personal Wireless Lighting**](http://2ak5ape.257.cz/), a smart lighting system for customizable and wireless home illumination - [**Ray-Ban Meta Smart glasses**](https://www.meta.com/de/en/legal/smart-glasses/third-party-notices-android/03/), a pair of smart glasses designed for capturing photos and videos with integrated connectivity and social features - [**Razer Synapse**](https://mysupport.razer.com/app/answers/detail/a_id/14146/~/open-source-software-for-razer-software), a unified configuration software enabling hardware customization for Razer devices +- [**Sharp Professional Displays**](https://jp.sharp/restricted/business/lcd-display/cms/images/source_pnla862/PN-LA652_752_862_LicenseInformation.pdf), a range of large-format interactive displays for business and education - [**Siemens SINEMA Remote Connect**](https://cache.industry.siemens.com/dl/files/790/109793790/att_1054961/v2/OSS_SINEMA-RC_86.pdf), a remote connectivity solution for monitoring and managing industrial networks and devices securely +- [**Skydio**](https://pages.skydio.com/rs/784-TUF-591/images/Open%20Source%20Software%20Notice%20v0.2.html), a manufacturer of autonomous drones for inspection, public safety, and defense applications - [**Sony PlayStation 4**](https://doc.dl.playstation.net/doc/ps4-oss/index.html), a gaming console developed by Sony that offers a wide range of games and multimedia entertainment features +- [**Sony Spatial Reality Display**](https://www.sony.co.jp/en/Products/Developer-Spatial-Reality-display/download/dcc-tools/blender-plugin/SpatiaRealityDisplayPluginforPreviewBL_Manual.pdf), a glasses-free stereoscopic 3D display and its plugins for Blender, 3ds Max, and ZBrush - [**Sony Virtual Webcam Driver for Remote Camera**](https://helpguide.sony.net/rc/vwd/v1/zh-cn/print.pdf), a software driver that enables the use of Sony cameras as virtual webcams for video conferencing and streaming +- [**Yamaha Clavinova**](https://usa.yamaha.com/files/download/other_assets/1/2298171/CLP-800_oss_license.pdf), a series of digital pianos combining acoustic piano feel with digital sound technology -## Operating Systems +## Operating Systems and Platforms - [**Apple iOS and macOS**](https://www.apple.com/macos), a family of operating systems developed by Apple, including iOS for mobile devices and macOS for desktop computers +- [**Chromium**](https://chromium.googlesource.com/chromium/src/+/main/third_party/nlohmann_json/), the open-source browser project that Google Chrome, Microsoft Edge, and many other browsers are built on, where the library is used as data container for on-device model execution - [**Google Fuchsia**](https://fuchsia.googlesource.com/third_party/json/), an open-source operating system developed by Google, designed to be secure, updatable, and adaptable across various devices +- [**LG webOS**](https://github.com/webosose/com.webos.service.camera), a Linux-based operating system used in LG smart TVs, signage, and embedded devices +- [**Microsoft Azure Linux**](https://github.com/microsoft/azurelinux), a Linux distribution developed by Microsoft for Azure infrastructure and edge workloads +- [**OpenHarmony**](https://github.com/openharmony/third_party_json), an open-source operating system for smart devices and the foundation of HarmonyOS - [**SerenityOS**](https://github.com/SerenityOS/serenity), an open-source operating system that aims to provide a simple and beautiful user experience with a focus on simplicity and elegance +- [**Windows Subsystem for Linux**](https://github.com/microsoft/WSL), a compatibility layer that runs Linux environments natively on Windows - [**Yocto**](http://ftp.emacinc.com/openembedded-sw/kirkstone-icop-5.15-kirkstone-6.0/archive-2024-10/pn8m-090t-ppc/licenses/nlohmann-json/), a Linux-based build system for creating custom operating systems and software distributions, tailored for embedded devices and IoT applications ## Development Tools and IDEs - [**Accentize SpectralBalance**](https://www.accentize.com/products/SpectralBalanceManual.pdf), an adaptive speech analysis tool designed to enhance audio quality by optimizing frequency balance in recordings +- [**Airbus Ghidralligator**](https://www.cyber.airbus.com/en/newsroom/stories/2025-06-ghidralligator), a Ghidra-based emulator from Airbus CyberSecurity used to fuzz and analyse embedded firmware +- [**Apache brpc**](https://github.com/apache/brpc/blob/master/src/butil/iobuf.h), an industrial-grade remote procedure call framework for C++ - [**Arm Compiler for Linux**](https://documentation-service.arm.com/static/66558e9d876c8d213b7843e4), a software development toolchain for compiling and optimizing applications on Arm-based Linux systems - [**BBEdit**](https://s3.amazonaws.com/BBSW-download/BBEdit_15.1.2_User_Manual.pdf), a professional text and code editor for macOS - [**CoderPad**](https://coderpad.io), a collaborative coding platform that enables real-time code interviews and assessments for developers; the library is included in every CoderPad instance and can be accessed with a simple `#include "json.hpp"` +- [**Codon**](https://github.com/exaloop/codon/blob/develop/jupyter/jupyter.h), an ahead-of-time compiler for a Python-like language - [**Compiler Explorer**](https://godbolt.org), a web-based tool that allows users to write, compile, and visualize the assembly output of code in various programming languages; the library is readily available and accessible with the directive `#include `. -- [**GitHub CodeQL**](https://github.com/github/codeql), a code analysis tool used for identifying security vulnerabilities and bugs in software through semantic queries +- [**Flutter**](https://github.com/flutter/flutter/blob/master/engine/src/flutter/impeller/compiler/reflector.cc), a UI toolkit for building natively compiled applications for mobile, web, and desktop from a single codebase +- [**Fraunhofer VVenC**](https://github.com/fraunhoferhhi/vvenc), a fast and efficient encoder for the Versatile Video Coding (H.266/VVC) standard +- [**GitHub CodeQL**](https://github.com/github/codeql/blob/main/shared/cpp/Diagnostics.h), a code analysis tool used for identifying security vulnerabilities and bugs in software through semantic queries +- [**GoPro ngfx**](https://github.com/gopro/ngfx), a low-level graphics abstraction and profiling framework developed by GoPro +- [**gRPC**](https://github.com/grpc/grpc/blob/master/tools/artifact_gen/utils.h), a high-performance universal remote procedure call framework - [**Hex-Rays**](https://docs.hex-rays.com/user-guide/user-interface/licenses), a reverse engineering toolset for analyzing and decompiling binaries, primarily used for security research and vulnerability analysis - [**ImHex**](https://github.com/WerWolv/ImHex), a hex editor designed for reverse engineering, providing advanced features for data analysis and manipulation +- [**Intel GITS**](https://github.com/intel/gits), a tool for capturing and replaying graphics API calls for debugging and performance analysis - [**Intel GPA Framework**](https://intel.github.io/gpasdk-doc/src/licenses.html), a suite of cross-platform tools for capturing, analyzing, and optimizing graphics applications across different APIs - [**Intopix**](https://www.intopix.com/software-licensing), a provider of advanced image processing and compression solutions used in software development and AV workflows - [**Java SE**](https://www.oracle.com/a/tech/docs/jdk8-lium.pdf), the core Java platform that provides the libraries and runtime needed to build and run general-purpose Java applications -- [**MKVToolNix**](https://mkvtoolnix.download/doc/README.md), a set of tools for creating, editing, and inspecting MKV (Matroska) multimedia container files - [**Meta Yoga**](https://github.com/facebook/yoga), a layout engine that facilitates flexible and efficient user interface design across multiple platforms -- [**NVIDIA Nsight Compute**](https://docs.nvidia.com/nsight-compute/2022.2/pdf/CopyrightAndLicenses.pdf), a performance analysis tool for CUDA applications that provides detailed insights into GPU performance metrics +- [**MKVToolNix**](https://mkvtoolnix.download/doc/README.md), a set of tools for creating, editing, and inspecting MKV (Matroska) multimedia container files +- [**MRTech IFF SDK**](https://mr-technologies.com/pub/iff-sdk-manual-2-0-1/iff-sdk-manual-2-0-1.pdf), an image processing SDK for machine vision applications with GPU-accelerated pipelines +- [**Nix**](https://github.com/NixOS/nix/blob/master/src/nix/build.cc), a purely functional package manager - [**Notepad++**](https://github.com/notepad-plus-plus/notepad-plus-plus), a free source code editor that supports various programming languages +- [**NVIDIA Nsight Compute**](https://docs.nvidia.com/nsight-compute/2022.2/pdf/CopyrightAndLicenses.pdf), a performance analysis tool for CUDA applications that provides detailed insights into GPU performance metrics +- [**openFrameworks**](https://github.com/openframeworks/openFrameworks/blob/master/libs/openFrameworks/utils/ofJson.h), a community-developed C++ toolkit for creative coding - [**OpenRGB**](https://gitlab.com/CalcProgrammer1/OpenRGB), an open source RGB lighting control that doesn't depend on manufacturer software - [**OpenTelemetry C++**](https://github.com/open-telemetry/opentelemetry-cpp), a library for collecting and exporting observability data in C++, enabling developers to implement distributed tracing and metrics in their application +- [**Oracle GraalVM**](https://docs.oracle.com/en/graalvm/jdk/21/docs/licensing-information/), a high-performance JDK distribution with ahead-of-time compilation and polyglot runtime support +- [**Philips amp-cucumber-cpp-runner**](https://github.com/philips-software/amp-cucumber-cpp-runner), a behaviour-driven development test runner for embedded C++ software developed at Philips - [**Qt Creator**](https://doc.qt.io/qtcreator/qtcreator-attribution-json-nlohmann.html), an IDE for developing applications using the Qt application framework +- [**Qt for MCUs**](https://doc.qt.io/QtForMCUs/quickultralite-attribution-nlohmann-json.html), a graphics framework for building fluid user interfaces on microcontrollers +- [**React Native**](https://github.com/react/react-native/blob/main/packages/react-native/ReactCxxPlatform/react/devsupport/PackagerConnection.cpp), a framework for building native mobile applications using React - [**Scanbot SDK**](https://docs.scanbot.io/barcode-scanner-sdk/web/third-party-libraries/), a software development kit (SDK) that provides tools for integrating advanced document scanning and barcode scanning capabilities into applications +- [**STMicroelectronics TouchGFX**](https://www.st.com/resource/en/additional_license_terms/additional-license-terms-x-cube-touchgfx.html), a graphical user interface framework shipped with STM32 microcontrollers for building embedded HMIs +- [**swagger-codegen**](https://github.com/swagger-api/swagger-codegen/blob/master/samples/server/petstore/pistache-server/model/Pet.h), a template-driven engine that generates API clients and server stubs from an OpenAPI specification +- [**Swoole**](https://github.com/swoole/swoole-src/blob/master/ext-src/swoole_admin_server.cc), a coroutine-based concurrency engine for PHP +- [**Tracy Profiler**](https://github.com/wolfpld/tracy/blob/master/profiler/src/profiler/TracyLlm.hpp), a real-time frame profiler for games and other applications +- [**WasmEdge**](https://github.com/WasmEdge/WasmEdge/blob/master/plugins/wasi_nn/GGML/tts/tts_core.cpp), a lightweight WebAssembly runtime for edge and cloud workloads +- [**x64dbg**](https://github.com/x64dbg/x64dbg/blob/development/src/cross/remote_table/TableRpcData.h), an open source user mode debugger for Windows, aimed at reverse engineering and malware analysis ## Machine Learning and AI +- [**Alibaba MNN**](https://github.com/alibaba/MNN), a lightweight deep learning inference engine for mobile and embedded devices +- [**AMD Gaia**](https://github.com/amd/gaia), an open-source framework for running generative AI applications locally on AMD hardware +- [**AMD Vitis AI (VAIP)**](https://github.com/amd/vaip), the execution provider stack that runs AI models on AMD Ryzen AI and adaptive computing devices - [**Apple Core ML Tools**](https://github.com/apple/coremltools), a set of tools for converting and configuring machine learning models for deployment in Apple's Core ML framework - [**Avular Mobile Robotics**](https://www.avular.com/licenses/nlohmann-json-3.9.1.txt), a platform for developing and deploying mobile robotics solutions +- [**FunASR**](https://github.com/modelscope/FunASR/blob/main/runtime/http/bin/asr_sessions.h), a speech recognition toolkit for training and deploying end-to-end models - [**Google gemma.cpp**](https://github.com/google/gemma.cpp), a lightweight C++ inference engine designed for running AI models from the Gemma family +- [**Google Magenta The Infinite Crate**](https://github.com/magenta/the-infinite-crate), an open-source generative AI plugin for digital audio workstations from Google's Magenta research team +- [**GPT4All**](https://github.com/nomic-ai/gpt4all/blob/main/gpt4all-chat/src/tool.h), a desktop application for running local large language models on consumer hardware +- [**Huawei MindSpore**](https://github.com/mindspore-ai/mindspore/blob/master/Third_Party_Open_Source_Software_Notice), a deep learning framework for training and inference across device, edge, and cloud +- [**KTransformers**](https://github.com/kvcache-ai/ktransformers/blob/main/archive/csrc/balance_serve/sched/model_config.h), a framework for heterogeneous large language model inference - [**llama.cpp**](https://github.com/ggerganov/llama.cpp), a C++ library designed for efficient inference of large language models (LLMs), enabling streamlined integration into applications +- [**LocalAI**](https://github.com/mudler/LocalAI/blob/master/backend/cpp/ds4/dsml_renderer.cpp), a self-hosted inference engine that exposes local models through an OpenAI-compatible API - [**MLX**](https://github.com/ml-explore/mlx), an array framework for machine learning on Apple Silicon - [**Mozilla llamafile**](https://github.com/Mozilla-Ocho/llamafile), a tool designed for distributing and executing large language models (LLMs) efficiently using a single file format - [**NVIDIA ACE**](https://docs.nvidia.com/ace/latest/index.html), a suite of real-time AI solutions designed for the development of interactive avatars and digital human applications, enabling scalable and sophisticated user interactions +- [**NVIDIA Instant NGP**](https://github.com/NVlabs/instant-ngp/blob/master/src/nerf_loader.cu), an implementation of instant neural graphics primitives for rapid scene reconstruction +- [**NVIDIA TensorRT**](https://github.com/NVIDIA/TensorRT), an SDK for high-performance deep learning inference, including its TensorRT-LLM extension for large language models +- [**NVIDIA TensorRT-LLM**](https://github.com/NVIDIA/TensorRT-LLM/blob/main/cpp/tensorrt_llm/common/safetensors.cpp), a toolkit for optimizing and serving large language model inference on GPUs +- [**ONNX Runtime**](https://github.com/microsoft/onnxruntime), a cross-platform inference and training accelerator for machine learning models +- [**OpenVINO**](https://github.com/openvinotoolkit/openvino), Intel's toolkit for optimizing and deploying deep learning inference across CPUs, GPUs, and NPUs +- [**PaddleOCR**](https://github.com/PaddlePaddle/PaddleOCR/blob/main/deploy/cpp_infer/src/modules/text_detection/result.cc), an optical character recognition toolkit that turns documents and images into structured data +- [**PaddlePaddle**](https://github.com/PaddlePaddle/Paddle/blob/develop/paddle/ap/src/axpr/anf_expr.cc), a deep learning framework for distributed training and inference - [**Peer**](https://support.peer.inc/hc/en-us/articles/17261335054235-Licenses), a platform offering personalized AI assistants for interactive learning and creative collaboration +- [**PyTorch**](https://github.com/pytorch/pytorch), a machine learning framework for building and training neural networks, widely used in research and production +- [**Qualcomm AI Engine Direct**](https://github.com/qualcomm/qai-appbuilder), a toolchain for building and running generative AI applications on Snapdragon devices +- [**sherpa-onnx**](https://github.com/k2-fsa/sherpa-onnx/blob/master/sherpa-onnx/csrc/sentence-piece-tokenizer.cc), a speech toolkit for on-device recognition, synthesis and speaker diarization - [**stable-diffusion.cpp**](https://github.com/leejet/stable-diffusion.cpp), a C++ implementation of the Stable Diffusion image generation model - [**TanvasTouch**](https://tanvas.co/tanvastouch-sdk-third-party-acknowledgments), a software development kit (SDK) that enables developers to create tactile experiences on touchscreens, allowing users to feel textures and physical sensations in a digital environment - [**TensorFlow**](https://github.com/tensorflow/tensorflow), a machine learning framework that facilitates the development and training of models, supporting data serialization and efficient data exchange between components +- [**whisper.cpp**](https://github.com/ggml-org/whisper.cpp), a C++ implementation of OpenAI's Whisper automatic speech recognition model ## Scientific Research and Analysis - [**BLACK**](https://www.black-sat.org/en/stable/installation/linux.html), a bounded linear temporal logic (LTL) satisfiability checker +- [**CERN ALICE O2**](https://github.com/AliceO2Group/AliceO2), the online-offline computing framework of the ALICE heavy-ion experiment at the Large Hadron Collider - [**CERN Atlas Athena**](https://gitlab.cern.ch/atlas/athena/-/blob/main/Control/PerformanceMonitoring/PerfMonComps/src/PerfMonMTSvc.h), a software framework used in the ATLAS experiment at the Large Hadron Collider (LHC) for performance monitoring +- [**CERN CMSSW**](https://github.com/cms-sw/cmssw), the offline software framework of the CMS experiment at the Large Hadron Collider +- [**CERN Gaudi**](https://gitlab.cern.ch/gaudi/Gaudi), the event-processing framework used by the LHCb and ATLAS experiments at the Large Hadron Collider - [**ICU**](https://github.com/unicode-org/icu), the International Components for Unicode, a mature library for software globalization and multilingual support - [**KAMERA**](https://github.com/Kitware/kamera), a platform for synchronized data collection and real-time deep learning to map marine species like polar bears and seals, aiding Arctic ecosystem research - [**KiCad**](https://gitlab.com/kicad/code/kicad/-/tree/master/thirdparty/nlohmann_json), a free and open-source software suite for electronic design automation +- [**LLNL ROSE**](https://github.com/llnl/rose), a compiler infrastructure from Lawrence Livermore National Laboratory for building source-to-source program analysis and transformation tools - [**Maple**](https://www.maplesoft.com/support/help/Maple/view.aspx?path=copyright), a symbolic and numeric computing environment for advanced mathematical modeling and analysis - [**MeVisLab**](https://mevislabdownloads.mevis.de/docs/current/MeVis/ThirdParty/Documentation/Publish/ThirdPartyReference/index.html), a software framework for medical image processing and visualization. +- [**MITK**](https://github.com/MITK/MITK), the Medical Imaging Interaction Toolkit, a framework for developing interactive medical image processing software - [**OpenPMD API**](https://openpmd-api.readthedocs.io/en/0.8.0-alpha/backends/json.html), a versatile programming interface for accessing and managing scientific data, designed to facilitate the efficient storage, retrieval, and sharing of simulation data across various applications and platforms +- [**ORNL DataFed**](https://github.com/ORNL/DataFed), a federated scientific data management system developed at Oak Ridge National Laboratory - [**ParaView**](https://github.com/Kitware/ParaView), an open-source tool for large-scale data visualization and analysis across various scientific domains - [**QGIS**](https://gitlab.b-data.ch/qgis/qgis/-/blob/backport-57658-to-release-3_34/external/nlohmann/json.hpp), a free and open-source geographic information system (GIS) application that allows users to create, edit, visualize, and analyze geospatial data across a variety of formats -- [**VTK**](https://github.com/Kitware/VTK), a software library for 3D computer graphics, image processing, and visualization +- [**Sandia InterSpec**](https://github.com/sandialabs/InterSpec), spectral radiation analysis software from Sandia National Laboratories for identifying radioactive isotopes - [**VolView**](https://github.com/Kitware/VolView), a lightweight application for interactive visualization and analysis of 3D medical imaging data. +- [**VTK**](https://github.com/Kitware/VTK), a software library for 3D computer graphics, image processing, and visualization ## Business and Productivity Software - [**ArcGIS PRO**](https://www.esri.com/content/dam/esrisites/en-us/media/legal/open-source-acknowledgements/arcgis-pro-2-8-attribution-report.html), a desktop geographic information system (GIS) application developed by Esri for mapping and spatial analysis - [**Autodesk Desktop**](https://damassets.autodesk.net/content/dam/autodesk/www/Company/legal-notices-trademarks/autodesk-desktop-platform-components/internal-autodesk-components-web-page-2023.pdf), a software platform developed by Autodesk for creating and managing desktop applications and services - [**Check Point**](https://www.checkpoint.com/about-us/copyright-and-trademarks/), a cybersecurity company specializing in threat prevention and network security solutions, offering a range of products designed to protect enterprises from cyber threats and ensure data integrity +- [**EasyEffects**](https://github.com/wwmm/easyeffects/blob/master/src/presets_manager.hpp), an audio effects processor for PipeWire offering limiting, compression and equalization +- [**espanso**](https://github.com/espanso/espanso/blob/dev/espanso-ui/src/win32/native.cpp), a cross-platform text expander +- [**Karabiner-Elements**](https://github.com/pqrs-org/Karabiner-Elements/blob/main/src/share/app_icon.hpp), a keyboard customizer for macOS +- [**MacType**](https://github.com/snowie2000/mactype/blob/directwrite/settings.h), a font rendering engine for Windows +- [**magicplan**](https://help.magicplan.app/acknowledgments), a mobile application for creating floor plans and interior designs using augmented reality - [**Microsoft Office for Mac**](https://officecdnmac.microsoft.com/pr/legal/mac/OfficeforMacAttributions.html), a suite of productivity applications developed by Microsoft for macOS, including tools for word processing, spreadsheets, and presentations - [**Microsoft Teams**](https://www.microsoft.com/microsoft-teams/), a team collaboration application offering workspace chat and video conferencing, file storage, and integration of proprietary and third-party applications and services +- [**MuseScore**](https://github.com/musescore/MuseScore), a free and open-source music notation and composition application +- [**NanaZip**](https://github.com/M2Team/NanaZip/blob/main/NanaZip.Codecs/NanaZip.Codecs.Archive.ElectronAsar.cpp), a 7-Zip derivative built for modern Windows - [**Nexthink Infinity**](https://docs.nexthink.com/legal/services-terms/experience-open-source-software-licenses/infinity-2022.8-software-licenses), a digital employee experience management platform for monitoring and improving IT performance - [**Sophos Connect Client**](https://docs.sophos.com/nsg/licenses/SophosConnect/SophosConnectAttribution.html), a secure VPN client from Sophos that allows remote users to connect to their corporate network, ensuring secure access to resources and data - [**Stonebranch**](https://stonebranchdocs.atlassian.net/wiki/spaces/UA77/pages/799545647/Licenses+for+Third-Party+Libraries), a cloud-based cybersecurity solution that integrates backup, disaster recovery, and cybersecurity features to protect data and ensure business continuity for organizations - [**Tablecruncher**](https://tablecruncher.com/), a data analysis tool that allows users to import, analyze, and visualize spreadsheet data, offering interactive features for better insights and decision-making -- [**magicplan**](https://help.magicplan.app/acknowledgments), a mobile application for creating floor plans and interior designs using augmented reality +- [**VNote**](https://github.com/vnotex/vnote/blob/master/src/core/services/notebookcoreservice.cpp), a Markdown-based note-taking application written in C++ ## Databases and Big Data - [**ADIOS2**](https://code.ornl.gov/ecpcitest/adios2/-/tree/pr4285_FFSUpstream/thirdparty/nlohmann_json?ref_type=heads), a data management framework designed for high-performance input and output operations +- [**Apache Doris**](https://github.com/apache/doris/blob/master/be/src/runtime/be_proc_monitor.cpp), a real-time analytical database for high-concurrency queries +- [**Claris FileMaker Server**](https://www.claris.com/company/legal/docs/acknowledgements/filemaker-server-macwin/claris_fms2025_acknowledgements_en.pdf), the server platform hosting FileMaker custom apps and databases, developed by Apple subsidiary Claris +- [**ClickHouse**](https://github.com/ClickHouse/ClickHouse), a column-oriented database management system for real-time analytical queries - [**Cribl Stream**](https://docs.cribl.io/stream/third-party-current-list/), a real-time data processing platform that enables organizations to collect, route, and transform observability data, enhancing visibility and insights into their systems - [**DB Browser for SQLite**](https://github.com/sqlitebrowser/sqlitebrowser), a visual open-source tool for creating, designing, and editing SQLite database files +- [**Manticore Search**](https://github.com/manticoresoftware/manticoresearch/blob/main/src/searchdhttpcompat.cpp), a database for search, offering full-text and vector queries +- [**Milvus**](https://github.com/milvus-io/milvus/blob/master/internal/core/src/query/PlanImpl.h), a cloud-native vector database built for embedding similarity search +- [**MongoDB**](https://github.com/mongodb/mongo/blob/master/src/mongo/replay/config_handler.cpp), a general-purpose document database - [**MySQL Connector/C++**](https://docs.oracle.com/cd/E17952_01/connector-cpp-9.1-license-com-en/license-opentelemetry-cpp-com.html), a C++ library for connecting and interacting with MySQL databases - [**MySQL NDB Cluster**](https://downloads.mysql.com/docs/licenses/cluster-9.0-com-en.pdf), a distributed database system that provides high availability and scalability for MySQL databases - [**MySQL Shell**](https://downloads.mysql.com/docs/licenses/mysql-shell-8.0-gpl-en.pdf), an advanced client and code editor for interacting with MySQL servers, supporting SQL, Python, and JavaScript -- [**PrestoDB**](https://github.com/prestodb/presto), a distributed SQL query engine designed for large-scale data analytics, originally developed by Facebook +- [**PrestoDB**](https://github.com/prestodb/presto/blob/master/presto-native-execution/presto_cpp/main/Announcer.cpp), a distributed SQL query engine designed for large-scale data analytics, originally developed by Facebook - [**ROOT Data Analysis Framework**](https://root.cern/doc/v614/classnlohmann_1_1basic__json.html), an open-source data analysis framework widely used in high-energy physics and other fields for data processing and visualization +- [**Typesense**](https://github.com/typesense/typesense/blob/v31/include/join.h), an open source typo-tolerant search engine +- [**Vearch**](https://github.com/jd-opensource/vearch), a distributed vector database developed at JD.com for similarity search and retrieval-augmented generation - [**WiredTiger**](https://github.com/wiredtiger/wiredtiger), a high-performance storage engine for databases, offering support for compression, concurrency, and checkpointing ## Simulation and Modeling +- [**Adobe Lagrange**](https://github.com/adobe/lagrange), a geometry processing library developed by Adobe for mesh manipulation and analysis - [**Arcturus HoloSuite**](https://www.datocms-assets.com/104353/1698904597-holosuite-third-party-software-credits-and-attributions-2.pdf), a software toolset for capturing, editing, and streaming volumetric video, featuring advanced compression technologies for high-quality 3D content creation - [**azul**](https://pure.tudelft.nl/ws/files/85338589/tgis.12673.pdf), a fast and efficient 3D city model viewer designed for visualizing urban environments and spatial data +- [**Bambu Studio**](https://github.com/bambulab/BambuStudio), a slicing and print management application for Bambu Lab 3D printers - [**Blender**](https://projects.blender.org/blender/blender/search?q=nlohmann), a free and open-source 3D creation suite for modeling, animation, rendering, and more - [**cpplot**](https://cpplot.readthedocs.io/en/latest/library_api/function_eigen_8h_1ac080eac0541014c5892a55e41bf785e6.html), a library for creating interactive graphs and charts in C++, which can be viewed in web browsers - [**Foundry Nuke**](https://learn.foundry.com/nuke/content/misc/studio_third_party_libraries.html), a powerful node-based digital compositing and visual effects application used in film and television post-production +- [**FreeCAD**](https://github.com/FreeCAD/FreeCAD), a free and open-source parametric 3D CAD modeler for product design and engineering - [**GAMS**](https://www.gams.com/47/docs/THIRDPARTY.html), a high-performance mathematical modeling system for optimization and decision support +- [**Keysight WirelessPro**](https://docs.keysight.com/display/engdocwirelesspro/WirelessPro+2026+Release+Notes), a simulation platform for 5G, 5G-Advanced, and 6G cellular network research - [**Kitware SMTK**](https://github.com/Kitware/SMTK), a software toolkit for managing simulation models and workflows in scientific and engineering applications - [**M-Star**](https://docs.mstarcfd.com/3_Licensing/thirdparty-licenses.html), a computational fluid dynamics software for simulating and analyzing fluid flow - [**MapleSim CAD Toolbox**](https://www.maplesoft.com/support/help/MapleSim/view.aspx?path=CADToolbox/copyright), a software extension for MapleSim that integrates CAD models, allowing users to import, manipulate, and analyze 3D CAD data within the MapleSim environment for enhanced modeling and simulation +- [**Microsoft AirSim**](https://github.com/microsoft/AirSim/blob/main/AirLib/include/common/Settings.hpp), a simulator for autonomous vehicles and drones built on Unreal Engine - [**NVIDIA Omniverse**](https://docs.omniverse.nvidia.com/composer/latest/common/product-licenses/usd-explorer/usd-explorer-2023.2.0-licenses-manifest.html), a platform for 3D content creation and collaboration that enables real-time simulations and interactive experiences across various industries +- [**OpenSCAD**](https://github.com/openscad/openscad/blob/master/src/core/AIClient.cc), a script-driven solid 3D CAD modeller +- [**OrcaSlicer**](https://github.com/SoftFever/OrcaSlicer), an open-source slicer supporting a wide range of consumer 3D printers - [**Pixar Renderman**](https://rmanwiki-26.pixar.com/space/REN26/19662083/Legal+Notice), a photorealistic 3D rendering software developed by Pixar, widely used in the film industry for creating high-quality visual effects and animations +- [**PrusaSlicer**](https://github.com/prusa3d/PrusaSlicer), the slicing software developed by Prusa Research for its 3D printers - [**ROS - Robot Operating System**](http://docs.ros.org/en/noetic/api/behaviortree_cpp/html/json_8hpp_source.html), a set of software libraries and tools that assist in developing robot applications - [**UBS**](https://www.ubs.com/), a multinational financial services and banking company @@ -161,18 +278,31 @@ the result of an internet search. If you know further customers of the library, - [**Acronis Cyber Protect Cloud**](https://care.acronis.com/s/article/59533-Third-party-software-used-in-Acronis-Cyber-Protect-Cloud?language=en_US), an all-in-one data protection solution that combines backup, disaster recovery, and cybersecurity to safeguard business data from threats like ransomware - [**Baereos**](https://gitlab.tiger-computing.co.uk/packages/bareos/-/blob/tiger/bullseye/third-party/CLI11/examples/json.cpp), a backup solution that provides data protection and recovery options for various environments, including physical and virtual systems - [**Bitdefender Home Scanner**](https://www.bitdefender.de/site/Main/view/home-scanner-open-source.html), a tool from Bitdefender that scans devices for malware and security threats, providing a safeguard against potential online dangers +- [**Cisco MLS++**](https://github.com/cisco/mlspp), an implementation of the Messaging Layer Security protocol for end-to-end encrypted group messaging - [**Citrix Provisioning**](https://docs.citrix.com/en-us/provisioning/2203-ltsr/downloads/pvs-third-party-notices-2203.pdf), a solution that streamlines the delivery of virtual desktops and applications by allowing administrators to manage and provision resources efficiently across multiple environments - [**Citrix Virtual Apps and Desktops**](https://docs.citrix.com/en-us/citrix-virtual-apps-desktops/2305/downloads/third-party-notices-apps-and-desktops.pdf), a solution from Citrix that delivers virtual apps and desktops - [**Cyberarc**](https://docs.cyberark.com/Downloads/Legal/Privileged%20Session%20Manager%20for%20SSH%20Third-Party%20Notices.pdf), a security solution that specializes in privileged access management, enabling organizations to control and monitor access to critical systems and data, thereby enhancing overall cybersecurity posture +- [**Deutsche Telekom sysrepo-plugins**](https://github.com/telekom/sysrepo-plugins), a collection of YANG datastore plugins used to manage network devices - [**Egnyte Desktop**](https://helpdesk.egnyte.com/hc/en-us/articles/360007071732-Third-Party-Software-Acknowledgements), a secure cloud storage solution designed for businesses, enabling file sharing, collaboration, and data management across teams while ensuring compliance and data protection - [**Elster**](https://www.secunet.com/en/about-us/press/article/elstersecure-bietet-komfortablen-login-ohne-passwort-dank-secunet-protect4use), a digital platform developed by German tax authorities for secure and efficient electronic tax filing and management using secunet protect4use +- [**Envoy**](https://github.com/envoyproxy/envoy), a cloud-native edge and service proxy that forms the data plane of many service meshes - [**Ethereum Solidity**](https://github.com/ethereum/solidity), a high-level, object-oriented programming language designed for implementing smart contracts on the Ethereum platform +- [**gVisor**](https://github.com/google/gvisor), an application kernel that provides a secure sandbox for running untrusted containers +- [**IBM Storage Virtualize**](https://public.dhe.ibm.com/systems/support/warranty/pdfs/stgoilc/SV_for_FS_7300_v8_7_0_Base_OILC.pdf), the software powering IBM FlashSystem enterprise storage arrays - [**Inciga**](https://fossies.org/linux/icinga2/third-party/nlohmann_json/json.hpp), a monitoring tool for IT infrastructure, designed to provide insights into system performance and availability through customizable dashboards and alerts - [**Intel Accelerator Management Daemon for VMware ESXi**](https://downloadmirror.intel.com/772507/THIRD-PARTY.txt), a management tool designed for monitoring and controlling Intel hardware accelerators within VMware ESXi environments, optimizing performance and resource allocation - [**Juniper Identity Management Service**](https://www.juniper.net/documentation/us/en/software/jims/jims-guide/jims-guide.pdf) +- [**Meta FBOSS**](https://github.com/facebook/fboss), the software stack that controls the network switches in Meta's data centers - [**Microsoft Azure IoT SDK**](https://library.e.abb.com/public/2779c5f85f30484192eb3cb3f666a201/IP%20Gateway%20Open%20License%20Declaration_9AKK108467A4095_Rev_C.pdf), a collection of tools and libraries to help developers connect, build, and deploy Internet of Things (IoT) solutions on the Azure cloud platform +- [**Microsoft Confidential Consortium Framework**](https://github.com/microsoft/CCF), a framework for building secure, highly available applications on trusted execution environments - [**Microsoft WinGet**](https://github.com/microsoft/winget-cli), a command-line utility included in the Windows Package Manager +- [**Mitsubishi Electric SECS/GEM**](https://dl.mitsubishielectric.com/dl/fa/document/manual/plc/sh082483eng/sh082483engi.pdf), the semiconductor equipment communication software running on Mitsubishi Electric C Controller and C intelligent function modules +- [**Moxa**](https://www.moxa.com/getmedia/fbe2a0c7-8dda-4b5b-a501-15e45adebb1f/moxa-foss-statement-for-da-720-series-win-10-ltsc-21h2-declaration-v1.0.pdf), a provider of industrial networking, computing, and automation infrastructure - [**plexusAV**](https://www.sisme.com/media/10994/manual_plexusav-p-avn-4-form8244-c.pdf), a high-performance AV-over-IP transceiver device capable of video encoding and decoding using the IPMX standard - [**Pointr**](https://docs-dev.pointr.tech/docs/8.x/Developer%20Portal/Open%20Source%20Licenses/), a platform for indoor positioning and navigation solutions, offering tools and SDKs for developers to create location-based applications - [**secunet protect4use**](https://www.secunet.com/en/about-us/press/article/elstersecure-bietet-komfortablen-login-ohne-passwort-dank-secunet-protect4use), a secure, passwordless multifactor authentication solution that transforms smartphones into digital keyrings, ensuring high security for online services and digital identities - [**Sencore MRD 7000**](https://www.foccusdigital.com/wp-content/uploads/2025/03/MRD-7000-Manual-8175V.pdf), a professional multi-channel receiver and decoder supporting UHD and HD stream decoding +- [**Siemens SINEC**](https://cache.industry.siemens.com/dl/files/917/109974917/att_1298783/v2/OSS_SINEC-NMS_99.pdf), a family of network management and infrastructure services for industrial networks +- [**Toshiba Industrial Servers**](https://www.global.toshiba/content/dam/toshiba/jp/products-solutions/industrial/computer/product/server/fs20000r/pdf/FS20000R_OSS_License_6E8C5817_rev0.pdf), the FS20000R series of industrial servers for factory automation and control systems +- [**Wazuh**](https://github.com/wazuh/wazuh/blob/main/src/data_provider/src/sysInfo.cpp), a security platform for threat detection, integrity monitoring and incident response +- [**ZeroTier**](https://github.com/zerotier/ZeroTierOne/blob/dev/osdep/OSUtils.hpp), a software-defined networking service that creates virtual Ethernet networks diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index f75872505..9a7698b2f 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -868,6 +868,12 @@ The size of an array or object in a [binary format](../features/binary_formats/i the size following `#` for [UBJSON](../features/binary_formats/ubjson.md)/[BJData](../features/binary_formats/bjdata.md), or the encoded length for [CBOR](../features/binary_formats/cbor.md). +The exception is also thrown for a [UBJSON](../features/binary_formats/ubjson.md) array of a type that is encoded by its +marker alone (`Z`, `T` or `F`) whose declared count exceeds 1,048,576 (`1 << 20`). Such an array has no payload, so its +count alone decides how much memory is allocated, and a handful of bytes would otherwise describe billions of values. +[`to_ubjson`](../api/basic_json/to_ubjson.md) writes longer arrays of these types without the size and type annotation, +so any value it produces can still be read back. + !!! failure "Example messages" ``` @@ -879,6 +885,9 @@ or the encoded length for [CBOR](../features/binary_formats/cbor.md). ``` [json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive map size ``` + ``` + [json.exception.out_of_range.408] syntax error while parsing UBJSON size: excessive array size + ``` ### json.exception.out_of_range.409 @@ -933,6 +942,49 @@ BSON stores the length of documents, arrays, strings, and binary values in a sig [`to_bson`](../api/basic_json/to_bson.md) produced documents with negative length prefixes that [`from_bson`](../api/basic_json/from_bson.md) rejected. +### json.exception.out_of_range.413 + +A JSON Patch `remove` operation cannot be applied because the target location's parent is neither an object nor an array. Per [RFC 6902](https://datatracker.ietf.org/doc/html/rfc6902), a `remove` target must reference a member of an existing object or an element of an existing array; a primitive value (string, number, boolean, etc.) or `null` has no members or elements to remove. + +!!! failure "Example message" + + ``` + cannot remove value: the JSON Patch 'remove' target's parent is of type number, but must be an object or array + ``` + +!!! note + + This exception was added in version 3.13.0. Before that, this situation was silently ignored (the `remove` operation had no effect). + +### json.exception.out_of_range.414 + +A JSON Patch `move` operation's `"from"` location is a proper prefix of its `"path"` location. Per [RFC 6902](https://datatracker.ietf.org/doc/html/rfc6902) (section 4.4), a location cannot be moved into one of its own children. + +!!! failure "Example message" + + ``` + cannot move value: 'from' path '/0' is a proper prefix of 'path' '/0/0' + ``` + +!!! note + + This exception was added in version 3.13.0. Before that, this situation could succeed with a corrupted result: for an array target, removing the "from" element before the "add" step shifted subsequent indices, so "path" silently re-resolved to a different element than intended. + +### json.exception.out_of_range.415 + +MessagePack's ext type and BSON's binary subtype are each stored in a single byte. This exception is thrown when serializing a +[`byte_container_with_subtype`](../api/byte_container_with_subtype/index.md) whose subtype exceeds 255. + +!!! failure "Example message" + + ``` + [json.exception.out_of_range.415] subtype 70000 is too large for the MessagePack ext type (max 255) + ``` + +!!! note + + This exception was added in version 3.13.0. Before that, subtypes above 255 were silently truncated modulo 256 instead of raising an error. + ## Further exceptions This exception is thrown in case of errors that cannot be classified with the @@ -965,3 +1017,19 @@ A JSON Patch operation 'test' failed. The unsuccessful operation is also printed ``` [json.exception.other_error.501] unsuccessful: {"op":"test","path":"/baz","value":"bar"} ``` + +### json.exception.other_error.502 + +[`to_ubjson`](../api/basic_json/to_ubjson.md) and [`to_bjdata`](../api/basic_json/to_bjdata.md) were called with +`use_type = true` but `use_size = false`. UBJSON requires a size marker (`#`) after a type marker (`$`). + +!!! failure "Example message" + + ``` + [json.exception.other_error.502] use_type requires use_size = true + ``` + +!!! note + + This exception was added in version 3.13.0. Before that, debug builds aborted on an assertion and release builds + wrote a `$` marker without `#`, which [`from_ubjson`](../api/basic_json/from_ubjson.md) then rejected. diff --git a/docs/mkdocs/docs/home/faq.md b/docs/mkdocs/docs/home/faq.md index 8394dcfc7..8b3602bd1 100644 --- a/docs/mkdocs/docs/home/faq.md +++ b/docs/mkdocs/docs/home/faq.md @@ -90,6 +90,54 @@ The library supports **Unicode input** as follows: In most cases, the parser is right to complain, because the input is not UTF-8 encoded. This is especially true for Microsoft Windows, where Latin-1 or ISO 8859-1 is often the standard encoding. +### NUL bytes in the input + +!!! question "Questions" + + - Why does `json::parse()` silently ignore part of my input? + - Why does a `std::string`/buffer with extra data after the JSON text parse without error, while a similar-looking string with extra text does not? + +A `'\0'` (NUL) byte anywhere in the input is treated the same as the real end of the input, rather than as an ordinary (and, outside of a string, invalid) byte. Everything from that byte onward is silently ignored, without a parse error — including further, otherwise well-formed JSON: + +```cpp +json::parse(std::string("123") + '\0'); // == 123, no error +json::parse(std::string("123") + '\0' + "true"); // == 123, the "true" is silently ignored too +``` + +This is different from any other unexpected trailing byte, which *does* raise [`parse_error.101`](../home/exceptions.md#jsonexceptionparse_error101): + +```cpp +json::parse("123x"); // throws parse_error.101: unexpected additional data +``` + +This falls out of the same convention used when no explicit input length is given at all: `json::parse(const char*)` already stops at the first NUL byte via `strlen()`, since a bare pointer has no length of its own. The library applies that same NUL-terminated-C-string convention uniformly, rather than only when a length is genuinely unavailable — so a `std::string`, iterator range, or container whose content happens to include a NUL byte is affected the same way a raw `const char*` would be. + +If your input may contain a trailing or embedded NUL that is **not** meant to signal the end of the JSON text — for instance, a fixed-size, zero-padded buffer — trim it yourself before calling `parse()`, since the library will otherwise silently stop there instead of raising an error: + +```cpp +s.resize(s.find('\0')); // drop everything from the first NUL onward, if any +json::parse(s); +``` + +**Opt-in strict handling (since version 3.13.0)** + +Manually trimming every input is easy to forget. If you define [`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md) to `1` before including the library, a `'\0'` byte is instead rejected like any other unexpected byte and raises `parse_error.101`, instead of being treated as end of input: + +```cpp +#define JSON_STRICT_NUL_HANDLING 1 +#include + +json::parse(std::string("123") + '\0'); // throws parse_error.101 instead of silently returning 123 +``` + +This macro defaults to `0` (disabled, preserving the behavior described above) to avoid breaking existing code that may depend on it, even unknowingly; it is planned to become the default in version 4.0.0. See [its documentation](../api/macros/json_strict_nul_handling.md) for details, including how it also affects `char` arrays such as string literals. + +Note that this is unrelated to an *unescaped* NUL byte occurring **inside** a quoted JSON string, which is a different, already-invalid case and is correctly rejected either way: + +```cpp +json::parse(std::string("\"") + '\0' + "\""); // throws parse_error.101: control character U+0000 (NUL) must be escaped to \u0000 +``` + ### Wide string handling !!! question diff --git a/docs/mkdocs/docs/images/customers.png b/docs/mkdocs/docs/images/customers.png index dbfa19148..146056f20 100644 Binary files a/docs/mkdocs/docs/images/customers.png and b/docs/mkdocs/docs/images/customers.png differ diff --git a/docs/mkdocs/docs/integration/cmake.md b/docs/mkdocs/docs/integration/cmake.md index a8a6d52b6..71512cbd5 100644 --- a/docs/mkdocs/docs/integration/cmake.md +++ b/docs/mkdocs/docs/integration/cmake.md @@ -198,6 +198,11 @@ Use the non-amalgamated version of the library. This option is `ON` by default. Treat the library headers like system headers (i.e., adding `SYSTEM` to the [`target_include_directories`](https://cmake.org/cmake/help/latest/command/target_include_directories.html) call) to check for this library by tools like Clang-Tidy. This option is `OFF` by default. +### `JSON_StrictNulHandling` + +Reject a `'\0'` (NUL) byte in the input instead of treating it as end of input, by defining the macro +[`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md). This option is `OFF` by default. + ### `JSON_Valgrind` Execute the test suite with [Valgrind](https://valgrind.org). This option is `OFF` by default. Depends on `JSON_BuildTests`. diff --git a/docs/mkdocs/docs/integration/migration_guide.md b/docs/mkdocs/docs/integration/migration_guide.md index cb1c62d7b..8b718d969 100644 --- a/docs/mkdocs/docs/integration/migration_guide.md +++ b/docs/mkdocs/docs/integration/migration_guide.md @@ -176,6 +176,12 @@ You can prepare existing code by already defining conversions with calls to [`get`](../api/basic_json/get.md), [`get_to`](../api/basic_json/get_to.md), [`get_ref`](../api/basic_json/get_ref.md), or [`get_ptr`](../api/basic_json/get_ptr.md). +!!! tip "Automatic migration" + + The community-maintained clang-tidy check `modernize-nlohmann-json-explicit-conversions` rewrites most implicit + conversions into calls to [`get`](../api/basic_json/get.md). It is not part of clang-tidy itself; see + [discussion #4610](https://github.com/nlohmann/json/discussions/4610) for how to build and use it. + === "Deprecated" ```cpp diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 2e1337f47..ffb0fae80 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -98,6 +98,7 @@ nav: - Types: - features/types/index.md - features/types/number_handling.md + - features/types/template_parameters.md - Integration: - integration/index.md - integration/migration_guide.md @@ -291,11 +292,14 @@ nav: - 'JSON_HAS_THREE_WAY_COMPARISON': api/macros/json_has_three_way_comparison.md - 'JSON_NOEXCEPTION': api/macros/json_noexception.md - 'JSON_NO_IO': api/macros/json_no_io.md + - 'JSON_NO_THREAD_LOCAL': api/macros/json_no_thread_local.md - 'JSON_SKIP_LIBRARY_VERSION_CHECK': api/macros/json_skip_library_version_check.md - 'JSON_SKIP_UNSUPPORTED_COMPILER_CHECK': api/macros/json_skip_unsupported_compiler_check.md + - 'JSON_STRICT_NUL_HANDLING': api/macros/json_strict_nul_handling.md - 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md - 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md - 'JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON': api/macros/json_use_legacy_discarded_value_comparison.md + - 'JSON_USE_SIMDUTF': api/macros/json_use_simdutf.md - 'NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_derived_type.md - 'NLOHMANN_DEFINE_TYPE_INTRUSIVE, NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_intrusive.md - 'NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_non_intrusive.md @@ -313,7 +317,9 @@ nav: - community/contribution_guidelines.md - community/quality_assurance.md - community/governance.md + - community/roadmap.md - community/security_policy.md + - community/assurance_case.md # Extras extra: @@ -393,7 +399,14 @@ plugins: - http://nlohmann.github.io/json/* - https://nlohmann.github.io/json/* - mailto:* - - privacy + - privacy: + # repology.org refuses requests from GitHub Actions runners, which made + # the privacy plugin abort the whole build when it could not download the + # package badges (the fetch fails, then reading the missing cache entry + # raises FileNotFoundError). Readers' browsers are served normally, so + # leave these badges as external references instead of self-hosting them. + assets_exclude: + - repology.org/* - llmstxt: markdown_description: > JSON for Modern C++ is a C++11 header-only library implementing a JSON diff --git a/docs/mkdocs/requirements.txt b/docs/mkdocs/requirements.txt index ef7e5a306..6396ec500 100644 --- a/docs/mkdocs/requirements.txt +++ b/docs/mkdocs/requirements.txt @@ -1,7 +1,7 @@ -wheel==0.47.0 +wheel==0.48.0 mkdocs==1.6.1 # documentation framework -mkdocs-git-revision-date-localized-plugin==1.5.3 # plugin "git-revision-date-localized" +mkdocs-git-revision-date-localized-plugin==1.6.0 # plugin "git-revision-date-localized" mkdocs-material==9.7.7 # theme for mkdocs mkdocs-material-extensions==1.3.1 # extensions mkdocs-minify-plugin==0.8.0 # plugin "minify" diff --git a/include/nlohmann/detail/abi_macros.hpp b/include/nlohmann/detail/abi_macros.hpp index 3e07a6a98..cca04e8ec 100644 --- a/include/nlohmann/detail/abi_macros.hpp +++ b/include/nlohmann/detail/abi_macros.hpp @@ -34,6 +34,10 @@ #define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0 #endif +#ifndef JSON_BRACE_INIT_COPY_SEMANTICS + #define JSON_BRACE_INIT_COPY_SEMANTICS 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -52,20 +56,27 @@ #define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS _bics +#else + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) json_abi ## a ## b ## c -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ - NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS) + NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/include/nlohmann/detail/conversions/from_json.hpp b/include/nlohmann/detail/conversions/from_json.hpp index 6856a0965..11e40f5f4 100644 --- a/include/nlohmann/detail/conversions/from_json.hpp +++ b/include/nlohmann/detail/conversions/from_json.hpp @@ -398,6 +398,17 @@ inline void from_json(const BasicJsonType& j, CompatibleArrayType& bin) } } +template +auto from_json_object_reserve(ConstructibleObjectType& obj, typename ConstructibleObjectType::size_type size, priority_tag<1> /*unused*/) +-> decltype(obj.reserve(size), void()) +{ + obj.reserve(size); +} + +template +inline void from_json_object_reserve(ConstructibleObjectType& /*obj*/, std::size_t /*size*/, priority_tag<0> /*unused*/) +{} + template::value, int> = 0> inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj) @@ -409,6 +420,7 @@ inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj) ConstructibleObjectType ret; const auto* inner_object = j.template get_ptr(); + from_json_object_reserve(ret, inner_object->size(), priority_tag<1> {}); for (const auto& p : *inner_object) { ret.emplace(p.first, p.second.template get()); diff --git a/include/nlohmann/detail/conversions/to_chars.hpp b/include/nlohmann/detail/conversions/to_chars.hpp index 70fb9b933..c0945ab9e 100644 --- a/include/nlohmann/detail/conversions/to_chars.hpp +++ b/include/nlohmann/detail/conversions/to_chars.hpp @@ -1075,8 +1075,8 @@ char* to_chars(char* first, const char* last, FloatType value) } #ifdef __GNUC__ -#pragma GCC diagnostic push -#pragma GCC diagnostic ignored "-Wfloat-equal" + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif if (value == 0) // +-0 { @@ -1087,7 +1087,7 @@ char* to_chars(char* first, const char* last, FloatType value) return first; } #ifdef __GNUC__ -#pragma GCC diagnostic pop + JSON_HEDLEY_DIAGNOSTIC_POP #endif JSON_ASSERT(last - first >= std::numeric_limits::max_digits10); diff --git a/include/nlohmann/detail/conversions/to_json.hpp b/include/nlohmann/detail/conversions/to_json.hpp index 5f8644700..491bb9873 100644 --- a/include/nlohmann/detail/conversions/to_json.hpp +++ b/include/nlohmann/detail/conversions/to_json.hpp @@ -471,6 +471,30 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = { std::get(t)... }; } +#if JSON_BRACE_INIT_COPY_SEMANTICS +// JSON_BRACE_INIT_COPY_SEMANTICS makes a one-element braced list copy its +// element instead of wrapping it, which would serialize std::tuple{5} as 5 +// rather than [5]. Build what the default deduction builds instead: an object +// if the element is a [string, value] pair, a one-element array otherwise. +template +inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence<0> /*unused*/) +{ + BasicJsonType element(std::get<0>(t)); + // same test as the initializer-list constructor, including the cast that + // keeps a string type constructible from 0 from selecting operator[](key) + const bool is_member = element.is_array() && element.size() == 2 + && element[static_cast(0)].is_string(); + if (is_member) + { + j = BasicJsonType::object({std::move(element)}); + } + else + { + j = BasicJsonType::array({std::move(element)}); + } +} +#endif + template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& /*unused*/, index_sequence<> /*unused*/) { diff --git a/include/nlohmann/detail/exceptions.hpp b/include/nlohmann/detail/exceptions.hpp index cb87bb93e..808a3b6bf 100644 --- a/include/nlohmann/detail/exceptions.hpp +++ b/include/nlohmann/detail/exceptions.hpp @@ -33,8 +33,8 @@ // code stumbling over this. See https://github.com/nlohmann/json/issues/4087 // for a discussion. #if defined(__clang__) - #pragma clang diagnostic push - #pragma clang diagnostic ignored "-Wweak-vtables" + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(clang diagnostic ignored "-Wweak-vtables") #endif NLOHMANN_JSON_NAMESPACE_BEGIN @@ -101,7 +101,10 @@ class exception : public std::exception { if (&element.second == current) { - tokens.emplace_back(element.first.c_str()); + // data() is null-terminated, so a key containing + // a null byte is cut short here rather than + // truncating the whole message at what() + tokens.emplace_back(element.first.data()); break; } } @@ -287,5 +290,5 @@ class other_error : public exception NLOHMANN_JSON_NAMESPACE_END #if defined(__clang__) - #pragma clang diagnostic pop + JSON_HEDLEY_DIAGNOSTIC_POP #endif diff --git a/include/nlohmann/detail/hash.hpp b/include/nlohmann/detail/hash.hpp index 61b3469f1..20a971886 100644 --- a/include/nlohmann/detail/hash.hpp +++ b/include/nlohmann/detail/hash.hpp @@ -11,8 +11,10 @@ #include // uint8_t #include // size_t #include // hash +#include // vector #include +#include #include NLOHMANN_JSON_NAMESPACE_BEGIN @@ -26,6 +28,9 @@ inline std::size_t combine(std::size_t seed, std::size_t h) noexcept return seed; } +template +std::size_t hash_iteratively(const BasicJsonType& j); + /*! @brief hash a JSON value @@ -33,12 +38,21 @@ The hash function tries to rely on std::hash where possible. Furthermore, the type of the JSON value is taken into account to have different hash values for null, 0, 0U, and false, etc. +Hashing an array or an object hashes its elements, which used to call this +function again once per nesting level, so a value nested deeply enough +exhausted the call stack and terminated the process. The descent is bounded +here: once @ref recursion_depth_limit levels have been entered, @ref +hash_iteratively hashes what is left without the call stack. A value nested +less deeply than that - all but a vanishing minority - is hashed exactly as +before, without allocating. + @tparam BasicJsonType basic_json specialization @param j JSON value to hash +@param depth nesting level of @a j, counted from the value passed by the caller @return hash value of j */ template -std::size_t hash(const BasicJsonType& j) +std::size_t hash(const BasicJsonType& j, const std::size_t depth = 0) { using string_t = typename BasicJsonType::string_t; using number_integer_t = typename BasicJsonType::number_integer_t; @@ -56,22 +70,32 @@ std::size_t hash(const BasicJsonType& j) case BasicJsonType::value_t::object: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j.items()) { const auto h = std::hash {}(element.key()); seed = combine(seed, h); - seed = combine(seed, hash(element.value())); + seed = combine(seed, hash(element.value(), depth + 1)); } return seed; } case BasicJsonType::value_t::array: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j) { - seed = combine(seed, hash(element)); + seed = combine(seed, hash(element, depth + 1)); } return seed; } @@ -114,7 +138,9 @@ std::size_t hash(const BasicJsonType& j) seed = combine(seed, static_cast(j.get_binary().subtype())); for (const auto byte : j.get_binary()) { - seed = combine(seed, std::hash {}(byte)); + // the cast is needed for binary types whose value type is not + // an integer (e.g., std::byte) + seed = combine(seed, std::hash {}(static_cast(byte))); } return seed; } @@ -125,5 +151,77 @@ std::size_t hash(const BasicJsonType& j) } } +/// an array or object whose elements @ref hash_iteratively is hashing +template +struct hash_frame +{ + hash_frame(const BasicJsonType* value_, std::size_t seed_) noexcept + : value(value_), position(value_->cbegin()), seed(seed_) + {} + + const BasicJsonType* value; + typename BasicJsonType::const_iterator position; + std::size_t seed; +}; + +/*! +@brief hash the array or object @a j without the call stack + +Computes the same value as @ref hash, keeping the arrays and objects it has +entered on an explicit stack instead of descending into them. Only reached for +values nested deeper than @ref recursion_depth_limit. + +@tparam BasicJsonType basic_json specialization +@param j array or object to hash +@return hash value of j +*/ +template +std::size_t hash_iteratively(const BasicJsonType& j) +{ + using string_t = typename BasicJsonType::string_t; + + std::vector> stack; + stack.emplace_back(&j, combine(static_cast(j.type()), j.size())); + + while (true) + { + // a copy, as entering an element below can reallocate the stack; the + // frame itself is only changed through stack.back() + const hash_frame frame = stack.back(); + + if (frame.position == frame.value->cend()) + { + // all elements are hashed: fold this value's hash into its parent's + // seed, exactly where the recursive version returns it + const std::size_t h = frame.seed; + stack.pop_back(); + if (stack.empty()) + { + return h; + } + stack.back().seed = combine(stack.back().seed, h); + continue; + } + + if (frame.value->is_object()) + { + stack.back().seed = combine(stack.back().seed, std::hash {}(frame.position.key())); + } + + // advance before entering the element, which pushes onto the stack + const BasicJsonType& element = *frame.position; + ++stack.back().position; + + if (element.is_structured()) + { + stack.emplace_back(&element, combine(static_cast(element.type()), element.size())); + } + else + { + stack.back().seed = combine(stack.back().seed, hash(element)); + } + } +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index 557d7669c..df46eea58 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -58,6 +58,26 @@ inline bool little_endianness(int num = 1) noexcept return *reinterpret_cast(&num) == 1; } +/*! +@brief largest element count accepted for a UBJSON container of a valueless type + +An element of type 'Z' (null), 'T' (true) or 'F' (false) is encoded by its +type marker alone, so an optimized container of one of those types has no +payload at all and its declared count is the only thing that decides how much +is allocated: `[$Z#L` followed by a large count turns some ten bytes of input +into that many values (see #2793, which reports 35 GB and 150 seconds). Every +other type costs at least one byte per element and is bounded by the end of +the input. + +This is a sanity bound rather than a security boundary, and it is far above +any container met in practice. @ref binary_writer falls back to the +unoptimized encoding for longer containers, so that a value serialized by +this library can always be read back. + +@sa https://github.com/nlohmann/json/issues/2793 +*/ +JSON_INLINE_VARIABLE constexpr std::size_t max_valueless_container_size = 1 << 20; + /////////////////// // binary reader // /////////////////// @@ -110,6 +130,7 @@ class binary_reader const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { sax = sax_; + container_stack.clear(); bool result = false; switch (format) @@ -159,6 +180,80 @@ class binary_reader } private: + //////////////////////// + // nested containers // + //////////////////////// + + /*! + @brief a container that has been opened and not closed yet + + The binary readers do not call themselves once per nesting level. Like + @ref parser::sax_parse_internal, which does the same for JSON text, they + keep the containers they are inside of on a heap-allocated stack, so that + the native call stack does not grow with the nesting depth of the input + and a deeply nested value is bounded by memory rather than by the stack + (see #5104). + + The members are ordered by decreasing alignment, which is the ordering that + keeps a struct from growing as members are added to it. + */ + struct container_frame + { + container_frame(const std::size_t remaining_, const bool is_object_, + const char_int_type type_marker_ = 0) noexcept + : remaining(remaining_), type_marker(type_marker_), is_object(is_object_) {} + + /// number of elements that have not been read yet, or npos when the + /// container is not sized and ends at a marker instead + std::size_t remaining; + /// BSON: value of chars_read before this document's size prefix, which + /// check_bson_document_size() needs once the document has been read + std::size_t start_position = 0; + /// UBJSON/BJData: the type marker of an optimized container, so that + /// its elements are read without one of their own; 0 otherwise + char_int_type type_marker; + /// BSON: the size this document declares, in bytes + std::int32_t declared_size = 0; + /// whether to close this container with end_object() or end_array() + bool is_object; + }; + + /*! + @brief open a nested array or object + + Emits the SAX start event and records the container. This is the only + place the binary readers start a container, so a check that rejects one + can be made here and is then guaranteed to run before the start event. + + @param[in] is_object whether an object (true) or an array (false) begins + @param[in] len number of elements the container declares + + @return whether the SAX parser accepted the start event + */ + bool enter_container(const bool is_object, const std::size_t len, + const char_int_type type_marker = 0) + { + if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->start_object(len) : !sax->start_array(len))) + { + return false; + } + + container_stack.emplace_back(len, is_object, type_marker); + return true; + } + + /// @copydoc enter_container + bool enter_array(const std::size_t len, const char_int_type type_marker = 0) + { + return enter_container(/*is_object*/false, len, type_marker); + } + + /// @copydoc enter_container + bool enter_object(const std::size_t len, const char_int_type type_marker = 0) + { + return enter_container(/*is_object*/true, len, type_marker); + } + ////////// // BSON // ////////// @@ -193,8 +288,10 @@ class binary_reader @brief Reads in a BSON-object and passes it to the SAX-parser. @return whether a valid BSON-value was passed to the SAX parser */ - bool parse_bson_internal() + bool open_bson_document(const bool is_object) { + // recorded before the size prefix is read, because + // check_bson_document_size() measures the document from here const std::size_t document_start = chars_read; std::int32_t document_size{}; if (!get_number(input_format_t::bson, document_size)) @@ -202,22 +299,91 @@ class binary_reader return false; } - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(detail::unknown_size()))) + if (JSON_HEDLEY_UNLIKELY(!enter_container(is_object, detail::unknown_size()))) { return false; } - if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_list(/*is_array*/false))) + container_frame& frame = container_stack.back(); + frame.start_position = document_start; + frame.declared_size = document_size; + return true; + } + + /*! + @brief read a BSON document and everything nested inside it + + Reads elements until the document that was begun here is complete, + resuming the enclosing document each time an embedded one ends, so that + the nesting depth of the input costs heap rather than native stack + (see #5104). + + @return whether reading the document succeeded + */ + bool parse_bson_internal() + { + if (JSON_HEDLEY_UNLIKELY(!open_bson_document(/*is_object*/true))) { return false; } - if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(document_start, document_size))) - { - return false; - } + // the key currently being read; hoisted out of the loop so that its + // capacity is reused across elements and across nesting levels + string_t key; - return sax->end_object(); + while (true) + { + const auto element_type = get(); + + if (element_type == 0) // end of the innermost document + { + // a copy, not a reference: it must stay valid across the + // pop_back() below, which destroys the container_stack + // element it would otherwise alias + const container_frame top = container_stack.back(); + + if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(top.start_position, top.declared_size))) + { + return false; + } + + container_stack.pop_back(); + if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + { + return false; + } + // the document begun here is complete once it is not inside one + if (container_stack.empty()) + { + return true; + } + continue; + } + + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::bson, "element list"))) + { + return false; + } + + const std::size_t element_type_parse_position = chars_read; + key.clear(); + if (JSON_HEDLEY_UNLIKELY(!get_bson_cstr(key))) + { + return false; + } + + // an array's elements are named "0", "1", ... in the wire format, + // and those names are not passed on + if (container_stack.back().is_object && !sax->key(key)) + { + return false; + } + + if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_internal(element_type, element_type_parse_position))) + { + return false; + } + } } /*! @@ -329,12 +495,12 @@ class binary_reader case 0x03: // object { - return parse_bson_internal(); + return open_bson_document(/*is_object*/true); } case 0x04: // array { - return parse_bson_array(); + return open_bson_document(/*is_object*/false); } case 0x05: // binary @@ -384,82 +550,7 @@ class binary_reader } } - /*! - @brief Read a BSON element list (as specified in the BSON-spec) - The same binary layout is used for objects and arrays, hence it must be - indicated with the argument @a is_array which one is expected - (true --> array, false --> object). - - @param[in] is_array Determines if the element list being read is to be - treated as an object (@a is_array == false), or as an - array (@a is_array == true). - @return whether a valid BSON-object/array was passed to the SAX parser - */ - bool parse_bson_element_list(const bool is_array) - { - string_t key; - - while (auto element_type = get()) - { - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::bson, "element list"))) - { - return false; - } - - const std::size_t element_type_parse_position = chars_read; - if (JSON_HEDLEY_UNLIKELY(!get_bson_cstr(key))) - { - return false; - } - - if (!is_array && !sax->key(key)) - { - return false; - } - - if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_internal(element_type, element_type_parse_position))) - { - return false; - } - - // get_bson_cstr only appends - key.clear(); - } - - return true; - } - - /*! - @brief Reads an array from the BSON input and passes it to the SAX-parser. - @return whether a valid BSON-array was passed to the SAX parser - */ - bool parse_bson_array() - { - const std::size_t document_start = chars_read; - std::int32_t document_size{}; - if (!get_number(input_format_t::bson, document_size)) - { - return false; - } - - if (JSON_HEDLEY_UNLIKELY(!sax->start_array(detail::unknown_size()))) - { - return false; - } - - if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_list(/*is_array*/true))) - { - return false; - } - - if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(document_start, document_size))) - { - return false; - } - - return sax->end_array(); - } ////////// // CBOR // @@ -491,9 +582,12 @@ class binary_reader @return whether a valid CBOR value was passed to the SAX parser */ - bool parse_cbor_internal(const bool get_char, - const cbor_tag_handler_t tag_handler) + bool parse_cbor_value(const bool get_char, + const cbor_tag_handler_t tag_handler, + bool& tag_pending) { + tag_pending = false; + switch (get_char ? get() : current) { // EOF @@ -685,37 +779,36 @@ class binary_reader case 0x95: case 0x96: case 0x97: - return get_cbor_array( - conditional_static_cast(static_cast(current) & 0x1Fu), tag_handler); + return enter_array(conditional_static_cast(static_cast(current) & 0x1Fu)); case 0x98: // array (one-byte uint8_t for n follows) { std::uint8_t len{}; - return get_number(input_format_t::cbor, len) && get_cbor_array(static_cast(len), tag_handler); + return get_number(input_format_t::cbor, len) && enter_array(static_cast(len)); } case 0x99: // array (two-byte uint16_t for n follow) { std::uint16_t len{}; - return get_number(input_format_t::cbor, len) && get_cbor_array(static_cast(len), tag_handler); + return get_number(input_format_t::cbor, len) && enter_array(static_cast(len)); } case 0x9A: // array (four-byte uint32_t for n follow) { std::uint32_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && get_cbor_array(size, tag_handler); + return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size); } case 0x9B: // array (eight-byte uint64_t for n follow) { std::uint64_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && get_cbor_array(size, tag_handler); + return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size); } case 0x9F: // array (indefinite length) - return get_cbor_array(detail::unknown_size(), tag_handler); + return enter_array(detail::unknown_size()); // map (0x00..0x17 pairs of data items follow) case 0xA0: @@ -742,36 +835,36 @@ class binary_reader case 0xB5: case 0xB6: case 0xB7: - return get_cbor_object(conditional_static_cast(static_cast(current) & 0x1Fu), tag_handler); + return enter_object(conditional_static_cast(static_cast(current) & 0x1Fu)); case 0xB8: // map (one-byte uint8_t for n follows) { std::uint8_t len{}; - return get_number(input_format_t::cbor, len) && get_cbor_object(static_cast(len), tag_handler); + return get_number(input_format_t::cbor, len) && enter_object(static_cast(len)); } case 0xB9: // map (two-byte uint16_t for n follow) { std::uint16_t len{}; - return get_number(input_format_t::cbor, len) && get_cbor_object(static_cast(len), tag_handler); + return get_number(input_format_t::cbor, len) && enter_object(static_cast(len)); } case 0xBA: // map (four-byte uint32_t for n follow) { std::uint32_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && get_cbor_object(size, tag_handler); + return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size); } case 0xBB: // map (eight-byte uint64_t for n follow) { std::uint64_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && get_cbor_object(size, tag_handler); + return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size); } case 0xBF: // map (indefinite length) - return get_cbor_object(detail::unknown_size(), tag_handler); + return enter_object(detail::unknown_size()); case 0xC0: // tagged item case 0xC1: @@ -855,7 +948,10 @@ class binary_reader default: break; } - return parse_cbor_internal(true, tag_handler); + // the tagged value follows; it is read by the loop in + // parse_cbor_internal() rather than by recursing here + tag_pending = true; + return true; } case cbor_tag_handler_t::store: @@ -905,7 +1001,11 @@ class binary_reader break; } default: - return parse_cbor_internal(true, tag_handler); + { + // as above, the tagged value is read by the caller + tag_pending = true; + return true; + } } get(); return get_cbor_binary(b) && sax->binary(b); @@ -996,23 +1096,21 @@ class binary_reader } /*! - @brief reads a CBOR string + @brief reads a definite-length CBOR string - This function first reads starting bytes to determine the expected - string length and then copies this number of bytes into a string. - Additionally, CBOR's strings with indefinite lengths are supported. + Reads everything @ref get_cbor_string accepts except the indefinite-length + form, which that function handles itself. The bytes are appended to @a + result, so consecutive chunks of an indefinite-length string can be read + into the same string. - @param[out] result created string + @param[out] result string the bytes are appended to @return whether string creation completed - */ - bool get_cbor_string(string_t& result) - { - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "string"))) - { - return false; - } + @pre @a current is not EOF + */ + bool get_cbor_string_chunk(string_t& result) + { switch (current) { // UTF-8 string (0x00..0x17 bytes follow) @@ -1068,20 +1166,6 @@ class binary_reader return get_number(input_format_t::cbor, len) && get_string(input_format_t::cbor, len, result); } - case 0x7F: // UTF-8 string (indefinite length) - { - while (get() != 0xFF) - { - string_t chunk; - if (!get_cbor_string(chunk)) - { - return false; - } - result.append(chunk); - } - return true; - } - default: { auto last_token = get_token_string(); @@ -1092,23 +1176,82 @@ class binary_reader } /*! - @brief reads a CBOR byte array + @brief reads a CBOR string This function first reads starting bytes to determine the expected - byte array length and then copies this number of bytes into the byte array. - Additionally, CBOR's byte arrays with indefinite lengths are supported. + string length and then copies this number of bytes into a string. + Additionally, CBOR's strings with indefinite lengths are supported. - @param[out] result created byte array + @param[out] result created string + + @return whether string creation completed + */ + bool get_cbor_string(string_t& result) + { + // number of indefinite-length strings that have been opened and not + // closed yet. RFC 8949, Section 3.2.3 does not permit nesting them, + // but this reader has always accepted it, so the open levels are + // counted instead of recursed through, which overflowed the stack for + // an input of repeated 0x7F bytes (see #5104). Every chunk is appended + // to the same result, so no per-level state is needed. + std::size_t open = 0; + + while (true) + { + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "string"))) + { + return false; + } + + if (current == 0x7F) // UTF-8 string (indefinite length) + { + ++open; + get(); + continue; + } + + // a break marker closes the innermost indefinite-length string; + // outside of one it is not a string and falls through to the error + if (open != 0 && current == 0xFF) + { + if (--open == 0) + { + return true; + } + get(); + continue; + } + + if (JSON_HEDLEY_UNLIKELY(!get_cbor_string_chunk(result))) + { + return false; + } + + if (open == 0) + { + return true; + } + + get(); + } + } + + /*! + @brief reads a definite-length CBOR byte array + + Reads everything @ref get_cbor_binary accepts except the indefinite-length + form, which that function handles itself. The bytes are appended to @a + result, so consecutive chunks of an indefinite-length byte array can be + read into the same byte array. + + @param[out] result byte array the bytes are appended to @return whether byte array creation completed - */ - bool get_cbor_binary(binary_t& result) - { - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "binary"))) - { - return false; - } + @pre @a current is not EOF + */ + bool get_cbor_binary_chunk(binary_t& result) + { switch (current) { // Binary data (0x00..0x17 bytes follow) @@ -1168,20 +1311,6 @@ class binary_reader get_binary(input_format_t::cbor, len, result); } - case 0x5F: // Binary data (indefinite length) - { - while (get() != 0xFF) - { - binary_t chunk; - if (!get_cbor_binary(chunk)) - { - return false; - } - result.insert(result.end(), chunk.begin(), chunk.end()); - } - return true; - } - default: { auto last_token = get_token_string(); @@ -1191,6 +1320,63 @@ class binary_reader } } + /*! + @brief reads a CBOR byte array + + This function first reads starting bytes to determine the expected + byte array length and then copies this number of bytes into the byte array. + Additionally, CBOR's byte arrays with indefinite lengths are supported. + + @param[out] result created byte array + + @return whether byte array creation completed + */ + bool get_cbor_binary(binary_t& result) + { + // the open indefinite-length byte arrays are counted rather than + // recursed through, for the reason given in @ref get_cbor_string + std::size_t open = 0; + + while (true) + { + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "binary"))) + { + return false; + } + + if (current == 0x5F) // Binary data (indefinite length) + { + ++open; + get(); + continue; + } + + // a break marker closes the innermost indefinite-length byte + // array; outside of one it falls through to the error below + if (open != 0 && current == 0xFF) + { + if (--open == 0) + { + return true; + } + get(); + continue; + } + + if (JSON_HEDLEY_UNLIKELY(!get_cbor_binary_chunk(result))) + { + return false; + } + + if (open == 0) + { + return true; + } + + get(); + } + } + /*! @brief narrow a definite CBOR array/map length to std::size_t @@ -1217,96 +1403,110 @@ class binary_reader } /*! - @param[in] len the length of the array or detail::unknown_size() for an - array of indefinite size + @brief read a CBOR value and everything nested inside it + + Reads values until the one that was begun here is complete, resuming the + enclosing container after each element, so that the nesting depth of the + input costs heap rather than native stack (see #5104). + + @param[in] get_char whether a new character should be retrieved from the + input (true) or whether the last read character + @a current should be considered instead @param[in] tag_handler how CBOR tags should be treated - @return whether array creation completed + + @return whether reading the value succeeded */ - bool get_cbor_array(const std::size_t len, - const cbor_tag_handler_t tag_handler) + bool parse_cbor_internal(const bool get_char, + const cbor_tag_handler_t tag_handler) { - if (JSON_HEDLEY_UNLIKELY(!sax->start_array(len))) - { - return false; - } + // whether the next value starts at a fresh byte or at the one already + // read into `current` + bool fetch = get_char; - if (len != detail::unknown_size()) + // the key currently being read; hoisted out of the loop so that its + // capacity is reused across elements and across nesting levels + string_t key; + + while (true) { - for (std::size_t i = 0; i < len; ++i) + if (!container_stack.empty()) { - if (JSON_HEDLEY_UNLIKELY(!parse_cbor_internal(true, tag_handler))) + // a copy, not a reference: it must stay valid across the + // pop_back() below, which destroys the container_stack element + // it would otherwise alias + const container_frame top = container_stack.back(); + bool at_end = false; + + if (top.remaining != npos) { - return false; + // definite length: the container ends once its elements + // have been read + at_end = (top.remaining == 0); + if (!at_end) + { + // claim the element about to be read + --container_stack.back().remaining; + if (top.is_object) + { + get(); + } + } + fetch = true; } - } - } - else - { - while (get() != 0xFF) - { - if (JSON_HEDLEY_UNLIKELY(!parse_cbor_internal(false, tag_handler))) + else { - return false; + // indefinite length: the container ends at a break marker. + // Testing for it consumes a byte, which is the first byte + // of the next element when it is not one. + at_end = (get() == 0xFF); + fetch = top.is_object; } - } - } - return sax->end_array(); - } - - /*! - @param[in] len the length of the object or detail::unknown_size() for an - object of indefinite size - @param[in] tag_handler how CBOR tags should be treated - @return whether object creation completed - */ - bool get_cbor_object(const std::size_t len, - const cbor_tag_handler_t tag_handler) - { - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(len))) - { - return false; - } - - if (len != 0) - { - string_t key; - if (len != detail::unknown_size()) - { - for (std::size_t i = 0; i < len; ++i) + if (at_end) { - get(); + container_stack.pop_back(); + if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + { + return false; + } + // the value begun here is complete once its container is + if (container_stack.empty()) + { + return true; + } + continue; + } + + if (top.is_object) + { + key.clear(); if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key))) { return false; } - - if (JSON_HEDLEY_UNLIKELY(!parse_cbor_internal(true, tag_handler))) - { - return false; - } - key.clear(); + fetch = true; } } - else - { - while (get() != 0xFF) - { - if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key))) - { - return false; - } - if (JSON_HEDLEY_UNLIKELY(!parse_cbor_internal(true, tag_handler))) - { - return false; - } - key.clear(); + // a tag is not a value of its own: read on until the tagged value + bool tag_pending = false; + do + { + if (JSON_HEDLEY_UNLIKELY(!parse_cbor_value(fetch, tag_handler, tag_pending))) + { + return false; } + fetch = true; + } + while (tag_pending); + + // a value that opened a container left it on the stack; one that + // did not, and that was not inside a container, was the whole value + if (container_stack.empty()) + { + return true; } } - - return sax->end_object(); } ///////////// @@ -1316,7 +1516,17 @@ class binary_reader /*! @return whether a valid MessagePack value was passed to the SAX parser */ - bool parse_msgpack_internal() + /*! + @brief read one MessagePack value + + Reads a single value and passes it to the SAX parser. A value that begins + a container is not read to its end: the container is opened with + @ref enter_container and its elements are read by + @ref parse_msgpack_internal, so that nesting does not consume native stack. + + @return whether reading the value succeeded + */ + bool parse_msgpack_value() { switch (get()) { @@ -1472,7 +1682,7 @@ class binary_reader case 0x8D: case 0x8E: case 0x8F: - return get_msgpack_object(conditional_static_cast(static_cast(current) & 0x0Fu)); + return enter_object(conditional_static_cast(static_cast(current) & 0x0Fu)); // fixarray case 0x90: @@ -1491,7 +1701,7 @@ class binary_reader case 0x9D: case 0x9E: case 0x9F: - return get_msgpack_array(conditional_static_cast(static_cast(current) & 0x0Fu)); + return enter_array(conditional_static_cast(static_cast(current) & 0x0Fu)); // fixstr case 0xA0: @@ -1622,25 +1832,25 @@ class binary_reader case 0xDC: // array 16 { std::uint16_t len{}; - return get_number(input_format_t::msgpack, len) && get_msgpack_array(static_cast(len)); + return get_number(input_format_t::msgpack, len) && enter_array(static_cast(len)); } case 0xDD: // array 32 { std::uint32_t len{}; - return get_number(input_format_t::msgpack, len) && get_msgpack_array(conditional_static_cast(len)); + return get_number(input_format_t::msgpack, len) && enter_array(conditional_static_cast(len)); } case 0xDE: // map 16 { std::uint16_t len{}; - return get_number(input_format_t::msgpack, len) && get_msgpack_object(static_cast(len)); + return get_number(input_format_t::msgpack, len) && enter_object(static_cast(len)); } case 0xDF: // map 32 { std::uint32_t len{}; - return get_number(input_format_t::msgpack, len) && get_msgpack_object(conditional_static_cast(len)); + return get_number(input_format_t::msgpack, len) && enter_object(conditional_static_cast(len)); } // negative fixint @@ -1888,55 +2098,69 @@ class binary_reader } /*! - @param[in] len the length of the array - @return whether array creation completed + @brief read a MessagePack value and everything nested inside it + + Reads values until the one that was begun here is complete, resuming the + enclosing container each time an element ends, so that the nesting depth + of the input costs heap rather than native stack (see #5104). + + @return whether reading the value succeeded */ - bool get_msgpack_array(const std::size_t len) + bool parse_msgpack_internal() { - if (JSON_HEDLEY_UNLIKELY(!sax->start_array(len))) - { - return false; - } - - for (std::size_t i = 0; i < len; ++i) - { - if (JSON_HEDLEY_UNLIKELY(!parse_msgpack_internal())) - { - return false; - } - } - - return sax->end_array(); - } - - /*! - @param[in] len the length of the object - @return whether object creation completed - */ - bool get_msgpack_object(const std::size_t len) - { - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(len))) - { - return false; - } - + // the key currently being read; hoisted out of the loop so that its + // capacity is reused across elements and across nesting levels string_t key; - for (std::size_t i = 0; i < len; ++i) + + while (true) { - get(); - if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key))) + if (!container_stack.empty()) + { + // copied out before anything can push onto the stack and + // invalidate a reference into it + const bool is_object = container_stack.back().is_object; + + if (container_stack.back().remaining == 0) + { + container_stack.pop_back(); + if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->end_object() : !sax->end_array())) + { + return false; + } + // the value begun here is complete once its container is + if (container_stack.empty()) + { + return true; + } + continue; + } + + // claim the element about to be read + --container_stack.back().remaining; + + if (is_object) + { + get(); + key.clear(); + if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key))) + { + return false; + } + } + } + + if (JSON_HEDLEY_UNLIKELY(!parse_msgpack_value())) { return false; } - if (JSON_HEDLEY_UNLIKELY(!parse_msgpack_internal())) + // a value that opened a container left it on the stack; one that + // did not, and that was not inside a container, was the whole value + if (container_stack.empty()) { - return false; + return true; } - key.clear(); } - - return sax->end_object(); } //////////// @@ -1952,7 +2176,103 @@ class binary_reader */ bool parse_ubjson_internal(const bool get_char = true) { - return get_ubjson_value(get_char ? get_ignore_noop() : current); + // the key currently being read; hoisted out of the loop so that its + // capacity is reused across elements and across nesting levels + string_t key; + + // the type marker of the value to read next + char_int_type prefix = get_char ? get_ignore_noop() : current; + + while (true) + { + const std::size_t depth = container_stack.size(); + + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_value(prefix))) + { + return false; + } + + // the value begun here is complete once it is not inside anything + if (container_stack.empty()) + { + return true; + } + + // a value was completed rather than a container opened; a + // container that ends at a marker needs the next byte to test + if (container_stack.size() == depth && container_stack.back().remaining == npos) + { + get_ignore_noop(); + } + + // advance to the next element, closing the containers that ended. + // top is a copy, not a reference: it must stay valid across the + // pop_back() below, which destroys the container_stack element it + // would otherwise alias. + for (;;) + { + const container_frame top = container_stack.back(); + + if (top.remaining != npos) + { + if (top.remaining != 0) + { + --container_stack.back().remaining; + if (top.is_object) + { + key.clear(); + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) + { + return false; + } + } + // an optimized container gives its elements no marker + prefix = (top.type_marker != 0) ? top.type_marker : get_ignore_noop(); + break; + } + } + // the end marker is compared against a literal rather than + // against a conditional expression, because char_int_type is + // unsigned for some input adapters and MSVC then reports the + // comparison as a signed/unsigned mismatch + else if (top.is_object ? (current != '}') : (current != ']')) + { + // a container that ends at a marker is never optimized, so + // every element carries its own marker; for an object the + // byte tested above is the first byte of the key + if (top.is_object) + { + key.clear(); + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key))) + { + return false; + } + prefix = get_ignore_noop(); + } + else + { + prefix = current; + } + break; + } + + container_stack.pop_back(); + if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + { + return false; + } + if (container_stack.empty()) + { + return true; + } + // the container that just ended was an element of the one + // below it, which may need the next byte for its own test + if (container_stack.back().remaining == npos) + { + get_ignore_noop(); + } + } + } } /*! @@ -2391,7 +2711,12 @@ class binary_reader { result.first = npos; // size result.second = 0; // type - bool is_ndarray = false; + // seed the flag with the caller's context: inside an ndarray dimension + // vector another ndarray is not allowed, and get_ubjson_size_value() + // rejects it up front instead of reading it and reporting afterwards. + // Seeding it with `false` made every '#' of a "[#[#[..." chain descend + // another level, which overflowed the stack (see #5104). + bool is_ndarray = inside_ndarray; get_ignore_noop(); @@ -2424,13 +2749,11 @@ class binary_reader } const bool is_error = get_ubjson_size_value(result.first, is_ndarray); - if (input_format == input_format_t::bjdata && is_ndarray) + // an ndarray was read here only if the flag flipped; when it was + // seeded true, get_ubjson_size_value() already rejected the nested + // dimension vector + if (input_format == input_format_t::bjdata && is_ndarray && !inside_ndarray) { - if (inside_ndarray) - { - return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read, - exception_message(input_format, "ndarray can not be recursive", "size"), nullptr)); - } result.second |= (1 << 8); // use bit 8 to indicate ndarray, all UBJSON and BJData markers should be ASCII letters } return is_error; @@ -2439,7 +2762,7 @@ class binary_reader if (current == '#') { const bool is_error = get_ubjson_size_value(result.first, is_ndarray); - if (input_format == input_format_t::bjdata && is_ndarray) + if (input_format == input_format_t::bjdata && is_ndarray && !inside_ndarray) { return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read, exception_message(input_format, "ndarray requires both type and size", "size"), nullptr)); @@ -2710,53 +3033,33 @@ class binary_reader if (size_and_type.first != npos) { - if (JSON_HEDLEY_UNLIKELY(!sax->start_array(size_and_type.first))) + // reading an element of a valueless type consumes no input, so the + // declared count alone decides how much is allocated; the check is + // made before the start event so that no container is opened that + // is then abandoned. See @ref max_valueless_container_size. + if (JSON_HEDLEY_UNLIKELY((size_and_type.second == 'Z' || size_and_type.second == 'T' || size_and_type.second == 'F') + && size_and_type.first > max_valueless_container_size)) + { + return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, + exception_message(input_format, "excessive array size", "size"), nullptr)); + } + + if (JSON_HEDLEY_UNLIKELY(!enter_array(size_and_type.first, size_and_type.second))) { return false; } - if (size_and_type.second != 0) + if (size_and_type.second == 'N') { - if (size_and_type.second != 'N') - { - for (std::size_t i = 0; i < size_and_type.first; ++i) - { - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_value(size_and_type.second))) - { - return false; - } - } - } - } - else - { - for (std::size_t i = 0; i < size_and_type.first; ++i) - { - if (JSON_HEDLEY_UNLIKELY(!parse_ubjson_internal())) - { - return false; - } - } - } - } - else - { - if (JSON_HEDLEY_UNLIKELY(!sax->start_array(detail::unknown_size()))) - { - return false; + // a no-op is not a value, so a container of them holds none; + // the declared size has already been passed to the SAX parser + container_stack.back().remaining = 0; } - while (current != ']') - { - if (JSON_HEDLEY_UNLIKELY(!parse_ubjson_internal(false))) - { - return false; - } - get_ignore_noop(); - } + return true; } - return sax->end_array(); + return enter_array(detail::unknown_size()); } /*! @@ -2778,68 +3081,12 @@ class binary_reader exception_message(input_format, "BJData object does not support ND-array size in optimized format", "object"), nullptr)); } - string_t key; if (size_and_type.first != npos) { - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(size_and_type.first))) - { - return false; - } - - if (size_and_type.second != 0) - { - for (std::size_t i = 0; i < size_and_type.first; ++i) - { - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) - { - return false; - } - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_value(size_and_type.second))) - { - return false; - } - key.clear(); - } - } - else - { - for (std::size_t i = 0; i < size_and_type.first; ++i) - { - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) - { - return false; - } - if (JSON_HEDLEY_UNLIKELY(!parse_ubjson_internal())) - { - return false; - } - key.clear(); - } - } - } - else - { - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(detail::unknown_size()))) - { - return false; - } - - while (current != '}') - { - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key))) - { - return false; - } - if (JSON_HEDLEY_UNLIKELY(!parse_ubjson_internal())) - { - return false; - } - get_ignore_noop(); - key.clear(); - } + return enter_object(size_and_type.first, size_and_type.second); } - return sax->end_object(); + return enter_object(detail::unknown_size()); } // Note, no reader for UBJSON binary types is implemented because they do @@ -2899,7 +3146,10 @@ class binary_reader number_string, out_of_range::create(406, concat("number overflow parsing '", number_string, '\''), nullptr)); } - return sax->number_float(parsed_float, std::move(number_string)); + // number_string is a std::string, while the SAX interface takes a + // string_t; convert explicitly, as the two are only implicitly + // convertible for some string types + return sax->number_float(parsed_float, string_t(number_string.data(), number_string.size())); } case token_type::uninitialized: case token_type::literal_true: @@ -3227,6 +3477,9 @@ class binary_reader /// the SAX parser json_sax_t* sax = nullptr; + /// the containers that have been opened and not closed yet; see @ref container_frame + std::vector container_stack{}; + // excluded markers in bjdata optimized type #define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \ make_array('F', 'H', 'N', 'S', 'T', 'Z', '[', '{') diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index 05d27f256..b0ce04eae 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -203,11 +203,31 @@ class input_stream_adapter // General-purpose iterator-based adapter. It might not be as fast as // theoretically possible for some containers, but it is extremely versatile. -// SentinelType defaults to IteratorType for backward compatibility, but may -// be a different type (e.g., a C++20 sentinel or counted_iterator). +// SentinelType defaults to IteratorType for backward compatibility, but may be +// a different type, e.g. a C++20 sentinel such as std::default_sentinel_t when +// IteratorType is a std::counted_iterator. template class iterator_input_adapter { + // Whether the number of elements between two positions can be computed in + // O(1): either the iterator and the sentinel have the same type (plain + // std::distance) or, in C++20, the sentinel is a sized sentinel for the + // iterator (std::ranges::distance), e.g. std::default_sentinel_t paired + // with std::counted_iterator. + // + // JSON_HAS_RANGES gates the C++20 branch: on standard libraries with an + // incomplete (libstdc++ < 11, see #4440) evaluating + // std::contiguous_iterator on a std::counted_iterator is a hard error + // instead of yielding false, and these traits are instantiated for every + // adapter. Such toolchains fall back to the pointer-only test and simply + // use the byte-at-a-time scanner. + static constexpr bool sentinel_is_sized = +#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) + std::is_same::value || std::sized_sentinel_for; +#else + std::is_same::value; +#endif + public: using char_type = typename std::iterator_traits::value_type; @@ -219,7 +239,7 @@ class iterator_input_adapter // in wide_string_input_adapter, which does not expose this). static constexpr bool supports_seek = std::is_same::iterator_category, std::random_access_iterator_tag>::value - && std::is_same::value + && sentinel_is_sized && sizeof(char_type) == 1; iterator_input_adapter(IteratorType first, SentinelType last) @@ -267,30 +287,60 @@ class iterator_input_adapter private: // whether IteratorType refers to a contiguous range and therefore supports // a std::memcpy fast path (pointers always do; in C++20 we can also detect - // library iterators such as those of std::vector and std::string). - // Computing the available element count needs either same-type iterators - // (plain std::distance) or, in C++20, a sized sentinel (std::ranges::distance), - // e.g. std::counted_iterator paired with std::default_sentinel_t. - static constexpr bool iterator_is_contiguous = -#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) - (std::is_same::value || std::sized_sentinel_for) - && (std::contiguous_iterator || std::is_pointer::value); + // library iterators such as those of std::vector and std::string). The + // available element count must also be computable in O(1), hence + // sentinel_is_sized. + static constexpr bool iterator_is_contiguous = sentinel_is_sized && +#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) + (std::contiguous_iterator || std::is_pointer::value); #else - std::is_same::value && std::is_pointer::value; + std::is_pointer::value; #endif + // number of unread elements in [current, end) + std::size_t remaining_count() const + { +#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) + // std::ranges::distance also supports sized sentinels of a different + // type (e.g. std::counted_iterator + std::default_sentinel_t) + return static_cast(std::ranges::distance(current, end)); +#else + return static_cast(std::distance(current, end)); +#endif + } + + public: + // Whether the remaining input is a single contiguous block of 1-byte + // elements that the lexer can inspect directly (used for the SWAR string + // fast path). + static constexpr bool supports_bulk_scan = + iterator_is_contiguous && sizeof(char_type) == 1; + + // Pointer to the next unread element; only valid when bulk_remaining() > 0. + const char_type* bulk_data() const + { + return &*current; + } + + // Number of unread elements available as one contiguous block. + std::size_t bulk_remaining() const + { + return remaining_count(); + } + + // Consume @a n elements previously inspected via bulk_data(). + void bulk_skip(std::size_t n) + { + std::advance(current, static_cast::difference_type>(n)); + } + + private: // contiguous fast path: bulk copy the remaining range with std::memcpy template std::size_t get_elements_impl(T* dest, std::size_t count, std::true_type /*contiguous*/) { const std::size_t wanted = count * sizeof(T); -#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) - // std::ranges::distance also supports sized sentinels of a different - // type (e.g. std::counted_iterator + std::default_sentinel_t) - const std::size_t available = static_cast(std::ranges::distance(current, end)) * sizeof(char_type); -#else - const std::size_t available = static_cast(std::distance(current, end)) * sizeof(char_type); -#endif + const std::size_t available = remaining_count() * sizeof(char_type); const std::size_t copied = (std::min)(wanted, available); if (JSON_HEDLEY_LIKELY(copied != 0)) { @@ -618,6 +668,46 @@ typename iterator_input_adapter_factory::adapter_typ return factory_type::create(first, last); } +// The element type a container's data() points at, cv-qualifiers removed. +// Ill-formed - and therefore SFINAE-friendly - for types without data(). +template +using container_data_t = typename std::remove_cv().data()) >::type >::type; + +// The container's own element type, cv-qualifiers removed. It is looked up on +// the bare type so it is also found when ContainerType is deduced as a +// reference by the forwarding-reference overload below. +template +using container_value_t = typename std::remove_cv < + typename std::remove_cv::type>::type::value_type >::type; + +// Detect a container that stores its elements contiguously as single bytes +// (std::string, std::vector, std::array, +// std::string_view, ...). Such inputs are wrapped in a pointer-based adapter so +// they benefit from the contiguous fast paths (bulk string scanning, memcpy for +// binary formats) in every C++ standard - not only in C++20, where the standard +// library iterators model std::contiguous_iterator and are detected directly. +// +// data() and size() on their own would be duck typing: they say nothing about +// size() counting the units data() points at, and reading [data(), data() + +// size()) as bytes would be wrong for a type where it does not. Requiring the +// container's own value_type to be that same single-byte element ties the two +// together; every contiguous standard container satisfies it. Anything else +// keeps the iterator-based adapter, which is always correct - only slower. +template +struct is_contiguous_byte_container : std::false_type {}; + +template +struct is_contiguous_byte_container < ContainerType, void_t < + container_data_t, + container_value_t, +decltype(std::declval().size()) >> + : std::integral_constant < bool, + std::is_pointer().data())>::value&& + std::is_integral>::value&& + sizeof(container_data_t) == 1 && + std::is_same, container_value_t>::value > {}; + // Convenience shorthand from container to iterator // Enables ADL on begin(container) and end(container) // Encloses the using declarations in namespace for not to leak them to outside scope @@ -645,12 +735,32 @@ struct container_input_adapter_factory< ContainerType, } // namespace container_input_adapter_factory_impl -template -typename container_input_adapter_factory_impl::container_input_adapter_factory::adapter_type input_adapter(ContainerType&& container) +// General container path (iterator-based). Contiguous single-byte containers +// are excluded here and routed through the pointer-based overload below. +template < typename ContainerType, + enable_if_t < !is_contiguous_byte_container::value, int > = 0 > +typename container_input_adapter_factory_impl::container_input_adapter_factory::adapter_type input_adapter(ContainerType && container) { return container_input_adapter_factory_impl::container_input_adapter_factory::create(std::forward(container)); } +// Contiguous single-byte containers (std::string, std::vector, ...) are +// wrapped in a pointer-based adapter so the contiguous fast paths apply in every +// standard. The pointer keeps the container's own element type (const char* for +// std::string, const std::uint8_t* for std::vector, ...), so the +// resulting char_type - and therefore the parsing behavior - is byte-for-byte +// identical to the iterator-based path; only the raw pointer additionally +// enables the bulk fast paths. The container outlives the adapter for the whole +// parse (temporaries live until the end of the full expression), exactly as the +// iterators it replaces did. +template < typename ContainerType, + enable_if_t < is_contiguous_byte_container::value, int > = 0 > +auto input_adapter(const ContainerType& container) +-> decltype(input_adapter(container.data(), container.data() + container.size())) +{ + return input_adapter(container.data(), container.data() + container.size()); +} + // specialization for std::string using string_input_adapter_type = decltype(input_adapter(std::declval())); @@ -700,6 +810,21 @@ contiguous_bytes_input_adapter input_adapter(CharT b) template auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) { +#if JSON_STRICT_NUL_HANDLING + // A `char` array from string-literal initialization (e.g. json::parse("123")) + // carries a trailing '\0' contributed by the compiler, not by the source + // text; drop exactly that one byte so it is not mistaken for real trailing + // data. Every other element type (unsigned char, std::uint8_t, ...) keeps + // the full extent unconditionally, since a trailing zero byte there is + // genuine data (e.g. CBOR/MessagePack). This intentionally does not + // strlen()-scan the array (as the pointer overload above does for a + // null-delimited string): for a `char` array that is not NUL-terminated + // within its bounds, that would read past the end of the array. + if (std::is_same::type, char>::value && N > 0 && array[N - 1] == 0) + { + return input_adapter(array, array + N - 1); + } +#endif return input_adapter(array, array + N); } diff --git a/include/nlohmann/detail/input/json_sax.hpp b/include/nlohmann/detail/input/json_sax.hpp index 8a98ee728..962913610 100644 --- a/include/nlohmann/detail/input/json_sax.hpp +++ b/include/nlohmann/detail/input/json_sax.hpp @@ -8,15 +8,17 @@ #pragma once +#include // find_if, min #include #include // string #include // enable_if_t -#include // move +#include // move, pair #include // vector #include #include #include +#include #include NLOHMANN_JSON_NAMESPACE_BEGIN @@ -150,6 +152,29 @@ constexpr std::size_t unknown_size() return (std::numeric_limits::max)(); } +/*! +@brief reserve capacity for @a len elements in array @a arr + +Reserving upfront avoids repeated reallocations while the elements are added, +but the reservation is capped so a bogus/hostile length (which is not bounded +by max_size(), unlike e.g. std::vector) cannot trigger an oversized allocation +for a small or truncated input. + +The overload below is selected for array types without reserve() (e.g., +std::deque), which are then left untouched. +*/ +template +auto reserve_array(ArrayType& arr, std::size_t len, priority_tag<1> /*unused*/) +-> decltype(arr.reserve(len), void()) +{ + constexpr std::size_t reserve_cap = 16384; + arr.reserve((std::min)(len, reserve_cap)); +} + +template +inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/) +{} + /*! @brief SAX implementation to create a JSON value from SAX events @@ -222,12 +247,16 @@ class json_sax_dom_parser bool string(string_t& val) { - handle_value(val); + // json_sax documents that the passed value may be moved from, + // so hand the buffer over instead of copying it + handle_value(std::move(val)); return true; } bool binary(binary_t& val) { + // json_sax documents that the passed value may be moved from, + // so hand the buffer over instead of copying it handle_value(std::move(val)); return true; } @@ -249,7 +278,7 @@ class json_sax_dom_parser if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); } return true; @@ -298,7 +327,12 @@ class json_sax_dom_parser if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + } + + if (len != detail::unknown_size()) + { + reserve_array(*ref_stack.back()->m_data.m_value.array, len, priority_tag<1> {}); } return true; @@ -532,12 +566,16 @@ class json_sax_dom_callback_parser bool string(string_t& val) { - handle_value(val); + // json_sax documents that the passed value may be moved from, + // so hand the buffer over instead of copying it + handle_value(std::move(val)); return true; } bool binary(binary_t& val) { + // json_sax documents that the passed value may be moved from, + // so hand the buffer over instead of copying it handle_value(std::move(val)); return true; } @@ -548,6 +586,11 @@ class json_sax_dom_callback_parser const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::object_start, discarded); keep_stack.push_back(keep); + // the key this object will be stored under, read before handle_value() + // may consume it; kept in lockstep with ref_stack so end_object() can + // find the object in its parent again + container_key_stack.push_back(current_key()); + auto val = handle_value(BasicJsonType::value_t::object, true); ref_stack.push_back(val.second); @@ -568,7 +611,7 @@ class json_sax_dom_callback_parser // check object limit if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); } } return true; @@ -581,11 +624,24 @@ class json_sax_dom_callback_parser // check callback for the key const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::key, k); key_keep_stack.push_back(keep); + // remember the key so a rejected value can be erased without searching + // the object for it (kept in lockstep with key_keep_stack) + key_stack.push_back(val); // add discarded value at the given key and store the reference for later if (keep && ref_stack.back()) { - object_element = &(ref_stack.back()->m_data.m_value.object->operator[](val) = discarded); + auto& obj = *ref_stack.back()->m_data.m_value.object; + const auto it = obj.find(val); + if (it != obj.end()) + { + // this is a duplicate key (legal in JSON); remember its + // current value so it can be restored later if the new + // value is rejected by the callback, instead of being + // erased together with the discarded placeholder + duplicate_key_stash.emplace_back(&(it->second), it->second); + } + object_element = &(obj[val] = discarded); } return true; @@ -597,13 +653,18 @@ class json_sax_dom_callback_parser { if (!callback(static_cast(ref_stack.size()) - 1, parse_event_t::object_end, *ref_stack.back())) { - // discard object - *ref_stack.back() = discarded; + // discard object, unless this slot holds a duplicate key's + // previous value pending restoration, in which case that + // value is restored instead of being discarded + if (!resolve_duplicate_key_stash(ref_stack.back(), true)) + { + *ref_stack.back() = discarded; #if JSON_DIAGNOSTIC_POSITIONS - // Set start/end positions for discarded object. - handle_diagnostic_positions_for_json_value(*ref_stack.back()); + // Set start/end positions for discarded object. + handle_diagnostic_positions_for_json_value(*ref_stack.back()); #endif + } } else { @@ -617,18 +678,25 @@ class json_sax_dom_callback_parser #endif ref_stack.back()->set_parents(); + // this object is finally, definitively kept; drop any + // pending duplicate-key stash entry for its slot since it + // can no longer be restored + resolve_duplicate_key_stash(ref_stack.back(), false); } } JSON_ASSERT(!ref_stack.empty()); JSON_ASSERT(!keep_stack.empty()); + JSON_ASSERT(!container_key_stack.empty()); ref_stack.pop_back(); keep_stack.pop_back(); + const string_t object_key = std::move(container_key_stack.back()); + container_key_stack.pop_back(); if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_structured()) { // remove discarded value - remove_discarded_value(*ref_stack.back()); + remove_discarded_value(*ref_stack.back(), object_key); } return true; @@ -639,6 +707,9 @@ class json_sax_dom_callback_parser const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::array_start, discarded); keep_stack.push_back(keep); + // see start_object() + container_key_stack.push_back(current_key()); + auto val = handle_value(BasicJsonType::value_t::array, true); ref_stack.push_back(val.second); @@ -659,7 +730,12 @@ class json_sax_dom_callback_parser // check array limit if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + } + + if (len != detail::unknown_size()) + { + reserve_array(*ref_stack.back()->m_data.m_value.array, len, priority_tag<1> {}); } } @@ -686,23 +762,35 @@ class json_sax_dom_callback_parser #endif ref_stack.back()->set_parents(); + // this array is finally, definitively kept; drop any + // pending duplicate-key stash entry for its slot since it + // can no longer be restored + resolve_duplicate_key_stash(ref_stack.back(), false); } else { - // discard array - *ref_stack.back() = discarded; + // discard array, unless this slot holds a duplicate key's + // previous value pending restoration, in which case that + // value is restored instead of being discarded + if (!resolve_duplicate_key_stash(ref_stack.back(), true)) + { + *ref_stack.back() = discarded; #if JSON_DIAGNOSTIC_POSITIONS - // Set start/end positions for discarded array. - handle_diagnostic_positions_for_json_value(*ref_stack.back()); + // Set start/end positions for discarded array. + handle_diagnostic_positions_for_json_value(*ref_stack.back()); #endif + } } } JSON_ASSERT(!ref_stack.empty()); JSON_ASSERT(!keep_stack.empty()); + JSON_ASSERT(!container_key_stack.empty()); ref_stack.pop_back(); keep_stack.pop_back(); + const string_t object_key = std::move(container_key_stack.back()); + container_key_stack.pop_back(); // remove discarded value if (!ref_stack.empty() && ref_stack.back()) @@ -716,7 +804,7 @@ class json_sax_dom_callback_parser // the array is either still stored under its key or was never // stored, leaving the placeholder key() wrote; both show up as // a discarded member of the parent object - remove_discarded_value(*ref_stack.back()); + remove_discarded_value(*ref_stack.back(), object_key); } } @@ -809,15 +897,92 @@ class json_sax_dom_callback_parser } #endif - /// remove the discarded value the callback rejected from its parent - static void remove_discarded_value(BasicJsonType& parent) + /// if there is a pending duplicate-key stash entry for this exact slot, + /// remove it from the stash; if restore_value is true, the stashed + /// previous value is moved back into the slot first (use this when the + /// new value at that slot was rejected); otherwise the stash entry is + /// simply dropped (use this when the new value was accepted, so it + /// correctly supersedes the old one and no restore should ever happen + /// for this slot again) + /// @return whether a matching stash entry was found (and processed) + bool resolve_duplicate_key_stash(BasicJsonType* slot, bool restore_value) { - for (auto it = parent.begin(); it != parent.end(); ++it) + const auto it = std::find_if(duplicate_key_stash.begin(), duplicate_key_stash.end(), + [slot](const std::pair& entry) { - if (it->is_discarded()) + return entry.first == slot; + }); + + if (it == duplicate_key_stash.end()) + { + return false; + } + + if (restore_value) + { + *slot = std::move(it->second); + } + duplicate_key_stash.erase(it); + return true; + } + + /*! + @brief the key the value now being handled will be stored under + + Empty unless the enclosing container is an object, in which case it is the + key of the pending key() event. Read before handle_value() consumes that + key, so it is also correct when the value never reaches its parent. + */ + string_t current_key() const + { + if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_object() + && !key_stack.empty()) + { + return key_stack.back(); + } + return string_t{}; + } + + /*! + @brief remove the discarded value the callback rejected from its parent, + unless it is a duplicate key's slot with a stashed previous value, in + which case that previous value is restored instead + + A rejected value can only ever be the one most recently added to @a parent: + the last element of an array, or the placeholder key() stored under @a key + in an object. Looking there directly makes this O(1) resp. O(log n), where + searching @a parent for it made a filtering parse quadratic in the number of + members of a single container. + + Finding no discarded value there means none was stored in the first place - + the callback rejected the value before it reached its parent - so there is + nothing to remove. + + @param[in,out] parent the container to remove the rejected value from + @param[in] key the key the value was stored under; unused for arrays + */ + void remove_discarded_value(BasicJsonType& parent, const string_t& key) + { + if (parent.is_array()) + { + auto& array = *parent.m_data.m_value.array; + if (!array.empty() && array.back().is_discarded()) { - parent.erase(it); - break; + array.pop_back(); + } + } + else if (parent.is_object()) + { + auto& object = *parent.m_data.m_value.object; + const auto it = object.find(key); + if (it != object.end() && it->second.is_discarded()) + { + // a duplicate key's slot has a stashed previous value that + // must be restored instead of being erased + if (!resolve_duplicate_key_stash(&it->second, true)) + { + object.erase(it); + } } } } @@ -867,11 +1032,14 @@ class json_sax_dom_callback_parser if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_object()) { JSON_ASSERT(!key_keep_stack.empty()); + JSON_ASSERT(!key_stack.empty()); const bool placeholder_stored = key_keep_stack.back(); key_keep_stack.pop_back(); + const string_t key = std::move(key_stack.back()); + key_stack.pop_back(); if (placeholder_stored) { - remove_discarded_value(*ref_stack.back()); + remove_discarded_value(*ref_stack.back(), key); } } return {false, nullptr}; @@ -904,8 +1072,10 @@ class json_sax_dom_callback_parser JSON_ASSERT(ref_stack.back()->is_object()); // check if we should store an element for the current key JSON_ASSERT(!key_keep_stack.empty()); + JSON_ASSERT(!key_stack.empty()); const bool store_element = key_keep_stack.back(); key_keep_stack.pop_back(); + key_stack.pop_back(); if (!store_element) { @@ -914,6 +1084,16 @@ class json_sax_dom_callback_parser JSON_ASSERT(object_element); *object_element = std::move(value); + if (!skip_callback) + { + // this scalar value finally, definitively replaces whatever was + // at this slot; drop any pending duplicate-key stash entry for + // it since it can no longer be restored (a container value at + // this slot is resolved later, in end_object()/end_array(), + // since skip_callback is true for the placeholder handling that + // happens here for those) + resolve_duplicate_key_stash(object_element, false); + } return {true, object_element}; } @@ -925,8 +1105,20 @@ class json_sax_dom_callback_parser std::vector keep_stack {}; // NOLINT(readability-redundant-member-init) /// stack to manage which object keys to keep std::vector key_keep_stack {}; // NOLINT(readability-redundant-member-init) + /// the keys key() stored a placeholder for, in lockstep with key_keep_stack + std::vector key_stack {}; // NOLINT(readability-redundant-member-init) + /// for each open container, the key it is stored under in its parent + /// object, in lockstep with ref_stack; unused where the parent is not an + /// object + std::vector container_key_stack {}; // NOLINT(readability-redundant-member-init) /// helper to hold the reference for the next object element BasicJsonType* object_element = nullptr; + /// stash of (slot pointer, previous value) for object members that + /// already existed when key() was called again for the same key + /// (duplicate keys); used to restore the previous value if the new + /// value is later rejected by the callback, instead of erasing the + /// member entirely + std::vector> duplicate_key_stash {}; /// whether a syntax error occurred bool errored = false; /// callback function diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index ff00facb4..754eb341e 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -19,7 +19,9 @@ #include // vector #include +#include #include +#include #include #include @@ -143,6 +145,25 @@ constexpr bool input_adapter_supports_lookahead(std::false_type /*detected*/) return false; } +// Detect whether an input adapter exposes a contiguous byte block that the +// lexer can scan directly (see iterator_input_adapter::supports_bulk_scan). +// Adapters without the flag - file, stream, wide-string, user-defined - fall +// back to the character-at-a-time string scanner. +template +using detect_supports_bulk_scan = decltype(InputAdapterType::supports_bulk_scan); + +template +constexpr bool input_adapter_supports_bulk_scan(std::true_type /*detected*/) +{ + return InputAdapterType::supports_bulk_scan; +} + +template +constexpr bool input_adapter_supports_bulk_scan(std::false_type /*detected*/) +{ + return false; +} + /*! @brief lexical analysis @@ -170,13 +191,22 @@ class lexer : public lexer_base static constexpr bool can_release_lookahead = input_adapter_supports_lookahead(is_detected {}); + /// whether string scanning may bulk-consume runs of ordinary characters + /// directly from a contiguous input buffer (SWAR fast path). This requires + /// the token to be reconstructible lazily (lazy_token_string), so bypassing + /// the per-character capture in get() cannot lose error diagnostics. + static constexpr bool bulk_scan = + lazy_token_string + && input_adapter_supports_bulk_scan(is_detected {}); + public: using token_type = typename lexer_base::token_type; - explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept + explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept : ia(std::move(adapter)) , ignore_comments(ignore_comments_) , decimal_point_char(static_cast(get_decimal_point())) + , discard_number_values(discard_number_values_) {} // deleted because of pointer members @@ -289,6 +319,40 @@ class lexer : public lexer_base return true; } + /// contiguous input: bulk-append the run of ordinary characters and complete + /// well-formed UTF-8 sequences starting at the current read position, leaving + /// the first byte that needs individual handling (the closing quote, an + /// escape, a control character, or an ill-formed UTF-8 byte) for get() + void scan_string_bulk(std::true_type /*bulk*/) + { + // a pending unget must be consumed through the normal path first + if (next_unget) + { + return; + } + const std::size_t remaining = ia.bulk_remaining(); + if (remaining == 0) + { + return; + } + const auto* const data = reinterpret_cast(ia.bulk_data()); + + const std::size_t pos = string_bulk_run(data, remaining); + if (pos == 0) + { + return; + } + token_buffer.append(reinterpret_cast(data), pos); + ia.bulk_skip(pos); + // the run contains no newline (all bytes < 0x20 are treated as special), + // so only the flat character counters advance + position.chars_read_total += pos; + position.chars_read_current_line += pos; + } + + /// streaming input: no bulk fast path + void scan_string_bulk(std::false_type /*bulk*/) const noexcept {} + /*! @brief scan a string literal @@ -314,6 +378,10 @@ class lexer : public lexer_base while (true) { + // bulk-consume ordinary characters from contiguous input, then + // handle the next special byte through the switch below + scan_string_bulk(std::integral_constant {}); + // get the next character switch (get()) { @@ -908,7 +976,9 @@ class lexer : public lexer_base case '\n': case '\r': case char_traits::eof(): +#if !JSON_STRICT_NUL_HANDLING case '\0': +#endif return true; default: @@ -926,8 +996,10 @@ class lexer : public lexer_base { switch (get()) { - case char_traits::eof(): +#if !JSON_STRICT_NUL_HANDLING case '\0': +#endif + case char_traits::eof(): { error_message = "invalid comment; missing closing '*/'"; return false; @@ -1032,6 +1104,12 @@ class lexer : public lexer_base // changed if minus sign, decimal point, or exponent is read token_type number_type = token_type::value_unsigned; + // offset just past the last mantissa byte in token_buffer (i.e. the + // index of 'e'/'E', or the whole token when there is no exponent). + // convert_number() uses it to count significant digits; npos means + // "not seen an exponent yet" and is resolved at scan_number_done + std::size_t mantissa_end = std::string::npos; + // state (init): we just found out we need to scan a number switch (current) { @@ -1217,6 +1295,9 @@ scan_number_decimal2: scan_number_exponent: // we just parsed an exponent number_type = token_type::value_float; + // this label is reached only right after the 'e'/'E' was appended (from + // the zero, any1, and decimal2 states), so the mantissa ends before it + mantissa_end = token_buffer.size() - 1; switch (get()) { case '+': @@ -1303,45 +1384,199 @@ scan_number_done: // we are done scanning a number) unget(); - char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) - errno = 0; + // no exponent was scanned: the mantissa spans the whole token + if (mantissa_end == std::string::npos) + { + mantissa_end = token_buffer.size(); + } - // try to parse integers first and fall back to floats + return convert_number(number_type, mantissa_end); + } + + /*! + @brief convert an already-validated integer token to its value + + The digit sequence in [first, last) has been validated by the caller, so a + dedicated parser can avoid the locale/errno overhead of std::strtoull. + + @return the token type on success; token_type::uninitialized if @a + number_type is not an integer type or the value does not fit, in + which case the caller falls back to the floating-point conversion + (matching the previous std::strtoull/std::strtoll behavior) + */ + token_type convert_integer(token_type number_type, const char* first, const char* last) + { if (number_type == token_type::value_unsigned) { - const auto x = std::strtoull(token_buffer.data(), &endptr, 10); - - // we checked the number format before - JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); - - if (errno != ERANGE) + if (parse_integer_unsigned(first, last, value_unsigned)) { - value_unsigned = static_cast(x); - if (value_unsigned == x) - { - return token_type::value_unsigned; - } + return token_type::value_unsigned; } } else if (number_type == token_type::value_integer) { - const auto x = std::strtoll(token_buffer.data(), &endptr, 10); - - // we checked the number format before - JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); - - if (errno != ERANGE) + if (parse_integer_signed(first, last, value_integer)) { - value_integer = static_cast(x); - if (value_integer == x) - { - return token_type::value_integer; - } + return token_type::value_integer; + } + } + + return token_type::uninitialized; + } + + /*! + @brief check whether Clinger's fast path can still succeed for this token + + parse_float_fast() needs a significand below 2^53. A mantissa with 17 or + more significant digits is at least 10^16 and therefore always exceeds it, + so calling the fast path would walk the token one extra time only to + decline before strtod has to run anyway. + + Significant digits are the mantissa's digits from the first nonzero one on; + the sign, the decimal point, leading zeros, and the exponent do not count. + The answer is derived from indices - the digits are not scanned again - so + this stays off the hot path of the number scanners. + + @param[in] mantissa_end offset just past the last mantissa byte in + token_buffer + @return false if parse_float_fast() is guaranteed to decline + */ + bool mantissa_fits_clinger(std::size_t mantissa_end) const + { + // 10^16 already exceeds 2^53, so 17 digits can never fit + constexpr std::size_t limit = 17; + + const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u; + const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u; + // the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so + // a leading zero can only be a lone "0", which is not significant + const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u; + JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero); + std::size_t digits = mantissa_end - neg - has_dot - lead_zero; + + if (JSON_HEDLEY_LIKELY(digits < limit)) + { + return true; + } + + // Only a number below 1 can carry further insignificant zeros, and only + // while the count stays at the limit does removing them change the + // answer - so this loop is skipped for all but a few tokens. Note + // token_buffer holds the locale's decimal point, so the fraction is + // located through decimal_point_position rather than by searching '.'. + if (lead_zero != 0) + { + JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit + for (std::size_t i = decimal_point_position + 1; + digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i) + { + --digits; + } + } + + return digits < limit; + } + + /*! + @brief convert the number text in token_buffer to its value and token type + + The digit sequence in token_buffer has already been validated (by the + scan_number() state machine or by the contiguous fast path) and holds the + locale decimal point in place of '.'. Integers are parsed first and fall + back to floating point on overflow. This is shared so both scanners produce + identical results. + + @param[in] mantissa_end offset just past the last mantissa byte in + token_buffer (the index of 'e'/'E', or + token_buffer.size() when there is no exponent); + used to skip Clinger's fast path when it cannot + possibly succeed - see mantissa_fits_clinger() + */ + token_type convert_number(token_type number_type, std::size_t mantissa_end) + { + // If the caller does not need the converted value (only whether the + // input is syntactically valid; see json_sax_acceptor/accept()), an + // unsigned/integer token can be reported without calling + // strtoull()/strtoll() at all, *provided* we can already tell from + // the digit count alone that the conversion cannot overflow 64 bits. + // Such tokens are always finite and are accepted unconditionally by + // the parser regardless of their actual value (parser::sax_parse_internal() + // never checks finiteness for value_unsigned/value_integer), so the + // classification below is all that is needed. + // + // A decimal number with up to 18 digits is always representable in + // both std::uint64_t and std::int64_t (18 nines is ~1e18, well below + // both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll() + // could not have set errno to ERANGE for it. Numbers with more digits + // (rare in practice) fall through to the exact code below, unchanged, + // so their handling -- including reclassification to value_float when + // the value overflows 64 bits, and rejection when it is not even + // finite as a double -- is bit-for-bit identical to before this + // optimization. + // + // Note this reasons about std::uint64_t/std::int64_t, not about + // number_unsigned_t/number_integer_t (BasicJsonType's own, possibly + // narrower, template parameters -- e.g. std::uint32_t). That is fine + // *only* because discard_number_values is exclusively set by + // accept() (see json.hpp), and accept() always parses through the + // library's own json_sax_acceptor -- never a user-supplied SAX + // consumer -- whose number_unsigned()/number_integer()/number_float() + // callbacks unconditionally discard their argument and return true. + // So for every caller that can reach this branch, neither the token + // classification below nor the eventual (possibly narrowed, and on + // this fast path left stale/unset) value_unsigned/value_integer is + // ever consulted -- an unsigned/integer token is accepted outright, + // and even a >18-digit token that this fast path deliberately falls + // through for is, once reclassified to value_float, still finite + // (and thus accepted) for any digit count that fits in number_unsigned_t + // or number_integer_t regardless of that type's width. If this + // function is ever taught to run with discard_number_values true for + // a caller that *does* read the converted value, this reasoning (and + // the fast path below) would need to be revisited. + if (discard_number_values) + { + constexpr std::size_t safe_digit_count = 18; + if (number_type == token_type::value_unsigned && token_buffer.size() <= safe_digit_count) + { + return token_type::value_unsigned; + } + if (number_type == token_type::value_integer && token_buffer.size() - 1 <= safe_digit_count) + { + return token_type::value_integer; + } + } + + const char* const num_begin = token_buffer.data(); + const char* const num_end = num_begin + token_buffer.size(); + + if (number_type != token_type::value_float) + { + const token_type integer_result = convert_integer(number_type, num_begin, num_end); + if (integer_result != token_type::uninitialized) + { + return integer_result; } } // this code is reached if we parse a floating-point number or if an - // integer conversion above failed + // integer conversion above overflowed. Prefer std::from_chars + // (Eisel-Lemire, locale-independent, correctly rounded) when available; + // otherwise the exact Clinger fast path (double only); otherwise the + // locale-aware strtof/strtod. + if (parse_float_from_chars(num_begin, num_end, value_float)) + { + return token_type::value_float; + } + // Skipping a fast path that cannot succeed is lossless and saves a full + // extra pass over the token's bytes, which otherwise shows up on + // high-precision inputs such as canada.json + if (mantissa_fits_clinger(mantissa_end) + && parse_float_fast(num_begin, num_end, decimal_point_char, value_float)) + { + return token_type::value_float; + } + + char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) strtof(value_float, token_buffer.data(), &endptr); // we checked the number format before @@ -1350,6 +1585,158 @@ scan_number_done: return token_type::value_float; } + /*! + @brief contiguous fast path for scanning a number + + Parses the whole number token straight from the input buffer, avoiding the + per-character get()/add() of scan_number(). On success it fills token_buffer + (with the locale decimal point substituted, as scan_number() does) and + returns the token type. On anything it does not fully recognize as a + well-formed number it makes no state change and returns + token_type::uninitialized, so the caller falls back to scan_number(), which + then produces the exact diagnostic. @a current is the first digit or the + leading minus (already read); the remaining bytes are taken from the adapter. + */ + token_type scan_number_bulk_contiguous() + { + // a pending unget offsets the buffer position from current; fall back + if (next_unget) + { + return token_type::uninitialized; + } + const std::size_t rem = ia.bulk_remaining(); + if (rem == 0) + { + // the first digit is the last input byte; let scan_number() finish + return token_type::uninitialized; + } + // the byte before the next unread one is current (contiguous input) + const char* const data = reinterpret_cast(ia.bulk_data()) - 1; + const std::size_t avail = rem + 1; + + // validate + classify the number extent (mirrors scan_number()'s grammar) + std::size_t i = 0; + std::size_t dot_index = std::string::npos; + token_type number_type = token_type::value_unsigned; + if (data[0] == '-') + { + number_type = token_type::value_integer; + i = 1; + if (i >= avail) + { + return token_type::uninitialized; + } + } + if (data[i] == '0') + { + ++i; + } + else if (data[i] >= '1' && data[i] <= '9') + { + ++i; + while (i < avail && data[i] >= '0' && data[i] <= '9') + { + ++i; + } + } + else + { + return token_type::uninitialized; + } + if (i < avail && data[i] == '.') + { + number_type = token_type::value_float; + dot_index = i; + ++i; + if (i >= avail || !(data[i] >= '0' && data[i] <= '9')) + { + return token_type::uninitialized; + } + while (i < avail && data[i] >= '0' && data[i] <= '9') + { + ++i; + } + } + // the mantissa ends here, whether or not an exponent part follows + const std::size_t mantissa_end = i; + if (i < avail && (data[i] == 'e' || data[i] == 'E')) + { + number_type = token_type::value_float; + ++i; + if (i < avail && (data[i] == '+' || data[i] == '-')) + { + ++i; + } + if (i >= avail || !(data[i] >= '0' && data[i] <= '9')) + { + return token_type::uninitialized; + } + while (i < avail && data[i] >= '0' && data[i] <= '9') + { + ++i; + } + } + const std::size_t len = i; + + // reset() records where this token starts (for diagnostics), so it has + // to run before the input position advances below + reset(); + + // An integer token needs no token_buffer: the SAX callbacks for + // number_integer/number_unsigned take only the value, and the overflow + // diagnostic rebuilds the text from the input. Convert straight from the + // input buffer and leave token_buffer empty. (JSON_DIAGNOSTIC_POSITIONS + // derives a number's start position from get_string().size(), so there + // the token still has to be materialized.) +#if !JSON_DIAGNOSTIC_POSITIONS + if (number_type != token_type::value_float) + { + const token_type integer_result = convert_integer(number_type, data, data + len); + if (JSON_HEDLEY_LIKELY(integer_result != token_type::uninitialized)) + { + ia.bulk_skip(len - 1); + position.chars_read_total += (len - 1); + position.chars_read_current_line += (len - 1); + return integer_result; + } + // The value does not fit an integer, so this token converts as a + // float. Recording that here keeps convert_number() below from + // repeating the integer attempt that just failed. + number_type = token_type::value_float; + } +#endif + + // materialize the token exactly as scan_number() would, substituting the + // locale decimal point so convert_number()'s strtof fallback stays valid. + // reset() already cleared token_buffer, so append() fills it (assign() is + // avoided because custom string_t types need not provide it) + token_buffer.append(reinterpret_cast(data), len); + if (dot_index != std::string::npos) + { + token_buffer[dot_index] = static_cast(decimal_point_char); + decimal_point_position = dot_index; + } + + ia.bulk_skip(len - 1); + position.chars_read_total += (len - 1); + position.chars_read_current_line += (len - 1); + + return convert_number(number_type, mantissa_end); + } + + /// contiguous input: try the number fast path, else the byte-path scanner + token_type scan_number_dispatch(std::true_type /*bulk*/) + { + const token_type t = scan_number_bulk_contiguous(); + return (t != token_type::uninitialized) ? t : scan_number(); + } + + /// streaming input: always use the byte-path scanner + token_type scan_number_dispatch(std::false_type /*bulk*/) + { + return scan_number(); + } + /*! @param[in] literal_text the literal text to expect @param[in] length the length of the passed literal text @@ -1417,8 +1804,7 @@ scan_number_done: */ char_int_type get() { - ++position.chars_read_total; - ++position.chars_read_current_line; + advance_position(); if (next_unget) { @@ -1430,6 +1816,23 @@ scan_number_done: current = ia.get_character(); } + return track_after_read(); + } + + /// shared head of get() / get_ignoring_pending_unget(): bump the + /// per-character position counters (line-count-on-'\n' bookkeeping is + /// handled afterwards, in track_after_read(), once `current` is known) + void advance_position() noexcept + { + ++position.chars_read_total; + ++position.chars_read_current_line; + } + + /// shared tail of get() / get_ignoring_pending_unget(): capture the + /// character for error messages (if needed) and update line/column + /// bookkeeping for the character now in `current` + char_int_type track_after_read() + { // seekable adapters reconstruct the token lazily on error (see // get_token_string), so the eager per-character copy is skipped capture_char(std::integral_constant {}); @@ -1437,12 +1840,38 @@ scan_number_done: if (current == '\n') { ++position.lines_read; + // remember the column the newline was read at: chars_read_current_line + // is about to be cleared, and a matching unget() cannot reconstruct it + chars_read_before_newline = position.chars_read_current_line; position.chars_read_current_line = 0; } return current; } + /*! + @brief like get(), but for call sites that can prove no unget() is pending + + get() has to check the `next_unget` flag on every call, because a + previous token may have ended with unget() (e.g. scan_number() always + ungets the character that terminated the number, so the next call to + scan() can see it again). skip_whitespace() reads that first, + possibly-ungotten character via a plain get(), but every further + character it reads is guaranteed to be a fresh read: nothing between + those calls invokes unget(). This variant skips the (otherwise always + false) next_unget branch for those calls; it is not a general + replacement for get(). + */ + char_int_type get_ignoring_pending_unget() + { + JSON_ASSERT(!next_unget); + + advance_position(); + current = ia.get_character(); + + return track_after_read(); + } + /// seekable adapter: nothing to capture, the token is rebuilt on error void capture_char(std::true_type /*lazy*/) const noexcept {} @@ -1470,12 +1899,20 @@ scan_number_done: --position.chars_read_total; // in case we "unget" a newline, we have to also decrement the lines_read + // and restore the column that get() cleared when it saw the newline; + // chars_read_current_line == 0 can only mean the last get() read one if (position.chars_read_current_line == 0) { if (position.lines_read > 0) { --position.lines_read; } + + // chars_read_before_newline counts the newline itself, which is the + // character being ungotten, hence the -1 + position.chars_read_current_line = (chars_read_before_newline > 0) + ? chars_read_before_newline - 1 + : 0; } else { @@ -1674,13 +2111,37 @@ scan_number_done: return true; } + /// whether `current` is one of the four JSON whitespace characters + bool current_is_whitespace() const noexcept + { + return current == ' ' || current == '\t' || current == '\n' || current == '\r'; + } + void skip_whitespace() { + // the first character may be a pending unget() left over from the + // previous token (see get_ignoring_pending_unget()); every + // subsequent character read by this loop is guaranteed fresh, since + // nothing below calls unget() + get(); + + if (!current_is_whitespace()) + { + return; + } + + // this is written as an if-guarded do-while (rather than a plain + // while loop) because that shape is what lets both GCC and Clang + // keep the input adapter's read pointer in a register across + // iterations; the equivalent while-loop measurably defeated that + // optimization in testing, turning long whitespace runs (e.g. the + // indentation of pretty-printed JSON) from a register-only loop + // into one that reloads the pointer from memory every character do { - get(); + get_ignoring_pending_unget(); } - while (current == ' ' || current == '\t' || current == '\n' || current == '\r'); + while (current_is_whitespace()); } token_type scan() @@ -1756,11 +2217,14 @@ scan_number_done: case '7': case '8': case '9': - return scan_number(); + return scan_number_dispatch(std::integral_constant {}); - // end of input (the null byte is needed when parsing from - // string literals) +#if !JSON_STRICT_NUL_HANDLING case '\0': +#endif + // end of input; by default, a null byte is also treated as end of + // input for backwards compatibility (see JSON_STRICT_NUL_HANDLING + // to opt into rejecting a null byte in the input instead) case char_traits::eof(): return token_type::end_of_input; @@ -1787,6 +2251,10 @@ scan_number_done: /// the start position of the current token position_t position {}; + /// the value chars_read_current_line had when the last newline was read, so + /// that unget() can restore the column instead of leaving it at 0 + std::size_t chars_read_before_newline = 0; + /// raw input token string for error messages; only populated for streaming /// adapters (seekable adapters reconstruct it lazily via token_string_start) std::vector token_string {}; @@ -1816,6 +2284,13 @@ scan_number_done: const char_int_type decimal_point_char = '.'; /// the position of the decimal point in the input std::size_t decimal_point_position = std::string::npos; + + /// whether the caller (e.g. accept()/json_sax_acceptor) only needs the + /// token classification and never looks at the converted numeric value; + /// when set, scan_number() may skip strtoull()/strtoll() for + /// value_unsigned/value_integer tokens whose digit count guarantees they + /// fit into 64 bits (see scan_number()) + const bool discard_number_values = false; }; } // namespace detail diff --git a/include/nlohmann/detail/input/number_parse.hpp b/include/nlohmann/detail/input/number_parse.hpp new file mode 100644 index 000000000..e50c3f67f --- /dev/null +++ b/include/nlohmann/detail/input/number_parse.hpp @@ -0,0 +1,302 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // array +#include // FLT_EVAL_METHOD +#include // size_t +#include // int64_t, uint64_t +#include // numeric_limits + +#include + +// std::from_chars lives in , but being in C++17 mode does not +// guarantee the header exists: GCC 7 sets __cplusplus to C++17 yet ships no +// (added in GCC 8; floating-point support in GCC 11). Guard the +// include with __has_include so such toolchains fall back to the scalar path. +#if defined(JSON_HAS_CPP_17) && defined(__has_include) + #if __has_include() + #include // from_chars (only used when __cpp_lib_to_chars is defined) + #include // errc + #endif +#endif + +// This file contains the value-conversion helpers used by the lexer to turn an +// already-validated number token into a value, without the locale/errno +// overhead of std::strtoull/std::strtod. They are free functions so the lexer +// stays focused on scanning; see lexer::convert_number(). + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief fast integer parser for an already-validated unsigned integer + +The number scanner has already checked that [first, last) is a valid JSON +integer, so this only needs to accumulate the digits and detect overflow. This +avoids the locale/errno machinery of std::strtoull, which dominates +integer-heavy inputs. + +@param[in] first pointer to the first character (a digit) +@param[in] last pointer past the last character +@param[out] value the parsed value on success +@return true if the value fit into @a NumberUnsignedType; false on overflow, in + which case the caller falls back to floating-point parsing (matching the + previous std::strtoull behavior) +*/ +template +bool parse_integer_unsigned(const char* first, const char* last, NumberUnsignedType& value) noexcept +{ + // accumulate in the widest unsigned type used by the previous strtoull + // path so the overflow behavior is unchanged for custom number types + std::uint64_t x = 0; + constexpr std::uint64_t cutoff = (std::numeric_limits::max)() / 10u; + constexpr std::uint64_t cutlim = (std::numeric_limits::max)() % 10u; + for (const char* p = first; p != last; ++p) + { + const auto digit = static_cast(static_cast(*p) - static_cast('0')); + if (JSON_HEDLEY_UNLIKELY(x > cutoff || (x == cutoff && digit > cutlim))) + { + return false; + } + x = (x * 10u) + digit; + } + value = static_cast(x); + // reject values that do not round-trip into a narrower NumberUnsignedType + return static_cast(value) == x; +} + +/*! +@brief fast integer parser for an already-validated negative integer + +@param[in] first pointer to the leading '-' +@param[in] last pointer past the last character +@param[out] value the parsed (negative) value on success +@return true on success; false on overflow (caller falls back to float) +*/ +template +bool parse_integer_signed(const char* first, const char* last, NumberIntegerType& value) noexcept +{ + // the state machine only reaches the signed path via a leading '-' + JSON_ASSERT(first != last && *first == '-'); + std::uint64_t magnitude = 0; + // |INT64_MIN| == INT64_MAX + 1; this is the largest admissible magnitude + constexpr std::uint64_t limit = static_cast((std::numeric_limits::max)()) + 1u; + for (const char* p = first + 1; p != last; ++p) + { + const auto digit = static_cast(static_cast(*p) - static_cast('0')); + if (JSON_HEDLEY_UNLIKELY(magnitude > (limit - digit) / 10u)) + { + return false; + } + magnitude = (magnitude * 10u) + digit; + } + const std::int64_t x = (magnitude == limit) + ? (std::numeric_limits::min)() + : -static_cast(magnitude); + value = static_cast(x); + // reject values that do not round-trip into a narrower NumberIntegerType + return static_cast(value) == x; +} + +/*! +@brief exact fast path for parsing a `double` (Clinger's algorithm) + +For the common case - at most 19 significant digits, a decimal exponent in +[-22, 22], and a significand below 2^53 - the value equals significand * +10^exp computed in IEEE-754 double arithmetic, which is exact under +round-to-nearest because both operands are exactly representable. This is the +same fast path used by fast_float/simdjson; the general cases are left to +std::strtod. The parser only activates for number_float_t == double; float and +long double keep the std::strtof/std::strtold paths (see the templated overload +below). + +@param[in] first pointer to the first character of the number +@param[in] last pointer past the last character +@param[in] decimal_point the (locale-dependent) decimal point character +@param[out] out the parsed value on success +@return true if the value was parsed exactly; false to fall back to strtod +*/ +template +bool parse_float_fast(const char* first, const char* last, DecimalPointType decimal_point, double& out) noexcept +{ +#if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0 + // Clinger's fast path is only exact when double operations are evaluated in + // true double precision. On platforms that keep intermediates in extended + // precision (e.g. the x87 FPU on 32-bit x86, where FLT_EVAL_METHOD == 2) the + // single significand * 10^scale step is double-rounded and can be 1 ULP off, + // so decline and let the caller fall back to the correctly-rounded + // std::from_chars / std::strtod path. + static_cast(first); + static_cast(last); + static_cast(decimal_point); + static_cast(out); + return false; +#else + static const std::array powers_of_ten = + { + { + 1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11, + 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22 + } + }; + + const char* p = first; + bool negative = false; + if (p != last && (*p == '-' || *p == '+')) + { + negative = (*p == '-'); + ++p; + } + + std::uint64_t significand = 0; + int num_digits = 0; + int fractional_digits = 0; + bool seen_dot = false; + bool any_digit = false; + for (; p != last; ++p) + { + const char c = *p; + if (c >= '0' && c <= '9') + { + any_digit = true; + if (JSON_HEDLEY_UNLIKELY(num_digits >= 19)) + { + return false; // significand may not fit into uint64_t + } + significand = (significand * 10u) + static_cast(c - '0'); + ++num_digits; + fractional_digits += static_cast(seen_dot); + } + else if (static_cast(c) == decimal_point) + { + if (JSON_HEDLEY_UNLIKELY(seen_dot)) + { + return false; + } + seen_dot = true; + } + else if (c == 'e' || c == 'E') + { + ++p; + break; + } + else + { + return false; + } + } + if (JSON_HEDLEY_UNLIKELY(!any_digit)) + { + return false; + } + + int exponent = 0; + if (p != last) // an exponent part remains + { + bool exp_negative = false; + if (p != last && (*p == '-' || *p == '+')) + { + exp_negative = (*p == '-'); + ++p; + } + bool any_exp_digit = false; + for (; p != last; ++p) + { + if (JSON_HEDLEY_UNLIKELY(*p < '0' || *p > '9')) + { + return false; + } + exponent = (exponent * 10) + (*p - '0'); + any_exp_digit = true; + if (JSON_HEDLEY_UNLIKELY(exponent > 9999)) + { + return false; + } + } + if (JSON_HEDLEY_UNLIKELY(!any_exp_digit)) + { + return false; + } + if (exp_negative) + { + exponent = -exponent; + } + } + + const int scale = exponent - fractional_digits; + if (JSON_HEDLEY_UNLIKELY(significand >= (static_cast(1) << 53))) + { + return false; // significand not exactly representable as double + } + + auto result = static_cast(significand); + if (scale >= 0) + { + if (JSON_HEDLEY_UNLIKELY(scale > 22)) + { + return false; + } + result *= powers_of_ten[static_cast(scale)]; + } + else + { + if (JSON_HEDLEY_UNLIKELY(-scale > 22)) + { + return false; + } + result /= powers_of_ten[static_cast(-scale)]; + } + out = negative ? -result : result; + return true; +#endif +} + +/// fast float path is only exact for `double`; decline for float/long double +template +bool parse_float_fast(const char* /*first*/, const char* /*last*/, DecimalPointType /*decimal_point*/, FloatType& /*out*/) noexcept +{ + return false; +} + +/*! +@brief parse a float with std::from_chars (Eisel-Lemire) when available + +std::from_chars is locale-independent, correctly rounded, and - via the +Eisel-Lemire algorithm in modern standard libraries - much faster than strtod +over the whole value range (not just the Clinger subset). It is used only when +__cpp_lib_to_chars indicates full floating-point support and only when it +consumes the entire token ([first, last)); a partial parse means the buffer +uses a non-'.' locale decimal point, in which case the caller falls back to the +locale-aware path. An under-/overflow (result_out_of_range) also declines, so +the caller's strtod fallback supplies the well-defined ±inf/0 result the parser +expects (side-stepping the P4168 divergence between implementations). + +@return true if the value was parsed exactly and fully; false to fall back +*/ +template +bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept +{ + // JSON_HAS_CPP_17 must gate the use as well as the include above: + // some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even + // in C++14 mode, where is not included. +#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars) + const auto result = std::from_chars(first, last, out); + return result.ec == std::errc() && result.ptr == last; +#else + static_cast(first); + static_cast(last); + static_cast(out); + return false; +#endif +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/input/parser.hpp b/include/nlohmann/detail/input/parser.hpp index ee0de4616..5fec57a70 100644 --- a/include/nlohmann/detail/input/parser.hpp +++ b/include/nlohmann/detail/input/parser.hpp @@ -72,9 +72,10 @@ class parser parser_callback_t cb = nullptr, const bool allow_exceptions_ = true, const bool ignore_comments = false, - const bool ignore_trailing_commas_ = false) + const bool ignore_trailing_commas_ = false, + const bool discard_number_values_ = false) : callback(std::move(cb)) - , m_lexer(std::move(adapter), ignore_comments) + , m_lexer(std::move(adapter), ignore_comments, discard_number_values_) , allow_exceptions(allow_exceptions_) , ignore_trailing_commas(ignore_trailing_commas_) { diff --git a/include/nlohmann/detail/input/string_scan.hpp b/include/nlohmann/detail/input/string_scan.hpp new file mode 100644 index 000000000..6af0e6c5d --- /dev/null +++ b/include/nlohmann/detail/input/string_scan.hpp @@ -0,0 +1,287 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // size_t +#include // uint64_t +#include // memcpy + +#include + +// Optional SIMD backend for bulk UTF-8 validation. This is an opt-in external +// dependency: nlohmann/json itself stays header-only and the C++11 scalar +// validator below is always available; defining JSON_USE_SIMDUTF additionally +// requires the simdutf headers on the include path and linking the simdutf +// library. See string_bulk_run(). +// +// simdutf.h itself requires C++17 - it rejects older standards with an #error - +// so the backend is only compiled in from C++17 on. Below that the macro has no +// effect and the scalar validator is used; it accepts and rejects exactly the +// same input, so only throughput differs. macro_scope.hpp is included above to +// have JSON_HAS_CPP_17 available for this test. +#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17) + #include +#endif + +// This file contains the byte-level string-scanning helpers used by the lexer's +// contiguous fast path. They operate purely on raw bytes (no dependency on the +// lexer's template parameters) so they are free functions, keeping the lexer +// itself focused on the state machine; see lexer::scan_string_bulk(). + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +// classify a single byte as needing individual string handling: the closing +// quote, an escape, a control character, or a non-ASCII (UTF-8) +// lead/continuation byte. Ordinary bytes (0x20..0x7F except '"' and '\\') are +// copied verbatim, which the bulk scanner does 8 bytes at a time. +inline bool is_string_special(unsigned char c) noexcept +{ + return c == '\"' || c == '\\' || c < 0x20u || c >= 0x80u; +} + +// SWAR helper: return a word whose high bit is set in every byte of @a v that +// is_string_special(); zero if the 8 bytes are all ordinary. +inline std::uint64_t swar_string_special(std::uint64_t v) noexcept +{ + constexpr std::uint64_t ones = 0x0101010101010101ull; + constexpr std::uint64_t high = 0x8080808080808080ull; + const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22) + const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C) + const std::uint64_t has_quote = (q - ones) & ~q & high; + const std::uint64_t has_backslash = (b - ones) & ~b & high; + const std::uint64_t has_control = (v - 0x2020202020202020ull) & ~v & high; // < 0x20 + const std::uint64_t has_non_ascii = v & high; // >= 0x80 + return has_quote | has_backslash | has_control | has_non_ascii; +} + +// return the index of the first is_string_special() byte in [data, data+n), or +// n if every byte is ordinary; scans 8 bytes at a time +inline std::size_t find_string_special(const unsigned char* data, std::size_t n) noexcept +{ + std::size_t i = 0; + for (; i + 8 <= n; i += 8) + { + std::uint64_t word = 0; + std::memcpy(&word, data + i, sizeof(word)); + if (swar_string_special(word) != 0) + { + // a special byte is in this word; locate it (endian-agnostic) + for (std::size_t j = 0; j < 8; ++j) + { + if (is_string_special(data[i + j])) + { + return i + j; + } + } + } + } + for (; i < n; ++i) + { + if (is_string_special(data[i])) + { + return i; + } + } + return n; +} + +// classify a byte as one the serializer must NOT copy verbatim when +// ensure_ascii is requested: the closing quote, an escape, a control character +// (< 0x20), DEL (0x7F), or any non-ASCII byte (>= 0x80). Everything else - +// printable ASCII except '"' and '\\' - is emitted unchanged. Note this differs +// from is_string_special() only in that 0x7F is also a stop (it is escaped as +// \u007f under ensure_ascii). +inline bool is_ascii_copyable(unsigned char c) noexcept +{ + return c >= 0x20u && c < 0x7Fu && c != '"' && c != '\\'; +} + +// return the index of the first byte in [data, data+n) that is NOT +// is_ascii_copyable(), or n if every byte can be copied verbatim; scans 8 bytes +// at a time. Used by the serializer's ensure_ascii fast path. +inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_t n) noexcept +{ + constexpr std::uint64_t ones = 0x0101010101010101ull; + constexpr std::uint64_t high = 0x8080808080808080ull; + std::size_t i = 0; + for (; i + 8 <= n; i += 8) + { + std::uint64_t v = 0; + std::memcpy(&v, data + i, sizeof(v)); + const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22) + const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C) + const std::uint64_t d = v ^ 0x7F7F7F7F7F7F7F7Full; // DEL (0x7F) + const std::uint64_t stop = ((q - ones) & ~q & high) // == '"' + | ((b - ones) & ~b & high) // == '\\' + | ((d - ones) & ~d & high) // == 0x7F + | ((v - 0x2020202020202020ull) & ~v & high) // < 0x20 + | (v & high); // >= 0x80 + if (stop != 0) + { + break; + } + } + for (; i < n; ++i) + { + if (!is_ascii_copyable(data[i])) + { + return i; + } + } + return n; +} + +// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its +// length (2..4) only when the bytes form a *well-formed* sequence using exactly +// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts +// precisely what the byte path accepts. Returns 0 for anything that is invalid, +// incomplete, or that the byte path must diagnose (the caller then defers to +// that path, keeping error messages unchanged). Lead bytes < 0x80 are handled +// by the caller and never passed here. +inline std::size_t validate_one_utf8(const unsigned char* data, std::size_t avail) noexcept +{ + const unsigned char c0 = data[0]; + if (c0 >= 0xC2 && c0 <= 0xDF) // U+0080..U+07FF + { + if (avail >= 2 && data[1] >= 0x80 && data[1] <= 0xBF) + { + return 2; + } + } + else if (c0 == 0xE0) // U+0800..U+0FFF + { + if (avail >= 3 && data[1] >= 0xA0 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF) + { + return 3; + } + } + else if ((c0 >= 0xE1 && c0 <= 0xEC) || c0 == 0xEE || c0 == 0xEF) // U+1000..U+CFFF, U+E000..U+FFFF + { + if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF) + { + return 3; + } + } + else if (c0 == 0xED) // U+D000..U+D7FF (excludes surrogates) + { + if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0x9F && data[2] >= 0x80 && data[2] <= 0xBF) + { + return 3; + } + } + else if (c0 == 0xF0) // U+10000..U+3FFFF + { + if (avail >= 4 && data[1] >= 0x90 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF) + { + return 4; + } + } + else if (c0 >= 0xF1 && c0 <= 0xF3) // U+40000..U+FFFFF + { + if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF) + { + return 4; + } + } + else if (c0 == 0xF4) // U+100000..U+10FFFF + { + if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0x8F && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF) + { + return 4; + } + } + return 0; // invalid, incomplete, or must be diagnosed by the byte path +} + +// Scalar (C++11) computation of the bulk run length: the number of leading +// bytes in [data, data+n) that are ordinary ASCII or complete well-formed UTF-8 +// sequences, stopping before the first byte that needs individual handling (the +// closing quote, an escape, a control character, or an ill-formed/truncated +// sequence). ASCII is skipped 8 bytes at a time. +inline std::size_t scalar_string_bulk_run(const unsigned char* data, std::size_t n) noexcept +{ + std::size_t pos = 0; + while (pos < n) + { + pos += find_string_special(data + pos, n - pos); + if (pos >= n || data[pos] < 0x80u) + { + break; // end of buffer, or a quote/escape/control byte + } + const std::size_t seq = validate_one_utf8(data + pos, n - pos); + if (seq == 0) + { + break; // ill-formed or truncated: let the byte path diagnose it + } + pos += seq; + } + return pos; +} + +#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17) +// Index of the first quote/escape/control byte in [data, data+n) (non-ASCII +// bytes are *not* stops here - the whole run is handed to simdutf), or n. +inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t n) noexcept +{ + constexpr std::uint64_t ones = 0x0101010101010101ull; + constexpr std::uint64_t high = 0x8080808080808080ull; + std::size_t i = 0; + for (; i + 8 <= n; i += 8) + { + std::uint64_t v = 0; + std::memcpy(&v, data + i, sizeof(v)); + const std::uint64_t q = v ^ 0x2222222222222222ull; + const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; + const std::uint64_t hit = ((q - ones) & ~q & high) + | ((b - ones) & ~b & high) + | ((v - 0x2020202020202020ull) & ~v & high); + if (hit != 0) + { + for (std::size_t j = 0; j < 8; ++j) + { + const unsigned char c = data[i + j]; + if (c == '\"' || c == '\\' || c < 0x20u) + { + return i + j; + } + } + } + } + for (; i < n; ++i) + { + const unsigned char c = data[i]; + if (c == '\"' || c == '\\' || c < 0x20u) + { + return i; + } + } + return n; +} +#endif + +// Backend-dispatched bulk run length. With JSON_USE_SIMDUTF the run up to the +// next delimiter is validated in one shot by simdutf; on the rare failure the +// scalar helper recomputes the exact valid prefix so the byte path still +// produces the precise diagnostic. Without it, the pure scalar path is used. +inline std::size_t string_bulk_run(const unsigned char* data, std::size_t n) noexcept +{ +#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17) + const std::size_t run = find_string_delimiter(data, n); + if (run != 0 && simdutf::validate_utf8(reinterpret_cast(data), run)) + { + return run; + } +#endif + return scalar_string_bulk_run(data, n); +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/iterators/iter_impl.hpp b/include/nlohmann/detail/iterators/iter_impl.hpp index 44448611a..22f3ffc39 100644 --- a/include/nlohmann/detail/iterators/iter_impl.hpp +++ b/include/nlohmann/detail/iterators/iter_impl.hpp @@ -88,8 +88,13 @@ class iter_impl // NOLINT(cppcoreguidelines-special-member-functions,hicpp-speci iter_impl() = default; ~iter_impl() = default; - iter_impl(iter_impl&&) noexcept = default; - iter_impl& operator=(iter_impl&&) noexcept = default; + // the exception specification is left to be computed rather than declared: + // an array or object type whose iterator is not nothrow move constructible + // (std::deque's is not before libstdc++ 11) would make a declared noexcept + // differ from the implicit one, which deletes the function -- and is an + // error outright with older compilers + iter_impl(iter_impl&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor,cppcoreguidelines-noexcept-move-operations) + iter_impl& operator=(iter_impl&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor,cppcoreguidelines-noexcept-move-operations) /*! @brief constructor for a given JSON instance diff --git a/include/nlohmann/detail/iterators/iteration_proxy.hpp b/include/nlohmann/detail/iterators/iteration_proxy.hpp index 99246d120..c8aa50dd2 100644 --- a/include/nlohmann/detail/iterators/iteration_proxy.hpp +++ b/include/nlohmann/detail/iterators/iteration_proxy.hpp @@ -18,6 +18,7 @@ #endif #include +#include #include #include #include @@ -206,10 +207,10 @@ NLOHMANN_JSON_NAMESPACE_END namespace std { +// Fix: https://github.com/nlohmann/json/issues/1401 #if defined(__clang__) - // Fix: https://github.com/nlohmann/json/issues/1401 - #pragma clang diagnostic push - #pragma clang diagnostic ignored "-Wmismatched-tags" + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(clang diagnostic ignored "-Wmismatched-tags") #endif template class tuple_size<::nlohmann::detail::iteration_proxy_value> // NOLINT(cert-dcl58-cpp) @@ -224,7 +225,7 @@ class tuple_element> ::nlohmann::detail::iteration_proxy_value> ())); }; #if defined(__clang__) - #pragma clang diagnostic pop + JSON_HEDLEY_DIAGNOSTIC_POP #endif } // namespace std diff --git a/include/nlohmann/detail/json_pointer.hpp b/include/nlohmann/detail/json_pointer.hpp index 576ba62cc..1540a8d6f 100644 --- a/include/nlohmann/detail/json_pointer.hpp +++ b/include/nlohmann/detail/json_pointer.hpp @@ -17,6 +17,7 @@ #endif // JSON_NO_IO #include // max #include // accumulate +#include // set #include // string #include // move #include // vector @@ -71,7 +72,7 @@ class json_pointer string_t{}, [](const string_t& a, const string_t& b) { - return detail::concat(a, '/', detail::escape(b)); + return detail::concat(a, '/', detail::escape(b)); }); } @@ -265,7 +266,7 @@ class json_pointer JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); } - const char* p = s.c_str(); + const char* p = s.data(); char* p_end = nullptr; // NOLINT(misc-const-correctness) errno = 0; // strtoull doesn't reset errno const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) @@ -300,19 +301,35 @@ class json_pointer } private: + /*! + @brief the reference token sequences that denote arrays + + @ref unflatten collects the pointer prefixes that have a reference token 0 + among their children; @ref get_and_create creates arrays exactly below + those prefixes and objects everywhere else. Deciding this up front keeps + the result independent of the order in which the flattened object is + iterated, which is unspecified for some object types. + */ + using array_parents_t = std::set>; + /*! @brief create and return a reference to the pointed to value @complexity Linear in the number of reference tokens. + @throw parse_error.106 if an array index begins with '0' @throw parse_error.109 if array index is not a number @throw type_error.313 if value cannot be unflattened */ template - BasicJsonType& get_and_create(BasicJsonType& j) const + BasicJsonType& get_and_create(BasicJsonType& j, const array_parents_t& array_parents) const { auto* result = &j; + // the reference tokens that have been consumed so far; used to look up + // whether the value to be created below is an array or an object + std::vector prefix; + // in case no reference tokens exist, return a reference to the JSON value // j which will be overwritten by a primitive value for (const auto& reference_token : reference_tokens) @@ -321,10 +338,11 @@ class json_pointer { case detail::value_t::null: { - if (reference_token == "0") + if (array_parents.find(prefix) != array_parents.end()) { - // start a new array if the reference token is 0 - result = &result->operator[](0); + // some reference token below this position is 0, so the + // value is an array + result = &result->operator[](array_index(reference_token)); } else { @@ -364,6 +382,8 @@ class json_pointer default: JSON_THROW(detail::type_error::create(313, "invalid value to unflatten", &j)); } + + prefix.push_back(reference_token); } return *result; @@ -748,6 +768,20 @@ class json_pointer } } + // the reference token consists only of digits at this point (cf. checks + // above); however, its numeric value might not be representable, in which + // case array_index() would throw out_of_range.404/410 -- contains() must + // not throw (see #5395), so such a reference token is treated as "not found" + errno = 0; // strtoull() does not reset errno on success + char* p_end = nullptr; // NOLINT(misc-const-correctness) + const unsigned long long magnitude = std::strtoull(reference_token.c_str(), &p_end, 10); // NOLINT(runtime/int) + if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX + || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) + { + // the array index cannot be represented as size_type + return false; + } + const auto idx = array_index(reference_token); if (idx >= ptr->size()) { @@ -823,7 +857,8 @@ class json_pointer { // use the text between the beginning of the reference token // (start) and the last slash (slash). - auto reference_token = reference_string.substr(start, slash - start); + const auto count = (slash == string_t::npos ? reference_string.size() : slash) - start; + auto reference_token = string_t(reference_string.data() + start, count); // check reference tokens are properly escaped for (std::size_t pos = reference_token.find_first_of('~'); @@ -939,6 +974,24 @@ class json_pointer BasicJsonType result; + // collect the pointer prefixes that have a reference token 0 among + // their children; the values below them are arrays, all others are + // objects (see array_parents_t) + array_parents_t array_parents; + for (const auto& element : *value.m_data.m_value.object) + { + json_pointer ptr(element.first); + std::vector prefix; + for (auto& reference_token : ptr.reference_tokens) + { + if (reference_token == "0") + { + array_parents.insert(prefix); + } + prefix.push_back(std::move(reference_token)); + } + } + // iterate the JSON object values for (const auto& element : *value.m_data.m_value.object) { @@ -951,7 +1004,7 @@ class json_pointer // that if the JSON pointer is "" (i.e., points to the whole value), // function get_and_create returns a reference to the result itself. // An assignment will then create a primitive value. - json_pointer(element.first).get_and_create(result) = element.second; + json_pointer(element.first).get_and_create(result, array_parents) = element.second; } return result; diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index 4682fd361..96fa165f5 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -186,6 +186,15 @@ #define JSON_NO_UNIQUE_ADDRESS #endif +// Clang targeting MinGW does not survive the thread_local storage the copy +// constructor uses to bound its descent: every test that copies a value +// segfaults with clang 11.0.1 and clang 18.1.8, while the same tests pass with +// GCC targeting MinGW and with every other toolchain the library is tested on. +// Copying works the same way without the counter, only more slowly. +#if !defined(JSON_NO_THREAD_LOCAL) && defined(__clang__) && defined(__MINGW32__) + #define JSON_NO_THREAD_LOCAL 1 +#endif + // disable documentation warnings on clang #if defined(__clang__) #pragma clang diagnostic push @@ -804,6 +813,6 @@ void templated_json_throw(ExceptionType exception) #define JSON_USE_GLOBAL_UDLS 1 #endif -#ifndef JSON_BRACE_INIT_COPY_SEMANTICS - #define JSON_BRACE_INIT_COPY_SEMANTICS 0 +#ifndef JSON_STRICT_NUL_HANDLING + #define JSON_STRICT_NUL_HANDLING 0 #endif diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index 2ac25a46d..afcbfc38b 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -26,7 +26,7 @@ #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION #undef JSON_USE_GLOBAL_UDLS -#undef JSON_BRACE_INIT_COPY_SEMANTICS +#undef JSON_STRICT_NUL_HANDLING #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH @@ -44,6 +44,7 @@ #undef JSON_HAS_STD_FORMAT #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON + #undef JSON_BRACE_INIT_COPY_SEMANTICS #endif #include diff --git a/include/nlohmann/detail/meta/type_traits.hpp b/include/nlohmann/detail/meta/type_traits.hpp index ebf6a2c26..6f8bf2a3d 100644 --- a/include/nlohmann/detail/meta/type_traits.hpp +++ b/include/nlohmann/detail/meta/type_traits.hpp @@ -172,17 +172,18 @@ struct has_to_json < BasicJsonType, T, enable_if_t < !is_basic_json::value >> template using detect_key_compare = typename T::key_compare; -template -struct has_key_compare : std::integral_constant::value> {}; - -// obtains the actual object key comparator +// obtains the actual object key comparator: object_t::key_compare if the +// object type defines it, and default_object_comparator_t otherwise +// +// note detected_or_t is used rather than std::conditional, because the latter +// names both of its type arguments eagerly; object_t::key_compare would then +// be a hard error for an object type that does not define it template struct actual_object_comparator { using object_t = typename BasicJsonType::object_t; using object_comparator_t = typename BasicJsonType::default_object_comparator_t; - using type = typename std::conditional < has_key_compare::value, - typename object_t::key_compare, object_comparator_t>::type; + using type = detected_or_t; }; template @@ -778,6 +779,22 @@ using has_erase_with_key_type = typename std::conditional < std::true_type, std::false_type >::type; +template +using detect_erase_with_iterator = decltype(std::declval().erase(std::declval())); + +// type trait to check if erase(iterator) returns void instead of the following +// iterator, as the object types that do not compute a successor the caller may +// not need do +template +using erase_returns_void = is_detected_exact; + +template +using detect_capacity = decltype(std::declval().capacity()); + +// type trait to check if a type has a capacity() member function +template +struct has_capacity : std::integral_constant::value> {}; + // a naive helper to check if a type is an ordered_map (exploits the fact that // ordered_map inherits capacity() from std::vector) template diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 35b5efa8a..eaa8d63b6 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -16,9 +16,14 @@ #include // memcpy #include // numeric_limits #include // string +#include // enable_if, is_constructible #include // move #include // vector +#ifdef _MSC_VER + #include // _byteswap_ushort, _byteswap_ulong, _byteswap_uint64 +#endif + #include #include #include @@ -39,10 +44,41 @@ enum class bjdata_version_t // binary writer // /////////////////// +/*! +@brief capacity hint for binary serialization into a std::vector + +Returns a *lower* bound on the number of bytes the serialization will produce, +so that writing an array/object of many elements does not start reallocating +from an empty buffer. Every array element occupies at least one byte in every +supported binary format, and every object entry at least two (a key of at least +one byte plus a value of at least one), plus one byte for the container header, +so the hint can never exceed the final size and the returned vector is never +left holding capacity the caller did not ask for. The buffer still grows +geometrically past the hint, so under-reserving only costs a few later +reallocations. Only the top-level element count is consulted (O(1), no walk of +the DOM); a single scalar, string, or binary value is written in one shot and +needs no hint. +*/ +template +std::size_t binary_reserve_hint(const BasicJsonType& j) +{ + if (j.is_array()) + { + return j.size() + 1; + } + + if (j.is_object()) + { + return (j.size() * 2) + 1; + } + + return 0; +} + /*! @brief serialization to CBOR and MessagePack values */ -template +template> class binary_writer { using string_t = typename BasicJsonType::string_t; @@ -53,12 +89,28 @@ class binary_writer /*! @brief create a binary writer + @param[in] sink output sink to write to (a value-type sink such as + output_vector_sink, or output_adapter_sink wrapping a + type-erased output adapter) + */ + explicit binary_writer(OutputSinkType sink) : oa(std::move(sink)) + {} + + /*! + @brief create a binary writer from a type-erased output adapter + + Convenience constructor for the default (output_adapter_sink) sink so the + `output_adapter`-based overloads keep constructing the writer directly from + an adapter. Constrained to sinks that can actually be built from an adapter, + so that a writer over some other sink type is not advertised as constructible + from one. + @param[in] adapter output adapter to write to */ - explicit binary_writer(output_adapter_t adapter) : oa(std::move(adapter)) - { - JSON_ASSERT(oa); - } + template < typename SinkType = OutputSinkType, + typename std::enable_if < std::is_constructible>::value, int >::type = 0 > + explicit binary_writer(output_adapter_t adapter) : oa(SinkType(std::move(adapter))) + {} /*! @param[in] j JSON value to serialize @@ -99,15 +151,15 @@ class binary_writer { case value_t::null: { - oa->write_character(to_char_type(0xF6)); + oa.write_character(to_char_type(0xF6)); break; } case value_t::boolean: { - oa->write_character(j.m_data.m_value.boolean - ? to_char_type(0xF5) - : to_char_type(0xF4)); + oa.write_character(j.m_data.m_value.boolean + ? to_char_type(0xF5) + : to_char_type(0xF4)); break; } @@ -124,22 +176,22 @@ class binary_writer } else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x18)); + oa.write_character(to_char_type(0x18)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x19)); + oa.write_character(to_char_type(0x19)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x1A)); + oa.write_character(to_char_type(0x1A)); write_number(static_cast(j.m_data.m_value.number_integer)); } else { - oa->write_character(to_char_type(0x1B)); + oa.write_character(to_char_type(0x1B)); write_number(static_cast(j.m_data.m_value.number_integer)); } } @@ -154,22 +206,22 @@ class binary_writer } else if (positive_number <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x38)); + oa.write_character(to_char_type(0x38)); write_number(static_cast(positive_number)); } else if (positive_number <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x39)); + oa.write_character(to_char_type(0x39)); write_number(static_cast(positive_number)); } else if (positive_number <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x3A)); + oa.write_character(to_char_type(0x3A)); write_number(static_cast(positive_number)); } else { - oa->write_character(to_char_type(0x3B)); + oa.write_character(to_char_type(0x3B)); write_number(static_cast(positive_number)); } } @@ -184,22 +236,22 @@ class binary_writer } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x18)); + oa.write_character(to_char_type(0x18)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x19)); + oa.write_character(to_char_type(0x19)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x1A)); + oa.write_character(to_char_type(0x1A)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } else { - oa->write_character(to_char_type(0x1B)); + oa.write_character(to_char_type(0x1B)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } break; @@ -210,16 +262,16 @@ class binary_writer if (std::isnan(j.m_data.m_value.number_float)) { // NaN is 0xf97e00 in CBOR - oa->write_character(to_char_type(0xF9)); - oa->write_character(to_char_type(0x7E)); - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0xF9)); + oa.write_character(to_char_type(0x7E)); + oa.write_character(to_char_type(0x00)); } else if (std::isinf(j.m_data.m_value.number_float)) { // Infinity is 0xf97c00, -Infinity is 0xf9fc00 - oa->write_character(to_char_type(0xf9)); - oa->write_character(j.m_data.m_value.number_float > 0 ? to_char_type(0x7C) : to_char_type(0xFC)); - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0xf9)); + oa.write_character(j.m_data.m_value.number_float > 0 ? to_char_type(0x7C) : to_char_type(0xFC)); + oa.write_character(to_char_type(0x00)); } else { @@ -238,31 +290,31 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x78)); + oa.write_character(to_char_type(0x78)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x79)); + oa.write_character(to_char_type(0x79)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x7A)); + oa.write_character(to_char_type(0x7A)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x7B)); + oa.write_character(to_char_type(0x7B)); write_number(static_cast(N)); } // LCOV_EXCL_STOP // step 2: write the string - oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->c_str()), - j.m_data.m_value.string->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.string->data()), + j.m_data.m_value.string->size()); break; } @@ -276,23 +328,23 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x98)); + oa.write_character(to_char_type(0x98)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x99)); + oa.write_character(to_char_type(0x99)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x9A)); + oa.write_character(to_char_type(0x9A)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x9B)); + oa.write_character(to_char_type(0x9B)); write_number(static_cast(N)); } // LCOV_EXCL_STOP @@ -339,31 +391,31 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x58)); + oa.write_character(to_char_type(0x58)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x59)); + oa.write_character(to_char_type(0x59)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x5A)); + oa.write_character(to_char_type(0x5A)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x5B)); + oa.write_character(to_char_type(0x5B)); write_number(static_cast(N)); } // LCOV_EXCL_STOP // step 2: write each element - oa->write_characters( - reinterpret_cast(j.m_data.m_value.binary->data()), - N); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.binary->data()), + N); break; } @@ -378,23 +430,23 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xB8)); + oa.write_character(to_char_type(0xB8)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xB9)); + oa.write_character(to_char_type(0xB9)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xBA)); + oa.write_character(to_char_type(0xBA)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xBB)); + oa.write_character(to_char_type(0xBB)); write_number(static_cast(N)); } // LCOV_EXCL_STOP @@ -423,15 +475,15 @@ class binary_writer { case value_t::null: // nil { - oa->write_character(to_char_type(0xC0)); + oa.write_character(to_char_type(0xC0)); break; } case value_t::boolean: // true and false { - oa->write_character(j.m_data.m_value.boolean - ? to_char_type(0xC3) - : to_char_type(0xC2)); + oa.write_character(j.m_data.m_value.boolean + ? to_char_type(0xC3) + : to_char_type(0xC2)); break; } @@ -450,25 +502,25 @@ class binary_writer else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 8 - oa->write_character(to_char_type(0xCC)); + oa.write_character(to_char_type(0xCC)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 16 - oa->write_character(to_char_type(0xCD)); + oa.write_character(to_char_type(0xCD)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 32 - oa->write_character(to_char_type(0xCE)); + oa.write_character(to_char_type(0xCE)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 64 - oa->write_character(to_char_type(0xCF)); + oa.write_character(to_char_type(0xCF)); write_number(static_cast(j.m_data.m_value.number_integer)); } } @@ -483,28 +535,28 @@ class binary_writer j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 8 - oa->write_character(to_char_type(0xD0)); + oa.write_character(to_char_type(0xD0)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 16 - oa->write_character(to_char_type(0xD1)); + oa.write_character(to_char_type(0xD1)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 32 - oa->write_character(to_char_type(0xD2)); + oa.write_character(to_char_type(0xD2)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 64 - oa->write_character(to_char_type(0xD3)); + oa.write_character(to_char_type(0xD3)); write_number(static_cast(j.m_data.m_value.number_integer)); } } @@ -521,25 +573,25 @@ class binary_writer else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 8 - oa->write_character(to_char_type(0xCC)); + oa.write_character(to_char_type(0xCC)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 16 - oa->write_character(to_char_type(0xCD)); + oa.write_character(to_char_type(0xCD)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 32 - oa->write_character(to_char_type(0xCE)); + oa.write_character(to_char_type(0xCE)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 64 - oa->write_character(to_char_type(0xCF)); + oa.write_character(to_char_type(0xCF)); write_number(static_cast(j.m_data.m_value.number_integer)); } break; @@ -563,26 +615,26 @@ class binary_writer else if (N <= (std::numeric_limits::max)()) { // str 8 - oa->write_character(to_char_type(0xD9)); + oa.write_character(to_char_type(0xD9)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // str 16 - oa->write_character(to_char_type(0xDA)); + oa.write_character(to_char_type(0xDA)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // str 32 - oa->write_character(to_char_type(0xDB)); + oa.write_character(to_char_type(0xDB)); write_number(static_cast(N)); } // step 2: write the string - oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->c_str()), - j.m_data.m_value.string->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.string->data()), + j.m_data.m_value.string->size()); break; } @@ -598,13 +650,13 @@ class binary_writer else if (N <= (std::numeric_limits::max)()) { // array 16 - oa->write_character(to_char_type(0xDC)); + oa.write_character(to_char_type(0xDC)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // array 32 - oa->write_character(to_char_type(0xDD)); + oa.write_character(to_char_type(0xDD)); write_number(static_cast(N)); } @@ -660,7 +712,7 @@ class binary_writer fixed = false; } - oa->write_character(to_char_type(output_type)); + oa.write_character(to_char_type(output_type)); if (!fixed) { write_number(static_cast(N)); @@ -672,7 +724,7 @@ class binary_writer ? 0xC8 // ext 16 : 0xC5; // bin 16 - oa->write_character(to_char_type(output_type)); + oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) @@ -681,20 +733,25 @@ class binary_writer ? 0xC9 // ext 32 : 0xC6; // bin 32 - oa->write_character(to_char_type(output_type)); + oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } // step 1.5: if this is an ext type, write the subtype if (use_ext) { + if (JSON_HEDLEY_UNLIKELY(j.m_data.m_value.binary->subtype() > (std::numeric_limits::max)())) + { + JSON_THROW(out_of_range::create(415, concat("subtype ", std::to_string(j.m_data.m_value.binary->subtype()), " is too large for the MessagePack ext type (max 255)"), &j)); + } + write_number(static_cast(j.m_data.m_value.binary->subtype())); } // step 2: write the byte string - oa->write_characters( - reinterpret_cast(j.m_data.m_value.binary->data()), - N); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.binary->data()), + N); break; } @@ -711,13 +768,13 @@ class binary_writer else if (N <= (std::numeric_limits::max)()) { // map 16 - oa->write_character(to_char_type(0xDE)); + oa.write_character(to_char_type(0xDE)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // map 32 - oa->write_character(to_char_type(0xDF)); + oa.write_character(to_char_type(0xDF)); write_number(static_cast(N)); } @@ -756,7 +813,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('Z')); + oa.write_character(to_char_type('Z')); } break; } @@ -765,9 +822,9 @@ class binary_writer { if (add_prefix) { - oa->write_character(j.m_data.m_value.boolean - ? to_char_type('T') - : to_char_type('F')); + oa.write_character(j.m_data.m_value.boolean + ? to_char_type('T') + : to_char_type('F')); } break; } @@ -794,12 +851,12 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('S')); + oa.write_character(to_char_type('S')); } write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata); - oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->c_str()), - j.m_data.m_value.string->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.string->data()), + j.m_data.m_value.string->size()); break; } @@ -807,13 +864,16 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('[')); + oa.write_character(to_char_type('[')); } bool prefix_required = true; if (use_type && !j.m_data.m_value.array->empty()) { - JSON_ASSERT(use_count); + if (!use_count) + { + JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); + } const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata); const bool same_prefix = std::all_of(j.begin() + 1, j.end(), [this, first_prefix, use_bjdata](const BasicJsonType & v) @@ -821,19 +881,27 @@ class binary_writer return ubjson_prefix(v, use_bjdata) == first_prefix; }); - std::vector bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type + // an optimized array of a valueless type carries no payload, so a + // reader has nothing but the declared count to bound the allocation + // by and refuses an excessive one. Write the unoptimized form for + // those, at one byte per element, so the result can be read back. + // Objects are not affected: every element is preceded by its key. + const bool valueless_type = (first_prefix == 'Z' || first_prefix == 'T' || first_prefix == 'F'); + const bool excessive_valueless = valueless_type + && j.m_data.m_value.array->size() > detail::max_valueless_container_size; - if (same_prefix && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) + if (same_prefix && !excessive_valueless + && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) { prefix_required = false; - oa->write_character(to_char_type('$')); - oa->write_character(first_prefix); + oa.write_character(to_char_type('$')); + oa.write_character(first_prefix); } } if (use_count) { - oa->write_character(to_char_type('#')); + oa.write_character(to_char_type('#')); write_number_with_ubjson_prefix(j.m_data.m_value.array->size(), true, use_bjdata); } @@ -844,7 +912,7 @@ class binary_writer if (!use_count) { - oa->write_character(to_char_type(']')); + oa.write_character(to_char_type(']')); } break; @@ -854,40 +922,45 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('[')); + oa.write_character(to_char_type('[')); } if (use_type && (bjdata_draft3 || !j.m_data.m_value.binary->empty())) { - JSON_ASSERT(use_count); - oa->write_character(to_char_type('$')); - oa->write_character(bjdata_draft3 ? 'B' : 'U'); + if (!use_count) + { + JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); + } + oa.write_character(to_char_type('$')); + oa.write_character(bjdata_draft3 ? 'B' : 'U'); } if (use_count) { - oa->write_character(to_char_type('#')); + oa.write_character(to_char_type('#')); write_number_with_ubjson_prefix(j.m_data.m_value.binary->size(), true, use_bjdata); } if (use_type) { - oa->write_characters( - reinterpret_cast(j.m_data.m_value.binary->data()), - j.m_data.m_value.binary->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.binary->data()), + j.m_data.m_value.binary->size()); } else { for (size_t i = 0; i < j.m_data.m_value.binary->size(); ++i) { - oa->write_character(to_char_type(bjdata_draft3 ? 'B' : 'U')); - oa->write_character(to_char_type(j.m_data.m_value.binary->data()[i])); + oa.write_character(to_char_type(bjdata_draft3 ? 'B' : 'U')); + // the cast is needed for binary types whose value type + // is not an integer (e.g., std::byte) + oa.write_character(to_char_type(static_cast(j.m_data.m_value.binary->data()[i]))); } } if (!use_count) { - oa->write_character(to_char_type(']')); + oa.write_character(to_char_type(']')); } break; @@ -905,13 +978,16 @@ class binary_writer if (add_prefix) { - oa->write_character(to_char_type('{')); + oa.write_character(to_char_type('{')); } bool prefix_required = true; if (use_type && !j.m_data.m_value.object->empty()) { - JSON_ASSERT(use_count); + if (!use_count) + { + JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); + } const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata); const bool same_prefix = std::all_of(j.begin(), j.end(), [this, first_prefix, use_bjdata](const BasicJsonType & v) @@ -919,34 +995,32 @@ class binary_writer return ubjson_prefix(v, use_bjdata) == first_prefix; }); - std::vector bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type - - if (same_prefix && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) + if (same_prefix && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) { prefix_required = false; - oa->write_character(to_char_type('$')); - oa->write_character(first_prefix); + oa.write_character(to_char_type('$')); + oa.write_character(first_prefix); } } if (use_count) { - oa->write_character(to_char_type('#')); + oa.write_character(to_char_type('#')); write_number_with_ubjson_prefix(j.m_data.m_value.object->size(), true, use_bjdata); } for (const auto& el : *j.m_data.m_value.object) { write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); - oa->write_characters( - reinterpret_cast(el.first.c_str()), - el.first.size()); + oa.write_characters( + reinterpret_cast(el.first.data()), + el.first.size()); write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); } if (!use_count) { - oa->write_character(to_char_type('}')); + oa.write_character(to_char_type('}')); } break; @@ -1000,10 +1074,13 @@ class binary_writer void write_bson_entry_header(const string_t& name, const std::uint8_t element_type) { - oa->write_character(to_char_type(element_type)); - oa->write_characters( - reinterpret_cast(name.c_str()), - name.size() + 1u); + oa.write_character(to_char_type(element_type)); + oa.write_characters( + reinterpret_cast(name.data()), + name.size()); + // the terminating null byte is written explicitly rather than taken + // from the buffer, so that string_t::data() need not be null-terminated + oa.write_character(to_char_type(0x00)); } /*! @@ -1013,7 +1090,7 @@ class binary_writer const bool value) { write_bson_entry_header(name, 0x08); - oa->write_character(value ? to_char_type(0x01) : to_char_type(0x00)); + oa.write_character(value ? to_char_type(0x01) : to_char_type(0x00)); } /*! @@ -1043,9 +1120,12 @@ class binary_writer write_bson_entry_header(name, 0x02); write_number(to_bson_length(value.size() + 1ul), true); - oa->write_characters( - reinterpret_cast(value.c_str()), - value.size() + 1); + oa.write_characters( + reinterpret_cast(value.data()), + value.size()); + // the terminating null byte is written explicitly rather than taken + // from the buffer, so that string_t::data() need not be null-terminated + oa.write_character(to_char_type(0x00)); } /*! @@ -1136,7 +1216,11 @@ class binary_writer const std::size_t embedded_document_size = std::accumulate(std::begin(value), std::end(value), static_cast(0), [&array_index](std::size_t result, const typename BasicJsonType::array_t::value_type & el) { - return result + calc_bson_element_size(std::to_string(array_index++), el); + // the index is built as a std::string, while calc_bson_element_size + // takes a string_t; convert explicitly, as the two are only + // implicitly convertible for some string types + const auto key = std::to_string(array_index++); + return result + calc_bson_element_size(string_t(key.data(), key.size()), el); }); return sizeof(std::int32_t) + embedded_document_size + 1ul; @@ -1163,10 +1247,14 @@ class binary_writer for (const auto& el : value) { - write_bson_element(std::to_string(array_index++), el); + // the index is built as a std::string, while write_bson_element takes + // a string_t; convert explicitly, as the two are only implicitly + // convertible for some string types + const auto key = std::to_string(array_index++); + write_bson_element(string_t(key.data(), key.size()), el); } - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0x00)); } /*! @@ -1178,9 +1266,15 @@ class binary_writer write_bson_entry_header(name, 0x05); write_number(to_bson_length(value.size()), true); + + if (value.has_subtype() && JSON_HEDLEY_UNLIKELY(value.subtype() > (std::numeric_limits::max)())) + { + JSON_THROW(out_of_range::create(415, concat("subtype ", std::to_string(value.subtype()), " is too large for the BSON binary subtype (max 255)"), nullptr)); + } + write_number(value.has_subtype() ? static_cast(value.subtype()) : static_cast(0x00)); - oa->write_characters(reinterpret_cast(value.data()), value.size()); + oa.write_characters(reinterpret_cast(value.data()), value.size()); } /*! @@ -1306,7 +1400,7 @@ class binary_writer write_bson_element(el.first, el.second); } - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0x00)); } ////////// @@ -1350,7 +1444,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(get_ubjson_float_prefix(n)); + oa.write_character(get_ubjson_float_prefix(n)); } write_number(n, use_bjdata); } @@ -1366,7 +1460,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('i')); // int8 + oa.write_character(to_char_type('i')); // int8 } write_number(static_cast(n), use_bjdata); } @@ -1374,7 +1468,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('U')); // uint8 + oa.write_character(to_char_type('U')); // uint8 } write_number(static_cast(n), use_bjdata); } @@ -1382,7 +1476,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('I')); // int16 + oa.write_character(to_char_type('I')); // int16 } write_number(static_cast(n), use_bjdata); } @@ -1390,7 +1484,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('u')); // uint16 - bjdata only + oa.write_character(to_char_type('u')); // uint16 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -1398,7 +1492,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('l')); // int32 + oa.write_character(to_char_type('l')); // int32 } write_number(static_cast(n), use_bjdata); } @@ -1406,7 +1500,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('m')); // uint32 - bjdata only + oa.write_character(to_char_type('m')); // uint32 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -1414,7 +1508,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('L')); // int64 + oa.write_character(to_char_type('L')); // int64 } write_number(static_cast(n), use_bjdata); } @@ -1422,7 +1516,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('M')); // uint64 - bjdata only + oa.write_character(to_char_type('M')); // uint64 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -1430,14 +1524,14 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('H')); // high-precision number + oa.write_character(to_char_type('H')); // high-precision number } const auto number = BasicJsonType(n).dump(); write_number_with_ubjson_prefix(number.size(), true, use_bjdata); for (std::size_t i = 0; i < number.size(); ++i) { - oa->write_character(to_char_type(static_cast(number[i]))); + oa.write_character(to_char_type(static_cast(number[i]))); } } } @@ -1454,7 +1548,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('i')); // int8 + oa.write_character(to_char_type('i')); // int8 } write_number(static_cast(n), use_bjdata); } @@ -1462,7 +1556,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('U')); // uint8 + oa.write_character(to_char_type('U')); // uint8 } write_number(static_cast(n), use_bjdata); } @@ -1470,7 +1564,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('I')); // int16 + oa.write_character(to_char_type('I')); // int16 } write_number(static_cast(n), use_bjdata); } @@ -1478,7 +1572,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('u')); // uint16 - bjdata only + oa.write_character(to_char_type('u')); // uint16 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -1486,7 +1580,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('l')); // int32 + oa.write_character(to_char_type('l')); // int32 } write_number(static_cast(n), use_bjdata); } @@ -1494,7 +1588,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('m')); // uint32 - bjdata only + oa.write_character(to_char_type('m')); // uint32 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -1502,7 +1596,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('L')); // int64 + oa.write_character(to_char_type('L')); // int64 } write_number(static_cast(n), use_bjdata); } @@ -1511,14 +1605,14 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('H')); // high-precision number + oa.write_character(to_char_type('H')); // high-precision number } const auto number = BasicJsonType(n).dump(); write_number_with_ubjson_prefix(number.size(), true, use_bjdata); for (std::size_t i = 0; i < number.size(); ++i) { - oa->write_character(to_char_type(static_cast(number[i]))); + oa.write_character(to_char_type(static_cast(number[i]))); } } // LCOV_EXCL_STOP @@ -1628,6 +1722,21 @@ class binary_writer } } + /*! + @brief whether BJData forbids @a marker as the type of an optimized array + or object + + Containers, strings, high-precision numbers, booleans and null cannot be + declared as the single type of an optimized container in BJData; such a + container is written unoptimized. The reader rejects them with the same + list (binary_reader::bjd_optimized_type_markers). + */ + static constexpr bool is_bjdata_excluded_type_marker(const CharType marker) noexcept + { + return marker == '[' || marker == '{' || marker == 'S' || marker == 'H' + || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; + } + static constexpr CharType get_ubjson_float_prefix(float /*unused*/) { return 'd'; // float 32 @@ -1638,6 +1747,20 @@ class binary_writer return 'D'; // float 64 } + /*! + @brief checks whether a JSON number fits into @a TargetType + @param[in] el a JSON number of either the signed or unsigned integer kind + @return whether @a el's value can be represented by @a TargetType without + wrapping, regardless of which of the two kinds it is stored as + */ + template + static bool bjdata_ndarray_value_in_range(const BasicJsonType& el) + { + return el.is_number_unsigned() + ? value_in_range_of(el.template get()) + : value_in_range_of(el.template get()); + } + /*! @return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid */ @@ -1649,6 +1772,16 @@ class binary_writer }; string_t key = "_ArrayType_"; + // the type name is looked up as a string below; a non-string + // annotation (e.g. a number, null, or an array) cannot name a known + // dtype, so it is treated the same as an unrecognized type name and + // falls back to a plain object encoding instead of throwing + // type_error.302 out of get() + if (!value.at(key).is_string()) + { + return true; + } + // use get() instead of static_cast to avoid an // ambiguous conversion under explicit instantiation on C++17 (see #4825) auto it = bjdtype.find(value.at(key).template get()); @@ -1658,9 +1791,39 @@ class binary_writer } CharType dtype = it->second; + // the 'B' (byte) marker is only defined from BJData Draft 3 onward; + // emitting it under an earlier draft would produce a stream that an + // earlier-draft reader rejects, so such an object falls back to a + // plain object encoding instead (see the "Binary values" section of + // the BJData documentation) + if (dtype == 'B' && bjdata_version < bjdata_version_t::draft3) + { + return true; + } + key = "_ArraySize_"; - std::size_t len = (value.at(key).empty() ? 0 : 1); - for (const auto& el : value.at(key)) + // the dimensions are written verbatim as the header length below, so a + // value that is not an array cannot produce a valid one: null emits 'Z' + // and an object emits '{', neither of which a reader accepts after '#'. + // Such an object is not a valid ndarray and falls back to a plain object. + if (!value.at(key).is_array()) + { + return true; + } + + // the reader only restores an annotated object from an ND-array header + // with at least two dimensions: an empty dimension vector, a single + // dimension, or a 1xN row vector is read back as a plain array, which + // would silently drop the annotation, so such an object falls back to + // a plain object encoding instead + const auto& dims = value.at(key); + if (dims.size() < 2 || (dims.size() == 2 && dims.at(0).is_number_integer() && dims.at(0).template get() == 1)) + { + return true; + } + + std::size_t len = 1; + for (const auto& el : dims) { // a dimension is read as an unsigned value below, so anything that // is not a non-negative integer is rejected: a non-integer entry @@ -1682,15 +1845,26 @@ class binary_writer return true; } const auto dim_size = static_cast(dim); - if (dim_size != 0 && len > (std::numeric_limits::max)() / dim_size) + + // the reader turns an ND-array with any zero dimension into an + // empty plain array, dropping the annotation, so keep the object + if (dim_size == 0) + { + return true; + } + if (len > (std::numeric_limits::max)() / dim_size) { return true; } len *= dim_size; } + // the elements are written from _ArrayData_ as a flat list, so it has + // to be an array: size() is 0 for null and 1 for any other scalar, and + // iterating an object visits its values, so any of these could match + // the dimensions by accident and be encoded as an unrelated ND-array key = "_ArrayData_"; - if (value.at(key).size() != len) + if (!value.at(key).is_array() || value.at(key).size() != len) { return true; } @@ -1713,10 +1887,64 @@ class binary_writer } } - oa->write_character('['); - oa->write_character('$'); - oa->write_character(dtype); - oa->write_character('#'); + // every element is cast to the (possibly narrower) C++ type matching + // dtype below; a value that does not fit that type would silently + // wrap (integers) or overflow to infinity (the "single" precision + // float) instead of being reported, so such an object falls back to + // a plain object encoding as well + for (const auto& el : value.at(key)) + { + bool in_range = true; + switch (dtype) + { + case 'U': + case 'C': + case 'B': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'i': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'u': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'I': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'm': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'l': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'M': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'L': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'd': + { + const auto dval = el.template get(); + in_range = !std::isfinite(dval) || + (dval >= static_cast(std::numeric_limits::lowest()) && + dval <= static_cast((std::numeric_limits::max)())); + break; + } + default: + // 'D' (double) already spans the full range of number_float_t + break; + } + if (!in_range) + { + return true; + } + } + + oa.write_character('['); + oa.write_character('$'); + oa.write_character(dtype); + oa.write_character('#'); key = "_ArraySize_"; write_ubjson(value.at(key), use_count, use_type, true, true, bjdata_version); @@ -1812,6 +2040,87 @@ class binary_writer On the other hand, BSON and BJData use little endian and should reorder on big endian systems. */ + // single-instruction byte swaps (compilers lower these to bswap/rev/movbe); + // used to emit big-endian numbers without a per-byte std::reverse loop + static std::uint16_t byte_swap(std::uint16_t x) noexcept + { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_bswap16(x); +#elif defined(_MSC_VER) + return _byteswap_ushort(x); +#else + return static_cast((x >> 8) | (x << 8)); +#endif + } + + static std::uint32_t byte_swap(std::uint32_t x) noexcept + { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_bswap32(x); +#elif defined(_MSC_VER) + return _byteswap_ulong(x); +#else + return ((x & 0x000000FFu) << 24) | ((x & 0x0000FF00u) << 8) + | ((x & 0x00FF0000u) >> 8) | ((x & 0xFF000000u) >> 24); +#endif + } + + static std::uint64_t byte_swap(std::uint64_t x) noexcept + { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_bswap64(x); +#elif defined(_MSC_VER) + return _byteswap_uint64(x); +#else + x = ((x & 0x00000000FFFFFFFFull) << 32) | ((x & 0xFFFFFFFF00000000ull) >> 32); + x = ((x & 0x0000FFFF0000FFFFull) << 16) | ((x & 0xFFFF0000FFFF0000ull) >> 16); + x = ((x & 0x00FF00FF00FF00FFull) << 8) | ((x & 0xFF00FF00FF00FF00ull) >> 8); + return x; +#endif + } + + /*! + @brief reverse the bytes of a buffer by byte-swapping it as UIntType + + Loading the buffer into an unsigned integer of the same width and swapping + that is what lets the compiler emit a single bswap/rev/movbe; reversing the + buffer element by element does not reliably get there (clang keeps a scalar + shuffle). The two memcpy calls are the only portable way to reinterpret the + bytes and are folded away by every optimizer. + */ + template + static void byte_swap_buffer(std::array& a) noexcept + { + static_assert(sizeof(UIntType) == N, "swap width must match the buffer size"); + UIntType v{}; + std::memcpy(&v, a.data(), sizeof(v)); + v = byte_swap(v); + std::memcpy(a.data(), &v, sizeof(v)); + } + + // reverse the bytes of a fixed-size buffer; a single byte_swap() for the + // common 2/4/8-byte number payloads, std::reverse for any other size + static void reverse_bytes(std::array& a) noexcept + { + byte_swap_buffer(a); + } + + static void reverse_bytes(std::array& a) noexcept + { + byte_swap_buffer(a); + } + + static void reverse_bytes(std::array& a) noexcept + { + byte_swap_buffer(a); + } + + template + static void reverse_bytes(std::array& a) noexcept + { + std::reverse(a.begin(), a.end()); + } + template void write_number(const NumberType n, const bool OutputIsLittleEndian = false) { @@ -1823,36 +2132,45 @@ class binary_writer if (is_little_endian != OutputIsLittleEndian) { // reverse byte order prior to conversion if necessary - std::reverse(vec.begin(), vec.end()); + reverse_bytes(vec); } - oa->write_characters(vec.data(), sizeof(NumberType)); + oa.write_characters(vec.data(), sizeof(NumberType)); } void write_compact_float(const number_float_t n, detail::input_format_t format) { #ifdef __GNUC__ -#pragma GCC diagnostic push -#pragma GCC diagnostic ignored "-Wfloat-equal" + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") +#endif + // When number_float_t is float, static_cast(n) is the identity and + // both branches below are intentionally identical (the "compact" float + // representation is the value itself). Only GCC diagnoses this, and only + // when the sink calls are inlined; clang has no such warning. + // (-Wduplicated-branches only exists from GCC 7 on; naming it on an older + // GCC would itself warn under -Wpragmas) +#if defined(__GNUC__) && !defined(__clang__) && (__GNUC__ >= 7) + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wduplicated-branches") #endif if (!std::isfinite(n) || ((static_cast(n) >= static_cast(std::numeric_limits::lowest()) && static_cast(n) <= static_cast((std::numeric_limits::max)()) && static_cast(static_cast(n)) == static_cast(n)))) { - oa->write_character(format == detail::input_format_t::cbor - ? get_cbor_float_prefix(static_cast(n)) - : get_msgpack_float_prefix(static_cast(n))); + oa.write_character(format == detail::input_format_t::cbor + ? get_cbor_float_prefix(static_cast(n)) + : get_msgpack_float_prefix(static_cast(n))); write_number(static_cast(n)); } else { - oa->write_character(format == detail::input_format_t::cbor - ? get_cbor_float_prefix(n) - : get_msgpack_float_prefix(n)); + oa.write_character(format == detail::input_format_t::cbor + ? get_cbor_float_prefix(n) + : get_msgpack_float_prefix(n)); write_number(n); } #ifdef __GNUC__ -#pragma GCC diagnostic pop + JSON_HEDLEY_DIAGNOSTIC_POP #endif } @@ -1915,7 +2233,7 @@ class binary_writer const bool is_little_endian = little_endianness(); /// the output - output_adapter_t oa = nullptr; + OutputSinkType oa; }; } // namespace detail diff --git a/include/nlohmann/detail/output/output_adapters.hpp b/include/nlohmann/detail/output/output_adapters.hpp index 94763e64a..7eb73121c 100644 --- a/include/nlohmann/detail/output/output_adapters.hpp +++ b/include/nlohmann/detail/output/output_adapters.hpp @@ -13,6 +13,7 @@ #include // back_inserter #include // shared_ptr, make_shared #include // basic_string +#include // move #include // vector #ifndef JSON_NO_IO @@ -44,22 +45,32 @@ template struct output_adapter_protocol template using output_adapter_t = std::shared_ptr>; -/// output adapter for byte vectors +/// @brief non-virtual output sink writing into a std::vector +/// +/// This sink is not part of the virtual output_adapter_protocol hierarchy: it is +/// passed to binary_writer by value as a template parameter, so +/// write_character()/write_characters() are ordinary (inlinable) calls with no +/// vtable lookup and no shared_ptr. It is used for the common +/// `to_cbor`/`to_msgpack`/... into a std::vector. output_vector_adapter below +/// wraps this same sink to provide the virtual interface. template> -class output_vector_adapter : public output_adapter_protocol +class output_vector_sink { public: - explicit output_vector_adapter(std::vector& vec) noexcept + explicit output_vector_sink(std::vector& vec) noexcept : v(vec) {} - void write_character(CharType c) override + void write_character(CharType c) { v.push_back(c); } - JSON_HEDLEY_NON_NULL(2) - void write_characters(const CharType* s, std::size_t length) override + // no JSON_HEDLEY_NON_NULL here: binary_writer legitimately passes a null + // pointer with length 0 for empty strings/binary values. Appending an empty + // range is a no-op; the type-erased path tolerates this via the (unattributed) + // virtual base, and the concrete sink must do the same. + void write_characters(const CharType* s, std::size_t length) { v.insert(v.end(), s, s + length); } @@ -68,6 +79,34 @@ class output_vector_adapter : public output_adapter_protocol std::vector& v; }; +/// output adapter for byte vectors +/// +/// The appending itself lives in output_vector_sink; this class only adds the +/// virtual output_adapter_protocol interface on top of it, so both the +/// type-erased and the templated path share one implementation. +template> +class output_vector_adapter : public output_adapter_protocol +{ + public: + explicit output_vector_adapter(std::vector& vec) noexcept + : sink(vec) + {} + + void write_character(CharType c) override + { + sink.write_character(c); + } + + JSON_HEDLEY_NON_NULL(2) + void write_characters(const CharType* s, std::size_t length) override + { + sink.write_characters(s, length); + } + + private: + output_vector_sink sink; +}; + #ifndef JSON_NO_IO /// output adapter for output streams template @@ -118,6 +157,39 @@ class output_string_adapter : public output_adapter_protocol StringType& str; }; +/// @brief output sink forwarding to a type-erased output adapter +/// +/// Wraps the polymorphic output_adapter_t so the same binary_writer template can +/// also target arbitrary adapters (output streams, strings, user-provided +/// adapters) via the `output_adapter`-based overloads. Each write still goes +/// through one virtual call, exactly as before; only the concrete sinks above +/// avoid it. +template +class output_adapter_sink +{ + public: + explicit output_adapter_sink(output_adapter_t adapter) + : oa(std::move(adapter)) + { + JSON_ASSERT(oa); + } + + void write_character(CharType c) + { + oa->write_character(c); + } + + // no JSON_HEDLEY_NON_NULL: forwards (null, 0) for empty payloads, exactly as + // the type-erased path already did before this sink existed + void write_characters(const CharType* s, std::size_t length) + { + oa->write_characters(s, length); + } + + private: + output_adapter_t oa; +}; + template> class output_adapter { diff --git a/include/nlohmann/detail/output/serializer.hpp b/include/nlohmann/detail/output/serializer.hpp index 0b608f8e2..7c38276ce 100644 --- a/include/nlohmann/detail/output/serializer.hpp +++ b/include/nlohmann/detail/output/serializer.hpp @@ -9,25 +9,28 @@ #pragma once -#include // reverse, remove, fill, find, none_of +#include // reverse, remove, fill, find, none_of, min #include // array #include // localeconv, lconv #include // labs, isfinite, isnan, signbit #include // size_t, ptrdiff_t #include // uint8_t #include // snprintf +#include // memcpy, memset #include // numeric_limits #include // string, char_traits -#include // setfill, setw #include // is_same #include // move +#include // vector #include #include +#include #include #include #include #include +#include #include #include @@ -60,18 +63,32 @@ class serializer public: /*! - @param[in] s output stream to serialize to + @param[in] s output adapter to serialize to; not owned by the serializer, + so it must outlive it (it lives at the call site) @param[in] ichar indentation character to use + @param[in] pretty_print_ whether the output shall be pretty-printed + @param[in] ensure_ascii_ If @a ensure_ascii_ is true, all non-ASCII + characters in the output are escaped with `\uXXXX` sequences, and the + result consists of ASCII characters only. + @param[in] indent_step_ the indent level @param[in] error_handler_ how to react on decoding errors + + None of @a pretty_print_, @a ensure_ascii_ and @a indent_step_ change over + the life of the serializer, so they are captured once here instead of + being threaded through every call to @ref dump, @ref dump_internal and + @ref dump_iteratively. */ - serializer(output_adapter_t s, const char ichar, + serializer(output_adapter_protocol& s, const char ichar, + const bool pretty_print_ = false, + const bool ensure_ascii_ = false, + const std::size_t indent_step_ = 0, error_handler_t error_handler_ = error_handler_t::strict) - : o(std::move(s)) - , loc(std::localeconv()) - , thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits::to_char_type(* (loc->thousands_sep))) - , decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits::to_char_type(* (loc->decimal_point))) + : o(&s) + , locale(std::localeconv()) , indent_char(ichar) - , indent_string(512, indent_char) + , pretty_print(pretty_print_) + , ensure_ascii(ensure_ascii_) + , indent_step(indent_step_) , error_handler(error_handler_) {} @@ -87,8 +104,8 @@ class serializer This function is called by the public member function dump and organizes the serialization internally. The indentation level is propagated as - additional parameter. In case of arrays and objects, the function is - called recursively. + additional parameter. Arrays and objects are serialized without recursion, + however deeply they are nested. - strings and object keys are escaped using `escape_string()` - integer numbers are converted implicitly via `operator<<` @@ -97,89 +114,109 @@ class serializer byte array @param[in] val value to serialize - @param[in] pretty_print whether the output shall be pretty-printed - @param[in] ensure_ascii If @a ensure_ascii is true, all non-ASCII characters - in the output are escaped with `\uXXXX` sequences, and the result consists - of ASCII characters only. - @param[in] indent_step the indent level @param[in] current_indent the current indent level (only used internally) */ void dump(const BasicJsonType& val, - const bool pretty_print, - const bool ensure_ascii, - const unsigned int indent_step, - const unsigned int current_indent = 0) + const std::size_t current_indent = 0) + { + dump_internal(val, current_indent); + flush(); + } + + JSON_PRIVATE_UNLESS_TESTED: + /*! + @brief worker for @ref dump + + Identical in behavior to the historical @ref dump, but writes into the + serializer's internal @ref write_buffer instead of issuing a virtual call + per token. The public @ref dump wraps this and flushes the buffer once the + top-level value has been serialized. + + Serializing a container descends into its elements, so a value nested deeply + enough used to exhaust the call stack and terminate the process with no + exception to catch. The descent is bounded here: once @ref recursion_depth_limit + levels have been entered, @ref dump_iteratively writes out what is left + without the call stack. A value nested less deeply than that - all but a + vanishing minority - is written by exactly the code that always wrote it. + + @sa https://github.com/nlohmann/json/issues/5387 + */ + void dump_internal(const BasicJsonType& val, + const std::size_t current_indent = 0, + const std::size_t depth = 0) { switch (val.m_data.m_type) { case value_t::object: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + dump_iteratively(val, current_indent); + return; + } + if (val.m_data.m_value.object->empty()) { - o->write_characters("{}", 2); + put_literal("{}"); return; } if (pretty_print) { - o->write_characters("{\n", 2); + put_literal("{\n"); // variable to hold indentation for recursive calls - const auto new_indent = current_indent + indent_step; - if (JSON_HEDLEY_UNLIKELY(indent_string.size() < new_indent)) - { - indent_string.resize(indent_string.size() * 2, ' '); - } + const auto new_indent = next_indent(current_indent, indent_step); // first n-1 elements auto i = val.m_data.m_value.object->cbegin(); for (std::size_t cnt = 0; cnt < val.m_data.m_value.object->size() - 1; ++cnt, ++i) { - o->write_characters(indent_string.c_str(), new_indent); - o->write_character('\"'); - dump_escaped(i->first, ensure_ascii); - o->write_characters("\": ", 3); - dump(i->second, true, ensure_ascii, indent_step, new_indent); - o->write_characters(",\n", 2); + put_indent(new_indent); + put_char('"'); + dump_escaped(i->first); + put_literal("\": "); + dump_internal(i->second, new_indent, depth + 1); + put_literal(",\n"); } // last element JSON_ASSERT(i != val.m_data.m_value.object->cend()); JSON_ASSERT(std::next(i) == val.m_data.m_value.object->cend()); - o->write_characters(indent_string.c_str(), new_indent); - o->write_character('\"'); - dump_escaped(i->first, ensure_ascii); - o->write_characters("\": ", 3); - dump(i->second, true, ensure_ascii, indent_step, new_indent); + put_indent(new_indent); + put_char('"'); + dump_escaped(i->first); + put_literal("\": "); + dump_internal(i->second, new_indent, depth + 1); - o->write_character('\n'); - o->write_characters(indent_string.c_str(), current_indent); - o->write_character('}'); + put_char('\n'); + put_indent(current_indent); + put_char('}'); } else { - o->write_character('{'); + put_char('{'); // first n-1 elements auto i = val.m_data.m_value.object->cbegin(); for (std::size_t cnt = 0; cnt < val.m_data.m_value.object->size() - 1; ++cnt, ++i) { - o->write_character('\"'); - dump_escaped(i->first, ensure_ascii); - o->write_characters("\":", 2); - dump(i->second, false, ensure_ascii, indent_step, current_indent); - o->write_character(','); + put_char('"'); + dump_escaped(i->first); + put_literal("\":"); + dump_internal(i->second, current_indent, depth + 1); + put_char(','); } // last element JSON_ASSERT(i != val.m_data.m_value.object->cend()); JSON_ASSERT(std::next(i) == val.m_data.m_value.object->cend()); - o->write_character('\"'); - dump_escaped(i->first, ensure_ascii); - o->write_characters("\":", 2); - dump(i->second, false, ensure_ascii, indent_step, current_indent); + put_char('"'); + dump_escaped(i->first); + put_literal("\":"); + dump_internal(i->second, current_indent, depth + 1); - o->write_character('}'); + put_char('}'); } return; @@ -187,58 +224,60 @@ class serializer case value_t::array: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + dump_iteratively(val, current_indent); + return; + } + if (val.m_data.m_value.array->empty()) { - o->write_characters("[]", 2); + put_literal("[]"); return; } if (pretty_print) { - o->write_characters("[\n", 2); + put_literal("[\n"); // variable to hold indentation for recursive calls - const auto new_indent = current_indent + indent_step; - if (JSON_HEDLEY_UNLIKELY(indent_string.size() < new_indent)) - { - indent_string.resize(indent_string.size() * 2, ' '); - } + const auto new_indent = next_indent(current_indent, indent_step); // first n-1 elements for (auto i = val.m_data.m_value.array->cbegin(); i != val.m_data.m_value.array->cend() - 1; ++i) { - o->write_characters(indent_string.c_str(), new_indent); - dump(*i, true, ensure_ascii, indent_step, new_indent); - o->write_characters(",\n", 2); + put_indent(new_indent); + dump_internal(*i, new_indent, depth + 1); + put_literal(",\n"); } // last element JSON_ASSERT(!val.m_data.m_value.array->empty()); - o->write_characters(indent_string.c_str(), new_indent); - dump(val.m_data.m_value.array->back(), true, ensure_ascii, indent_step, new_indent); + put_indent(new_indent); + dump_internal(val.m_data.m_value.array->back(), new_indent, depth + 1); - o->write_character('\n'); - o->write_characters(indent_string.c_str(), current_indent); - o->write_character(']'); + put_char('\n'); + put_indent(current_indent); + put_char(']'); } else { - o->write_character('['); + put_char('['); // first n-1 elements for (auto i = val.m_data.m_value.array->cbegin(); i != val.m_data.m_value.array->cend() - 1; ++i) { - dump(*i, false, ensure_ascii, indent_step, current_indent); - o->write_character(','); + dump_internal(*i, current_indent, depth + 1); + put_char(','); } // last element JSON_ASSERT(!val.m_data.m_value.array->empty()); - dump(val.m_data.m_value.array->back(), false, ensure_ascii, indent_step, current_indent); + dump_internal(val.m_data.m_value.array->back(), current_indent, depth + 1); - o->write_character(']'); + put_char(']'); } return; @@ -246,9 +285,9 @@ class serializer case value_t::string: { - o->write_character('\"'); - dump_escaped(*val.m_data.m_value.string, ensure_ascii); - o->write_character('\"'); + put_char('"'); + dump_escaped(*val.m_data.m_value.string); + put_char('"'); return; } @@ -256,70 +295,66 @@ class serializer { if (pretty_print) { - o->write_characters("{\n", 2); + put_literal("{\n"); // variable to hold indentation for recursive calls - const auto new_indent = current_indent + indent_step; - if (JSON_HEDLEY_UNLIKELY(indent_string.size() < new_indent)) - { - indent_string.resize(indent_string.size() * 2, ' '); - } + const auto new_indent = next_indent(current_indent, indent_step); - o->write_characters(indent_string.c_str(), new_indent); + put_indent(new_indent); - o->write_characters("\"bytes\": [", 10); + put_literal("\"bytes\": ["); if (!val.m_data.m_value.binary->empty()) { for (auto i = val.m_data.m_value.binary->cbegin(); i != val.m_data.m_value.binary->cend() - 1; ++i) { - dump_integer(*i); - o->write_characters(", ", 2); + dump_byte(*i); + put_literal(", "); } - dump_integer(val.m_data.m_value.binary->back()); + dump_byte(val.m_data.m_value.binary->back()); } - o->write_characters("],\n", 3); - o->write_characters(indent_string.c_str(), new_indent); + put_literal("],\n"); + put_indent(new_indent); - o->write_characters("\"subtype\": ", 11); + put_literal("\"subtype\": "); if (val.m_data.m_value.binary->has_subtype()) { dump_integer(val.m_data.m_value.binary->subtype()); } else { - o->write_characters("null", 4); + put_literal("null"); } - o->write_character('\n'); - o->write_characters(indent_string.c_str(), current_indent); - o->write_character('}'); + put_char('\n'); + put_indent(current_indent); + put_char('}'); } else { - o->write_characters("{\"bytes\":[", 10); + put_literal("{\"bytes\":["); if (!val.m_data.m_value.binary->empty()) { for (auto i = val.m_data.m_value.binary->cbegin(); i != val.m_data.m_value.binary->cend() - 1; ++i) { - dump_integer(*i); - o->write_character(','); + dump_byte(*i); + put_char(','); } - dump_integer(val.m_data.m_value.binary->back()); + dump_byte(val.m_data.m_value.binary->back()); } - o->write_characters("],\"subtype\":", 12); + put_literal("],\"subtype\":"); if (val.m_data.m_value.binary->has_subtype()) { dump_integer(val.m_data.m_value.binary->subtype()); - o->write_character('}'); + put_char('}'); } else { - o->write_characters("null}", 5); + put_literal("null}"); } } return; @@ -329,11 +364,11 @@ class serializer { if (val.m_data.m_value.boolean) { - o->write_characters("true", 4); + put_literal("true"); } else { - o->write_characters("false", 5); + put_literal("false"); } return; } @@ -358,13 +393,13 @@ class serializer case value_t::discarded: { - o->write_characters("", 11); + put_literal(""); return; } case value_t::null: { - o->write_characters("null", 4); + put_literal("null"); return; } @@ -373,6 +408,360 @@ class serializer } } + private: + /*! + @brief write out @a val and everything below it without the call stack + + Emits the same bytes as @ref dump_internal, keeping the containers it has + entered on an explicit stack instead of descending into them. Only reached + for values nested deeper than @ref recursion_depth_limit, which is why it is not + written for speed: walking every value this way measured up to 20% slower on + object-heavy documents than letting the compiler drive the descent. + */ + void dump_iteratively(const BasicJsonType& val, + const std::size_t current_indent = 0) + { + // Scalars, empty containers and binary values are written by dump_value + // alone, so nothing is allocated for them: only a container with + // elements is ever pushed. + std::vector stack; + + dump_value(val, current_indent, stack); + + while (!stack.empty()) + { + dump_frame& frame = stack.back(); + + if (frame.value->m_data.m_type == value_t::object) + { + const auto* object = frame.value->m_data.m_value.object; + + if (frame.object_it == object->cend()) + { + if (pretty_print) + { + put_char('\n'); + put_indent(frame.current_indent); + } + + put_char('}'); + stack.pop_back(); + continue; + } + + // the separator goes in front of every element but the first, + // which puts exactly one between each pair and none at the end + if (frame.object_it != object->cbegin()) + { + if (pretty_print) + { + put_literal(",\n"); + } + else + { + put_char(','); + } + } + + if (pretty_print) + { + put_indent(frame.child_indent); + } + + put_char('"'); + dump_escaped(frame.object_it->first); + + if (pretty_print) + { + put_literal("\": "); + } + else + { + put_literal("\":"); + } + + const BasicJsonType& element = frame.object_it->second; + ++frame.object_it; + + // read everything needed from the frame before this: entering a + // container pushes another one and can move them all + const std::size_t element_indent = frame.child_indent; + dump_value(element, element_indent, stack); + } + else + { + const auto* array = frame.value->m_data.m_value.array; + + if (frame.array_it == array->cend()) + { + if (pretty_print) + { + put_char('\n'); + put_indent(frame.current_indent); + } + + put_char(']'); + stack.pop_back(); + continue; + } + + if (frame.array_it != array->cbegin()) + { + if (pretty_print) + { + put_literal(",\n"); + } + else + { + put_char(','); + } + } + + if (pretty_print) + { + put_indent(frame.child_indent); + } + + const BasicJsonType& element = *frame.array_it; + ++frame.array_it; + + // see above + const std::size_t element_indent = frame.child_indent; + dump_value(element, element_indent, stack); + } + } + } + + private: + /// @brief a container that has been opened but not closed yet + struct dump_frame + { + dump_frame(const BasicJsonType* value_, const std::size_t current_indent_, + const std::size_t child_indent_) noexcept + : value(value_) + , current_indent(current_indent_) + , child_indent(child_indent_) + {} + + /// the object or array being serialized + const BasicJsonType* value; + /// the element to serialize next; which of the two is live follows from + /// the type of @a value. They are kept side by side rather than in a + /// union, which would need its special members written out by hand, see + /// detail/iterators/internal_iterator.hpp + typename BasicJsonType::object_t::const_iterator object_it{}; + typename BasicJsonType::array_t::const_iterator array_it{}; + /// the indentation of the container itself, used by its closing bracket + std::size_t current_indent; + /// the indentation of the container's elements + std::size_t child_indent; + }; + + /*! + @brief serialize the value @a val, but not the elements of a container + + An object or array with elements is opened and pushed onto @a stack for + @ref dump_internal to walk; everything else - including a binary value, + which looks like an object but has no elements to descend into - is written + out here in full. + */ + void dump_value(const BasicJsonType& val, + const std::size_t current_indent, + std::vector& stack) + { + switch (val.m_data.m_type) + { + case value_t::object: + { + if (val.m_data.m_value.object->empty()) + { + put_literal("{}"); + return; + } + + std::size_t child_indent = current_indent; + + if (pretty_print) + { + put_literal("{\n"); + child_indent = next_indent(current_indent, indent_step); + } + else + { + put_char('{'); + } + + stack.emplace_back(&val, current_indent, child_indent); + stack.back().object_it = val.m_data.m_value.object->cbegin(); + return; + } + + case value_t::array: + { + if (val.m_data.m_value.array->empty()) + { + put_literal("[]"); + return; + } + + std::size_t child_indent = current_indent; + + if (pretty_print) + { + put_literal("[\n"); + child_indent = next_indent(current_indent, indent_step); + } + else + { + put_char('['); + } + + stack.emplace_back(&val, current_indent, child_indent); + stack.back().array_it = val.m_data.m_value.array->cbegin(); + return; + } + + case value_t::string: + { + put_char('"'); + dump_escaped(*val.m_data.m_value.string); + put_char('"'); + return; + } + + case value_t::binary: + { + if (pretty_print) + { + put_literal("{\n"); + + // variable to hold indentation for the bytes + const auto new_indent = next_indent(current_indent, indent_step); + + put_indent(new_indent); + + put_literal("\"bytes\": ["); + + if (!val.m_data.m_value.binary->empty()) + { + for (auto i = val.m_data.m_value.binary->cbegin(); + i != val.m_data.m_value.binary->cend() - 1; ++i) + { + dump_byte(*i); + put_literal(", "); + } + dump_byte(val.m_data.m_value.binary->back()); + } + + put_literal("],\n"); + put_indent(new_indent); + + put_literal("\"subtype\": "); + if (val.m_data.m_value.binary->has_subtype()) + { + dump_integer(val.m_data.m_value.binary->subtype()); + } + else + { + put_literal("null"); + } + put_char('\n'); + put_indent(current_indent); + put_char('}'); + } + else + { + put_literal("{\"bytes\":["); + + if (!val.m_data.m_value.binary->empty()) + { + for (auto i = val.m_data.m_value.binary->cbegin(); + i != val.m_data.m_value.binary->cend() - 1; ++i) + { + dump_byte(*i); + put_char(','); + } + dump_byte(val.m_data.m_value.binary->back()); + } + + put_literal("],\"subtype\":"); + if (val.m_data.m_value.binary->has_subtype()) + { + dump_integer(val.m_data.m_value.binary->subtype()); + put_char('}'); + } + else + { + put_literal("null}"); + } + } + return; + } + + case value_t::boolean: + { + if (val.m_data.m_value.boolean) + { + put_literal("true"); + } + else + { + put_literal("false"); + } + return; + } + + case value_t::number_integer: + { + dump_integer(val.m_data.m_value.number_integer); + return; + } + + case value_t::number_unsigned: + { + dump_integer(val.m_data.m_value.number_unsigned); + return; + } + + case value_t::number_float: + { + dump_float(val.m_data.m_value.number_float); + return; + } + + case value_t::discarded: + { + put_literal(""); + return; + } + + case value_t::null: + { + put_literal("null"); + return; + } + + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + } + } + + + /*! + @brief the indentation level to use for the children of the current value + + A very large @a indent_step can wrap the unsigned accumulation on deep + nesting, which would silently truncate the indentation. Far harder to reach + now that the accumulator is a std::size_t, but still reachable where that is + 32 bits wide. + */ + static std::size_t next_indent(const std::size_t current_indent, const std::size_t indent_step) + { + const std::size_t new_indent = current_indent + indent_step; + JSON_ASSERT(new_indent >= current_indent); + return new_indent; + } + JSON_PRIVATE_UNLESS_TESTED: /*! @brief dump escaped string @@ -383,12 +772,32 @@ class serializer representation. The escaped string is written to output stream @a o. @param[in] s the string to escape - @param[in] ensure_ascii whether to escape non-ASCII characters with - \uXXXX sequences @complexity Linear in the length of string @a s. */ - void dump_escaped(const string_t& s, const bool ensure_ascii) + void dump_escaped(const string_t& s) + { + // dispatch once here rather than test the flag inside the loop: it does + // not change while a string is written, and folding it lets each of the + // two scanners be inlined into a loop of its own + if (ensure_ascii) + { + dump_escaped_impl(s); + } + else + { + dump_escaped_impl(s); + } + } + + /*! + @brief worker for @ref dump_escaped + + @a ensure_ascii is a template parameter here so that the branch on it is + resolved once, outside the loop; see @ref dump_escaped. + */ + template + void dump_escaped_impl(const string_t& s) { std::uint32_t codepoint{}; std::uint8_t state = UTF8_ACCEPT; @@ -400,6 +809,56 @@ class serializer for (std::size_t i = 0; i < s.size(); ++i) { + // Fast path: at a character boundary (state == UTF8_ACCEPT), + // bulk-copy the longest run of bytes that need no escaping using a + // SWAR scanner shared with the lexer's contiguous path. The scanner + // stops exactly at the first byte dump_escaped would handle + // individually, so that byte is left to the byte-at-a-time path + // below, keeping escaping output and error diagnostics unchanged. + // + // - EnsureAscii == false: string_bulk_run() copies ordinary bytes + // and complete well-formed UTF-8, stopping at a quote, backslash, + // control character (< 0x20), or ill-formed/truncated sequence. + // - EnsureAscii == true: only printable ASCII may be copied + // verbatim; find_ascii_copyable_run() additionally stops at 0x7F + // and every non-ASCII byte (>= 0x80), which must be \u-escaped. + if (state == UTF8_ACCEPT) + { + const auto* const data = reinterpret_cast(s.data()); + // A run can only be non-empty when the very first byte is one + // the scanner may copy, so test that single byte before paying + // for the scan. Without it, text whose characters all have to be + // escaped - CJK under ensure_ascii, where every byte is >= 0x80 - + // runs the scanner once per character only to be told zero. + std::size_t run = 0; + if (!EnsureAscii) + { + run = string_bulk_run(data + i, s.size() - i); + } + else if (is_ascii_copyable(data[i])) + { + run = find_ascii_copyable_run(data + i, s.size() - i); + } + if (run != 0) + { + // emit any bytes still pending in string_buffer first to + // preserve output order, then write the run directly + if (bytes != 0) + { + put_buffer(string_buffer, bytes); + bytes = 0; + } + put_string(s, i, i + run); + bytes_after_last_accept = 0; + undumped_chars = 0; + i += run; + if (i >= s.size()) + { + break; + } + } + } + const auto byte = static_cast(s[i]); switch (decode(state, codepoint, byte)) @@ -446,7 +905,7 @@ class serializer case 0x22: // quotation mark { string_buffer[bytes++] = '\\'; - string_buffer[bytes++] = '\"'; + string_buffer[bytes++] = '"'; break; } @@ -460,8 +919,8 @@ class serializer default: { // escape control characters (0x00..0x1F) or, if - // ensure_ascii parameter is used, non-ASCII characters - if ((codepoint <= 0x1F) || (ensure_ascii && (codepoint >= 0x7F))) + // EnsureAscii parameter is used, non-ASCII characters + if ((codepoint <= 0x1F) || (EnsureAscii && (codepoint >= 0x7F))) { if (codepoint <= 0xFFFF) { @@ -488,7 +947,7 @@ class serializer // written ("\uxxxx\uxxxx\0") for one code point if (string_buffer.size() - bytes < 13) { - o->write_characters(string_buffer.data(), bytes); + put_buffer(string_buffer, bytes); bytes = 0; } @@ -526,7 +985,7 @@ class serializer if (error_handler == error_handler_t::replace) { // add a replacement character - if (ensure_ascii) + if (EnsureAscii) { string_buffer[bytes++] = '\\'; string_buffer[bytes++] = 'u'; @@ -547,7 +1006,7 @@ class serializer // written ("\uxxxx\uxxxx\0") for one code point if (string_buffer.size() - bytes < 13) { - o->write_characters(string_buffer.data(), bytes); + put_buffer(string_buffer, bytes); bytes = 0; } @@ -569,7 +1028,7 @@ class serializer default: // decode found yet incomplete multibyte code point { - if (!ensure_ascii) + if (!EnsureAscii) { // code point will not be escaped - copy byte to buffer string_buffer[bytes++] = s[i]; @@ -586,7 +1045,7 @@ class serializer // write buffer if (bytes > 0) { - o->write_characters(string_buffer.data(), bytes); + put_buffer(string_buffer, bytes); } } else @@ -596,28 +1055,28 @@ class serializer { case error_handler_t::strict: { - JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast(s.back() | 0))), nullptr)); + JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast(s[s.size() - 1] | 0))), nullptr)); } case error_handler_t::ignore: { // write all accepted bytes - o->write_characters(string_buffer.data(), bytes_after_last_accept); + put_buffer(string_buffer, bytes_after_last_accept); break; } case error_handler_t::replace: { // write all accepted bytes - o->write_characters(string_buffer.data(), bytes_after_last_accept); + put_buffer(string_buffer, bytes_after_last_accept); // add a replacement character - if (ensure_ascii) + if (EnsureAscii) { - o->write_characters("\\ufffd", 6); + put_literal("\\ufffd"); } else { - o->write_characters("\xEF\xBF\xBD", 3); + put_literal("\xEF\xBF\xBD"); } break; } @@ -628,6 +1087,160 @@ class serializer } } + private: + /*! + @brief append a single character to the write buffer + + Structural characters ('{', '"', ',', ...) previously went straight to the + output adapter, one virtual call each. Buffering them and flushing in bulk + turns those many indirect calls into a single memcpy plus an occasional + flush, which dominates the cost of serializing object/array-heavy values. + */ + void put_char(char c) + { + if (JSON_HEDLEY_UNLIKELY(write_buffer_pos == write_buffer.size())) + { + flush(); + } + write_buffer[write_buffer_pos++] = c; + } + + /*! + @brief append @a indent indentation characters to the write buffer + + Writes the indentation straight into the buffer instead of copying it out of + a pre-grown indentation string, so no auxiliary string has to be sized, + resized, or kept in sync with the deepest nesting level reached. + + An indentation wider than the buffer is emitted by filling the buffer with + the indentation character once and flushing that same content repeatedly: + flushing does not disturb what the buffer holds, so re-filling it between + flushes would be redundant work. + */ + void put_indent(std::size_t indent) + { + // closing braces at the outermost level ask for no indentation at all + if (indent == 0) + { + return; + } + + const std::size_t capacity = write_buffer.size(); + + // fill whatever room is left in the buffer; this is the whole job + // whenever the indentation is narrower than the buffer, which is the + // case for every sane indent_step + const std::size_t head = (std::min)(indent, capacity - write_buffer_pos); + std::memset(write_buffer.data() + write_buffer_pos, indent_char, head); + write_buffer_pos += head; + indent -= head; + + if (JSON_HEDLEY_LIKELY(indent == 0)) + { + return; + } + + // the buffer is full and the remainder spans whole buffer-fulls: flush + // what is pending, then fill the buffer with the indentation character + // exactly once and hand the same bytes to the adapter as often as needed + flush(); + std::memset(write_buffer.data(), indent_char, capacity); + + while (indent >= capacity) + { + write_buffer_pos = capacity; + flush(); + indent -= capacity; + } + + // the buffer still holds indentation characters throughout, so the tail + // only has to be claimed, not written again + write_buffer_pos = indent; + } + + /*! + @brief append a string literal to the write buffer + + The length comes from the array bound rather than a hand-written count, so + it cannot drift out of sync with the literal. A literal always fits into the + buffer (checked at compile time), so unlike @ref put_string this needs no + write-through path for oversized runs. + */ + template + void put_literal(const char (&s)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + { + static_assert(N >= 2, "put_literal expects a non-empty string literal"); + // the array bound counts the terminating NUL, which is not written + constexpr std::size_t length = N - 1; + static_assert(length < write_buffer_size, "string literal must fit into the write buffer"); + + if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + length > write_buffer.size())) + { + flush(); + } + std::memcpy(write_buffer.data() + write_buffer_pos, s, length); + write_buffer_pos += length; + } + + /*! + @brief append the characters of @a str in [@a start, @a end) + + The only way to append a run of characters: @a str carries its own bound, + so the range can be checked against it, which a bare pointer plus a count + could not do. Runs that do not fit the buffer are written straight through + the output adapter (after flushing what is pending), so large string and + number payloads are not copied an extra time. + */ + template + void put_string(const StringType& str, std::size_t start, std::size_t end) + { + JSON_ASSERT(start <= end); + JSON_ASSERT(end <= str.size()); + + const char* const s = str.data() + start; + const std::size_t length = end - start; + + if (JSON_HEDLEY_UNLIKELY(length >= write_buffer.size())) + { + flush(); + o->write_characters(s, length); + return; + } + if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + length > write_buffer.size())) + { + flush(); + } + std::memcpy(write_buffer.data() + write_buffer_pos, s, length); + write_buffer_pos += length; + } + + /*! + @brief append the first @a length characters of a fixed-size buffer + */ + template + void put_buffer(const std::array& buffer, std::size_t length) + { + put_string(buffer, 0, length); + } + + JSON_PRIVATE_UNLESS_TESTED: + /*! + @brief flush the write buffer to the output adapter + + Writing zero characters is a well-defined no-op for every output adapter, so + the buffered length is passed through unconditionally (no empty-guard branch + to leave uncovered). + + @note dump_escaped() and dump_integer()/dump_float() write into the internal + write buffer; callers that invoke them directly (rather than through the + public dump()) must call flush() before inspecting the output. + */ + void flush() + { + o->write_characters(write_buffer.data(), write_buffer_pos); + write_buffer_pos = 0; + } + private: /*! @brief count digits @@ -703,6 +1316,19 @@ class serializer pos += 6; } + /*! + @brief convert a single element of a binary value to its byte value + + The elements of a binary value are dumped as the numbers 0..255, regardless + of the value type of the configured BinaryType: that type may be signed + (`char`), unsigned (`std::uint8_t`), or not an integer at all + (`std::byte`), none of which @ref dump_integer can handle uniformly. + */ + static std::uint8_t to_byte_value(binary_char_t x) noexcept + { + return static_cast(x); + } + // templates to avoid warnings about useless casts template ::value, int> = 0> bool is_negative_number(NumberType x) @@ -716,6 +1342,64 @@ class serializer return false; } + /*! + @brief write the decimal representation of the byte @a value + + A binary value's bytes are always in [0, 255], so writing one needs neither + the digit counting nor the 64-bit arithmetic that @ref dump_integer does for + an arbitrary number, and the three digits it takes at most are written + straight into the write buffer. + + Any byte type that is not a plain unsigned byte is converted to its + @ref to_byte_value "byte value" and left to @ref dump_integer, so a signed + or non-integral BinaryType::value_type (`char`, `std::byte`, ...) still + dumps as 0..255. + */ + template + void dump_byte(const ByteType value) + { + dump_byte(value, std::integral_constant < bool, + std::is_unsigned::value && sizeof(ByteType) == 1 + && !std::is_same::value > {}); + } + + template + void dump_byte(const ByteType value, std::false_type /*is_plain_byte*/) + { + dump_integer(to_byte_value(value)); + } + + template + void dump_byte(const ByteType value, std::true_type /*is_plain_byte*/) + { + if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + 3 > write_buffer.size())) + { + flush(); + } + + const auto byte = static_cast(value); + // Accumulate the offset in a local and store it back once. Writing + // through write_buffer[] is a char write, which may alias any object, + // so with the member updated in place the compiler has to reload and + // store it around every digit - measured 2.4x slower on a dump of a + // multi-megabyte binary value. + std::size_t pos = write_buffer_pos; + + if (byte >= 100) + { + write_buffer[pos++] = static_cast('0' + (byte / 100)); + write_buffer[pos++] = static_cast('0' + ((byte / 10) % 10)); + } + else if (byte >= 10) + { + write_buffer[pos++] = static_cast('0' + (byte / 10)); + } + + write_buffer[pos++] = static_cast('0' + (byte % 10)); + + write_buffer_pos = pos; + } + /*! @brief dump an integer @@ -728,8 +1412,7 @@ class serializer template < typename NumberType, detail::enable_if_t < std::is_integral::value || std::is_same::value || - std::is_same::value || - std::is_same::value, + std::is_same::value, int > = 0 > void dump_integer(NumberType x) { @@ -752,7 +1435,7 @@ class serializer // special case for "0" if (x == 0) { - o->write_character('0'); + put_char('0'); return; } @@ -805,7 +1488,7 @@ class serializer *(--buffer_ptr) = static_cast('0' + abs_value); } - o->write_characters(number_buffer.data(), n_chars); + put_buffer(number_buffer, n_chars); } /*! @@ -821,7 +1504,7 @@ class serializer // NaN / inf if (!std::isfinite(x)) { - o->write_characters("null", 4); + put_literal("null"); return; } @@ -842,7 +1525,7 @@ class serializer auto* begin = number_buffer.data(); auto* end = ::nlohmann::detail::to_chars(begin, begin + number_buffer.size(), x); - o->write_characters(begin, static_cast(end - begin)); + put_buffer(number_buffer, static_cast(end - begin)); } JSON_HEDLEY_NON_NULL(1) @@ -873,27 +1556,27 @@ class serializer JSON_ASSERT(static_cast(len) < number_buffer.size()); // erase thousands separators - if (thousands_sep != '\0') + if (locale.thousands_sep != '\0') { // NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::remove returns an iterator, see https://github.com/nlohmann/json/issues/3081 - const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, thousands_sep); + const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, locale.thousands_sep); std::fill(end, number_buffer.end(), '\0'); JSON_ASSERT((end - number_buffer.begin()) <= len); len = (end - number_buffer.begin()); } // convert decimal point to '.' - if (decimal_point != '\0' && decimal_point != '.') + if (locale.decimal_point != '\0' && locale.decimal_point != '.') { // NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::find returns an iterator, see https://github.com/nlohmann/json/issues/3081 - const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), decimal_point); + const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), locale.decimal_point); if (dec_pos != number_buffer.end()) { *dec_pos = '.'; } } - o->write_characters(number_buffer.data(), static_cast(len)); + put_buffer(number_buffer, static_cast(len)); // determine if we need to append ".0" const bool value_is_int_like = @@ -905,7 +1588,7 @@ class serializer if (value_is_int_like) { - o->write_characters(".0", 2); + put_literal(".0"); } } @@ -992,29 +1675,53 @@ class serializer } private: - /// the output of the serializer - output_adapter_t o = nullptr; + /// the locale's thousand separator and decimal point characters + struct locale_chars + { + explicit locale_chars(const std::lconv* loc) noexcept + : thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits::to_char_type(* (loc->thousands_sep))) + , decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits::to_char_type(* (loc->decimal_point))) + {} + + const char thousands_sep; + const char decimal_point; + }; + + /// the output of the serializer (non-owning; the adapter lives at the call site) + output_adapter_protocol* o = nullptr; /// a (hopefully) large enough character buffer std::array number_buffer{{}}; - /// the locale - const std::lconv* loc = nullptr; - /// the locale's thousand separator character - const char thousands_sep = '\0'; - /// the locale's decimal point character - const char decimal_point = '\0'; + /// computed once from std::localeconv() at construction; @ref + /// locale_chars keeps std::localeconv()'s pointer from having to be held + /// past the constructor, while still letting these stay const + const locale_chars locale; /// string buffer std::array string_buffer{{}}; /// the indentation character const char indent_char; - /// the indentation string - string_t indent_string; + + /// whether to pretty-print the output + const bool pretty_print; + + /// whether to escape non-ASCII characters with \uXXXX sequences + const bool ensure_ascii; + + /// the indent level + const std::size_t indent_step; /// error_handler how to react on decoding errors const error_handler_t error_handler; + + /// buffer collecting output before it is flushed to the output adapter, so + /// that the many small structural writes become few bulk writes + static constexpr std::size_t write_buffer_size = 1024; + std::array write_buffer{{}}; + /// number of valid bytes currently held in @ref write_buffer + std::size_t write_buffer_pos = 0; }; } // namespace detail diff --git a/include/nlohmann/detail/recursion_depth_limit.hpp b/include/nlohmann/detail/recursion_depth_limit.hpp new file mode 100644 index 000000000..fe3bd8026 --- /dev/null +++ b/include/nlohmann/detail/recursion_depth_limit.hpp @@ -0,0 +1,35 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // size_t + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief the number of nesting levels an operation recurses into + +Operations that walk a value (serializing, hashing, merging, ...) recurse once +per nesting level, which is fastest, but a value nested deeply enough would +exhaust the call stack. So they recurse only this many levels deep and finish +whatever lies below with an explicit stack. All of them share this limit. + +@sa https://github.com/nlohmann/json/issues/5387 +*/ +constexpr std::size_t recursion_depth_limit() noexcept +{ + return 128; +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/string_escape.hpp b/include/nlohmann/detail/string_escape.hpp index 7715dde7a..0d24a56bc 100644 --- a/include/nlohmann/detail/string_escape.hpp +++ b/include/nlohmann/detail/string_escape.hpp @@ -8,50 +8,56 @@ #pragma once +#include // size_t + #include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail { -/*! -@brief replace all occurrences of a substring by another string - -@param[in,out] s the string to manipulate; changed so that all - occurrences of @a f are replaced with @a t -@param[in] f the substring to replace with @a t -@param[in] t the string to replace @a f - -@pre The search string @a f must not be empty. **This precondition is -enforced with an assertion.** - -@since version 2.0.0 -*/ -template -inline void replace_substring(StringType& s, const StringType& f, - const StringType& t) -{ - JSON_ASSERT(!f.empty()); - for (auto pos = s.find(f); // find the first occurrence of f - pos != StringType::npos; // make sure f was found - s.replace(pos, f.size(), t), // replace with t, and - pos = s.find(f, pos + t.size())) // find the next occurrence of f - {} -} - /*! * @brief string escaping as described in RFC 6901 (Sect. 4) * @param[in] s string to escape * @return escaped string * * Note the order of escaping "~" to "~0" and "/" to "~1" is important. + * + * The string is rebuilt in a single pass, appending whole runs between the + * characters that need escaping. Scanning with find_first_of() keeps the + * common case -- nothing to escape -- as fast as a single search, while + * repeated replace() calls would move the tail of the string once per + * escaped character. */ template -inline StringType escape(StringType s) +inline StringType escape(const StringType& s) { - replace_substring(s, StringType{"~"}, StringType{"~0"}); - replace_substring(s, StringType{"/"}, StringType{"~1"}); - return s; + auto next_special = [&s](std::size_t from) + { + const auto tilde = s.find_first_of('~', from); + const auto slash = s.find_first_of('/', from); + return tilde < slash ? tilde : slash; // npos is the largest value + }; + + auto pos = next_special(0); + if (pos == StringType::npos) + { + return s; + } + + StringType result; + result.reserve(s.size() + 2); + + std::size_t run = 0; + while (pos != StringType::npos) + { + result.append(s.data() + run, pos - run); + result.append(s[pos] == '~' ? "~0" : "~1", 2); + run = pos + 1; + pos = next_special(run); + } + result.append(s.data() + run, s.size() - run); + return result; } /*! @@ -60,12 +66,43 @@ inline StringType escape(StringType s) * @return unescaped string * * Note the order of escaping "~1" to "/" and "~0" to "~" is important. + * + * Rebuilt in a single pass, see @ref escape. A "~" that is followed by + * neither "0" nor "1" is passed through unchanged; @ref json_pointer rejects + * such input before it gets here. */ template inline void unescape(StringType& s) { - replace_substring(s, StringType{"~1"}, StringType{"/"}); - replace_substring(s, StringType{"~0"}, StringType{"~"}); + auto pos = s.find_first_of('~', 0); + if (pos == StringType::npos) + { + return; + } + + StringType result; + result.reserve(s.size()); + + std::size_t run = 0; + while (pos != StringType::npos) + { + result.append(s.data() + run, pos - run); + + const auto next = pos + 1; + if (next < s.size() && (s[next] == '0' || s[next] == '1')) + { + result.append(s[next] == '0' ? "~" : "/", 1); + run = pos + 2; + } + else + { + result.append("~", 1); + run = pos + 1; + } + pos = s.find_first_of('~', run); + } + result.append(s.data() + run, s.size() - run); + s = result; } } // namespace detail diff --git a/include/nlohmann/detail/value_t.hpp b/include/nlohmann/detail/value_t.hpp index 06aefa5ac..9fc217bcc 100644 --- a/include/nlohmann/detail/value_t.hpp +++ b/include/nlohmann/detail/value_t.hpp @@ -9,9 +9,12 @@ #pragma once #include // array +#include // isnan, ldexp, trunc #include // size_t #include // uint8_t +#include // numeric_limits #include // string +#include // is_signed #include #if JSON_HAS_THREE_WAY_COMPARISON @@ -114,5 +117,67 @@ inline bool operator<(const value_t lhs, const value_t rhs) noexcept } #endif + +/*! +@brief compare an integer with a floating point number without precision loss + +Widening the integer to the floating point type loses precision beyond the +float's mantissa, which makes equality intransitive: both 2^63-2 and 2^63-1 +round to 2^63, so each compares equal to that float while differing from each +other. Ordering built on that is not a strict weak ordering, so sorting such +values, or using them as keys in an ordered container, is undefined behavior. + +Returns a value to be compared against zero with the original operator, which +reproduces the exact ordering. A NaN operand is returned as is, so comparing it +against zero keeps NaN's semantics: false for the relational operators and +unordered for `<=>`. +*/ +template +FloatType compare_integer_with_float(const IntegerType i, const FloatType f) noexcept +{ + const auto ordered = [](int c) noexcept + { + return static_cast(c); + }; + + if (std::isnan(f)) + { + return f; + } + + // values of IntegerType lie in [-bound, bound) when signed and in + // [0, bound) when unsigned; digits excludes the sign bit, so bound is a + // power of two that the float represents exactly + const FloatType bound = std::ldexp(static_cast(1), std::numeric_limits::digits); + if (f >= bound) + { + return ordered(-1); + } + if (std::is_signed::value ? (f < -bound) : (f < static_cast(0))) + { + return ordered(1); + } + + // f is now within the integer's range, so truncating it is exact + const FloatType truncated = std::trunc(f); + const auto as_integer = static_cast(truncated); + if (i != as_integer) + { + return ordered(i < as_integer ? -1 : 1); + } + + // the integer parts agree, so any fractional part decides + const FloatType fraction = f - truncated; + if (fraction > static_cast(0)) + { + return ordered(-1); + } + if (fraction < static_cast(0)) + { + return ordered(1); + } + return ordered(0); +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 235e6b737..09dca4694 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -28,14 +28,14 @@ #pragma GCC diagnostic ignored "-Wignored-attributes" #endif -#include // all_of, find, for_each +#include // all_of, find, for_each, none_of #include // nullptr_t, ptrdiff_t, size_t #include // hash, less #include // initializer_list #ifndef JSON_NO_IO #include // istream, ostream #endif // JSON_NO_IO -#include // random_access_iterator_tag +#include // make_move_iterator, random_access_iterator_tag #include // unique_ptr #include // string, stoi, to_string #include // declval, forward, move, pair, swap @@ -68,6 +68,7 @@ #include #include #include +#include #include #include #include @@ -140,7 +141,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec friend ::nlohmann::detail::serializer; template friend class ::nlohmann::detail::iter_impl; - template + template friend class ::nlohmann::detail::binary_writer; template friend class ::nlohmann::detail::binary_reader; @@ -164,11 +165,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::parser_callback_tcb = nullptr, const bool allow_exceptions = true, const bool ignore_comments = false, - const bool ignore_trailing_commas = false + const bool ignore_trailing_commas = false, + const bool discard_number_values = false ) { return ::nlohmann::detail::parser(std::move(adapter), - std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas, discard_number_values); } private: @@ -187,6 +189,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec template using binary_reader = ::nlohmann::detail::binary_reader; template using binary_writer = ::nlohmann::detail::binary_writer; + // binary_writer over a concrete (non-virtual) sink appending into a std::vector, + // used by the vector-returning to_* overloads + template using vector_binary_writer = + ::nlohmann::detail::binary_writer>; + template static vector_binary_writer vector_writer(std::vector& v) + { + return vector_binary_writer(::nlohmann::detail::output_vector_sink(v)); + } JSON_PRIVATE_UNLESS_TESTED: using serializer = ::nlohmann::detail::serializer; @@ -403,6 +413,18 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @} + // Two template parameter requirements that would otherwise be silently + // violated: neither produces a diagnostic of its own, and both corrupt + // values rather than failing. + + static_assert(sizeof(typename BinaryType::value_type) == 1, + "BinaryType::value_type must be exactly one byte wide, " + "because the binary readers and writers reinterpret the container's storage as raw bytes"); + + static_assert(sizeof(NumberUnsignedType) >= sizeof(NumberIntegerType), + "NumberUnsignedType must be at least as wide as NumberIntegerType, " + "because it has to hold the absolute value of every NumberIntegerType value"); + private: /// helper for exception-safe object creation @@ -783,21 +805,76 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return it; } - reference set_parent(reference j, std::size_t old_capacity = detail::unknown_size()) + /// @brief erase an element from the object and return the following one + /// Not every map returns an iterator from erase(iterator): some containers + /// (e.g., Abseil's hash maps) return void to avoid computing a successor + /// the caller may not need. Compute it before erasing for those. + template < typename It, detail::enable_if_t < + !detail::erase_returns_void::value, int > = 0 > + typename object_t::iterator erase_from_object(It pos) + { + return m_data.m_value.object->erase(pos); + } + + template < typename It, detail::enable_if_t < + detail::erase_returns_void::value, int > = 0 > + typename object_t::iterator erase_from_object(It pos) + { + auto next = std::next(pos); + m_data.m_value.object->erase(pos); + return next; + } + + /// @brief the capacity of the stored array, or unknown_size() + /// Only JSON_DIAGNOSTICS uses the value, to detect a reallocation that + /// would invalidate the parent pointers. Array types that do not have a + /// capacity() member function report unknown_size(), which is treated as + /// "the elements may have moved". +#if JSON_DIAGNOSTICS + template < typename A = array_t, detail::enable_if_t < detail::has_capacity::value, int > = 0 > + std::size_t array_capacity() const noexcept + { + return m_data.m_value.array->capacity(); + } + + template < typename A = array_t, detail::enable_if_t < !detail::has_capacity::value, int > = 0 > + std::size_t array_capacity() const noexcept + { + return detail::unknown_size(); + } +#else + static constexpr std::size_t array_capacity() noexcept + { + return detail::unknown_size(); + } +#endif + + /// @brief set the parent of a value that has just been added to an array + /// @param j the added value + /// @param old_capacity the value @ref array_capacity() returned before the + /// insertion + reference set_parent_after_array_insert(reference j, std::size_t old_capacity) { #if JSON_DIAGNOSTICS - if (old_capacity != detail::unknown_size()) + // see https://github.com/nlohmann/json/issues/2838 + JSON_ASSERT(type() == value_t::array); + if (JSON_HEDLEY_UNLIKELY(old_capacity == detail::unknown_size() + || array_capacity() != old_capacity)) { - // see https://github.com/nlohmann/json/issues/2838 - JSON_ASSERT(type() == value_t::array); - if (JSON_HEDLEY_UNLIKELY(m_data.m_value.array->capacity() != old_capacity)) - { - // capacity has changed: update all parents - set_parents(); - return j; - } + // the capacity has changed, or the array type does not let us tell: + // the elements may have moved, so update all parents + set_parents(); + return j; } +#else + static_cast(old_capacity); +#endif + return set_parent(j); + } + reference set_parent(reference j) + { +#if JSON_DIAGNOSTICS // ordered_json uses a vector internally, so pointers could have // been invalidated; see https://github.com/nlohmann/json/issues/2962 #ifdef JSON_HEDLEY_MSVC_VERSION @@ -816,11 +893,377 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec j.m_parent = this; #else static_cast(j); - static_cast(old_capacity); #endif return j; } +#ifndef JSON_NO_THREAD_LOCAL + /// the number of levels an operation descends into before it finishes the + /// value below it without the call stack + static constexpr std::uint8_t nesting_depth_limit() + { + return 128; + } + + /*! + @brief how many levels the operation going on in this thread has descended into + + Copying a value and comparing two values share this count. The library never + nests one inside the other - copying a value does not compare one, and + comparing two values does not copy them - and where user code nests them + anyway, sharing the count only ends a descent sooner than it had to, which + costs a little speed and is never wrong. + + A byte is enough: the count never exceeds the limit by more than the single + level that notices the limit has been reached. + */ + static std::uint8_t& nesting_depth() noexcept + { + static thread_local std::uint8_t depth = 0; // NOLINT(misc-use-internal-linkage) + return depth; + } +#endif + + /*! + @brief counts one level of a bounded descent for as long as it runs, and + reports whether the descent was still within the limit when it began + + Looks the count up and tests it against the limit itself, rather than + leaving that to the caller: either way it is reached exactly once, so + there is nothing to be gained by making the caller do it. + + Does nothing and is never @ref okay without thread-local storage, where no + descent can be bounded at all: a caller that only descends while this says + it may always ends up finishing without the call stack, exactly as if every + value were nested past the limit. + */ + class nesting_depth_guard + { + public: + nesting_depth_guard() noexcept +#ifdef JSON_NO_THREAD_LOCAL + : m_okay(false) +#else + : m_okay(nesting_depth() < nesting_depth_limit()) +#endif + { +#ifndef JSON_NO_THREAD_LOCAL + ++nesting_depth(); +#endif + } + + ~nesting_depth_guard() + { +#ifndef JSON_NO_THREAD_LOCAL + --nesting_depth(); +#endif + } + + nesting_depth_guard(const nesting_depth_guard&) = delete; + nesting_depth_guard& operator=(const nesting_depth_guard&) = delete; + nesting_depth_guard(nesting_depth_guard&&) = delete; + nesting_depth_guard& operator=(nesting_depth_guard&&) = delete; + + bool okay() const noexcept + { + return m_okay; + } + + private: + bool m_okay; + }; + + /// an entry of the iterative deep copy's worklist: a structured value and + /// the value that is to become its copy + using copy_worklist_t = std::vector>; + + /// scratch space to build the key skeleton of an object copy in one go + using copy_scratch_t = std::vector>; + + /// @brief copy everything of @a src into @a dst but its type and value + static void copy_metadata(const basic_json& src, basic_json& dst) + { + // a custom base class is only required to be copy-constructible and + // move-assignable, so the copy has to go through a temporary + static_cast(dst) = json_base_class_t(static_cast(src)); + +#if JSON_DIAGNOSTIC_POSITIONS + dst.start_position = src.start_position; + dst.end_position = src.end_position; +#endif + } + + /*! + @brief copy the value of @a src into @a dst, which must not be structured + + Objects and arrays are left alone: creating those is the one thing the copy + constructor and @ref copy_shallow do differently from one another, and it is + the reason copying a value can descend at all. + */ + /// @note inlined on purpose: both callers have already told an object or an + /// array apart from the rest, and letting the compiler fold that test + /// into this switch is worth a few percent when copying a value made + /// mostly of numbers + JSON_HEDLEY_ALWAYS_INLINE + static void copy_leaf_value(const basic_json& src, basic_json& dst) + { + switch (src.m_data.m_type) + { + case value_t::string: + { + dst.m_data.m_value = *src.m_data.m_value.string; + break; + } + + case value_t::binary: + { + dst.m_data.m_value = *src.m_data.m_value.binary; + break; + } + + case value_t::boolean: + { + dst.m_data.m_value = src.m_data.m_value.boolean; + break; + } + + case value_t::number_integer: + { + dst.m_data.m_value = src.m_data.m_value.number_integer; + break; + } + + case value_t::number_unsigned: + { + dst.m_data.m_value = src.m_data.m_value.number_unsigned; + break; + } + + case value_t::number_float: + { + dst.m_data.m_value = src.m_data.m_value.number_float; + break; + } + + case value_t::object: + case value_t::array: + case value_t::null: + case value_t::discarded: + default: + break; + } + } + + /*! + @brief copy everything of @a src into the null value @a dst but the children + + Objects and arrays are not copied here; they are appended to @a worklist to + be created later by @ref copy_iteratively. Until that happens, @a dst remains + a null value, so that a partially built copy can be destroyed at any point + without ever violating the class invariants. + */ + static void copy_shallow(const basic_json& src, basic_json& dst, copy_worklist_t& worklist) + { + copy_metadata(src, dst); + + if (src.m_data.m_type == value_t::object || src.m_data.m_type == value_t::array) + { + // defer: dst stays a null value until its container exists + worklist.emplace_back(&src, &dst); + return; + } + + copy_leaf_value(src, dst); + + // only now that the value exists may the type be set: had the creation + // of the value thrown, dst would have been left as a valid null value + dst.m_data.m_type = src.m_data.m_type; + } + + /// @brief create the copy of the array @a src in @a dst + /// @note structured elements are appended to @a worklist instead + static void copy_array_level(const basic_json& src, basic_json& dst, copy_worklist_t& worklist) + { + const array_t& src_array = *src.m_data.m_value.array; + + // create all elements up front: growing the array afterwards could + // invalidate the pointers that are handed to the worklist; resize() + // rather than the fill constructor, because not every array type + // provides the latter (e.g., ones without a matching allocator-aware + // fill constructor) + dst.m_data.m_value.array = create(); + dst.m_data.m_value.array->resize(src_array.size()); + + auto dst_it = dst.m_data.m_value.array->begin(); + for (auto src_it = src_array.cbegin(); src_it != src_array.cend(); ++src_it, ++dst_it) + { + copy_shallow(*src_it, *dst_it, worklist); + } + } + + /// @brief create the copy of the object @a src in @a dst + /// @note structured values are appended to @a worklist instead + static void copy_object_level(const basic_json& src, basic_json& dst, + copy_worklist_t& worklist, copy_scratch_t& scratch) + { + const object_t& src_object = *src.m_data.m_value.object; + + // build the complete key skeleton and hand it to the object's range + // constructor: adding the keys one by one would be quadratic for object + // types that are backed by a vector, such as nlohmann::ordered_map + scratch.clear(); + scratch.reserve(src_object.size()); + for (const auto& element : src_object) + { + scratch.emplace_back(element.first, basic_json()); + } + + dst.m_data.m_value.object = create(std::make_move_iterator(scratch.begin()), + std::make_move_iterator(scratch.end())); + scratch.clear(); + + // pair every value of the copy with its counterpart in the original; + // both are enumerated in the same order for every object type with a + // deterministic order, so the lookup is only needed for exotic ones + auto src_it = src_object.cbegin(); + for (auto& element : *dst.m_data.m_value.object) + { + if (JSON_HEDLEY_LIKELY(src_it != src_object.cend() && src_it->first == element.first)) + { + copy_shallow(src_it->second, element.second, worklist); + ++src_it; + } + else + { + const auto found = src_object.find(element.first); + JSON_ASSERT(found != src_object.cend()); + copy_shallow(found->second, element.second, worklist); + } + } + } + + /*! + @brief deep-copy the object or array @a src into this value without recursing + + The values whose copy has not been created yet are kept on an explicit + worklist rather than on the call stack. This is only reached for values + nested deeper than @ref nesting_depth_limit levels, which is why it copies + every container by hand instead of letting the container do it: the fast + ways of doing so would descend into the elements and defeat the purpose. + */ + void copy_iteratively(const basic_json& src) + { + copy_worklist_t worklist; + copy_scratch_t scratch; + + const basic_json* src_value = &src; + basic_json* dst_value = this; + + for (;;) + { + if (src_value->m_data.m_type == value_t::array) + { + copy_array_level(*src_value, *dst_value, worklist); + } + else + { + copy_object_level(*src_value, *dst_value, worklist, scratch); + } + + // the container is complete and will not be modified again + dst_value->set_parents(); + + if (worklist.empty()) + { + break; + } + + const auto& next = worklist.back(); + src_value = next.first; + dst_value = next.second; + worklist.pop_back(); + + // the value stops being a null value exactly here + dst_value->m_data.m_type = src_value->m_data.m_type; + } + } + + /*! + @brief copy one level of the object or array @a src into this value + + The container copies its own elements, which is the fastest way to fill it. + Every element that is structured itself comes back to @ref copy_structured. + */ + void copy_level(const basic_json& src) + { + if (m_data.m_type == value_t::object) + { + m_data.m_value = *src.m_data.m_value.object; + } + else + { + m_data.m_value = *src.m_data.m_value.array; + } + + set_parents(); + } + + /*! + @brief deep-copy the object or array @a src into this value + + Copying a container copies its elements, so a value nested deeply enough + used to exhaust the call stack. The descent is bounded here: the first + @ref nesting_depth_limit levels are copied by the containers themselves, just + as they always were, and anything below that is copied without the call + stack by @ref copy_iteratively. Copying a value can therefore no longer + exhaust the stack, however deeply it is nested, just like destroying one + cannot since #1436. + + Nothing has to be scanned or built by hand to reach that: a value that is + not nested deeper than the limit - all but a vanishing minority - is copied + exactly as it was before, and this whole detour costs it one counter. + + @sa https://github.com/nlohmann/json/issues/5387 + */ + void copy_structured(const basic_json& src) + { + const nesting_depth_guard guard; + + if (JSON_HEDLEY_LIKELY(guard.okay())) + { + copy_level(src); + return; + } + + // Finish this value without descending any further. It is completed + // before this returns, so a copy made by a custom base class - or by + // anything else that runs while a copy is going on - is unaffected by + // the copy it is nested in. + copy_iteratively(src); + } + + + /// @brief restore the parent pointers after erasing from an object + /// ordered_json keeps its members in a vector, and erasing a member + /// re-constructs every member after it in place, which resets their + /// parent pointers + void set_parents_after_object_erase() + { +#if JSON_DIAGNOSTICS +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning(push ) +#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr +#endif + if (detail::is_ordered_map::value) + { + set_parents(); + } +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning( pop ) +#endif +#endif + } + public: ////////////////////////// // JSON parser callback // @@ -1200,60 +1643,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // check of passed value is valid other.assert_invariant(); - switch (m_data.m_type) + if (m_data.m_type == value_t::object || m_data.m_type == value_t::array) { - case value_t::object: - { - m_data.m_value = *other.m_data.m_value.object; - break; - } - - case value_t::array: - { - m_data.m_value = *other.m_data.m_value.array; - break; - } - - case value_t::string: - { - m_data.m_value = *other.m_data.m_value.string; - break; - } - - case value_t::boolean: - { - m_data.m_value = other.m_data.m_value.boolean; - break; - } - - case value_t::number_integer: - { - m_data.m_value = other.m_data.m_value.number_integer; - break; - } - - case value_t::number_unsigned: - { - m_data.m_value = other.m_data.m_value.number_unsigned; - break; - } - - case value_t::number_float: - { - m_data.m_value = other.m_data.m_value.number_float; - break; - } - - case value_t::binary: - { - m_data.m_value = *other.m_data.m_value.binary; - break; - } - - case value_t::null: - case value_t::discarded: - default: - break; + // copying the container directly would call this constructor again + // for every element, once per nesting level + copy_structured(other); + } + else + { + copy_leaf_value(other, *this); } set_parents(); @@ -1335,21 +1733,26 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief serialization /// @sa https://json.nlohmann.me/api/basic_json/dump/ + JSON_HEDLEY_WARN_UNUSED_RESULT string_t dump(const int indent = -1, const char indent_char = ' ', const bool ensure_ascii = false, const error_handler_t error_handler = error_handler_t::strict) const { string_t result; - serializer s(detail::output_adapter(result), indent_char, error_handler); + detail::output_string_adapter string_adapter(result); if (indent >= 0) { - s.dump(*this, true, ensure_ascii, static_cast(indent)); + serializer s(string_adapter, indent_char, + true, ensure_ascii, static_cast(indent), error_handler); + s.dump(*this); } else { - s.dump(*this, false, ensure_ascii, 0); + serializer s(string_adapter, indent_char, + false, ensure_ascii, 0, error_handler); + s.dump(*this); } return result; @@ -1357,6 +1760,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return the type of the JSON value (explicit) /// @sa https://json.nlohmann.me/api/basic_json/type/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr value_t type() const noexcept { return m_data.m_type; @@ -1364,6 +1768,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether type is primitive /// @sa https://json.nlohmann.me/api/basic_json/is_primitive/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_primitive() const noexcept { return is_null() || is_string() || is_boolean() || is_number() || is_binary(); @@ -1371,6 +1776,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether type is structured /// @sa https://json.nlohmann.me/api/basic_json/is_structured/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_structured() const noexcept { return is_array() || is_object(); @@ -1378,6 +1784,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is null /// @sa https://json.nlohmann.me/api/basic_json/is_null/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_null() const noexcept { return m_data.m_type == value_t::null; @@ -1385,6 +1792,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is a boolean /// @sa https://json.nlohmann.me/api/basic_json/is_boolean/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_boolean() const noexcept { return m_data.m_type == value_t::boolean; @@ -1392,6 +1800,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is a number /// @sa https://json.nlohmann.me/api/basic_json/is_number/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_number() const noexcept { return is_number_integer() || is_number_float(); @@ -1399,6 +1808,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is an integer number /// @sa https://json.nlohmann.me/api/basic_json/is_number_integer/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_number_integer() const noexcept { return m_data.m_type == value_t::number_integer || m_data.m_type == value_t::number_unsigned; @@ -1406,6 +1816,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is an unsigned integer number /// @sa https://json.nlohmann.me/api/basic_json/is_number_unsigned/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_number_unsigned() const noexcept { return m_data.m_type == value_t::number_unsigned; @@ -1413,6 +1824,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is a floating-point number /// @sa https://json.nlohmann.me/api/basic_json/is_number_float/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_number_float() const noexcept { return m_data.m_type == value_t::number_float; @@ -1420,6 +1832,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is an object /// @sa https://json.nlohmann.me/api/basic_json/is_object/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_object() const noexcept { return m_data.m_type == value_t::object; @@ -1427,6 +1840,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is an array /// @sa https://json.nlohmann.me/api/basic_json/is_array/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_array() const noexcept { return m_data.m_type == value_t::array; @@ -1434,6 +1848,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is a string /// @sa https://json.nlohmann.me/api/basic_json/is_string/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_string() const noexcept { return m_data.m_type == value_t::string; @@ -1441,6 +1856,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is a binary array /// @sa https://json.nlohmann.me/api/basic_json/is_binary/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_binary() const noexcept { return m_data.m_type == value_t::binary; @@ -1448,6 +1864,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is discarded /// @sa https://json.nlohmann.me/api/basic_json/is_discarded/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_discarded() const noexcept { return m_data.m_type == value_t::discarded; @@ -2009,22 +2426,17 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec reference at(size_type idx) { // at only works for arrays - if (JSON_HEDLEY_LIKELY(is_array())) - { - JSON_TRY - { - return set_parent(m_data.m_value.array->at(idx)); - } - JSON_CATCH (std::out_of_range&) - { - // create a better exception explanation - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } // cppcheck-suppress[missingReturn] - } - else + if (JSON_HEDLEY_UNLIKELY(!is_array())) { JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); } + + if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) + { + JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); + } + + return set_parent((*m_data.m_value.array)[idx]); } /// @brief access specified array element with bounds checking @@ -2032,22 +2444,17 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const_reference at(size_type idx) const { // at only works for arrays - if (JSON_HEDLEY_LIKELY(is_array())) - { - JSON_TRY - { - return m_data.m_value.array->at(idx); - } - JSON_CATCH (std::out_of_range&) - { - // create a better exception explanation - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } // cppcheck-suppress[missingReturn] - } - else + if (JSON_HEDLEY_UNLIKELY(!is_array())) { JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); } + + if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) + { + JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); + } + + return (*m_data.m_value.array)[idx]; } /// @brief access specified object element with bounds checking @@ -2147,12 +2554,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec #if JSON_DIAGNOSTICS // remember array size & capacity before resizing const auto old_size = m_data.m_value.array->size(); - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); #endif m_data.m_value.array->resize(idx + 1); #if JSON_DIAGNOSTICS - if (JSON_HEDLEY_UNLIKELY(m_data.m_value.array->capacity() != old_capacity)) + if (JSON_HEDLEY_UNLIKELY(old_capacity == detail::unknown_size() + || array_capacity() != old_capacity)) { // capacity has changed: update all parents set_parents(); @@ -2543,7 +2951,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { - result.m_it.object_iterator = m_data.m_value.object->erase(pos.m_it.object_iterator); + result.m_it.object_iterator = erase_from_object(pos.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -2616,6 +3025,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { result.m_it.object_iterator = m_data.m_value.object->erase(first.m_it.object_iterator, last.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -2646,7 +3056,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - return m_data.m_value.object->erase(std::forward(key)); + const auto erased = m_data.m_value.object->erase(std::forward(key)); + set_parents_after_object_erase(); + return erased; } template < typename KeyType, detail::enable_if_t < @@ -2663,6 +3075,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); + set_parents_after_object_erase(); return 1; } return 0; @@ -2779,6 +3192,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief returns the number of occurrences of a key in a JSON object /// @sa https://json.nlohmann.me/api/basic_json/count/ + JSON_HEDLEY_WARN_UNUSED_RESULT size_type count(const typename object_t::key_type& key) const { // return 0 for all nonobject types @@ -2789,6 +3203,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/count/ template::value, int> = 0> + JSON_HEDLEY_WARN_UNUSED_RESULT size_type count(KeyType && key) const { // return 0 for all nonobject types @@ -2797,6 +3212,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief check the existence of an element in a JSON object /// @sa https://json.nlohmann.me/api/basic_json/contains/ + JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(const typename object_t::key_type& key) const { return is_object() && m_data.m_value.object->find(key) != m_data.m_value.object->end(); @@ -2806,6 +3222,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/contains/ template::value, int> = 0> + JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(KeyType && key) const { return is_object() && m_data.m_value.object->find(std::forward(key)) != m_data.m_value.object->end(); @@ -2813,12 +3230,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief check the existence of an element in a JSON object given a JSON pointer /// @sa https://json.nlohmann.me/api/basic_json/contains/ + JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(const json_pointer& ptr) const { return ptr.contains(this); } template::value, int> = 0> + JSON_HEDLEY_WARN_UNUSED_RESULT JSON_HEDLEY_DEPRECATED_FOR(3.11.0, basic_json::json_pointer or nlohmann::json_pointer) // NOLINT(readability/alt_tokens) bool contains(const typename ::nlohmann::json_pointer& ptr) const { @@ -2974,6 +3393,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief checks whether the container is empty. /// @sa https://json.nlohmann.me/api/basic_json/empty/ + JSON_HEDLEY_WARN_UNUSED_RESULT bool empty() const noexcept { switch (m_data.m_type) @@ -3013,6 +3433,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief returns the number of elements /// @sa https://json.nlohmann.me/api/basic_json/size/ + JSON_HEDLEY_WARN_UNUSED_RESULT size_type size() const noexcept { switch (m_data.m_type) @@ -3052,6 +3473,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief returns the maximum possible number of elements /// @sa https://json.nlohmann.me/api/basic_json/max_size/ + JSON_HEDLEY_WARN_UNUSED_RESULT size_type max_size() const noexcept { switch (m_data.m_type) @@ -3173,9 +3595,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // add the element to the array (move semantics) - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); m_data.m_value.array->push_back(std::move(val)); - set_parent(m_data.m_value.array->back(), old_capacity); + set_parent_after_array_insert(m_data.m_value.array->back(), old_capacity); // if val is moved from, basic_json move constructor marks it null, so we do not call the destructor } @@ -3206,9 +3628,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // add the element to the array - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); m_data.m_value.array->push_back(val); - set_parent(m_data.m_value.array->back(), old_capacity); + set_parent_after_array_insert(m_data.m_value.array->back(), old_capacity); } /// @brief add an object to an array @@ -3294,9 +3716,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // add the element to the array (perfect forwarding) - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); m_data.m_value.array->emplace_back(std::forward(args)...); - return set_parent(m_data.m_value.array->back(), old_capacity); + return set_parent_after_array_insert(m_data.m_value.array->back(), old_capacity); } /// @brief add an object to an object if key does not exist @@ -3375,7 +3797,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/insert/ iterator insert(const_iterator pos, basic_json&& val) // NOLINT(performance-unnecessary-value-param) { - return insert(pos, val); + return insert(std::move(pos), val); } /// @brief inserts copies of element into array @@ -3425,6 +3847,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(invalid_iterator::create(211, "passed iterators may not belong to container", this)); } + // passed iterators must belong to arrays + if (JSON_HEDLEY_UNLIKELY(!first.m_object->is_array())) + { + JSON_THROW(invalid_iterator::create(202, "iterators first and last must point to arrays", this)); + } + // insert to array and return iterator return insert_iterator(pos, first.m_it.array_iterator, last.m_it.array_iterator); } @@ -3511,27 +3939,117 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(312, detail::concat("cannot use update() with ", first.m_object->type_name()), first.m_object)); } + update_members(first, last, merge_objects, 0); + } + + private: + /// @brief an object @ref update_members_iteratively or @ref + /// merge_patch_iteratively is merging into, and the members still to merge + struct merge_frame + { + merge_frame(basic_json* target_, const_iterator position_, const_iterator last_) noexcept + : target(target_), position(std::move(position_)), last(std::move(last_)) + {} + + basic_json* target; + const_iterator position; + const_iterator last; + }; + + /*! + @brief the members loop of @ref update, for this object and range + + Merging a nested object calls this function again, once per nesting + level, so a value nested deeply enough used to exhaust the call stack and + terminate the process. The descent is bounded here: once @ref + detail::recursion_depth_limit levels have been entered, @ref + update_members_iteratively merges what is left without the call stack. + + @param[in] depth nesting level of this object, counted from the object + @ref update was called on + */ + void update_members(const const_iterator& first, const const_iterator& last, const bool merge_objects, const std::size_t depth) + { + if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit())) + { + update_members_iteratively(first, last); + return; + } + for (auto it = first; it != last; ++it) { if (merge_objects && it.value().is_object()) { - auto it2 = m_data.m_value.object->find(it.key()); - if (it2 != m_data.m_value.object->end()) + const auto it2 = m_data.m_value.object->find(it.key()); + // Only recurse when the existing value is itself an object. + // Otherwise overwrite, matching the documented "all other values + // are overwritten as usual" behavior (see #5402). + if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { - it2->second.update(it.value(), true); -#if JSON_DIAGNOSTICS - it2->second.set_parents(); -#endif + it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); continue; } } - m_data.m_value.object->operator[](it.key()) = it.value(); -#if JSON_DIAGNOSTICS - m_data.m_value.object->operator[](it.key()).m_parent = this; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + set_parent(m_data.m_value.object->operator[](it.key()) = it.value()); } } + /*! + @brief merge @a first to @a last into this object without the call stack + + Does the same as @ref update_members with `merge_objects` set, keeping the + objects whose merge was interrupted by a nested one on an explicit stack + instead of descending into them. A nested object is still merged + completely before the next member, in the same order as the recursive + version. Only reached for values nested deeper than @ref + detail::recursion_depth_limit. + */ + void update_members_iteratively(const_iterator first, const_iterator last) + { + std::vector stack; + + basic_json* target = this; + while (true) + { + if (first == last) + { + if (stack.empty()) + { + break; + } + + // a nested object is merged: continue with its parent + target = stack.back().target; + first = stack.back().position; + last = stack.back().last; + stack.pop_back(); + continue; + } + + if (first.value().is_object()) + { + const auto it2 = target->m_data.m_value.object->find(first.key()); + if (it2 != target->m_data.m_value.object->end() && it2->second.is_object()) + { + const basic_json& source = first.value(); + ++first; + stack.emplace_back(target, first, last); + target = &it2->second; + first = source.cbegin(); + last = source.cend(); + continue; + } + } + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + target->set_parent(target->m_data.m_value.object->operator[](first.key()) = first.value()); + ++first; + } + } + + public: /// @brief exchanges the values /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(reference other) noexcept ( @@ -3544,6 +4062,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec std::swap(m_data.m_type, other.m_data.m_type); std::swap(m_data.m_value, other.m_data.m_value); +#if JSON_DIAGNOSTIC_POSITIONS + std::swap(start_position, other.start_position); + std::swap(end_position, other.end_position); +#endif + set_parents(); other.set_parents(); assert_invariant(); @@ -3570,6 +4093,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { using std::swap; swap(*(m_data.m_value.array), other); + set_parents(); } else { @@ -3586,6 +4110,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { using std::swap; swap(*(m_data.m_value.object), other); + set_parents(); } else { @@ -3700,19 +4225,19 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } \ else if (lhs_type == value_t::number_integer && rhs_type == value_t::number_float) \ { \ - return static_cast(lhs.m_data.m_value.number_integer) op rhs.m_data.m_value.number_float; \ + return (detail::compare_integer_with_float(lhs.m_data.m_value.number_integer, rhs.m_data.m_value.number_float)) op (static_cast(0)); \ } \ else if (lhs_type == value_t::number_float && rhs_type == value_t::number_integer) \ { \ - return lhs.m_data.m_value.number_float op static_cast(rhs.m_data.m_value.number_integer); \ + return (static_cast(0)) op (detail::compare_integer_with_float(rhs.m_data.m_value.number_integer, lhs.m_data.m_value.number_float)); \ } \ else if (lhs_type == value_t::number_unsigned && rhs_type == value_t::number_float) \ { \ - return static_cast(lhs.m_data.m_value.number_unsigned) op rhs.m_data.m_value.number_float; \ + return (detail::compare_integer_with_float(lhs.m_data.m_value.number_unsigned, rhs.m_data.m_value.number_float)) op (static_cast(0)); \ } \ else if (lhs_type == value_t::number_float && rhs_type == value_t::number_unsigned) \ { \ - return lhs.m_data.m_value.number_float op static_cast(rhs.m_data.m_value.number_unsigned); \ + return (static_cast(0)) op (detail::compare_integer_with_float(rhs.m_data.m_value.number_unsigned, lhs.m_data.m_value.number_float)); \ } \ else if (lhs_type == value_t::number_unsigned && rhs_type == value_t::number_integer) \ { \ @@ -3767,13 +4292,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec bool operator==(const_reference rhs) const noexcept { #ifdef __GNUC__ -#pragma GCC diagnostic push -#pragma GCC diagnostic ignored "-Wfloat-equal" + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif const_reference lhs = *this; JSON_IMPLEMENT_OPERATOR( ==, true, false, false) #ifdef __GNUC__ -#pragma GCC diagnostic pop + JSON_HEDLEY_DIAGNOSTIC_POP #endif } @@ -3860,12 +4385,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec friend bool operator==(const_reference lhs, const_reference rhs) noexcept { #ifdef __GNUC__ -#pragma GCC diagnostic push -#pragma GCC diagnostic ignored "-Wfloat-equal" + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif JSON_IMPLEMENT_OPERATOR( ==, true, false, false) #ifdef __GNUC__ -#pragma GCC diagnostic pop + JSON_HEDLEY_DIAGNOSTIC_POP #endif } @@ -4052,8 +4577,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec o.width(0); // do the actual serialization - serializer s(detail::output_adapter(o), o.fill()); - s.dump(j, pretty_print, false, static_cast(indentation)); + detail::output_stream_adapter stream_adapter(o); + serializer s(stream_adapter, o.fill(), + pretty_print, false, static_cast(indentation)); + s.dump(j); return o; } @@ -4126,22 +4653,24 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief check if the input is valid JSON /// @sa https://json.nlohmann.me/api/basic_json/accept/ template + JSON_HEDLEY_WARN_UNUSED_RESULT static bool accept(InputType&& i, const bool ignore_comments = false, const bool ignore_trailing_commas = false) { - return parser(detail::input_adapter(std::forward(i)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true); + return parser(detail::input_adapter(std::forward(i)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true); } /// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support) /// @sa https://json.nlohmann.me/api/basic_json/accept/ template::value, int> = 0> + JSON_HEDLEY_WARN_UNUSED_RESULT static bool accept(IteratorType first, SentinelType last, const bool ignore_comments = false, const bool ignore_trailing_commas = false) { - return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true); + return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true); } JSON_HEDLEY_WARN_UNUSED_RESULT @@ -4150,7 +4679,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_comments = false, const bool ignore_trailing_commas = false) { - return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true); + return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true); } /// @brief generate SAX events @@ -4236,6 +4765,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return the type as string /// @sa https://json.nlohmann.me/api/basic_json/type_name/ + JSON_HEDLEY_WARN_UNUSED_RESULT JSON_HEDLEY_RETURNS_NON_NULL const char* type_name() const noexcept { @@ -4337,7 +4867,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_cbor(const basic_json& j) { std::vector result; - to_cbor(j, result); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_cbor(j); return result; } @@ -4360,7 +4891,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_msgpack(const basic_json& j) { std::vector result; - to_msgpack(j, result); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_msgpack(j); return result; } @@ -4385,7 +4917,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool use_type = false) { std::vector result; - to_ubjson(j, result, use_size, use_type); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_ubjson(j, use_size, use_type); return result; } @@ -4413,7 +4946,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bjdata_version_t version = bjdata_version_t::draft2) { std::vector result; - to_bjdata(j, result, use_size, use_type, version); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_ubjson(j, use_size, use_type, true, true, version); return result; } @@ -4440,7 +4974,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_bson(const basic_json& j) { std::vector result; - to_bson(j, result); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_bson(j); return result; } @@ -4470,8 +5005,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in CBOR format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -4487,8 +5025,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } template @@ -4513,8 +5054,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - const bool res = binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in MessagePack format @@ -4528,8 +5072,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in MessagePack format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -4544,8 +5091,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } template @@ -4568,8 +5118,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - const bool res = binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in UBJSON format @@ -4583,8 +5136,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in UBJSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -4599,8 +5155,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } template @@ -4623,8 +5182,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - const bool res = binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in BJData format @@ -4638,8 +5200,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in BJData format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -4654,8 +5219,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in BSON format @@ -4669,8 +5237,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in BSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -4685,8 +5256,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } template @@ -4709,8 +5283,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - const bool res = binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @} @@ -4936,6 +5513,36 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // note erase performs range check parent.erase(json_pointer::template array_index(last_path)); } + else + { + // the parent of a "remove" target must be an object or array + // (see #5396) + JSON_THROW(out_of_range::create(413, detail::concat("cannot remove value: the JSON Patch 'remove' target's parent is of type ", parent.type_name(), ", but must be an object or array"), &parent)); + } + }; + + // RFC 6902 (section 4.4) forbids "from" from being a proper prefix + // of "path" for a "move" operation: a location cannot be moved into + // one of its own children. Compares reference tokens (already + // unescaped by json_pointer's parser) rather than the raw pointer + // strings, since a token may itself contain an escaped '/' or '~' + // that would defeat a naive string-prefix comparison. "from" equal + // to "path" is *not* a proper prefix and must return false. + const auto is_proper_prefix = [](const json_pointer & from, const json_pointer & to) + { + const auto from_size = from.reference_tokens.size(); + if (from_size >= to.reference_tokens.size()) + { + return false; + } + for (std::size_t i = 0; i < from_size; ++i) + { + if (!(from.reference_tokens[i] == to.reference_tokens[i])) + { + return false; + } + } + return true; }; // type check: top level value must be an array @@ -5013,6 +5620,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const auto from_path = get_value("move", "from", true).template get(); json_pointer from_ptr(from_path); + if (JSON_HEDLEY_UNLIKELY(is_proper_prefix(from_ptr, ptr))) + { + JSON_THROW(out_of_range::create(414, detail::concat("cannot move value: 'from' path '", from_path, "' is a proper prefix of 'path' '", path, "'"), &result)); + } + // the "from" location must exist - use at() basic_json const v = result.at(from_ptr); @@ -5125,19 +5737,17 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // We now reached the end of at least one array // in a second pass, traverse the remaining elements - // remove my remaining elements - const auto end_index = static_cast(result.size()); - while (i < source.size()) + // remove my remaining elements, highest index first; appending + // in that order avoids the quadratic reinsertion done before + for (std::size_t j = source.size(); j > i; --j) { - // add operations in reverse order to avoid invalid - // indices - result.insert(result.begin() + end_index, object( + result.push_back(object( { {"op", "remove"}, - {"path", detail::concat(path, '/', detail::to_string(i))} + {"path", detail::concat(path, '/', detail::to_string(j - 1))} })); - ++i; } + i = source.size(); // add other remaining elements while (i < target.size()) @@ -5156,34 +5766,139 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { - // first pass: traverse this object's elements + // first pass: record, for every source key, whether it is + // common to both objects (in source's iteration order) or + // was deleted (i.e., in source but not in target) -- this is + // a by-product of the target.find() call already needed to + // tell the two cases apart, so it adds no extra lookups. The + // "remove" ops themselves are emitted later, interleaved + // with the recursive per-key diffs in the fast path below, + // to match source's original iteration order (as the + // original, pre-reordering-aware implementation did) instead + // of grouping all removes before all recursive diffs. + std::vector common_keys_source_order; for (auto it = source.cbegin(); it != source.cend(); ++it) { - // escape the key name to be used in a JSON patch - const auto path_key = detail::concat(path, '/', detail::escape(it.key())); - if (target.find(it.key()) != target.end()) { - // recursive call to compare object values at key it - auto temp_diff = diff(it.value(), target[it.key()], path_key); - result.insert(result.end(), temp_diff.begin(), temp_diff.end()); + common_keys_source_order.push_back(it.key()); + } + } + + // second pass: find keys that were added (i.e., in target but + // not in source), and record the keys common to both, in + // target's iteration order -- again a by-product of the + // source.find() call already needed to detect added keys. At + // the same time, determine whether every added key comes + // after every common key in target's order (a precondition + // for the fast path below, which only ever appends new keys + // at the very end): for an object_t whose iteration order is + // a pure function of the key set (e.g. the default std::map, + // which always iterates in sorted key order), the order + // check further below is always true and this whole + // mechanism is effectively a no-op; it only matters for a + // reorderable object_t such as the one backing `ordered_json`. + // patch ops for keys that were added (i.e., in target but not + // in source); built here so the fast path below can reuse + // them without a second source.find() per target key. Only + // used by the fast path -- the slow (reordering) path + // rebuilds "add" ops for every key itself. + std::vector common_keys_target_order; + basic_json added_ops(value_t::array); + bool new_keys_form_suffix = true; + bool seen_new_key = false; + for (auto it = target.cbegin(); it != target.cend(); ++it) + { + if (source.find(it.key()) == source.end()) + { + seen_new_key = true; + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); + added_ops.push_back( + { + {"op", "add"}, {"path", path_key}, + {"value", it.value()} + }); } else { - // found a key that is not in o -> remove it + common_keys_target_order.push_back(it.key()); + if (seen_new_key) + { + new_keys_form_suffix = false; + } + } + } + + if (common_keys_source_order == common_keys_target_order && new_keys_form_suffix) + { + // fast path: order of common keys already matches (or the + // object_t's iteration order does not depend on + // insertion history), so a plain per-key recursive diff + // is correct and minimal, as before. common_keys_source_order + // is, by construction, the subsequence of source's keys + // that are common to both objects, in source's iteration + // order -- so it can be walked in lockstep with `source` + // using a cheap key comparison instead of another lookup. + // Deleted keys (those source keys not in common_keys_source_order) + // are interleaved here too, in source's original order, to + // match the historical (pre-reordering-aware) output order. + auto common_it = common_keys_source_order.cbegin(); + for (auto it = source.cbegin(); it != source.cend(); ++it) + { + if (common_it != common_keys_source_order.cend() && it.key() == *common_it) + { + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); + auto temp_diff = diff(it.value(), target[it.key()], path_key); + result.insert(result.end(), temp_diff.begin(), temp_diff.end()); + ++common_it; + } + else + { + // found a key that is not in target -> remove it + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); + result.push_back(object( + { + {"op", "remove"}, {"path", path_key} + })); + } + } + + // append the "add" ops for brand-new keys collected above + // during the pass over target -- no second source.find() + // per target key needed + result.insert(result.end(), added_ops.begin(), added_ops.end()); + } + else + { + // slow path: the common keys are in a different relative + // order in source and target (only possible for a + // reorderable object_t like ordered_map). Building a + // minimal reordering patch is a nontrivial (LCS-like) + // problem; instead, remove every source key -- both + // deleted keys (which must be removed regardless) and + // common keys (removed so they can be re-added in + // target's order) -- and re-add every key that should + // remain, with its final target value, in target's + // order. basic_json::patch()'s "add" operation on an + // object uses operator[], which appends at the end for a + // vector-backed insertion-ordered map when the key does + // not already exist -- so removing a key and then adding + // it moves it to the end, fixing its position. + for (auto it = source.cbegin(); it != source.cend(); ++it) + { + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); result.push_back(object( { {"op", "remove"}, {"path", path_key} })); } - } - // second pass: traverse other object's elements - for (auto it = target.cbegin(); it != target.cend(); ++it) - { - if (source.find(it.key()) == source.end()) + // add every key that is either common (just removed + // above) or brand new, in target's iteration order, so + // that the final order after applying the patch matches + // target exactly + for (auto it = target.cbegin(); it != target.cend(); ++it) { - // found a key that is not in this -> add it const auto path_key = detail::concat(path, '/', detail::escape(it.key())); result.push_back( { @@ -5229,9 +5944,30 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief applies a JSON Merge Patch /// @sa https://json.nlohmann.me/api/basic_json/merge_patch/ void merge_patch(const basic_json& apply_patch) + { + apply_merge_patch(apply_patch, 0); + } + + private: + /*! + @brief @ref merge_patch, for a patch at nesting level @a depth + + Applying a nested object calls this function again, once per nesting + level, so a patch nested deeply enough used to exhaust the call stack and + terminate the process. The descent is bounded here: once @ref + detail::recursion_depth_limit levels have been entered, @ref + merge_patch_iteratively applies what is left without the call stack. + */ + void apply_merge_patch(const basic_json& apply_patch, const std::size_t depth) { if (apply_patch.is_object()) { + if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit())) + { + merge_patch_iteratively(apply_patch); + return; + } + if (!is_object()) { *this = object(); @@ -5244,7 +5980,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } else { - operator[](it.key()).merge_patch(it.value()); + operator[](it.key()).apply_merge_patch(it.value(), depth + 1); } } } @@ -5254,6 +5990,62 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } } + /*! + @brief apply @a apply_patch to this value without the call stack + + Does the same as @ref merge_patch, keeping the objects being patched on an + explicit stack instead of descending into them. A nested object is still + patched completely before the next member, in the same order as the + recursive version. Only reached for patches nested deeper than @ref + detail::recursion_depth_limit. + */ + void merge_patch_iteratively(const basic_json& apply_patch) + { + std::vector stack; + + // patch `target` with `patch`, or start patching it member by member + const auto apply = [&stack](basic_json & target, const basic_json & patch) + { + if (patch.is_object()) + { + if (!target.is_object()) + { + target = basic_json::object(); + } + stack.emplace_back(&target, patch.cbegin(), patch.cend()); + } + else + { + target = patch; + } + }; + + apply(*this, apply_patch); + while (!stack.empty()) + { + // a copy, as applying a member below can reallocate the stack; + // the frame itself is only changed through stack.back() + const merge_frame frame = stack.back(); + if (frame.position == frame.last) + { + stack.pop_back(); + continue; + } + + const const_iterator member = frame.position; + ++stack.back().position; + if (member.value().is_null()) + { + frame.target->erase(member.key()); + } + else + { + apply(frame.target->operator[](member.key()), member.value()); + } + } + } + + public: /// @} }; diff --git a/include/nlohmann/thirdparty/hedley/hedley_undef.hpp b/include/nlohmann/thirdparty/hedley/hedley_undef.hpp index 1b8bd4338..e4d9838cb 100644 --- a/include/nlohmann/thirdparty/hedley/hedley_undef.hpp +++ b/include/nlohmann/thirdparty/hedley/hedley_undef.hpp @@ -17,7 +17,7 @@ #undef JSON_HEDLEY_CLANG_HAS_ATTRIBUTE #undef JSON_HEDLEY_CLANG_HAS_BUILTIN #undef JSON_HEDLEY_CLANG_HAS_CPP_ATTRIBUTE -#undef JSON_HEDLEY_CLANG_HAS_DECLSPEC_DECLSPEC_ATTRIBUTE +#undef JSON_HEDLEY_CLANG_HAS_DECLSPEC_ATTRIBUTE #undef JSON_HEDLEY_CLANG_HAS_EXTENSION #undef JSON_HEDLEY_CLANG_HAS_FEATURE #undef JSON_HEDLEY_CLANG_HAS_WARNING @@ -108,7 +108,10 @@ #undef JSON_HEDLEY_PELLES_VERSION_CHECK #undef JSON_HEDLEY_PGI_VERSION #undef JSON_HEDLEY_PGI_VERSION_CHECK +#undef JSON_HEDLEY_PRAGMA #undef JSON_HEDLEY_PREDICT +#undef JSON_HEDLEY_PREDICT_FALSE +#undef JSON_HEDLEY_PREDICT_TRUE #undef JSON_HEDLEY_PRINTF_FORMAT #undef JSON_HEDLEY_PRIVATE #undef JSON_HEDLEY_PUBLIC diff --git a/nlohmann_json.natvis b/nlohmann_json.natvis index 09a46d67d..2eccbe17c 100644 --- a/nlohmann_json.natvis +++ b/nlohmann_json.natvis @@ -215,6 +215,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -275,4 +395,604 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 0cbf9a339..7cf463902 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -28,14 +28,14 @@ #pragma GCC diagnostic ignored "-Wignored-attributes" #endif -#include // all_of, find, for_each +#include // all_of, find, for_each, none_of #include // nullptr_t, ptrdiff_t, size_t #include // hash, less #include // initializer_list #ifndef JSON_NO_IO #include // istream, ostream #endif // JSON_NO_IO -#include // random_access_iterator_tag +#include // make_move_iterator, random_access_iterator_tag #include // unique_ptr #include // string, stoi, to_string #include // declval, forward, move, pair, swap @@ -91,6 +91,10 @@ #define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0 #endif +#ifndef JSON_BRACE_INIT_COPY_SEMANTICS + #define JSON_BRACE_INIT_COPY_SEMANTICS 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -109,20 +113,27 @@ #define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS _bics +#else + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) json_abi ## a ## b ## c -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ - NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS) + NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ @@ -223,9 +234,12 @@ #include // array +#include // isnan, ldexp, trunc #include // size_t #include // uint8_t +#include // numeric_limits #include // string +#include // is_signed // #include // __ _____ _____ _____ @@ -2561,6 +2575,15 @@ JSON_HEDLEY_DIAGNOSTIC_POP #define JSON_NO_UNIQUE_ADDRESS #endif +// Clang targeting MinGW does not survive the thread_local storage the copy +// constructor uses to bound its descent: every test that copies a value +// segfaults with clang 11.0.1 and clang 18.1.8, while the same tests pass with +// GCC targeting MinGW and with every other toolchain the library is tested on. +// Copying works the same way without the counter, only more slowly. +#if !defined(JSON_NO_THREAD_LOCAL) && defined(__clang__) && defined(__MINGW32__) + #define JSON_NO_THREAD_LOCAL 1 +#endif + // disable documentation warnings on clang #if defined(__clang__) #pragma clang diagnostic push @@ -3179,8 +3202,8 @@ void templated_json_throw(ExceptionType exception) #define JSON_USE_GLOBAL_UDLS 1 #endif -#ifndef JSON_BRACE_INIT_COPY_SEMANTICS - #define JSON_BRACE_INIT_COPY_SEMANTICS 0 +#ifndef JSON_STRICT_NUL_HANDLING + #define JSON_STRICT_NUL_HANDLING 0 #endif #if JSON_HAS_THREE_WAY_COMPARISON @@ -3283,6 +3306,68 @@ inline bool operator<(const value_t lhs, const value_t rhs) noexcept } #endif + +/*! +@brief compare an integer with a floating point number without precision loss + +Widening the integer to the floating point type loses precision beyond the +float's mantissa, which makes equality intransitive: both 2^63-2 and 2^63-1 +round to 2^63, so each compares equal to that float while differing from each +other. Ordering built on that is not a strict weak ordering, so sorting such +values, or using them as keys in an ordered container, is undefined behavior. + +Returns a value to be compared against zero with the original operator, which +reproduces the exact ordering. A NaN operand is returned as is, so comparing it +against zero keeps NaN's semantics: false for the relational operators and +unordered for `<=>`. +*/ +template +FloatType compare_integer_with_float(const IntegerType i, const FloatType f) noexcept +{ + const auto ordered = [](int c) noexcept + { + return static_cast(c); + }; + + if (std::isnan(f)) + { + return f; + } + + // values of IntegerType lie in [-bound, bound) when signed and in + // [0, bound) when unsigned; digits excludes the sign bit, so bound is a + // power of two that the float represents exactly + const FloatType bound = std::ldexp(static_cast(1), std::numeric_limits::digits); + if (f >= bound) + { + return ordered(-1); + } + if (std::is_signed::value ? (f < -bound) : (f < static_cast(0))) + { + return ordered(1); + } + + // f is now within the integer's range, so truncating it is exact + const FloatType truncated = std::trunc(f); + const auto as_integer = static_cast(truncated); + if (i != as_integer) + { + return ordered(i < as_integer ? -1 : 1); + } + + // the integer parts agree, so any fractional part decides + const FloatType fraction = f - truncated; + if (fraction > static_cast(0)) + { + return ordered(-1); + } + if (fraction < static_cast(0)) + { + return ordered(1); + } + return ordered(0); +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -3297,6 +3382,8 @@ NLOHMANN_JSON_NAMESPACE_END +#include // size_t + // #include @@ -3304,44 +3391,48 @@ NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail { -/*! -@brief replace all occurrences of a substring by another string - -@param[in,out] s the string to manipulate; changed so that all - occurrences of @a f are replaced with @a t -@param[in] f the substring to replace with @a t -@param[in] t the string to replace @a f - -@pre The search string @a f must not be empty. **This precondition is -enforced with an assertion.** - -@since version 2.0.0 -*/ -template -inline void replace_substring(StringType& s, const StringType& f, - const StringType& t) -{ - JSON_ASSERT(!f.empty()); - for (auto pos = s.find(f); // find the first occurrence of f - pos != StringType::npos; // make sure f was found - s.replace(pos, f.size(), t), // replace with t, and - pos = s.find(f, pos + t.size())) // find the next occurrence of f - {} -} - /*! * @brief string escaping as described in RFC 6901 (Sect. 4) * @param[in] s string to escape * @return escaped string * * Note the order of escaping "~" to "~0" and "/" to "~1" is important. + * + * The string is rebuilt in a single pass, appending whole runs between the + * characters that need escaping. Scanning with find_first_of() keeps the + * common case -- nothing to escape -- as fast as a single search, while + * repeated replace() calls would move the tail of the string once per + * escaped character. */ template -inline StringType escape(StringType s) +inline StringType escape(const StringType& s) { - replace_substring(s, StringType{"~"}, StringType{"~0"}); - replace_substring(s, StringType{"/"}, StringType{"~1"}); - return s; + auto next_special = [&s](std::size_t from) + { + const auto tilde = s.find_first_of('~', from); + const auto slash = s.find_first_of('/', from); + return tilde < slash ? tilde : slash; // npos is the largest value + }; + + auto pos = next_special(0); + if (pos == StringType::npos) + { + return s; + } + + StringType result; + result.reserve(s.size() + 2); + + std::size_t run = 0; + while (pos != StringType::npos) + { + result.append(s.data() + run, pos - run); + result.append(s[pos] == '~' ? "~0" : "~1", 2); + run = pos + 1; + pos = next_special(run); + } + result.append(s.data() + run, s.size() - run); + return result; } /*! @@ -3350,12 +3441,43 @@ inline StringType escape(StringType s) * @return unescaped string * * Note the order of escaping "~1" to "/" and "~0" to "~" is important. + * + * Rebuilt in a single pass, see @ref escape. A "~" that is followed by + * neither "0" nor "1" is passed through unchanged; @ref json_pointer rejects + * such input before it gets here. */ template inline void unescape(StringType& s) { - replace_substring(s, StringType{"~1"}, StringType{"/"}); - replace_substring(s, StringType{"~0"}, StringType{"~"}); + auto pos = s.find_first_of('~', 0); + if (pos == StringType::npos) + { + return; + } + + StringType result; + result.reserve(s.size()); + + std::size_t run = 0; + while (pos != StringType::npos) + { + result.append(s.data() + run, pos - run); + + const auto next = pos + 1; + if (next < s.size() && (s[next] == '0' || s[next] == '1')) + { + result.append(s[next] == '0' ? "~" : "/", 1); + run = pos + 2; + } + else + { + result.append("~", 1); + run = pos + 1; + } + pos = s.find_first_of('~', run); + } + result.append(s.data() + run, s.size() - run); + s = result; } } // namespace detail @@ -3935,17 +4057,18 @@ struct has_to_json < BasicJsonType, T, enable_if_t < !is_basic_json::value >> template using detect_key_compare = typename T::key_compare; -template -struct has_key_compare : std::integral_constant::value> {}; - -// obtains the actual object key comparator +// obtains the actual object key comparator: object_t::key_compare if the +// object type defines it, and default_object_comparator_t otherwise +// +// note detected_or_t is used rather than std::conditional, because the latter +// names both of its type arguments eagerly; object_t::key_compare would then +// be a hard error for an object type that does not define it template struct actual_object_comparator { using object_t = typename BasicJsonType::object_t; using object_comparator_t = typename BasicJsonType::default_object_comparator_t; - using type = typename std::conditional < has_key_compare::value, - typename object_t::key_compare, object_comparator_t>::type; + using type = detected_or_t; }; template @@ -4541,6 +4664,22 @@ using has_erase_with_key_type = typename std::conditional < std::true_type, std::false_type >::type; +template +using detect_erase_with_iterator = decltype(std::declval().erase(std::declval())); + +// type trait to check if erase(iterator) returns void instead of the following +// iterator, as the object types that do not compute a successor the caller may +// not need do +template +using erase_returns_void = is_detected_exact; + +template +using detect_capacity = decltype(std::declval().capacity()); + +// type trait to check if a type has a capacity() member function +template +struct has_capacity : std::integral_constant::value> {}; + // a naive helper to check if a type is an ordered_map (exploits the fact that // ordered_map inherits capacity() from std::vector) template @@ -4880,8 +5019,8 @@ NLOHMANN_JSON_NAMESPACE_END // code stumbling over this. See https://github.com/nlohmann/json/issues/4087 // for a discussion. #if defined(__clang__) - #pragma clang diagnostic push - #pragma clang diagnostic ignored "-Wweak-vtables" + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(clang diagnostic ignored "-Wweak-vtables") #endif NLOHMANN_JSON_NAMESPACE_BEGIN @@ -4948,7 +5087,10 @@ class exception : public std::exception { if (&element.second == current) { - tokens.emplace_back(element.first.c_str()); + // data() is null-terminated, so a key containing + // a null byte is cut short here rather than + // truncating the whole message at what() + tokens.emplace_back(element.first.data()); break; } } @@ -5134,7 +5276,7 @@ class other_error : public exception NLOHMANN_JSON_NAMESPACE_END #if defined(__clang__) - #pragma clang diagnostic pop + JSON_HEDLEY_DIAGNOSTIC_POP #endif // #include @@ -5628,6 +5770,17 @@ inline void from_json(const BasicJsonType& j, CompatibleArrayType& bin) } } +template +auto from_json_object_reserve(ConstructibleObjectType& obj, typename ConstructibleObjectType::size_type size, priority_tag<1> /*unused*/) +-> decltype(obj.reserve(size), void()) +{ + obj.reserve(size); +} + +template +inline void from_json_object_reserve(ConstructibleObjectType& /*obj*/, std::size_t /*size*/, priority_tag<0> /*unused*/) +{} + template::value, int> = 0> inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj) @@ -5639,6 +5792,7 @@ inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj) ConstructibleObjectType ret; const auto* inner_object = j.template get_ptr(); + from_json_object_reserve(ret, inner_object->size(), priority_tag<1> {}); for (const auto& p : *inner_object) { ret.emplace(p.first, p.second.template get()); @@ -5910,6 +6064,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -6139,10 +6295,10 @@ NLOHMANN_JSON_NAMESPACE_END namespace std { +// Fix: https://github.com/nlohmann/json/issues/1401 #if defined(__clang__) - // Fix: https://github.com/nlohmann/json/issues/1401 - #pragma clang diagnostic push - #pragma clang diagnostic ignored "-Wmismatched-tags" + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(clang diagnostic ignored "-Wmismatched-tags") #endif template class tuple_size<::nlohmann::detail::iteration_proxy_value> // NOLINT(cert-dcl58-cpp) @@ -6157,7 +6313,7 @@ class tuple_element> ::nlohmann::detail::iteration_proxy_value> ())); }; #if defined(__clang__) - #pragma clang diagnostic pop + JSON_HEDLEY_DIAGNOSTIC_POP #endif } // namespace std @@ -6618,6 +6774,30 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = { std::get(t)... }; } +#if JSON_BRACE_INIT_COPY_SEMANTICS +// JSON_BRACE_INIT_COPY_SEMANTICS makes a one-element braced list copy its +// element instead of wrapping it, which would serialize std::tuple{5} as 5 +// rather than [5]. Build what the default deduction builds instead: an object +// if the element is a [string, value] pair, a one-element array otherwise. +template +inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence<0> /*unused*/) +{ + BasicJsonType element(std::get<0>(t)); + // same test as the initializer-list constructor, including the cast that + // keeps a string type constructible from 0 from selecting operator[](key) + const bool is_member = element.is_array() && element.size() == 2 + && element[static_cast(0)].is_string(); + if (is_member) + { + j = BasicJsonType::object({std::move(element)}); + } + else + { + j = BasicJsonType::array({std::move(element)}); + } +} +#endif + template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& /*unused*/, index_sequence<> /*unused*/) { @@ -6848,9 +7028,48 @@ NLOHMANN_JSON_NAMESPACE_END #include // uint8_t #include // size_t #include // hash +#include // vector // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // size_t + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief the number of nesting levels an operation recurses into + +Operations that walk a value (serializing, hashing, merging, ...) recurse once +per nesting level, which is fastest, but a value nested deeply enough would +exhaust the call stack. So they recurse only this many levels deep and finish +whatever lies below with an explicit stack. All of them share this limit. + +@sa https://github.com/nlohmann/json/issues/5387 +*/ +constexpr std::size_t recursion_depth_limit() noexcept +{ + return 128; +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // #include @@ -6865,6 +7084,9 @@ inline std::size_t combine(std::size_t seed, std::size_t h) noexcept return seed; } +template +std::size_t hash_iteratively(const BasicJsonType& j); + /*! @brief hash a JSON value @@ -6872,12 +7094,21 @@ The hash function tries to rely on std::hash where possible. Furthermore, the type of the JSON value is taken into account to have different hash values for null, 0, 0U, and false, etc. +Hashing an array or an object hashes its elements, which used to call this +function again once per nesting level, so a value nested deeply enough +exhausted the call stack and terminated the process. The descent is bounded +here: once @ref recursion_depth_limit levels have been entered, @ref +hash_iteratively hashes what is left without the call stack. A value nested +less deeply than that - all but a vanishing minority - is hashed exactly as +before, without allocating. + @tparam BasicJsonType basic_json specialization @param j JSON value to hash +@param depth nesting level of @a j, counted from the value passed by the caller @return hash value of j */ template -std::size_t hash(const BasicJsonType& j) +std::size_t hash(const BasicJsonType& j, const std::size_t depth = 0) { using string_t = typename BasicJsonType::string_t; using number_integer_t = typename BasicJsonType::number_integer_t; @@ -6895,22 +7126,32 @@ std::size_t hash(const BasicJsonType& j) case BasicJsonType::value_t::object: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j.items()) { const auto h = std::hash {}(element.key()); seed = combine(seed, h); - seed = combine(seed, hash(element.value())); + seed = combine(seed, hash(element.value(), depth + 1)); } return seed; } case BasicJsonType::value_t::array: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j) { - seed = combine(seed, hash(element)); + seed = combine(seed, hash(element, depth + 1)); } return seed; } @@ -6953,7 +7194,9 @@ std::size_t hash(const BasicJsonType& j) seed = combine(seed, static_cast(j.get_binary().subtype())); for (const auto byte : j.get_binary()) { - seed = combine(seed, std::hash {}(byte)); + // the cast is needed for binary types whose value type is not + // an integer (e.g., std::byte) + seed = combine(seed, std::hash {}(static_cast(byte))); } return seed; } @@ -6964,6 +7207,78 @@ std::size_t hash(const BasicJsonType& j) } } +/// an array or object whose elements @ref hash_iteratively is hashing +template +struct hash_frame +{ + hash_frame(const BasicJsonType* value_, std::size_t seed_) noexcept + : value(value_), position(value_->cbegin()), seed(seed_) + {} + + const BasicJsonType* value; + typename BasicJsonType::const_iterator position; + std::size_t seed; +}; + +/*! +@brief hash the array or object @a j without the call stack + +Computes the same value as @ref hash, keeping the arrays and objects it has +entered on an explicit stack instead of descending into them. Only reached for +values nested deeper than @ref recursion_depth_limit. + +@tparam BasicJsonType basic_json specialization +@param j array or object to hash +@return hash value of j +*/ +template +std::size_t hash_iteratively(const BasicJsonType& j) +{ + using string_t = typename BasicJsonType::string_t; + + std::vector> stack; + stack.emplace_back(&j, combine(static_cast(j.type()), j.size())); + + while (true) + { + // a copy, as entering an element below can reallocate the stack; the + // frame itself is only changed through stack.back() + const hash_frame frame = stack.back(); + + if (frame.position == frame.value->cend()) + { + // all elements are hashed: fold this value's hash into its parent's + // seed, exactly where the recursive version returns it + const std::size_t h = frame.seed; + stack.pop_back(); + if (stack.empty()) + { + return h; + } + stack.back().seed = combine(stack.back().seed, h); + continue; + } + + if (frame.value->is_object()) + { + stack.back().seed = combine(stack.back().seed, std::hash {}(frame.position.key())); + } + + // advance before entering the element, which pushes onto the stack + const BasicJsonType& element = *frame.position; + ++stack.back().position; + + if (element.is_structured()) + { + stack.emplace_back(&element, combine(static_cast(element.type()), element.size())); + } + else + { + stack.back().seed = combine(stack.back().seed, hash(element)); + } + } +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -7206,11 +7521,31 @@ class input_stream_adapter // General-purpose iterator-based adapter. It might not be as fast as // theoretically possible for some containers, but it is extremely versatile. -// SentinelType defaults to IteratorType for backward compatibility, but may -// be a different type (e.g., a C++20 sentinel or counted_iterator). +// SentinelType defaults to IteratorType for backward compatibility, but may be +// a different type, e.g. a C++20 sentinel such as std::default_sentinel_t when +// IteratorType is a std::counted_iterator. template class iterator_input_adapter { + // Whether the number of elements between two positions can be computed in + // O(1): either the iterator and the sentinel have the same type (plain + // std::distance) or, in C++20, the sentinel is a sized sentinel for the + // iterator (std::ranges::distance), e.g. std::default_sentinel_t paired + // with std::counted_iterator. + // + // JSON_HAS_RANGES gates the C++20 branch: on standard libraries with an + // incomplete (libstdc++ < 11, see #4440) evaluating + // std::contiguous_iterator on a std::counted_iterator is a hard error + // instead of yielding false, and these traits are instantiated for every + // adapter. Such toolchains fall back to the pointer-only test and simply + // use the byte-at-a-time scanner. + static constexpr bool sentinel_is_sized = +#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) + std::is_same::value || std::sized_sentinel_for; +#else + std::is_same::value; +#endif + public: using char_type = typename std::iterator_traits::value_type; @@ -7222,7 +7557,7 @@ class iterator_input_adapter // in wide_string_input_adapter, which does not expose this). static constexpr bool supports_seek = std::is_same::iterator_category, std::random_access_iterator_tag>::value - && std::is_same::value + && sentinel_is_sized && sizeof(char_type) == 1; iterator_input_adapter(IteratorType first, SentinelType last) @@ -7270,30 +7605,60 @@ class iterator_input_adapter private: // whether IteratorType refers to a contiguous range and therefore supports // a std::memcpy fast path (pointers always do; in C++20 we can also detect - // library iterators such as those of std::vector and std::string). - // Computing the available element count needs either same-type iterators - // (plain std::distance) or, in C++20, a sized sentinel (std::ranges::distance), - // e.g. std::counted_iterator paired with std::default_sentinel_t. - static constexpr bool iterator_is_contiguous = -#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) - (std::is_same::value || std::sized_sentinel_for) - && (std::contiguous_iterator || std::is_pointer::value); + // library iterators such as those of std::vector and std::string). The + // available element count must also be computable in O(1), hence + // sentinel_is_sized. + static constexpr bool iterator_is_contiguous = sentinel_is_sized && +#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) + (std::contiguous_iterator || std::is_pointer::value); #else - std::is_same::value && std::is_pointer::value; + std::is_pointer::value; #endif + // number of unread elements in [current, end) + std::size_t remaining_count() const + { +#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) + // std::ranges::distance also supports sized sentinels of a different + // type (e.g. std::counted_iterator + std::default_sentinel_t) + return static_cast(std::ranges::distance(current, end)); +#else + return static_cast(std::distance(current, end)); +#endif + } + + public: + // Whether the remaining input is a single contiguous block of 1-byte + // elements that the lexer can inspect directly (used for the SWAR string + // fast path). + static constexpr bool supports_bulk_scan = + iterator_is_contiguous && sizeof(char_type) == 1; + + // Pointer to the next unread element; only valid when bulk_remaining() > 0. + const char_type* bulk_data() const + { + return &*current; + } + + // Number of unread elements available as one contiguous block. + std::size_t bulk_remaining() const + { + return remaining_count(); + } + + // Consume @a n elements previously inspected via bulk_data(). + void bulk_skip(std::size_t n) + { + std::advance(current, static_cast::difference_type>(n)); + } + + private: // contiguous fast path: bulk copy the remaining range with std::memcpy template std::size_t get_elements_impl(T* dest, std::size_t count, std::true_type /*contiguous*/) { const std::size_t wanted = count * sizeof(T); -#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) - // std::ranges::distance also supports sized sentinels of a different - // type (e.g. std::counted_iterator + std::default_sentinel_t) - const std::size_t available = static_cast(std::ranges::distance(current, end)) * sizeof(char_type); -#else - const std::size_t available = static_cast(std::distance(current, end)) * sizeof(char_type); -#endif + const std::size_t available = remaining_count() * sizeof(char_type); const std::size_t copied = (std::min)(wanted, available); if (JSON_HEDLEY_LIKELY(copied != 0)) { @@ -7621,6 +7986,46 @@ typename iterator_input_adapter_factory::adapter_typ return factory_type::create(first, last); } +// The element type a container's data() points at, cv-qualifiers removed. +// Ill-formed - and therefore SFINAE-friendly - for types without data(). +template +using container_data_t = typename std::remove_cv().data()) >::type >::type; + +// The container's own element type, cv-qualifiers removed. It is looked up on +// the bare type so it is also found when ContainerType is deduced as a +// reference by the forwarding-reference overload below. +template +using container_value_t = typename std::remove_cv < + typename std::remove_cv::type>::type::value_type >::type; + +// Detect a container that stores its elements contiguously as single bytes +// (std::string, std::vector, std::array, +// std::string_view, ...). Such inputs are wrapped in a pointer-based adapter so +// they benefit from the contiguous fast paths (bulk string scanning, memcpy for +// binary formats) in every C++ standard - not only in C++20, where the standard +// library iterators model std::contiguous_iterator and are detected directly. +// +// data() and size() on their own would be duck typing: they say nothing about +// size() counting the units data() points at, and reading [data(), data() + +// size()) as bytes would be wrong for a type where it does not. Requiring the +// container's own value_type to be that same single-byte element ties the two +// together; every contiguous standard container satisfies it. Anything else +// keeps the iterator-based adapter, which is always correct - only slower. +template +struct is_contiguous_byte_container : std::false_type {}; + +template +struct is_contiguous_byte_container < ContainerType, void_t < + container_data_t, + container_value_t, +decltype(std::declval().size()) >> + : std::integral_constant < bool, + std::is_pointer().data())>::value&& + std::is_integral>::value&& + sizeof(container_data_t) == 1 && + std::is_same, container_value_t>::value > {}; + // Convenience shorthand from container to iterator // Enables ADL on begin(container) and end(container) // Encloses the using declarations in namespace for not to leak them to outside scope @@ -7648,12 +8053,32 @@ struct container_input_adapter_factory< ContainerType, } // namespace container_input_adapter_factory_impl -template -typename container_input_adapter_factory_impl::container_input_adapter_factory::adapter_type input_adapter(ContainerType&& container) +// General container path (iterator-based). Contiguous single-byte containers +// are excluded here and routed through the pointer-based overload below. +template < typename ContainerType, + enable_if_t < !is_contiguous_byte_container::value, int > = 0 > +typename container_input_adapter_factory_impl::container_input_adapter_factory::adapter_type input_adapter(ContainerType && container) { return container_input_adapter_factory_impl::container_input_adapter_factory::create(std::forward(container)); } +// Contiguous single-byte containers (std::string, std::vector, ...) are +// wrapped in a pointer-based adapter so the contiguous fast paths apply in every +// standard. The pointer keeps the container's own element type (const char* for +// std::string, const std::uint8_t* for std::vector, ...), so the +// resulting char_type - and therefore the parsing behavior - is byte-for-byte +// identical to the iterator-based path; only the raw pointer additionally +// enables the bulk fast paths. The container outlives the adapter for the whole +// parse (temporaries live until the end of the full expression), exactly as the +// iterators it replaces did. +template < typename ContainerType, + enable_if_t < is_contiguous_byte_container::value, int > = 0 > +auto input_adapter(const ContainerType& container) +-> decltype(input_adapter(container.data(), container.data() + container.size())) +{ + return input_adapter(container.data(), container.data() + container.size()); +} + // specialization for std::string using string_input_adapter_type = decltype(input_adapter(std::declval())); @@ -7703,6 +8128,21 @@ contiguous_bytes_input_adapter input_adapter(CharT b) template auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) { +#if JSON_STRICT_NUL_HANDLING + // A `char` array from string-literal initialization (e.g. json::parse("123")) + // carries a trailing '\0' contributed by the compiler, not by the source + // text; drop exactly that one byte so it is not mistaken for real trailing + // data. Every other element type (unsigned char, std::uint8_t, ...) keeps + // the full extent unconditionally, since a trailing zero byte there is + // genuine data (e.g. CBOR/MessagePack). This intentionally does not + // strlen()-scan the array (as the pointer overload above does for a + // null-delimited string): for a `char` array that is not NUL-terminated + // within its bounds, that would read past the end of the array. + if (std::is_same::type, char>::value && N > 0 && array[N - 1] == 0) + { + return input_adapter(array, array + N - 1); + } +#endif return input_adapter(array, array + N); } @@ -7751,10 +8191,11 @@ NLOHMANN_JSON_NAMESPACE_END +#include // find_if, min #include #include // string #include // enable_if_t -#include // move +#include // move, pair #include // vector // #include @@ -7782,8 +8223,603 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // array +#include // FLT_EVAL_METHOD +#include // size_t +#include // int64_t, uint64_t +#include // numeric_limits + +// #include + + +// std::from_chars lives in , but being in C++17 mode does not +// guarantee the header exists: GCC 7 sets __cplusplus to C++17 yet ships no +// (added in GCC 8; floating-point support in GCC 11). Guard the +// include with __has_include so such toolchains fall back to the scalar path. +#if defined(JSON_HAS_CPP_17) && defined(__has_include) + #if __has_include() + #include // from_chars (only used when __cpp_lib_to_chars is defined) + #include // errc + #endif +#endif + +// This file contains the value-conversion helpers used by the lexer to turn an +// already-validated number token into a value, without the locale/errno +// overhead of std::strtoull/std::strtod. They are free functions so the lexer +// stays focused on scanning; see lexer::convert_number(). + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief fast integer parser for an already-validated unsigned integer + +The number scanner has already checked that [first, last) is a valid JSON +integer, so this only needs to accumulate the digits and detect overflow. This +avoids the locale/errno machinery of std::strtoull, which dominates +integer-heavy inputs. + +@param[in] first pointer to the first character (a digit) +@param[in] last pointer past the last character +@param[out] value the parsed value on success +@return true if the value fit into @a NumberUnsignedType; false on overflow, in + which case the caller falls back to floating-point parsing (matching the + previous std::strtoull behavior) +*/ +template +bool parse_integer_unsigned(const char* first, const char* last, NumberUnsignedType& value) noexcept +{ + // accumulate in the widest unsigned type used by the previous strtoull + // path so the overflow behavior is unchanged for custom number types + std::uint64_t x = 0; + constexpr std::uint64_t cutoff = (std::numeric_limits::max)() / 10u; + constexpr std::uint64_t cutlim = (std::numeric_limits::max)() % 10u; + for (const char* p = first; p != last; ++p) + { + const auto digit = static_cast(static_cast(*p) - static_cast('0')); + if (JSON_HEDLEY_UNLIKELY(x > cutoff || (x == cutoff && digit > cutlim))) + { + return false; + } + x = (x * 10u) + digit; + } + value = static_cast(x); + // reject values that do not round-trip into a narrower NumberUnsignedType + return static_cast(value) == x; +} + +/*! +@brief fast integer parser for an already-validated negative integer + +@param[in] first pointer to the leading '-' +@param[in] last pointer past the last character +@param[out] value the parsed (negative) value on success +@return true on success; false on overflow (caller falls back to float) +*/ +template +bool parse_integer_signed(const char* first, const char* last, NumberIntegerType& value) noexcept +{ + // the state machine only reaches the signed path via a leading '-' + JSON_ASSERT(first != last && *first == '-'); + std::uint64_t magnitude = 0; + // |INT64_MIN| == INT64_MAX + 1; this is the largest admissible magnitude + constexpr std::uint64_t limit = static_cast((std::numeric_limits::max)()) + 1u; + for (const char* p = first + 1; p != last; ++p) + { + const auto digit = static_cast(static_cast(*p) - static_cast('0')); + if (JSON_HEDLEY_UNLIKELY(magnitude > (limit - digit) / 10u)) + { + return false; + } + magnitude = (magnitude * 10u) + digit; + } + const std::int64_t x = (magnitude == limit) + ? (std::numeric_limits::min)() + : -static_cast(magnitude); + value = static_cast(x); + // reject values that do not round-trip into a narrower NumberIntegerType + return static_cast(value) == x; +} + +/*! +@brief exact fast path for parsing a `double` (Clinger's algorithm) + +For the common case - at most 19 significant digits, a decimal exponent in +[-22, 22], and a significand below 2^53 - the value equals significand * +10^exp computed in IEEE-754 double arithmetic, which is exact under +round-to-nearest because both operands are exactly representable. This is the +same fast path used by fast_float/simdjson; the general cases are left to +std::strtod. The parser only activates for number_float_t == double; float and +long double keep the std::strtof/std::strtold paths (see the templated overload +below). + +@param[in] first pointer to the first character of the number +@param[in] last pointer past the last character +@param[in] decimal_point the (locale-dependent) decimal point character +@param[out] out the parsed value on success +@return true if the value was parsed exactly; false to fall back to strtod +*/ +template +bool parse_float_fast(const char* first, const char* last, DecimalPointType decimal_point, double& out) noexcept +{ +#if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0 + // Clinger's fast path is only exact when double operations are evaluated in + // true double precision. On platforms that keep intermediates in extended + // precision (e.g. the x87 FPU on 32-bit x86, where FLT_EVAL_METHOD == 2) the + // single significand * 10^scale step is double-rounded and can be 1 ULP off, + // so decline and let the caller fall back to the correctly-rounded + // std::from_chars / std::strtod path. + static_cast(first); + static_cast(last); + static_cast(decimal_point); + static_cast(out); + return false; +#else + static const std::array powers_of_ten = + { + { + 1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11, + 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22 + } + }; + + const char* p = first; + bool negative = false; + if (p != last && (*p == '-' || *p == '+')) + { + negative = (*p == '-'); + ++p; + } + + std::uint64_t significand = 0; + int num_digits = 0; + int fractional_digits = 0; + bool seen_dot = false; + bool any_digit = false; + for (; p != last; ++p) + { + const char c = *p; + if (c >= '0' && c <= '9') + { + any_digit = true; + if (JSON_HEDLEY_UNLIKELY(num_digits >= 19)) + { + return false; // significand may not fit into uint64_t + } + significand = (significand * 10u) + static_cast(c - '0'); + ++num_digits; + fractional_digits += static_cast(seen_dot); + } + else if (static_cast(c) == decimal_point) + { + if (JSON_HEDLEY_UNLIKELY(seen_dot)) + { + return false; + } + seen_dot = true; + } + else if (c == 'e' || c == 'E') + { + ++p; + break; + } + else + { + return false; + } + } + if (JSON_HEDLEY_UNLIKELY(!any_digit)) + { + return false; + } + + int exponent = 0; + if (p != last) // an exponent part remains + { + bool exp_negative = false; + if (p != last && (*p == '-' || *p == '+')) + { + exp_negative = (*p == '-'); + ++p; + } + bool any_exp_digit = false; + for (; p != last; ++p) + { + if (JSON_HEDLEY_UNLIKELY(*p < '0' || *p > '9')) + { + return false; + } + exponent = (exponent * 10) + (*p - '0'); + any_exp_digit = true; + if (JSON_HEDLEY_UNLIKELY(exponent > 9999)) + { + return false; + } + } + if (JSON_HEDLEY_UNLIKELY(!any_exp_digit)) + { + return false; + } + if (exp_negative) + { + exponent = -exponent; + } + } + + const int scale = exponent - fractional_digits; + if (JSON_HEDLEY_UNLIKELY(significand >= (static_cast(1) << 53))) + { + return false; // significand not exactly representable as double + } + + auto result = static_cast(significand); + if (scale >= 0) + { + if (JSON_HEDLEY_UNLIKELY(scale > 22)) + { + return false; + } + result *= powers_of_ten[static_cast(scale)]; + } + else + { + if (JSON_HEDLEY_UNLIKELY(-scale > 22)) + { + return false; + } + result /= powers_of_ten[static_cast(-scale)]; + } + out = negative ? -result : result; + return true; +#endif +} + +/// fast float path is only exact for `double`; decline for float/long double +template +bool parse_float_fast(const char* /*first*/, const char* /*last*/, DecimalPointType /*decimal_point*/, FloatType& /*out*/) noexcept +{ + return false; +} + +/*! +@brief parse a float with std::from_chars (Eisel-Lemire) when available + +std::from_chars is locale-independent, correctly rounded, and - via the +Eisel-Lemire algorithm in modern standard libraries - much faster than strtod +over the whole value range (not just the Clinger subset). It is used only when +__cpp_lib_to_chars indicates full floating-point support and only when it +consumes the entire token ([first, last)); a partial parse means the buffer +uses a non-'.' locale decimal point, in which case the caller falls back to the +locale-aware path. An under-/overflow (result_out_of_range) also declines, so +the caller's strtod fallback supplies the well-defined ±inf/0 result the parser +expects (side-stepping the P4168 divergence between implementations). + +@return true if the value was parsed exactly and fully; false to fall back +*/ +template +bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept +{ + // JSON_HAS_CPP_17 must gate the use as well as the include above: + // some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even + // in C++14 mode, where is not included. +#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars) + const auto result = std::from_chars(first, last, out); + return result.ec == std::errc() && result.ptr == last; +#else + static_cast(first); + static_cast(last); + static_cast(out); + return false; +#endif +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // size_t +#include // uint64_t +#include // memcpy + +// #include + + +// Optional SIMD backend for bulk UTF-8 validation. This is an opt-in external +// dependency: nlohmann/json itself stays header-only and the C++11 scalar +// validator below is always available; defining JSON_USE_SIMDUTF additionally +// requires the simdutf headers on the include path and linking the simdutf +// library. See string_bulk_run(). +// +// simdutf.h itself requires C++17 - it rejects older standards with an #error - +// so the backend is only compiled in from C++17 on. Below that the macro has no +// effect and the scalar validator is used; it accepts and rejects exactly the +// same input, so only throughput differs. macro_scope.hpp is included above to +// have JSON_HAS_CPP_17 available for this test. +#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17) + #include +#endif + +// This file contains the byte-level string-scanning helpers used by the lexer's +// contiguous fast path. They operate purely on raw bytes (no dependency on the +// lexer's template parameters) so they are free functions, keeping the lexer +// itself focused on the state machine; see lexer::scan_string_bulk(). + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +// classify a single byte as needing individual string handling: the closing +// quote, an escape, a control character, or a non-ASCII (UTF-8) +// lead/continuation byte. Ordinary bytes (0x20..0x7F except '"' and '\\') are +// copied verbatim, which the bulk scanner does 8 bytes at a time. +inline bool is_string_special(unsigned char c) noexcept +{ + return c == '\"' || c == '\\' || c < 0x20u || c >= 0x80u; +} + +// SWAR helper: return a word whose high bit is set in every byte of @a v that +// is_string_special(); zero if the 8 bytes are all ordinary. +inline std::uint64_t swar_string_special(std::uint64_t v) noexcept +{ + constexpr std::uint64_t ones = 0x0101010101010101ull; + constexpr std::uint64_t high = 0x8080808080808080ull; + const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22) + const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C) + const std::uint64_t has_quote = (q - ones) & ~q & high; + const std::uint64_t has_backslash = (b - ones) & ~b & high; + const std::uint64_t has_control = (v - 0x2020202020202020ull) & ~v & high; // < 0x20 + const std::uint64_t has_non_ascii = v & high; // >= 0x80 + return has_quote | has_backslash | has_control | has_non_ascii; +} + +// return the index of the first is_string_special() byte in [data, data+n), or +// n if every byte is ordinary; scans 8 bytes at a time +inline std::size_t find_string_special(const unsigned char* data, std::size_t n) noexcept +{ + std::size_t i = 0; + for (; i + 8 <= n; i += 8) + { + std::uint64_t word = 0; + std::memcpy(&word, data + i, sizeof(word)); + if (swar_string_special(word) != 0) + { + // a special byte is in this word; locate it (endian-agnostic) + for (std::size_t j = 0; j < 8; ++j) + { + if (is_string_special(data[i + j])) + { + return i + j; + } + } + } + } + for (; i < n; ++i) + { + if (is_string_special(data[i])) + { + return i; + } + } + return n; +} + +// classify a byte as one the serializer must NOT copy verbatim when +// ensure_ascii is requested: the closing quote, an escape, a control character +// (< 0x20), DEL (0x7F), or any non-ASCII byte (>= 0x80). Everything else - +// printable ASCII except '"' and '\\' - is emitted unchanged. Note this differs +// from is_string_special() only in that 0x7F is also a stop (it is escaped as +// \u007f under ensure_ascii). +inline bool is_ascii_copyable(unsigned char c) noexcept +{ + return c >= 0x20u && c < 0x7Fu && c != '"' && c != '\\'; +} + +// return the index of the first byte in [data, data+n) that is NOT +// is_ascii_copyable(), or n if every byte can be copied verbatim; scans 8 bytes +// at a time. Used by the serializer's ensure_ascii fast path. +inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_t n) noexcept +{ + constexpr std::uint64_t ones = 0x0101010101010101ull; + constexpr std::uint64_t high = 0x8080808080808080ull; + std::size_t i = 0; + for (; i + 8 <= n; i += 8) + { + std::uint64_t v = 0; + std::memcpy(&v, data + i, sizeof(v)); + const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22) + const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C) + const std::uint64_t d = v ^ 0x7F7F7F7F7F7F7F7Full; // DEL (0x7F) + const std::uint64_t stop = ((q - ones) & ~q & high) // == '"' + | ((b - ones) & ~b & high) // == '\\' + | ((d - ones) & ~d & high) // == 0x7F + | ((v - 0x2020202020202020ull) & ~v & high) // < 0x20 + | (v & high); // >= 0x80 + if (stop != 0) + { + break; + } + } + for (; i < n; ++i) + { + if (!is_ascii_copyable(data[i])) + { + return i; + } + } + return n; +} + +// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its +// length (2..4) only when the bytes form a *well-formed* sequence using exactly +// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts +// precisely what the byte path accepts. Returns 0 for anything that is invalid, +// incomplete, or that the byte path must diagnose (the caller then defers to +// that path, keeping error messages unchanged). Lead bytes < 0x80 are handled +// by the caller and never passed here. +inline std::size_t validate_one_utf8(const unsigned char* data, std::size_t avail) noexcept +{ + const unsigned char c0 = data[0]; + if (c0 >= 0xC2 && c0 <= 0xDF) // U+0080..U+07FF + { + if (avail >= 2 && data[1] >= 0x80 && data[1] <= 0xBF) + { + return 2; + } + } + else if (c0 == 0xE0) // U+0800..U+0FFF + { + if (avail >= 3 && data[1] >= 0xA0 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF) + { + return 3; + } + } + else if ((c0 >= 0xE1 && c0 <= 0xEC) || c0 == 0xEE || c0 == 0xEF) // U+1000..U+CFFF, U+E000..U+FFFF + { + if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF) + { + return 3; + } + } + else if (c0 == 0xED) // U+D000..U+D7FF (excludes surrogates) + { + if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0x9F && data[2] >= 0x80 && data[2] <= 0xBF) + { + return 3; + } + } + else if (c0 == 0xF0) // U+10000..U+3FFFF + { + if (avail >= 4 && data[1] >= 0x90 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF) + { + return 4; + } + } + else if (c0 >= 0xF1 && c0 <= 0xF3) // U+40000..U+FFFFF + { + if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF) + { + return 4; + } + } + else if (c0 == 0xF4) // U+100000..U+10FFFF + { + if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0x8F && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF) + { + return 4; + } + } + return 0; // invalid, incomplete, or must be diagnosed by the byte path +} + +// Scalar (C++11) computation of the bulk run length: the number of leading +// bytes in [data, data+n) that are ordinary ASCII or complete well-formed UTF-8 +// sequences, stopping before the first byte that needs individual handling (the +// closing quote, an escape, a control character, or an ill-formed/truncated +// sequence). ASCII is skipped 8 bytes at a time. +inline std::size_t scalar_string_bulk_run(const unsigned char* data, std::size_t n) noexcept +{ + std::size_t pos = 0; + while (pos < n) + { + pos += find_string_special(data + pos, n - pos); + if (pos >= n || data[pos] < 0x80u) + { + break; // end of buffer, or a quote/escape/control byte + } + const std::size_t seq = validate_one_utf8(data + pos, n - pos); + if (seq == 0) + { + break; // ill-formed or truncated: let the byte path diagnose it + } + pos += seq; + } + return pos; +} + +#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17) +// Index of the first quote/escape/control byte in [data, data+n) (non-ASCII +// bytes are *not* stops here - the whole run is handed to simdutf), or n. +inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t n) noexcept +{ + constexpr std::uint64_t ones = 0x0101010101010101ull; + constexpr std::uint64_t high = 0x8080808080808080ull; + std::size_t i = 0; + for (; i + 8 <= n; i += 8) + { + std::uint64_t v = 0; + std::memcpy(&v, data + i, sizeof(v)); + const std::uint64_t q = v ^ 0x2222222222222222ull; + const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; + const std::uint64_t hit = ((q - ones) & ~q & high) + | ((b - ones) & ~b & high) + | ((v - 0x2020202020202020ull) & ~v & high); + if (hit != 0) + { + for (std::size_t j = 0; j < 8; ++j) + { + const unsigned char c = data[i + j]; + if (c == '\"' || c == '\\' || c < 0x20u) + { + return i + j; + } + } + } + } + for (; i < n; ++i) + { + const unsigned char c = data[i]; + if (c == '\"' || c == '\\' || c < 0x20u) + { + return i; + } + } + return n; +} +#endif + +// Backend-dispatched bulk run length. With JSON_USE_SIMDUTF the run up to the +// next delimiter is validated in one shot by simdutf; on the rare failure the +// scalar helper recomputes the exact valid prefix so the byte path still +// produces the precise diagnostic. Without it, the pure scalar path is used. +inline std::size_t string_bulk_run(const unsigned char* data, std::size_t n) noexcept +{ +#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17) + const std::size_t run = find_string_delimiter(data, n); + if (run != 0 && simdutf::validate_utf8(reinterpret_cast(data), run)) + { + return run; + } +#endif + return scalar_string_bulk_run(data, n); +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // #include // #include @@ -7909,6 +8945,25 @@ constexpr bool input_adapter_supports_lookahead(std::false_type /*detected*/) return false; } +// Detect whether an input adapter exposes a contiguous byte block that the +// lexer can scan directly (see iterator_input_adapter::supports_bulk_scan). +// Adapters without the flag - file, stream, wide-string, user-defined - fall +// back to the character-at-a-time string scanner. +template +using detect_supports_bulk_scan = decltype(InputAdapterType::supports_bulk_scan); + +template +constexpr bool input_adapter_supports_bulk_scan(std::true_type /*detected*/) +{ + return InputAdapterType::supports_bulk_scan; +} + +template +constexpr bool input_adapter_supports_bulk_scan(std::false_type /*detected*/) +{ + return false; +} + /*! @brief lexical analysis @@ -7936,13 +8991,22 @@ class lexer : public lexer_base static constexpr bool can_release_lookahead = input_adapter_supports_lookahead(is_detected {}); + /// whether string scanning may bulk-consume runs of ordinary characters + /// directly from a contiguous input buffer (SWAR fast path). This requires + /// the token to be reconstructible lazily (lazy_token_string), so bypassing + /// the per-character capture in get() cannot lose error diagnostics. + static constexpr bool bulk_scan = + lazy_token_string + && input_adapter_supports_bulk_scan(is_detected {}); + public: using token_type = typename lexer_base::token_type; - explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept + explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept : ia(std::move(adapter)) , ignore_comments(ignore_comments_) , decimal_point_char(static_cast(get_decimal_point())) + , discard_number_values(discard_number_values_) {} // deleted because of pointer members @@ -8055,6 +9119,40 @@ class lexer : public lexer_base return true; } + /// contiguous input: bulk-append the run of ordinary characters and complete + /// well-formed UTF-8 sequences starting at the current read position, leaving + /// the first byte that needs individual handling (the closing quote, an + /// escape, a control character, or an ill-formed UTF-8 byte) for get() + void scan_string_bulk(std::true_type /*bulk*/) + { + // a pending unget must be consumed through the normal path first + if (next_unget) + { + return; + } + const std::size_t remaining = ia.bulk_remaining(); + if (remaining == 0) + { + return; + } + const auto* const data = reinterpret_cast(ia.bulk_data()); + + const std::size_t pos = string_bulk_run(data, remaining); + if (pos == 0) + { + return; + } + token_buffer.append(reinterpret_cast(data), pos); + ia.bulk_skip(pos); + // the run contains no newline (all bytes < 0x20 are treated as special), + // so only the flat character counters advance + position.chars_read_total += pos; + position.chars_read_current_line += pos; + } + + /// streaming input: no bulk fast path + void scan_string_bulk(std::false_type /*bulk*/) const noexcept {} + /*! @brief scan a string literal @@ -8080,6 +9178,10 @@ class lexer : public lexer_base while (true) { + // bulk-consume ordinary characters from contiguous input, then + // handle the next special byte through the switch below + scan_string_bulk(std::integral_constant {}); + // get the next character switch (get()) { @@ -8674,7 +9776,9 @@ class lexer : public lexer_base case '\n': case '\r': case char_traits::eof(): +#if !JSON_STRICT_NUL_HANDLING case '\0': +#endif return true; default: @@ -8692,8 +9796,10 @@ class lexer : public lexer_base { switch (get()) { - case char_traits::eof(): +#if !JSON_STRICT_NUL_HANDLING case '\0': +#endif + case char_traits::eof(): { error_message = "invalid comment; missing closing '*/'"; return false; @@ -8798,6 +9904,12 @@ class lexer : public lexer_base // changed if minus sign, decimal point, or exponent is read token_type number_type = token_type::value_unsigned; + // offset just past the last mantissa byte in token_buffer (i.e. the + // index of 'e'/'E', or the whole token when there is no exponent). + // convert_number() uses it to count significant digits; npos means + // "not seen an exponent yet" and is resolved at scan_number_done + std::size_t mantissa_end = std::string::npos; + // state (init): we just found out we need to scan a number switch (current) { @@ -8983,6 +10095,9 @@ scan_number_decimal2: scan_number_exponent: // we just parsed an exponent number_type = token_type::value_float; + // this label is reached only right after the 'e'/'E' was appended (from + // the zero, any1, and decimal2 states), so the mantissa ends before it + mantissa_end = token_buffer.size() - 1; switch (get()) { case '+': @@ -9069,45 +10184,199 @@ scan_number_done: // we are done scanning a number) unget(); - char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) - errno = 0; + // no exponent was scanned: the mantissa spans the whole token + if (mantissa_end == std::string::npos) + { + mantissa_end = token_buffer.size(); + } - // try to parse integers first and fall back to floats + return convert_number(number_type, mantissa_end); + } + + /*! + @brief convert an already-validated integer token to its value + + The digit sequence in [first, last) has been validated by the caller, so a + dedicated parser can avoid the locale/errno overhead of std::strtoull. + + @return the token type on success; token_type::uninitialized if @a + number_type is not an integer type or the value does not fit, in + which case the caller falls back to the floating-point conversion + (matching the previous std::strtoull/std::strtoll behavior) + */ + token_type convert_integer(token_type number_type, const char* first, const char* last) + { if (number_type == token_type::value_unsigned) { - const auto x = std::strtoull(token_buffer.data(), &endptr, 10); - - // we checked the number format before - JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); - - if (errno != ERANGE) + if (parse_integer_unsigned(first, last, value_unsigned)) { - value_unsigned = static_cast(x); - if (value_unsigned == x) - { - return token_type::value_unsigned; - } + return token_type::value_unsigned; } } else if (number_type == token_type::value_integer) { - const auto x = std::strtoll(token_buffer.data(), &endptr, 10); - - // we checked the number format before - JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); - - if (errno != ERANGE) + if (parse_integer_signed(first, last, value_integer)) { - value_integer = static_cast(x); - if (value_integer == x) - { - return token_type::value_integer; - } + return token_type::value_integer; + } + } + + return token_type::uninitialized; + } + + /*! + @brief check whether Clinger's fast path can still succeed for this token + + parse_float_fast() needs a significand below 2^53. A mantissa with 17 or + more significant digits is at least 10^16 and therefore always exceeds it, + so calling the fast path would walk the token one extra time only to + decline before strtod has to run anyway. + + Significant digits are the mantissa's digits from the first nonzero one on; + the sign, the decimal point, leading zeros, and the exponent do not count. + The answer is derived from indices - the digits are not scanned again - so + this stays off the hot path of the number scanners. + + @param[in] mantissa_end offset just past the last mantissa byte in + token_buffer + @return false if parse_float_fast() is guaranteed to decline + */ + bool mantissa_fits_clinger(std::size_t mantissa_end) const + { + // 10^16 already exceeds 2^53, so 17 digits can never fit + constexpr std::size_t limit = 17; + + const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u; + const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u; + // the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so + // a leading zero can only be a lone "0", which is not significant + const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u; + JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero); + std::size_t digits = mantissa_end - neg - has_dot - lead_zero; + + if (JSON_HEDLEY_LIKELY(digits < limit)) + { + return true; + } + + // Only a number below 1 can carry further insignificant zeros, and only + // while the count stays at the limit does removing them change the + // answer - so this loop is skipped for all but a few tokens. Note + // token_buffer holds the locale's decimal point, so the fraction is + // located through decimal_point_position rather than by searching '.'. + if (lead_zero != 0) + { + JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit + for (std::size_t i = decimal_point_position + 1; + digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i) + { + --digits; + } + } + + return digits < limit; + } + + /*! + @brief convert the number text in token_buffer to its value and token type + + The digit sequence in token_buffer has already been validated (by the + scan_number() state machine or by the contiguous fast path) and holds the + locale decimal point in place of '.'. Integers are parsed first and fall + back to floating point on overflow. This is shared so both scanners produce + identical results. + + @param[in] mantissa_end offset just past the last mantissa byte in + token_buffer (the index of 'e'/'E', or + token_buffer.size() when there is no exponent); + used to skip Clinger's fast path when it cannot + possibly succeed - see mantissa_fits_clinger() + */ + token_type convert_number(token_type number_type, std::size_t mantissa_end) + { + // If the caller does not need the converted value (only whether the + // input is syntactically valid; see json_sax_acceptor/accept()), an + // unsigned/integer token can be reported without calling + // strtoull()/strtoll() at all, *provided* we can already tell from + // the digit count alone that the conversion cannot overflow 64 bits. + // Such tokens are always finite and are accepted unconditionally by + // the parser regardless of their actual value (parser::sax_parse_internal() + // never checks finiteness for value_unsigned/value_integer), so the + // classification below is all that is needed. + // + // A decimal number with up to 18 digits is always representable in + // both std::uint64_t and std::int64_t (18 nines is ~1e18, well below + // both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll() + // could not have set errno to ERANGE for it. Numbers with more digits + // (rare in practice) fall through to the exact code below, unchanged, + // so their handling -- including reclassification to value_float when + // the value overflows 64 bits, and rejection when it is not even + // finite as a double -- is bit-for-bit identical to before this + // optimization. + // + // Note this reasons about std::uint64_t/std::int64_t, not about + // number_unsigned_t/number_integer_t (BasicJsonType's own, possibly + // narrower, template parameters -- e.g. std::uint32_t). That is fine + // *only* because discard_number_values is exclusively set by + // accept() (see json.hpp), and accept() always parses through the + // library's own json_sax_acceptor -- never a user-supplied SAX + // consumer -- whose number_unsigned()/number_integer()/number_float() + // callbacks unconditionally discard their argument and return true. + // So for every caller that can reach this branch, neither the token + // classification below nor the eventual (possibly narrowed, and on + // this fast path left stale/unset) value_unsigned/value_integer is + // ever consulted -- an unsigned/integer token is accepted outright, + // and even a >18-digit token that this fast path deliberately falls + // through for is, once reclassified to value_float, still finite + // (and thus accepted) for any digit count that fits in number_unsigned_t + // or number_integer_t regardless of that type's width. If this + // function is ever taught to run with discard_number_values true for + // a caller that *does* read the converted value, this reasoning (and + // the fast path below) would need to be revisited. + if (discard_number_values) + { + constexpr std::size_t safe_digit_count = 18; + if (number_type == token_type::value_unsigned && token_buffer.size() <= safe_digit_count) + { + return token_type::value_unsigned; + } + if (number_type == token_type::value_integer && token_buffer.size() - 1 <= safe_digit_count) + { + return token_type::value_integer; + } + } + + const char* const num_begin = token_buffer.data(); + const char* const num_end = num_begin + token_buffer.size(); + + if (number_type != token_type::value_float) + { + const token_type integer_result = convert_integer(number_type, num_begin, num_end); + if (integer_result != token_type::uninitialized) + { + return integer_result; } } // this code is reached if we parse a floating-point number or if an - // integer conversion above failed + // integer conversion above overflowed. Prefer std::from_chars + // (Eisel-Lemire, locale-independent, correctly rounded) when available; + // otherwise the exact Clinger fast path (double only); otherwise the + // locale-aware strtof/strtod. + if (parse_float_from_chars(num_begin, num_end, value_float)) + { + return token_type::value_float; + } + // Skipping a fast path that cannot succeed is lossless and saves a full + // extra pass over the token's bytes, which otherwise shows up on + // high-precision inputs such as canada.json + if (mantissa_fits_clinger(mantissa_end) + && parse_float_fast(num_begin, num_end, decimal_point_char, value_float)) + { + return token_type::value_float; + } + + char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) strtof(value_float, token_buffer.data(), &endptr); // we checked the number format before @@ -9116,6 +10385,158 @@ scan_number_done: return token_type::value_float; } + /*! + @brief contiguous fast path for scanning a number + + Parses the whole number token straight from the input buffer, avoiding the + per-character get()/add() of scan_number(). On success it fills token_buffer + (with the locale decimal point substituted, as scan_number() does) and + returns the token type. On anything it does not fully recognize as a + well-formed number it makes no state change and returns + token_type::uninitialized, so the caller falls back to scan_number(), which + then produces the exact diagnostic. @a current is the first digit or the + leading minus (already read); the remaining bytes are taken from the adapter. + */ + token_type scan_number_bulk_contiguous() + { + // a pending unget offsets the buffer position from current; fall back + if (next_unget) + { + return token_type::uninitialized; + } + const std::size_t rem = ia.bulk_remaining(); + if (rem == 0) + { + // the first digit is the last input byte; let scan_number() finish + return token_type::uninitialized; + } + // the byte before the next unread one is current (contiguous input) + const char* const data = reinterpret_cast(ia.bulk_data()) - 1; + const std::size_t avail = rem + 1; + + // validate + classify the number extent (mirrors scan_number()'s grammar) + std::size_t i = 0; + std::size_t dot_index = std::string::npos; + token_type number_type = token_type::value_unsigned; + if (data[0] == '-') + { + number_type = token_type::value_integer; + i = 1; + if (i >= avail) + { + return token_type::uninitialized; + } + } + if (data[i] == '0') + { + ++i; + } + else if (data[i] >= '1' && data[i] <= '9') + { + ++i; + while (i < avail && data[i] >= '0' && data[i] <= '9') + { + ++i; + } + } + else + { + return token_type::uninitialized; + } + if (i < avail && data[i] == '.') + { + number_type = token_type::value_float; + dot_index = i; + ++i; + if (i >= avail || !(data[i] >= '0' && data[i] <= '9')) + { + return token_type::uninitialized; + } + while (i < avail && data[i] >= '0' && data[i] <= '9') + { + ++i; + } + } + // the mantissa ends here, whether or not an exponent part follows + const std::size_t mantissa_end = i; + if (i < avail && (data[i] == 'e' || data[i] == 'E')) + { + number_type = token_type::value_float; + ++i; + if (i < avail && (data[i] == '+' || data[i] == '-')) + { + ++i; + } + if (i >= avail || !(data[i] >= '0' && data[i] <= '9')) + { + return token_type::uninitialized; + } + while (i < avail && data[i] >= '0' && data[i] <= '9') + { + ++i; + } + } + const std::size_t len = i; + + // reset() records where this token starts (for diagnostics), so it has + // to run before the input position advances below + reset(); + + // An integer token needs no token_buffer: the SAX callbacks for + // number_integer/number_unsigned take only the value, and the overflow + // diagnostic rebuilds the text from the input. Convert straight from the + // input buffer and leave token_buffer empty. (JSON_DIAGNOSTIC_POSITIONS + // derives a number's start position from get_string().size(), so there + // the token still has to be materialized.) +#if !JSON_DIAGNOSTIC_POSITIONS + if (number_type != token_type::value_float) + { + const token_type integer_result = convert_integer(number_type, data, data + len); + if (JSON_HEDLEY_LIKELY(integer_result != token_type::uninitialized)) + { + ia.bulk_skip(len - 1); + position.chars_read_total += (len - 1); + position.chars_read_current_line += (len - 1); + return integer_result; + } + // The value does not fit an integer, so this token converts as a + // float. Recording that here keeps convert_number() below from + // repeating the integer attempt that just failed. + number_type = token_type::value_float; + } +#endif + + // materialize the token exactly as scan_number() would, substituting the + // locale decimal point so convert_number()'s strtof fallback stays valid. + // reset() already cleared token_buffer, so append() fills it (assign() is + // avoided because custom string_t types need not provide it) + token_buffer.append(reinterpret_cast(data), len); + if (dot_index != std::string::npos) + { + token_buffer[dot_index] = static_cast(decimal_point_char); + decimal_point_position = dot_index; + } + + ia.bulk_skip(len - 1); + position.chars_read_total += (len - 1); + position.chars_read_current_line += (len - 1); + + return convert_number(number_type, mantissa_end); + } + + /// contiguous input: try the number fast path, else the byte-path scanner + token_type scan_number_dispatch(std::true_type /*bulk*/) + { + const token_type t = scan_number_bulk_contiguous(); + return (t != token_type::uninitialized) ? t : scan_number(); + } + + /// streaming input: always use the byte-path scanner + token_type scan_number_dispatch(std::false_type /*bulk*/) + { + return scan_number(); + } + /*! @param[in] literal_text the literal text to expect @param[in] length the length of the passed literal text @@ -9183,8 +10604,7 @@ scan_number_done: */ char_int_type get() { - ++position.chars_read_total; - ++position.chars_read_current_line; + advance_position(); if (next_unget) { @@ -9196,6 +10616,23 @@ scan_number_done: current = ia.get_character(); } + return track_after_read(); + } + + /// shared head of get() / get_ignoring_pending_unget(): bump the + /// per-character position counters (line-count-on-'\n' bookkeeping is + /// handled afterwards, in track_after_read(), once `current` is known) + void advance_position() noexcept + { + ++position.chars_read_total; + ++position.chars_read_current_line; + } + + /// shared tail of get() / get_ignoring_pending_unget(): capture the + /// character for error messages (if needed) and update line/column + /// bookkeeping for the character now in `current` + char_int_type track_after_read() + { // seekable adapters reconstruct the token lazily on error (see // get_token_string), so the eager per-character copy is skipped capture_char(std::integral_constant {}); @@ -9203,12 +10640,38 @@ scan_number_done: if (current == '\n') { ++position.lines_read; + // remember the column the newline was read at: chars_read_current_line + // is about to be cleared, and a matching unget() cannot reconstruct it + chars_read_before_newline = position.chars_read_current_line; position.chars_read_current_line = 0; } return current; } + /*! + @brief like get(), but for call sites that can prove no unget() is pending + + get() has to check the `next_unget` flag on every call, because a + previous token may have ended with unget() (e.g. scan_number() always + ungets the character that terminated the number, so the next call to + scan() can see it again). skip_whitespace() reads that first, + possibly-ungotten character via a plain get(), but every further + character it reads is guaranteed to be a fresh read: nothing between + those calls invokes unget(). This variant skips the (otherwise always + false) next_unget branch for those calls; it is not a general + replacement for get(). + */ + char_int_type get_ignoring_pending_unget() + { + JSON_ASSERT(!next_unget); + + advance_position(); + current = ia.get_character(); + + return track_after_read(); + } + /// seekable adapter: nothing to capture, the token is rebuilt on error void capture_char(std::true_type /*lazy*/) const noexcept {} @@ -9236,12 +10699,20 @@ scan_number_done: --position.chars_read_total; // in case we "unget" a newline, we have to also decrement the lines_read + // and restore the column that get() cleared when it saw the newline; + // chars_read_current_line == 0 can only mean the last get() read one if (position.chars_read_current_line == 0) { if (position.lines_read > 0) { --position.lines_read; } + + // chars_read_before_newline counts the newline itself, which is the + // character being ungotten, hence the -1 + position.chars_read_current_line = (chars_read_before_newline > 0) + ? chars_read_before_newline - 1 + : 0; } else { @@ -9440,13 +10911,37 @@ scan_number_done: return true; } + /// whether `current` is one of the four JSON whitespace characters + bool current_is_whitespace() const noexcept + { + return current == ' ' || current == '\t' || current == '\n' || current == '\r'; + } + void skip_whitespace() { + // the first character may be a pending unget() left over from the + // previous token (see get_ignoring_pending_unget()); every + // subsequent character read by this loop is guaranteed fresh, since + // nothing below calls unget() + get(); + + if (!current_is_whitespace()) + { + return; + } + + // this is written as an if-guarded do-while (rather than a plain + // while loop) because that shape is what lets both GCC and Clang + // keep the input adapter's read pointer in a register across + // iterations; the equivalent while-loop measurably defeated that + // optimization in testing, turning long whitespace runs (e.g. the + // indentation of pretty-printed JSON) from a register-only loop + // into one that reloads the pointer from memory every character do { - get(); + get_ignoring_pending_unget(); } - while (current == ' ' || current == '\t' || current == '\n' || current == '\r'); + while (current_is_whitespace()); } token_type scan() @@ -9522,11 +11017,14 @@ scan_number_done: case '7': case '8': case '9': - return scan_number(); + return scan_number_dispatch(std::integral_constant {}); - // end of input (the null byte is needed when parsing from - // string literals) +#if !JSON_STRICT_NUL_HANDLING case '\0': +#endif + // end of input; by default, a null byte is also treated as end of + // input for backwards compatibility (see JSON_STRICT_NUL_HANDLING + // to opt into rejecting a null byte in the input instead) case char_traits::eof(): return token_type::end_of_input; @@ -9553,6 +11051,10 @@ scan_number_done: /// the start position of the current token position_t position {}; + /// the value chars_read_current_line had when the last newline was read, so + /// that unget() can restore the column instead of leaving it at 0 + std::size_t chars_read_before_newline = 0; + /// raw input token string for error messages; only populated for streaming /// adapters (seekable adapters reconstruct it lazily via token_string_start) std::vector token_string {}; @@ -9582,6 +11084,13 @@ scan_number_done: const char_int_type decimal_point_char = '.'; /// the position of the decimal point in the input std::size_t decimal_point_position = std::string::npos; + + /// whether the caller (e.g. accept()/json_sax_acceptor) only needs the + /// token classification and never looks at the converted numeric value; + /// when set, scan_number() may skip strtoull()/strtoll() for + /// value_unsigned/value_integer tokens whose digit count guarantees they + /// fit into 64 bits (see scan_number()) + const bool discard_number_values = false; }; } // namespace detail @@ -9589,6 +11098,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include NLOHMANN_JSON_NAMESPACE_BEGIN @@ -9723,6 +11234,29 @@ constexpr std::size_t unknown_size() return (std::numeric_limits::max)(); } +/*! +@brief reserve capacity for @a len elements in array @a arr + +Reserving upfront avoids repeated reallocations while the elements are added, +but the reservation is capped so a bogus/hostile length (which is not bounded +by max_size(), unlike e.g. std::vector) cannot trigger an oversized allocation +for a small or truncated input. + +The overload below is selected for array types without reserve() (e.g., +std::deque), which are then left untouched. +*/ +template +auto reserve_array(ArrayType& arr, std::size_t len, priority_tag<1> /*unused*/) +-> decltype(arr.reserve(len), void()) +{ + constexpr std::size_t reserve_cap = 16384; + arr.reserve((std::min)(len, reserve_cap)); +} + +template +inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/) +{} + /*! @brief SAX implementation to create a JSON value from SAX events @@ -9795,12 +11329,16 @@ class json_sax_dom_parser bool string(string_t& val) { - handle_value(val); + // json_sax documents that the passed value may be moved from, + // so hand the buffer over instead of copying it + handle_value(std::move(val)); return true; } bool binary(binary_t& val) { + // json_sax documents that the passed value may be moved from, + // so hand the buffer over instead of copying it handle_value(std::move(val)); return true; } @@ -9822,7 +11360,7 @@ class json_sax_dom_parser if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); } return true; @@ -9871,7 +11409,12 @@ class json_sax_dom_parser if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + } + + if (len != detail::unknown_size()) + { + reserve_array(*ref_stack.back()->m_data.m_value.array, len, priority_tag<1> {}); } return true; @@ -10105,12 +11648,16 @@ class json_sax_dom_callback_parser bool string(string_t& val) { - handle_value(val); + // json_sax documents that the passed value may be moved from, + // so hand the buffer over instead of copying it + handle_value(std::move(val)); return true; } bool binary(binary_t& val) { + // json_sax documents that the passed value may be moved from, + // so hand the buffer over instead of copying it handle_value(std::move(val)); return true; } @@ -10121,6 +11668,11 @@ class json_sax_dom_callback_parser const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::object_start, discarded); keep_stack.push_back(keep); + // the key this object will be stored under, read before handle_value() + // may consume it; kept in lockstep with ref_stack so end_object() can + // find the object in its parent again + container_key_stack.push_back(current_key()); + auto val = handle_value(BasicJsonType::value_t::object, true); ref_stack.push_back(val.second); @@ -10141,7 +11693,7 @@ class json_sax_dom_callback_parser // check object limit if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); } } return true; @@ -10154,11 +11706,24 @@ class json_sax_dom_callback_parser // check callback for the key const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::key, k); key_keep_stack.push_back(keep); + // remember the key so a rejected value can be erased without searching + // the object for it (kept in lockstep with key_keep_stack) + key_stack.push_back(val); // add discarded value at the given key and store the reference for later if (keep && ref_stack.back()) { - object_element = &(ref_stack.back()->m_data.m_value.object->operator[](val) = discarded); + auto& obj = *ref_stack.back()->m_data.m_value.object; + const auto it = obj.find(val); + if (it != obj.end()) + { + // this is a duplicate key (legal in JSON); remember its + // current value so it can be restored later if the new + // value is rejected by the callback, instead of being + // erased together with the discarded placeholder + duplicate_key_stash.emplace_back(&(it->second), it->second); + } + object_element = &(obj[val] = discarded); } return true; @@ -10170,13 +11735,18 @@ class json_sax_dom_callback_parser { if (!callback(static_cast(ref_stack.size()) - 1, parse_event_t::object_end, *ref_stack.back())) { - // discard object - *ref_stack.back() = discarded; + // discard object, unless this slot holds a duplicate key's + // previous value pending restoration, in which case that + // value is restored instead of being discarded + if (!resolve_duplicate_key_stash(ref_stack.back(), true)) + { + *ref_stack.back() = discarded; #if JSON_DIAGNOSTIC_POSITIONS - // Set start/end positions for discarded object. - handle_diagnostic_positions_for_json_value(*ref_stack.back()); + // Set start/end positions for discarded object. + handle_diagnostic_positions_for_json_value(*ref_stack.back()); #endif + } } else { @@ -10190,18 +11760,25 @@ class json_sax_dom_callback_parser #endif ref_stack.back()->set_parents(); + // this object is finally, definitively kept; drop any + // pending duplicate-key stash entry for its slot since it + // can no longer be restored + resolve_duplicate_key_stash(ref_stack.back(), false); } } JSON_ASSERT(!ref_stack.empty()); JSON_ASSERT(!keep_stack.empty()); + JSON_ASSERT(!container_key_stack.empty()); ref_stack.pop_back(); keep_stack.pop_back(); + const string_t object_key = std::move(container_key_stack.back()); + container_key_stack.pop_back(); if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_structured()) { // remove discarded value - remove_discarded_value(*ref_stack.back()); + remove_discarded_value(*ref_stack.back(), object_key); } return true; @@ -10212,6 +11789,9 @@ class json_sax_dom_callback_parser const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::array_start, discarded); keep_stack.push_back(keep); + // see start_object() + container_key_stack.push_back(current_key()); + auto val = handle_value(BasicJsonType::value_t::array, true); ref_stack.push_back(val.second); @@ -10232,7 +11812,12 @@ class json_sax_dom_callback_parser // check array limit if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + } + + if (len != detail::unknown_size()) + { + reserve_array(*ref_stack.back()->m_data.m_value.array, len, priority_tag<1> {}); } } @@ -10259,23 +11844,35 @@ class json_sax_dom_callback_parser #endif ref_stack.back()->set_parents(); + // this array is finally, definitively kept; drop any + // pending duplicate-key stash entry for its slot since it + // can no longer be restored + resolve_duplicate_key_stash(ref_stack.back(), false); } else { - // discard array - *ref_stack.back() = discarded; + // discard array, unless this slot holds a duplicate key's + // previous value pending restoration, in which case that + // value is restored instead of being discarded + if (!resolve_duplicate_key_stash(ref_stack.back(), true)) + { + *ref_stack.back() = discarded; #if JSON_DIAGNOSTIC_POSITIONS - // Set start/end positions for discarded array. - handle_diagnostic_positions_for_json_value(*ref_stack.back()); + // Set start/end positions for discarded array. + handle_diagnostic_positions_for_json_value(*ref_stack.back()); #endif + } } } JSON_ASSERT(!ref_stack.empty()); JSON_ASSERT(!keep_stack.empty()); + JSON_ASSERT(!container_key_stack.empty()); ref_stack.pop_back(); keep_stack.pop_back(); + const string_t object_key = std::move(container_key_stack.back()); + container_key_stack.pop_back(); // remove discarded value if (!ref_stack.empty() && ref_stack.back()) @@ -10289,7 +11886,7 @@ class json_sax_dom_callback_parser // the array is either still stored under its key or was never // stored, leaving the placeholder key() wrote; both show up as // a discarded member of the parent object - remove_discarded_value(*ref_stack.back()); + remove_discarded_value(*ref_stack.back(), object_key); } } @@ -10382,15 +11979,92 @@ class json_sax_dom_callback_parser } #endif - /// remove the discarded value the callback rejected from its parent - static void remove_discarded_value(BasicJsonType& parent) + /// if there is a pending duplicate-key stash entry for this exact slot, + /// remove it from the stash; if restore_value is true, the stashed + /// previous value is moved back into the slot first (use this when the + /// new value at that slot was rejected); otherwise the stash entry is + /// simply dropped (use this when the new value was accepted, so it + /// correctly supersedes the old one and no restore should ever happen + /// for this slot again) + /// @return whether a matching stash entry was found (and processed) + bool resolve_duplicate_key_stash(BasicJsonType* slot, bool restore_value) { - for (auto it = parent.begin(); it != parent.end(); ++it) + const auto it = std::find_if(duplicate_key_stash.begin(), duplicate_key_stash.end(), + [slot](const std::pair& entry) { - if (it->is_discarded()) + return entry.first == slot; + }); + + if (it == duplicate_key_stash.end()) + { + return false; + } + + if (restore_value) + { + *slot = std::move(it->second); + } + duplicate_key_stash.erase(it); + return true; + } + + /*! + @brief the key the value now being handled will be stored under + + Empty unless the enclosing container is an object, in which case it is the + key of the pending key() event. Read before handle_value() consumes that + key, so it is also correct when the value never reaches its parent. + */ + string_t current_key() const + { + if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_object() + && !key_stack.empty()) + { + return key_stack.back(); + } + return string_t{}; + } + + /*! + @brief remove the discarded value the callback rejected from its parent, + unless it is a duplicate key's slot with a stashed previous value, in + which case that previous value is restored instead + + A rejected value can only ever be the one most recently added to @a parent: + the last element of an array, or the placeholder key() stored under @a key + in an object. Looking there directly makes this O(1) resp. O(log n), where + searching @a parent for it made a filtering parse quadratic in the number of + members of a single container. + + Finding no discarded value there means none was stored in the first place - + the callback rejected the value before it reached its parent - so there is + nothing to remove. + + @param[in,out] parent the container to remove the rejected value from + @param[in] key the key the value was stored under; unused for arrays + */ + void remove_discarded_value(BasicJsonType& parent, const string_t& key) + { + if (parent.is_array()) + { + auto& array = *parent.m_data.m_value.array; + if (!array.empty() && array.back().is_discarded()) { - parent.erase(it); - break; + array.pop_back(); + } + } + else if (parent.is_object()) + { + auto& object = *parent.m_data.m_value.object; + const auto it = object.find(key); + if (it != object.end() && it->second.is_discarded()) + { + // a duplicate key's slot has a stashed previous value that + // must be restored instead of being erased + if (!resolve_duplicate_key_stash(&it->second, true)) + { + object.erase(it); + } } } } @@ -10440,11 +12114,14 @@ class json_sax_dom_callback_parser if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_object()) { JSON_ASSERT(!key_keep_stack.empty()); + JSON_ASSERT(!key_stack.empty()); const bool placeholder_stored = key_keep_stack.back(); key_keep_stack.pop_back(); + const string_t key = std::move(key_stack.back()); + key_stack.pop_back(); if (placeholder_stored) { - remove_discarded_value(*ref_stack.back()); + remove_discarded_value(*ref_stack.back(), key); } } return {false, nullptr}; @@ -10477,8 +12154,10 @@ class json_sax_dom_callback_parser JSON_ASSERT(ref_stack.back()->is_object()); // check if we should store an element for the current key JSON_ASSERT(!key_keep_stack.empty()); + JSON_ASSERT(!key_stack.empty()); const bool store_element = key_keep_stack.back(); key_keep_stack.pop_back(); + key_stack.pop_back(); if (!store_element) { @@ -10487,6 +12166,16 @@ class json_sax_dom_callback_parser JSON_ASSERT(object_element); *object_element = std::move(value); + if (!skip_callback) + { + // this scalar value finally, definitively replaces whatever was + // at this slot; drop any pending duplicate-key stash entry for + // it since it can no longer be restored (a container value at + // this slot is resolved later, in end_object()/end_array(), + // since skip_callback is true for the placeholder handling that + // happens here for those) + resolve_duplicate_key_stash(object_element, false); + } return {true, object_element}; } @@ -10498,8 +12187,20 @@ class json_sax_dom_callback_parser std::vector keep_stack {}; // NOLINT(readability-redundant-member-init) /// stack to manage which object keys to keep std::vector key_keep_stack {}; // NOLINT(readability-redundant-member-init) + /// the keys key() stored a placeholder for, in lockstep with key_keep_stack + std::vector key_stack {}; // NOLINT(readability-redundant-member-init) + /// for each open container, the key it is stored under in its parent + /// object, in lockstep with ref_stack; unused where the parent is not an + /// object + std::vector container_key_stack {}; // NOLINT(readability-redundant-member-init) /// helper to hold the reference for the next object element BasicJsonType* object_element = nullptr; + /// stash of (slot pointer, previous value) for object members that + /// already existed when key() was called again for the same key + /// (duplicate keys); used to restore the previous value if the new + /// value is later rejected by the callback, instead of erasing the + /// member entirely + std::vector> duplicate_key_stash {}; /// whether a syntax error occurred bool errored = false; /// callback function @@ -10790,6 +12491,26 @@ inline bool little_endianness(int num = 1) noexcept return *reinterpret_cast(&num) == 1; } +/*! +@brief largest element count accepted for a UBJSON container of a valueless type + +An element of type 'Z' (null), 'T' (true) or 'F' (false) is encoded by its +type marker alone, so an optimized container of one of those types has no +payload at all and its declared count is the only thing that decides how much +is allocated: `[$Z#L` followed by a large count turns some ten bytes of input +into that many values (see #2793, which reports 35 GB and 150 seconds). Every +other type costs at least one byte per element and is bounded by the end of +the input. + +This is a sanity bound rather than a security boundary, and it is far above +any container met in practice. @ref binary_writer falls back to the +unoptimized encoding for longer containers, so that a value serialized by +this library can always be read back. + +@sa https://github.com/nlohmann/json/issues/2793 +*/ +JSON_INLINE_VARIABLE constexpr std::size_t max_valueless_container_size = 1 << 20; + /////////////////// // binary reader // /////////////////// @@ -10842,6 +12563,7 @@ class binary_reader const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { sax = sax_; + container_stack.clear(); bool result = false; switch (format) @@ -10891,6 +12613,80 @@ class binary_reader } private: + //////////////////////// + // nested containers // + //////////////////////// + + /*! + @brief a container that has been opened and not closed yet + + The binary readers do not call themselves once per nesting level. Like + @ref parser::sax_parse_internal, which does the same for JSON text, they + keep the containers they are inside of on a heap-allocated stack, so that + the native call stack does not grow with the nesting depth of the input + and a deeply nested value is bounded by memory rather than by the stack + (see #5104). + + The members are ordered by decreasing alignment, which is the ordering that + keeps a struct from growing as members are added to it. + */ + struct container_frame + { + container_frame(const std::size_t remaining_, const bool is_object_, + const char_int_type type_marker_ = 0) noexcept + : remaining(remaining_), type_marker(type_marker_), is_object(is_object_) {} + + /// number of elements that have not been read yet, or npos when the + /// container is not sized and ends at a marker instead + std::size_t remaining; + /// BSON: value of chars_read before this document's size prefix, which + /// check_bson_document_size() needs once the document has been read + std::size_t start_position = 0; + /// UBJSON/BJData: the type marker of an optimized container, so that + /// its elements are read without one of their own; 0 otherwise + char_int_type type_marker; + /// BSON: the size this document declares, in bytes + std::int32_t declared_size = 0; + /// whether to close this container with end_object() or end_array() + bool is_object; + }; + + /*! + @brief open a nested array or object + + Emits the SAX start event and records the container. This is the only + place the binary readers start a container, so a check that rejects one + can be made here and is then guaranteed to run before the start event. + + @param[in] is_object whether an object (true) or an array (false) begins + @param[in] len number of elements the container declares + + @return whether the SAX parser accepted the start event + */ + bool enter_container(const bool is_object, const std::size_t len, + const char_int_type type_marker = 0) + { + if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->start_object(len) : !sax->start_array(len))) + { + return false; + } + + container_stack.emplace_back(len, is_object, type_marker); + return true; + } + + /// @copydoc enter_container + bool enter_array(const std::size_t len, const char_int_type type_marker = 0) + { + return enter_container(/*is_object*/false, len, type_marker); + } + + /// @copydoc enter_container + bool enter_object(const std::size_t len, const char_int_type type_marker = 0) + { + return enter_container(/*is_object*/true, len, type_marker); + } + ////////// // BSON // ////////// @@ -10925,8 +12721,10 @@ class binary_reader @brief Reads in a BSON-object and passes it to the SAX-parser. @return whether a valid BSON-value was passed to the SAX parser */ - bool parse_bson_internal() + bool open_bson_document(const bool is_object) { + // recorded before the size prefix is read, because + // check_bson_document_size() measures the document from here const std::size_t document_start = chars_read; std::int32_t document_size{}; if (!get_number(input_format_t::bson, document_size)) @@ -10934,22 +12732,91 @@ class binary_reader return false; } - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(detail::unknown_size()))) + if (JSON_HEDLEY_UNLIKELY(!enter_container(is_object, detail::unknown_size()))) { return false; } - if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_list(/*is_array*/false))) + container_frame& frame = container_stack.back(); + frame.start_position = document_start; + frame.declared_size = document_size; + return true; + } + + /*! + @brief read a BSON document and everything nested inside it + + Reads elements until the document that was begun here is complete, + resuming the enclosing document each time an embedded one ends, so that + the nesting depth of the input costs heap rather than native stack + (see #5104). + + @return whether reading the document succeeded + */ + bool parse_bson_internal() + { + if (JSON_HEDLEY_UNLIKELY(!open_bson_document(/*is_object*/true))) { return false; } - if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(document_start, document_size))) - { - return false; - } + // the key currently being read; hoisted out of the loop so that its + // capacity is reused across elements and across nesting levels + string_t key; - return sax->end_object(); + while (true) + { + const auto element_type = get(); + + if (element_type == 0) // end of the innermost document + { + // a copy, not a reference: it must stay valid across the + // pop_back() below, which destroys the container_stack + // element it would otherwise alias + const container_frame top = container_stack.back(); + + if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(top.start_position, top.declared_size))) + { + return false; + } + + container_stack.pop_back(); + if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + { + return false; + } + // the document begun here is complete once it is not inside one + if (container_stack.empty()) + { + return true; + } + continue; + } + + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::bson, "element list"))) + { + return false; + } + + const std::size_t element_type_parse_position = chars_read; + key.clear(); + if (JSON_HEDLEY_UNLIKELY(!get_bson_cstr(key))) + { + return false; + } + + // an array's elements are named "0", "1", ... in the wire format, + // and those names are not passed on + if (container_stack.back().is_object && !sax->key(key)) + { + return false; + } + + if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_internal(element_type, element_type_parse_position))) + { + return false; + } + } } /*! @@ -11061,12 +12928,12 @@ class binary_reader case 0x03: // object { - return parse_bson_internal(); + return open_bson_document(/*is_object*/true); } case 0x04: // array { - return parse_bson_array(); + return open_bson_document(/*is_object*/false); } case 0x05: // binary @@ -11116,82 +12983,7 @@ class binary_reader } } - /*! - @brief Read a BSON element list (as specified in the BSON-spec) - The same binary layout is used for objects and arrays, hence it must be - indicated with the argument @a is_array which one is expected - (true --> array, false --> object). - - @param[in] is_array Determines if the element list being read is to be - treated as an object (@a is_array == false), or as an - array (@a is_array == true). - @return whether a valid BSON-object/array was passed to the SAX parser - */ - bool parse_bson_element_list(const bool is_array) - { - string_t key; - - while (auto element_type = get()) - { - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::bson, "element list"))) - { - return false; - } - - const std::size_t element_type_parse_position = chars_read; - if (JSON_HEDLEY_UNLIKELY(!get_bson_cstr(key))) - { - return false; - } - - if (!is_array && !sax->key(key)) - { - return false; - } - - if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_internal(element_type, element_type_parse_position))) - { - return false; - } - - // get_bson_cstr only appends - key.clear(); - } - - return true; - } - - /*! - @brief Reads an array from the BSON input and passes it to the SAX-parser. - @return whether a valid BSON-array was passed to the SAX parser - */ - bool parse_bson_array() - { - const std::size_t document_start = chars_read; - std::int32_t document_size{}; - if (!get_number(input_format_t::bson, document_size)) - { - return false; - } - - if (JSON_HEDLEY_UNLIKELY(!sax->start_array(detail::unknown_size()))) - { - return false; - } - - if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_list(/*is_array*/true))) - { - return false; - } - - if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(document_start, document_size))) - { - return false; - } - - return sax->end_array(); - } ////////// // CBOR // @@ -11223,9 +13015,12 @@ class binary_reader @return whether a valid CBOR value was passed to the SAX parser */ - bool parse_cbor_internal(const bool get_char, - const cbor_tag_handler_t tag_handler) + bool parse_cbor_value(const bool get_char, + const cbor_tag_handler_t tag_handler, + bool& tag_pending) { + tag_pending = false; + switch (get_char ? get() : current) { // EOF @@ -11417,37 +13212,36 @@ class binary_reader case 0x95: case 0x96: case 0x97: - return get_cbor_array( - conditional_static_cast(static_cast(current) & 0x1Fu), tag_handler); + return enter_array(conditional_static_cast(static_cast(current) & 0x1Fu)); case 0x98: // array (one-byte uint8_t for n follows) { std::uint8_t len{}; - return get_number(input_format_t::cbor, len) && get_cbor_array(static_cast(len), tag_handler); + return get_number(input_format_t::cbor, len) && enter_array(static_cast(len)); } case 0x99: // array (two-byte uint16_t for n follow) { std::uint16_t len{}; - return get_number(input_format_t::cbor, len) && get_cbor_array(static_cast(len), tag_handler); + return get_number(input_format_t::cbor, len) && enter_array(static_cast(len)); } case 0x9A: // array (four-byte uint32_t for n follow) { std::uint32_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && get_cbor_array(size, tag_handler); + return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size); } case 0x9B: // array (eight-byte uint64_t for n follow) { std::uint64_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && get_cbor_array(size, tag_handler); + return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size); } case 0x9F: // array (indefinite length) - return get_cbor_array(detail::unknown_size(), tag_handler); + return enter_array(detail::unknown_size()); // map (0x00..0x17 pairs of data items follow) case 0xA0: @@ -11474,36 +13268,36 @@ class binary_reader case 0xB5: case 0xB6: case 0xB7: - return get_cbor_object(conditional_static_cast(static_cast(current) & 0x1Fu), tag_handler); + return enter_object(conditional_static_cast(static_cast(current) & 0x1Fu)); case 0xB8: // map (one-byte uint8_t for n follows) { std::uint8_t len{}; - return get_number(input_format_t::cbor, len) && get_cbor_object(static_cast(len), tag_handler); + return get_number(input_format_t::cbor, len) && enter_object(static_cast(len)); } case 0xB9: // map (two-byte uint16_t for n follow) { std::uint16_t len{}; - return get_number(input_format_t::cbor, len) && get_cbor_object(static_cast(len), tag_handler); + return get_number(input_format_t::cbor, len) && enter_object(static_cast(len)); } case 0xBA: // map (four-byte uint32_t for n follow) { std::uint32_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && get_cbor_object(size, tag_handler); + return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size); } case 0xBB: // map (eight-byte uint64_t for n follow) { std::uint64_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && get_cbor_object(size, tag_handler); + return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size); } case 0xBF: // map (indefinite length) - return get_cbor_object(detail::unknown_size(), tag_handler); + return enter_object(detail::unknown_size()); case 0xC0: // tagged item case 0xC1: @@ -11587,7 +13381,10 @@ class binary_reader default: break; } - return parse_cbor_internal(true, tag_handler); + // the tagged value follows; it is read by the loop in + // parse_cbor_internal() rather than by recursing here + tag_pending = true; + return true; } case cbor_tag_handler_t::store: @@ -11637,7 +13434,11 @@ class binary_reader break; } default: - return parse_cbor_internal(true, tag_handler); + { + // as above, the tagged value is read by the caller + tag_pending = true; + return true; + } } get(); return get_cbor_binary(b) && sax->binary(b); @@ -11728,23 +13529,21 @@ class binary_reader } /*! - @brief reads a CBOR string + @brief reads a definite-length CBOR string - This function first reads starting bytes to determine the expected - string length and then copies this number of bytes into a string. - Additionally, CBOR's strings with indefinite lengths are supported. + Reads everything @ref get_cbor_string accepts except the indefinite-length + form, which that function handles itself. The bytes are appended to @a + result, so consecutive chunks of an indefinite-length string can be read + into the same string. - @param[out] result created string + @param[out] result string the bytes are appended to @return whether string creation completed - */ - bool get_cbor_string(string_t& result) - { - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "string"))) - { - return false; - } + @pre @a current is not EOF + */ + bool get_cbor_string_chunk(string_t& result) + { switch (current) { // UTF-8 string (0x00..0x17 bytes follow) @@ -11800,20 +13599,6 @@ class binary_reader return get_number(input_format_t::cbor, len) && get_string(input_format_t::cbor, len, result); } - case 0x7F: // UTF-8 string (indefinite length) - { - while (get() != 0xFF) - { - string_t chunk; - if (!get_cbor_string(chunk)) - { - return false; - } - result.append(chunk); - } - return true; - } - default: { auto last_token = get_token_string(); @@ -11824,23 +13609,82 @@ class binary_reader } /*! - @brief reads a CBOR byte array + @brief reads a CBOR string This function first reads starting bytes to determine the expected - byte array length and then copies this number of bytes into the byte array. - Additionally, CBOR's byte arrays with indefinite lengths are supported. + string length and then copies this number of bytes into a string. + Additionally, CBOR's strings with indefinite lengths are supported. - @param[out] result created byte array + @param[out] result created string + + @return whether string creation completed + */ + bool get_cbor_string(string_t& result) + { + // number of indefinite-length strings that have been opened and not + // closed yet. RFC 8949, Section 3.2.3 does not permit nesting them, + // but this reader has always accepted it, so the open levels are + // counted instead of recursed through, which overflowed the stack for + // an input of repeated 0x7F bytes (see #5104). Every chunk is appended + // to the same result, so no per-level state is needed. + std::size_t open = 0; + + while (true) + { + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "string"))) + { + return false; + } + + if (current == 0x7F) // UTF-8 string (indefinite length) + { + ++open; + get(); + continue; + } + + // a break marker closes the innermost indefinite-length string; + // outside of one it is not a string and falls through to the error + if (open != 0 && current == 0xFF) + { + if (--open == 0) + { + return true; + } + get(); + continue; + } + + if (JSON_HEDLEY_UNLIKELY(!get_cbor_string_chunk(result))) + { + return false; + } + + if (open == 0) + { + return true; + } + + get(); + } + } + + /*! + @brief reads a definite-length CBOR byte array + + Reads everything @ref get_cbor_binary accepts except the indefinite-length + form, which that function handles itself. The bytes are appended to @a + result, so consecutive chunks of an indefinite-length byte array can be + read into the same byte array. + + @param[out] result byte array the bytes are appended to @return whether byte array creation completed - */ - bool get_cbor_binary(binary_t& result) - { - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "binary"))) - { - return false; - } + @pre @a current is not EOF + */ + bool get_cbor_binary_chunk(binary_t& result) + { switch (current) { // Binary data (0x00..0x17 bytes follow) @@ -11900,20 +13744,6 @@ class binary_reader get_binary(input_format_t::cbor, len, result); } - case 0x5F: // Binary data (indefinite length) - { - while (get() != 0xFF) - { - binary_t chunk; - if (!get_cbor_binary(chunk)) - { - return false; - } - result.insert(result.end(), chunk.begin(), chunk.end()); - } - return true; - } - default: { auto last_token = get_token_string(); @@ -11923,6 +13753,63 @@ class binary_reader } } + /*! + @brief reads a CBOR byte array + + This function first reads starting bytes to determine the expected + byte array length and then copies this number of bytes into the byte array. + Additionally, CBOR's byte arrays with indefinite lengths are supported. + + @param[out] result created byte array + + @return whether byte array creation completed + */ + bool get_cbor_binary(binary_t& result) + { + // the open indefinite-length byte arrays are counted rather than + // recursed through, for the reason given in @ref get_cbor_string + std::size_t open = 0; + + while (true) + { + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "binary"))) + { + return false; + } + + if (current == 0x5F) // Binary data (indefinite length) + { + ++open; + get(); + continue; + } + + // a break marker closes the innermost indefinite-length byte + // array; outside of one it falls through to the error below + if (open != 0 && current == 0xFF) + { + if (--open == 0) + { + return true; + } + get(); + continue; + } + + if (JSON_HEDLEY_UNLIKELY(!get_cbor_binary_chunk(result))) + { + return false; + } + + if (open == 0) + { + return true; + } + + get(); + } + } + /*! @brief narrow a definite CBOR array/map length to std::size_t @@ -11949,96 +13836,110 @@ class binary_reader } /*! - @param[in] len the length of the array or detail::unknown_size() for an - array of indefinite size + @brief read a CBOR value and everything nested inside it + + Reads values until the one that was begun here is complete, resuming the + enclosing container after each element, so that the nesting depth of the + input costs heap rather than native stack (see #5104). + + @param[in] get_char whether a new character should be retrieved from the + input (true) or whether the last read character + @a current should be considered instead @param[in] tag_handler how CBOR tags should be treated - @return whether array creation completed + + @return whether reading the value succeeded */ - bool get_cbor_array(const std::size_t len, - const cbor_tag_handler_t tag_handler) + bool parse_cbor_internal(const bool get_char, + const cbor_tag_handler_t tag_handler) { - if (JSON_HEDLEY_UNLIKELY(!sax->start_array(len))) - { - return false; - } + // whether the next value starts at a fresh byte or at the one already + // read into `current` + bool fetch = get_char; - if (len != detail::unknown_size()) + // the key currently being read; hoisted out of the loop so that its + // capacity is reused across elements and across nesting levels + string_t key; + + while (true) { - for (std::size_t i = 0; i < len; ++i) + if (!container_stack.empty()) { - if (JSON_HEDLEY_UNLIKELY(!parse_cbor_internal(true, tag_handler))) + // a copy, not a reference: it must stay valid across the + // pop_back() below, which destroys the container_stack element + // it would otherwise alias + const container_frame top = container_stack.back(); + bool at_end = false; + + if (top.remaining != npos) { - return false; + // definite length: the container ends once its elements + // have been read + at_end = (top.remaining == 0); + if (!at_end) + { + // claim the element about to be read + --container_stack.back().remaining; + if (top.is_object) + { + get(); + } + } + fetch = true; } - } - } - else - { - while (get() != 0xFF) - { - if (JSON_HEDLEY_UNLIKELY(!parse_cbor_internal(false, tag_handler))) + else { - return false; + // indefinite length: the container ends at a break marker. + // Testing for it consumes a byte, which is the first byte + // of the next element when it is not one. + at_end = (get() == 0xFF); + fetch = top.is_object; } - } - } - return sax->end_array(); - } - - /*! - @param[in] len the length of the object or detail::unknown_size() for an - object of indefinite size - @param[in] tag_handler how CBOR tags should be treated - @return whether object creation completed - */ - bool get_cbor_object(const std::size_t len, - const cbor_tag_handler_t tag_handler) - { - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(len))) - { - return false; - } - - if (len != 0) - { - string_t key; - if (len != detail::unknown_size()) - { - for (std::size_t i = 0; i < len; ++i) + if (at_end) { - get(); + container_stack.pop_back(); + if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + { + return false; + } + // the value begun here is complete once its container is + if (container_stack.empty()) + { + return true; + } + continue; + } + + if (top.is_object) + { + key.clear(); if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key))) { return false; } - - if (JSON_HEDLEY_UNLIKELY(!parse_cbor_internal(true, tag_handler))) - { - return false; - } - key.clear(); + fetch = true; } } - else - { - while (get() != 0xFF) - { - if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key))) - { - return false; - } - if (JSON_HEDLEY_UNLIKELY(!parse_cbor_internal(true, tag_handler))) - { - return false; - } - key.clear(); + // a tag is not a value of its own: read on until the tagged value + bool tag_pending = false; + do + { + if (JSON_HEDLEY_UNLIKELY(!parse_cbor_value(fetch, tag_handler, tag_pending))) + { + return false; } + fetch = true; + } + while (tag_pending); + + // a value that opened a container left it on the stack; one that + // did not, and that was not inside a container, was the whole value + if (container_stack.empty()) + { + return true; } } - - return sax->end_object(); } ///////////// @@ -12048,7 +13949,17 @@ class binary_reader /*! @return whether a valid MessagePack value was passed to the SAX parser */ - bool parse_msgpack_internal() + /*! + @brief read one MessagePack value + + Reads a single value and passes it to the SAX parser. A value that begins + a container is not read to its end: the container is opened with + @ref enter_container and its elements are read by + @ref parse_msgpack_internal, so that nesting does not consume native stack. + + @return whether reading the value succeeded + */ + bool parse_msgpack_value() { switch (get()) { @@ -12204,7 +14115,7 @@ class binary_reader case 0x8D: case 0x8E: case 0x8F: - return get_msgpack_object(conditional_static_cast(static_cast(current) & 0x0Fu)); + return enter_object(conditional_static_cast(static_cast(current) & 0x0Fu)); // fixarray case 0x90: @@ -12223,7 +14134,7 @@ class binary_reader case 0x9D: case 0x9E: case 0x9F: - return get_msgpack_array(conditional_static_cast(static_cast(current) & 0x0Fu)); + return enter_array(conditional_static_cast(static_cast(current) & 0x0Fu)); // fixstr case 0xA0: @@ -12354,25 +14265,25 @@ class binary_reader case 0xDC: // array 16 { std::uint16_t len{}; - return get_number(input_format_t::msgpack, len) && get_msgpack_array(static_cast(len)); + return get_number(input_format_t::msgpack, len) && enter_array(static_cast(len)); } case 0xDD: // array 32 { std::uint32_t len{}; - return get_number(input_format_t::msgpack, len) && get_msgpack_array(conditional_static_cast(len)); + return get_number(input_format_t::msgpack, len) && enter_array(conditional_static_cast(len)); } case 0xDE: // map 16 { std::uint16_t len{}; - return get_number(input_format_t::msgpack, len) && get_msgpack_object(static_cast(len)); + return get_number(input_format_t::msgpack, len) && enter_object(static_cast(len)); } case 0xDF: // map 32 { std::uint32_t len{}; - return get_number(input_format_t::msgpack, len) && get_msgpack_object(conditional_static_cast(len)); + return get_number(input_format_t::msgpack, len) && enter_object(conditional_static_cast(len)); } // negative fixint @@ -12620,55 +14531,69 @@ class binary_reader } /*! - @param[in] len the length of the array - @return whether array creation completed + @brief read a MessagePack value and everything nested inside it + + Reads values until the one that was begun here is complete, resuming the + enclosing container each time an element ends, so that the nesting depth + of the input costs heap rather than native stack (see #5104). + + @return whether reading the value succeeded */ - bool get_msgpack_array(const std::size_t len) + bool parse_msgpack_internal() { - if (JSON_HEDLEY_UNLIKELY(!sax->start_array(len))) - { - return false; - } - - for (std::size_t i = 0; i < len; ++i) - { - if (JSON_HEDLEY_UNLIKELY(!parse_msgpack_internal())) - { - return false; - } - } - - return sax->end_array(); - } - - /*! - @param[in] len the length of the object - @return whether object creation completed - */ - bool get_msgpack_object(const std::size_t len) - { - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(len))) - { - return false; - } - + // the key currently being read; hoisted out of the loop so that its + // capacity is reused across elements and across nesting levels string_t key; - for (std::size_t i = 0; i < len; ++i) + + while (true) { - get(); - if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key))) + if (!container_stack.empty()) + { + // copied out before anything can push onto the stack and + // invalidate a reference into it + const bool is_object = container_stack.back().is_object; + + if (container_stack.back().remaining == 0) + { + container_stack.pop_back(); + if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->end_object() : !sax->end_array())) + { + return false; + } + // the value begun here is complete once its container is + if (container_stack.empty()) + { + return true; + } + continue; + } + + // claim the element about to be read + --container_stack.back().remaining; + + if (is_object) + { + get(); + key.clear(); + if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key))) + { + return false; + } + } + } + + if (JSON_HEDLEY_UNLIKELY(!parse_msgpack_value())) { return false; } - if (JSON_HEDLEY_UNLIKELY(!parse_msgpack_internal())) + // a value that opened a container left it on the stack; one that + // did not, and that was not inside a container, was the whole value + if (container_stack.empty()) { - return false; + return true; } - key.clear(); } - - return sax->end_object(); } //////////// @@ -12684,7 +14609,103 @@ class binary_reader */ bool parse_ubjson_internal(const bool get_char = true) { - return get_ubjson_value(get_char ? get_ignore_noop() : current); + // the key currently being read; hoisted out of the loop so that its + // capacity is reused across elements and across nesting levels + string_t key; + + // the type marker of the value to read next + char_int_type prefix = get_char ? get_ignore_noop() : current; + + while (true) + { + const std::size_t depth = container_stack.size(); + + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_value(prefix))) + { + return false; + } + + // the value begun here is complete once it is not inside anything + if (container_stack.empty()) + { + return true; + } + + // a value was completed rather than a container opened; a + // container that ends at a marker needs the next byte to test + if (container_stack.size() == depth && container_stack.back().remaining == npos) + { + get_ignore_noop(); + } + + // advance to the next element, closing the containers that ended. + // top is a copy, not a reference: it must stay valid across the + // pop_back() below, which destroys the container_stack element it + // would otherwise alias. + for (;;) + { + const container_frame top = container_stack.back(); + + if (top.remaining != npos) + { + if (top.remaining != 0) + { + --container_stack.back().remaining; + if (top.is_object) + { + key.clear(); + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) + { + return false; + } + } + // an optimized container gives its elements no marker + prefix = (top.type_marker != 0) ? top.type_marker : get_ignore_noop(); + break; + } + } + // the end marker is compared against a literal rather than + // against a conditional expression, because char_int_type is + // unsigned for some input adapters and MSVC then reports the + // comparison as a signed/unsigned mismatch + else if (top.is_object ? (current != '}') : (current != ']')) + { + // a container that ends at a marker is never optimized, so + // every element carries its own marker; for an object the + // byte tested above is the first byte of the key + if (top.is_object) + { + key.clear(); + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key))) + { + return false; + } + prefix = get_ignore_noop(); + } + else + { + prefix = current; + } + break; + } + + container_stack.pop_back(); + if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + { + return false; + } + if (container_stack.empty()) + { + return true; + } + // the container that just ended was an element of the one + // below it, which may need the next byte for its own test + if (container_stack.back().remaining == npos) + { + get_ignore_noop(); + } + } + } } /*! @@ -13123,7 +15144,12 @@ class binary_reader { result.first = npos; // size result.second = 0; // type - bool is_ndarray = false; + // seed the flag with the caller's context: inside an ndarray dimension + // vector another ndarray is not allowed, and get_ubjson_size_value() + // rejects it up front instead of reading it and reporting afterwards. + // Seeding it with `false` made every '#' of a "[#[#[..." chain descend + // another level, which overflowed the stack (see #5104). + bool is_ndarray = inside_ndarray; get_ignore_noop(); @@ -13156,13 +15182,11 @@ class binary_reader } const bool is_error = get_ubjson_size_value(result.first, is_ndarray); - if (input_format == input_format_t::bjdata && is_ndarray) + // an ndarray was read here only if the flag flipped; when it was + // seeded true, get_ubjson_size_value() already rejected the nested + // dimension vector + if (input_format == input_format_t::bjdata && is_ndarray && !inside_ndarray) { - if (inside_ndarray) - { - return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read, - exception_message(input_format, "ndarray can not be recursive", "size"), nullptr)); - } result.second |= (1 << 8); // use bit 8 to indicate ndarray, all UBJSON and BJData markers should be ASCII letters } return is_error; @@ -13171,7 +15195,7 @@ class binary_reader if (current == '#') { const bool is_error = get_ubjson_size_value(result.first, is_ndarray); - if (input_format == input_format_t::bjdata && is_ndarray) + if (input_format == input_format_t::bjdata && is_ndarray && !inside_ndarray) { return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read, exception_message(input_format, "ndarray requires both type and size", "size"), nullptr)); @@ -13442,53 +15466,33 @@ class binary_reader if (size_and_type.first != npos) { - if (JSON_HEDLEY_UNLIKELY(!sax->start_array(size_and_type.first))) + // reading an element of a valueless type consumes no input, so the + // declared count alone decides how much is allocated; the check is + // made before the start event so that no container is opened that + // is then abandoned. See @ref max_valueless_container_size. + if (JSON_HEDLEY_UNLIKELY((size_and_type.second == 'Z' || size_and_type.second == 'T' || size_and_type.second == 'F') + && size_and_type.first > max_valueless_container_size)) + { + return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, + exception_message(input_format, "excessive array size", "size"), nullptr)); + } + + if (JSON_HEDLEY_UNLIKELY(!enter_array(size_and_type.first, size_and_type.second))) { return false; } - if (size_and_type.second != 0) + if (size_and_type.second == 'N') { - if (size_and_type.second != 'N') - { - for (std::size_t i = 0; i < size_and_type.first; ++i) - { - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_value(size_and_type.second))) - { - return false; - } - } - } - } - else - { - for (std::size_t i = 0; i < size_and_type.first; ++i) - { - if (JSON_HEDLEY_UNLIKELY(!parse_ubjson_internal())) - { - return false; - } - } - } - } - else - { - if (JSON_HEDLEY_UNLIKELY(!sax->start_array(detail::unknown_size()))) - { - return false; + // a no-op is not a value, so a container of them holds none; + // the declared size has already been passed to the SAX parser + container_stack.back().remaining = 0; } - while (current != ']') - { - if (JSON_HEDLEY_UNLIKELY(!parse_ubjson_internal(false))) - { - return false; - } - get_ignore_noop(); - } + return true; } - return sax->end_array(); + return enter_array(detail::unknown_size()); } /*! @@ -13510,68 +15514,12 @@ class binary_reader exception_message(input_format, "BJData object does not support ND-array size in optimized format", "object"), nullptr)); } - string_t key; if (size_and_type.first != npos) { - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(size_and_type.first))) - { - return false; - } - - if (size_and_type.second != 0) - { - for (std::size_t i = 0; i < size_and_type.first; ++i) - { - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) - { - return false; - } - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_value(size_and_type.second))) - { - return false; - } - key.clear(); - } - } - else - { - for (std::size_t i = 0; i < size_and_type.first; ++i) - { - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) - { - return false; - } - if (JSON_HEDLEY_UNLIKELY(!parse_ubjson_internal())) - { - return false; - } - key.clear(); - } - } - } - else - { - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(detail::unknown_size()))) - { - return false; - } - - while (current != '}') - { - if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key))) - { - return false; - } - if (JSON_HEDLEY_UNLIKELY(!parse_ubjson_internal())) - { - return false; - } - get_ignore_noop(); - key.clear(); - } + return enter_object(size_and_type.first, size_and_type.second); } - return sax->end_object(); + return enter_object(detail::unknown_size()); } // Note, no reader for UBJSON binary types is implemented because they do @@ -13631,7 +15579,10 @@ class binary_reader number_string, out_of_range::create(406, concat("number overflow parsing '", number_string, '\''), nullptr)); } - return sax->number_float(parsed_float, std::move(number_string)); + // number_string is a std::string, while the SAX interface takes a + // string_t; convert explicitly, as the two are only implicitly + // convertible for some string types + return sax->number_float(parsed_float, string_t(number_string.data(), number_string.size())); } case token_type::uninitialized: case token_type::literal_true: @@ -13959,6 +15910,9 @@ class binary_reader /// the SAX parser json_sax_t* sax = nullptr; + /// the containers that have been opened and not closed yet; see @ref container_frame + std::vector container_stack{}; + // excluded markers in bjdata optimized type #define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \ make_array('F', 'H', 'N', 'S', 'T', 'Z', '[', '{') @@ -14088,9 +16042,10 @@ class parser parser_callback_t cb = nullptr, const bool allow_exceptions_ = true, const bool ignore_comments = false, - const bool ignore_trailing_commas_ = false) + const bool ignore_trailing_commas_ = false, + const bool discard_number_values_ = false) : callback(std::move(cb)) - , m_lexer(std::move(adapter), ignore_comments) + , m_lexer(std::move(adapter), ignore_comments, discard_number_values_) , allow_exceptions(allow_exceptions_) , ignore_trailing_commas(ignore_trailing_commas_) { @@ -14850,8 +16805,13 @@ class iter_impl // NOLINT(cppcoreguidelines-special-member-functions,hicpp-speci iter_impl() = default; ~iter_impl() = default; - iter_impl(iter_impl&&) noexcept = default; - iter_impl& operator=(iter_impl&&) noexcept = default; + // the exception specification is left to be computed rather than declared: + // an array or object type whose iterator is not nothrow move constructible + // (std::deque's is not before libstdc++ 11) would make a declared noexcept + // differ from the implicit one, which deletes the function -- and is an + // error outright with older compilers + iter_impl(iter_impl&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor,cppcoreguidelines-noexcept-move-operations) + iter_impl& operator=(iter_impl&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor,cppcoreguidelines-noexcept-move-operations) /*! @brief constructor for a given JSON instance @@ -15732,6 +17692,7 @@ NLOHMANN_JSON_NAMESPACE_END #endif // JSON_NO_IO #include // max #include // accumulate +#include // set #include // string #include // move #include // vector @@ -15791,7 +17752,7 @@ class json_pointer string_t{}, [](const string_t& a, const string_t& b) { - return detail::concat(a, '/', detail::escape(b)); + return detail::concat(a, '/', detail::escape(b)); }); } @@ -15985,7 +17946,7 @@ class json_pointer JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); } - const char* p = s.c_str(); + const char* p = s.data(); char* p_end = nullptr; // NOLINT(misc-const-correctness) errno = 0; // strtoull doesn't reset errno const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) @@ -16020,19 +17981,35 @@ class json_pointer } private: + /*! + @brief the reference token sequences that denote arrays + + @ref unflatten collects the pointer prefixes that have a reference token 0 + among their children; @ref get_and_create creates arrays exactly below + those prefixes and objects everywhere else. Deciding this up front keeps + the result independent of the order in which the flattened object is + iterated, which is unspecified for some object types. + */ + using array_parents_t = std::set>; + /*! @brief create and return a reference to the pointed to value @complexity Linear in the number of reference tokens. + @throw parse_error.106 if an array index begins with '0' @throw parse_error.109 if array index is not a number @throw type_error.313 if value cannot be unflattened */ template - BasicJsonType& get_and_create(BasicJsonType& j) const + BasicJsonType& get_and_create(BasicJsonType& j, const array_parents_t& array_parents) const { auto* result = &j; + // the reference tokens that have been consumed so far; used to look up + // whether the value to be created below is an array or an object + std::vector prefix; + // in case no reference tokens exist, return a reference to the JSON value // j which will be overwritten by a primitive value for (const auto& reference_token : reference_tokens) @@ -16041,10 +18018,11 @@ class json_pointer { case detail::value_t::null: { - if (reference_token == "0") + if (array_parents.find(prefix) != array_parents.end()) { - // start a new array if the reference token is 0 - result = &result->operator[](0); + // some reference token below this position is 0, so the + // value is an array + result = &result->operator[](array_index(reference_token)); } else { @@ -16084,6 +18062,8 @@ class json_pointer default: JSON_THROW(detail::type_error::create(313, "invalid value to unflatten", &j)); } + + prefix.push_back(reference_token); } return *result; @@ -16468,6 +18448,20 @@ class json_pointer } } + // the reference token consists only of digits at this point (cf. checks + // above); however, its numeric value might not be representable, in which + // case array_index() would throw out_of_range.404/410 -- contains() must + // not throw (see #5395), so such a reference token is treated as "not found" + errno = 0; // strtoull() does not reset errno on success + char* p_end = nullptr; // NOLINT(misc-const-correctness) + const unsigned long long magnitude = std::strtoull(reference_token.c_str(), &p_end, 10); // NOLINT(runtime/int) + if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX + || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) + { + // the array index cannot be represented as size_type + return false; + } + const auto idx = array_index(reference_token); if (idx >= ptr->size()) { @@ -16543,7 +18537,8 @@ class json_pointer { // use the text between the beginning of the reference token // (start) and the last slash (slash). - auto reference_token = reference_string.substr(start, slash - start); + const auto count = (slash == string_t::npos ? reference_string.size() : slash) - start; + auto reference_token = string_t(reference_string.data() + start, count); // check reference tokens are properly escaped for (std::size_t pos = reference_token.find_first_of('~'); @@ -16659,6 +18654,24 @@ class json_pointer BasicJsonType result; + // collect the pointer prefixes that have a reference token 0 among + // their children; the values below them are arrays, all others are + // objects (see array_parents_t) + array_parents_t array_parents; + for (const auto& element : *value.m_data.m_value.object) + { + json_pointer ptr(element.first); + std::vector prefix; + for (auto& reference_token : ptr.reference_tokens) + { + if (reference_token == "0") + { + array_parents.insert(prefix); + } + prefix.push_back(std::move(reference_token)); + } + } + // iterate the JSON object values for (const auto& element : *value.m_data.m_value.object) { @@ -16671,7 +18684,7 @@ class json_pointer // that if the JSON pointer is "" (i.e., points to the whole value), // function get_and_create returns a reference to the result itself. // An assignment will then create a primitive value. - json_pointer(element.first).get_and_create(result) = element.second; + json_pointer(element.first).get_and_create(result, array_parents) = element.second; } return result; @@ -16946,9 +18959,14 @@ NLOHMANN_JSON_NAMESPACE_END #include // memcpy #include // numeric_limits #include // string +#include // enable_if, is_constructible #include // move #include // vector +#ifdef _MSC_VER + #include // _byteswap_ushort, _byteswap_ulong, _byteswap_uint64 +#endif + // #include // #include @@ -16969,6 +18987,7 @@ NLOHMANN_JSON_NAMESPACE_END #include // back_inserter #include // shared_ptr, make_shared #include // basic_string +#include // move #include // vector #ifndef JSON_NO_IO @@ -17001,22 +19020,32 @@ template struct output_adapter_protocol template using output_adapter_t = std::shared_ptr>; -/// output adapter for byte vectors +/// @brief non-virtual output sink writing into a std::vector +/// +/// This sink is not part of the virtual output_adapter_protocol hierarchy: it is +/// passed to binary_writer by value as a template parameter, so +/// write_character()/write_characters() are ordinary (inlinable) calls with no +/// vtable lookup and no shared_ptr. It is used for the common +/// `to_cbor`/`to_msgpack`/... into a std::vector. output_vector_adapter below +/// wraps this same sink to provide the virtual interface. template> -class output_vector_adapter : public output_adapter_protocol +class output_vector_sink { public: - explicit output_vector_adapter(std::vector& vec) noexcept + explicit output_vector_sink(std::vector& vec) noexcept : v(vec) {} - void write_character(CharType c) override + void write_character(CharType c) { v.push_back(c); } - JSON_HEDLEY_NON_NULL(2) - void write_characters(const CharType* s, std::size_t length) override + // no JSON_HEDLEY_NON_NULL here: binary_writer legitimately passes a null + // pointer with length 0 for empty strings/binary values. Appending an empty + // range is a no-op; the type-erased path tolerates this via the (unattributed) + // virtual base, and the concrete sink must do the same. + void write_characters(const CharType* s, std::size_t length) { v.insert(v.end(), s, s + length); } @@ -17025,6 +19054,34 @@ class output_vector_adapter : public output_adapter_protocol std::vector& v; }; +/// output adapter for byte vectors +/// +/// The appending itself lives in output_vector_sink; this class only adds the +/// virtual output_adapter_protocol interface on top of it, so both the +/// type-erased and the templated path share one implementation. +template> +class output_vector_adapter : public output_adapter_protocol +{ + public: + explicit output_vector_adapter(std::vector& vec) noexcept + : sink(vec) + {} + + void write_character(CharType c) override + { + sink.write_character(c); + } + + JSON_HEDLEY_NON_NULL(2) + void write_characters(const CharType* s, std::size_t length) override + { + sink.write_characters(s, length); + } + + private: + output_vector_sink sink; +}; + #ifndef JSON_NO_IO /// output adapter for output streams template @@ -17075,6 +19132,39 @@ class output_string_adapter : public output_adapter_protocol StringType& str; }; +/// @brief output sink forwarding to a type-erased output adapter +/// +/// Wraps the polymorphic output_adapter_t so the same binary_writer template can +/// also target arbitrary adapters (output streams, strings, user-provided +/// adapters) via the `output_adapter`-based overloads. Each write still goes +/// through one virtual call, exactly as before; only the concrete sinks above +/// avoid it. +template +class output_adapter_sink +{ + public: + explicit output_adapter_sink(output_adapter_t adapter) + : oa(std::move(adapter)) + { + JSON_ASSERT(oa); + } + + void write_character(CharType c) + { + oa->write_character(c); + } + + // no JSON_HEDLEY_NON_NULL: forwards (null, 0) for empty payloads, exactly as + // the type-erased path already did before this sink existed + void write_characters(const CharType* s, std::size_t length) + { + oa->write_characters(s, length); + } + + private: + output_adapter_t oa; +}; + template> class output_adapter { @@ -17121,10 +19211,41 @@ enum class bjdata_version_t // binary writer // /////////////////// +/*! +@brief capacity hint for binary serialization into a std::vector + +Returns a *lower* bound on the number of bytes the serialization will produce, +so that writing an array/object of many elements does not start reallocating +from an empty buffer. Every array element occupies at least one byte in every +supported binary format, and every object entry at least two (a key of at least +one byte plus a value of at least one), plus one byte for the container header, +so the hint can never exceed the final size and the returned vector is never +left holding capacity the caller did not ask for. The buffer still grows +geometrically past the hint, so under-reserving only costs a few later +reallocations. Only the top-level element count is consulted (O(1), no walk of +the DOM); a single scalar, string, or binary value is written in one shot and +needs no hint. +*/ +template +std::size_t binary_reserve_hint(const BasicJsonType& j) +{ + if (j.is_array()) + { + return j.size() + 1; + } + + if (j.is_object()) + { + return (j.size() * 2) + 1; + } + + return 0; +} + /*! @brief serialization to CBOR and MessagePack values */ -template +template> class binary_writer { using string_t = typename BasicJsonType::string_t; @@ -17135,12 +19256,28 @@ class binary_writer /*! @brief create a binary writer + @param[in] sink output sink to write to (a value-type sink such as + output_vector_sink, or output_adapter_sink wrapping a + type-erased output adapter) + */ + explicit binary_writer(OutputSinkType sink) : oa(std::move(sink)) + {} + + /*! + @brief create a binary writer from a type-erased output adapter + + Convenience constructor for the default (output_adapter_sink) sink so the + `output_adapter`-based overloads keep constructing the writer directly from + an adapter. Constrained to sinks that can actually be built from an adapter, + so that a writer over some other sink type is not advertised as constructible + from one. + @param[in] adapter output adapter to write to */ - explicit binary_writer(output_adapter_t adapter) : oa(std::move(adapter)) - { - JSON_ASSERT(oa); - } + template < typename SinkType = OutputSinkType, + typename std::enable_if < std::is_constructible>::value, int >::type = 0 > + explicit binary_writer(output_adapter_t adapter) : oa(SinkType(std::move(adapter))) + {} /*! @param[in] j JSON value to serialize @@ -17181,15 +19318,15 @@ class binary_writer { case value_t::null: { - oa->write_character(to_char_type(0xF6)); + oa.write_character(to_char_type(0xF6)); break; } case value_t::boolean: { - oa->write_character(j.m_data.m_value.boolean - ? to_char_type(0xF5) - : to_char_type(0xF4)); + oa.write_character(j.m_data.m_value.boolean + ? to_char_type(0xF5) + : to_char_type(0xF4)); break; } @@ -17206,22 +19343,22 @@ class binary_writer } else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x18)); + oa.write_character(to_char_type(0x18)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x19)); + oa.write_character(to_char_type(0x19)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x1A)); + oa.write_character(to_char_type(0x1A)); write_number(static_cast(j.m_data.m_value.number_integer)); } else { - oa->write_character(to_char_type(0x1B)); + oa.write_character(to_char_type(0x1B)); write_number(static_cast(j.m_data.m_value.number_integer)); } } @@ -17236,22 +19373,22 @@ class binary_writer } else if (positive_number <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x38)); + oa.write_character(to_char_type(0x38)); write_number(static_cast(positive_number)); } else if (positive_number <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x39)); + oa.write_character(to_char_type(0x39)); write_number(static_cast(positive_number)); } else if (positive_number <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x3A)); + oa.write_character(to_char_type(0x3A)); write_number(static_cast(positive_number)); } else { - oa->write_character(to_char_type(0x3B)); + oa.write_character(to_char_type(0x3B)); write_number(static_cast(positive_number)); } } @@ -17266,22 +19403,22 @@ class binary_writer } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x18)); + oa.write_character(to_char_type(0x18)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x19)); + oa.write_character(to_char_type(0x19)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x1A)); + oa.write_character(to_char_type(0x1A)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } else { - oa->write_character(to_char_type(0x1B)); + oa.write_character(to_char_type(0x1B)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } break; @@ -17292,16 +19429,16 @@ class binary_writer if (std::isnan(j.m_data.m_value.number_float)) { // NaN is 0xf97e00 in CBOR - oa->write_character(to_char_type(0xF9)); - oa->write_character(to_char_type(0x7E)); - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0xF9)); + oa.write_character(to_char_type(0x7E)); + oa.write_character(to_char_type(0x00)); } else if (std::isinf(j.m_data.m_value.number_float)) { // Infinity is 0xf97c00, -Infinity is 0xf9fc00 - oa->write_character(to_char_type(0xf9)); - oa->write_character(j.m_data.m_value.number_float > 0 ? to_char_type(0x7C) : to_char_type(0xFC)); - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0xf9)); + oa.write_character(j.m_data.m_value.number_float > 0 ? to_char_type(0x7C) : to_char_type(0xFC)); + oa.write_character(to_char_type(0x00)); } else { @@ -17320,31 +19457,31 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x78)); + oa.write_character(to_char_type(0x78)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x79)); + oa.write_character(to_char_type(0x79)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x7A)); + oa.write_character(to_char_type(0x7A)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x7B)); + oa.write_character(to_char_type(0x7B)); write_number(static_cast(N)); } // LCOV_EXCL_STOP // step 2: write the string - oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->c_str()), - j.m_data.m_value.string->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.string->data()), + j.m_data.m_value.string->size()); break; } @@ -17358,23 +19495,23 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x98)); + oa.write_character(to_char_type(0x98)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x99)); + oa.write_character(to_char_type(0x99)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x9A)); + oa.write_character(to_char_type(0x9A)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x9B)); + oa.write_character(to_char_type(0x9B)); write_number(static_cast(N)); } // LCOV_EXCL_STOP @@ -17421,31 +19558,31 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x58)); + oa.write_character(to_char_type(0x58)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x59)); + oa.write_character(to_char_type(0x59)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x5A)); + oa.write_character(to_char_type(0x5A)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x5B)); + oa.write_character(to_char_type(0x5B)); write_number(static_cast(N)); } // LCOV_EXCL_STOP // step 2: write each element - oa->write_characters( - reinterpret_cast(j.m_data.m_value.binary->data()), - N); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.binary->data()), + N); break; } @@ -17460,23 +19597,23 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xB8)); + oa.write_character(to_char_type(0xB8)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xB9)); + oa.write_character(to_char_type(0xB9)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xBA)); + oa.write_character(to_char_type(0xBA)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xBB)); + oa.write_character(to_char_type(0xBB)); write_number(static_cast(N)); } // LCOV_EXCL_STOP @@ -17505,15 +19642,15 @@ class binary_writer { case value_t::null: // nil { - oa->write_character(to_char_type(0xC0)); + oa.write_character(to_char_type(0xC0)); break; } case value_t::boolean: // true and false { - oa->write_character(j.m_data.m_value.boolean - ? to_char_type(0xC3) - : to_char_type(0xC2)); + oa.write_character(j.m_data.m_value.boolean + ? to_char_type(0xC3) + : to_char_type(0xC2)); break; } @@ -17532,25 +19669,25 @@ class binary_writer else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 8 - oa->write_character(to_char_type(0xCC)); + oa.write_character(to_char_type(0xCC)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 16 - oa->write_character(to_char_type(0xCD)); + oa.write_character(to_char_type(0xCD)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 32 - oa->write_character(to_char_type(0xCE)); + oa.write_character(to_char_type(0xCE)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 64 - oa->write_character(to_char_type(0xCF)); + oa.write_character(to_char_type(0xCF)); write_number(static_cast(j.m_data.m_value.number_integer)); } } @@ -17565,28 +19702,28 @@ class binary_writer j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 8 - oa->write_character(to_char_type(0xD0)); + oa.write_character(to_char_type(0xD0)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 16 - oa->write_character(to_char_type(0xD1)); + oa.write_character(to_char_type(0xD1)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 32 - oa->write_character(to_char_type(0xD2)); + oa.write_character(to_char_type(0xD2)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 64 - oa->write_character(to_char_type(0xD3)); + oa.write_character(to_char_type(0xD3)); write_number(static_cast(j.m_data.m_value.number_integer)); } } @@ -17603,25 +19740,25 @@ class binary_writer else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 8 - oa->write_character(to_char_type(0xCC)); + oa.write_character(to_char_type(0xCC)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 16 - oa->write_character(to_char_type(0xCD)); + oa.write_character(to_char_type(0xCD)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 32 - oa->write_character(to_char_type(0xCE)); + oa.write_character(to_char_type(0xCE)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 64 - oa->write_character(to_char_type(0xCF)); + oa.write_character(to_char_type(0xCF)); write_number(static_cast(j.m_data.m_value.number_integer)); } break; @@ -17645,26 +19782,26 @@ class binary_writer else if (N <= (std::numeric_limits::max)()) { // str 8 - oa->write_character(to_char_type(0xD9)); + oa.write_character(to_char_type(0xD9)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // str 16 - oa->write_character(to_char_type(0xDA)); + oa.write_character(to_char_type(0xDA)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // str 32 - oa->write_character(to_char_type(0xDB)); + oa.write_character(to_char_type(0xDB)); write_number(static_cast(N)); } // step 2: write the string - oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->c_str()), - j.m_data.m_value.string->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.string->data()), + j.m_data.m_value.string->size()); break; } @@ -17680,13 +19817,13 @@ class binary_writer else if (N <= (std::numeric_limits::max)()) { // array 16 - oa->write_character(to_char_type(0xDC)); + oa.write_character(to_char_type(0xDC)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // array 32 - oa->write_character(to_char_type(0xDD)); + oa.write_character(to_char_type(0xDD)); write_number(static_cast(N)); } @@ -17742,7 +19879,7 @@ class binary_writer fixed = false; } - oa->write_character(to_char_type(output_type)); + oa.write_character(to_char_type(output_type)); if (!fixed) { write_number(static_cast(N)); @@ -17754,7 +19891,7 @@ class binary_writer ? 0xC8 // ext 16 : 0xC5; // bin 16 - oa->write_character(to_char_type(output_type)); + oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) @@ -17763,20 +19900,25 @@ class binary_writer ? 0xC9 // ext 32 : 0xC6; // bin 32 - oa->write_character(to_char_type(output_type)); + oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } // step 1.5: if this is an ext type, write the subtype if (use_ext) { + if (JSON_HEDLEY_UNLIKELY(j.m_data.m_value.binary->subtype() > (std::numeric_limits::max)())) + { + JSON_THROW(out_of_range::create(415, concat("subtype ", std::to_string(j.m_data.m_value.binary->subtype()), " is too large for the MessagePack ext type (max 255)"), &j)); + } + write_number(static_cast(j.m_data.m_value.binary->subtype())); } // step 2: write the byte string - oa->write_characters( - reinterpret_cast(j.m_data.m_value.binary->data()), - N); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.binary->data()), + N); break; } @@ -17793,13 +19935,13 @@ class binary_writer else if (N <= (std::numeric_limits::max)()) { // map 16 - oa->write_character(to_char_type(0xDE)); + oa.write_character(to_char_type(0xDE)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // map 32 - oa->write_character(to_char_type(0xDF)); + oa.write_character(to_char_type(0xDF)); write_number(static_cast(N)); } @@ -17838,7 +19980,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('Z')); + oa.write_character(to_char_type('Z')); } break; } @@ -17847,9 +19989,9 @@ class binary_writer { if (add_prefix) { - oa->write_character(j.m_data.m_value.boolean - ? to_char_type('T') - : to_char_type('F')); + oa.write_character(j.m_data.m_value.boolean + ? to_char_type('T') + : to_char_type('F')); } break; } @@ -17876,12 +20018,12 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('S')); + oa.write_character(to_char_type('S')); } write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata); - oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->c_str()), - j.m_data.m_value.string->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.string->data()), + j.m_data.m_value.string->size()); break; } @@ -17889,13 +20031,16 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('[')); + oa.write_character(to_char_type('[')); } bool prefix_required = true; if (use_type && !j.m_data.m_value.array->empty()) { - JSON_ASSERT(use_count); + if (!use_count) + { + JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); + } const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata); const bool same_prefix = std::all_of(j.begin() + 1, j.end(), [this, first_prefix, use_bjdata](const BasicJsonType & v) @@ -17903,19 +20048,27 @@ class binary_writer return ubjson_prefix(v, use_bjdata) == first_prefix; }); - std::vector bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type + // an optimized array of a valueless type carries no payload, so a + // reader has nothing but the declared count to bound the allocation + // by and refuses an excessive one. Write the unoptimized form for + // those, at one byte per element, so the result can be read back. + // Objects are not affected: every element is preceded by its key. + const bool valueless_type = (first_prefix == 'Z' || first_prefix == 'T' || first_prefix == 'F'); + const bool excessive_valueless = valueless_type + && j.m_data.m_value.array->size() > detail::max_valueless_container_size; - if (same_prefix && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) + if (same_prefix && !excessive_valueless + && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) { prefix_required = false; - oa->write_character(to_char_type('$')); - oa->write_character(first_prefix); + oa.write_character(to_char_type('$')); + oa.write_character(first_prefix); } } if (use_count) { - oa->write_character(to_char_type('#')); + oa.write_character(to_char_type('#')); write_number_with_ubjson_prefix(j.m_data.m_value.array->size(), true, use_bjdata); } @@ -17926,7 +20079,7 @@ class binary_writer if (!use_count) { - oa->write_character(to_char_type(']')); + oa.write_character(to_char_type(']')); } break; @@ -17936,40 +20089,45 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('[')); + oa.write_character(to_char_type('[')); } if (use_type && (bjdata_draft3 || !j.m_data.m_value.binary->empty())) { - JSON_ASSERT(use_count); - oa->write_character(to_char_type('$')); - oa->write_character(bjdata_draft3 ? 'B' : 'U'); + if (!use_count) + { + JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); + } + oa.write_character(to_char_type('$')); + oa.write_character(bjdata_draft3 ? 'B' : 'U'); } if (use_count) { - oa->write_character(to_char_type('#')); + oa.write_character(to_char_type('#')); write_number_with_ubjson_prefix(j.m_data.m_value.binary->size(), true, use_bjdata); } if (use_type) { - oa->write_characters( - reinterpret_cast(j.m_data.m_value.binary->data()), - j.m_data.m_value.binary->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.binary->data()), + j.m_data.m_value.binary->size()); } else { for (size_t i = 0; i < j.m_data.m_value.binary->size(); ++i) { - oa->write_character(to_char_type(bjdata_draft3 ? 'B' : 'U')); - oa->write_character(to_char_type(j.m_data.m_value.binary->data()[i])); + oa.write_character(to_char_type(bjdata_draft3 ? 'B' : 'U')); + // the cast is needed for binary types whose value type + // is not an integer (e.g., std::byte) + oa.write_character(to_char_type(static_cast(j.m_data.m_value.binary->data()[i]))); } } if (!use_count) { - oa->write_character(to_char_type(']')); + oa.write_character(to_char_type(']')); } break; @@ -17987,13 +20145,16 @@ class binary_writer if (add_prefix) { - oa->write_character(to_char_type('{')); + oa.write_character(to_char_type('{')); } bool prefix_required = true; if (use_type && !j.m_data.m_value.object->empty()) { - JSON_ASSERT(use_count); + if (!use_count) + { + JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); + } const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata); const bool same_prefix = std::all_of(j.begin(), j.end(), [this, first_prefix, use_bjdata](const BasicJsonType & v) @@ -18001,34 +20162,32 @@ class binary_writer return ubjson_prefix(v, use_bjdata) == first_prefix; }); - std::vector bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type - - if (same_prefix && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) + if (same_prefix && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) { prefix_required = false; - oa->write_character(to_char_type('$')); - oa->write_character(first_prefix); + oa.write_character(to_char_type('$')); + oa.write_character(first_prefix); } } if (use_count) { - oa->write_character(to_char_type('#')); + oa.write_character(to_char_type('#')); write_number_with_ubjson_prefix(j.m_data.m_value.object->size(), true, use_bjdata); } for (const auto& el : *j.m_data.m_value.object) { write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); - oa->write_characters( - reinterpret_cast(el.first.c_str()), - el.first.size()); + oa.write_characters( + reinterpret_cast(el.first.data()), + el.first.size()); write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); } if (!use_count) { - oa->write_character(to_char_type('}')); + oa.write_character(to_char_type('}')); } break; @@ -18082,10 +20241,13 @@ class binary_writer void write_bson_entry_header(const string_t& name, const std::uint8_t element_type) { - oa->write_character(to_char_type(element_type)); - oa->write_characters( - reinterpret_cast(name.c_str()), - name.size() + 1u); + oa.write_character(to_char_type(element_type)); + oa.write_characters( + reinterpret_cast(name.data()), + name.size()); + // the terminating null byte is written explicitly rather than taken + // from the buffer, so that string_t::data() need not be null-terminated + oa.write_character(to_char_type(0x00)); } /*! @@ -18095,7 +20257,7 @@ class binary_writer const bool value) { write_bson_entry_header(name, 0x08); - oa->write_character(value ? to_char_type(0x01) : to_char_type(0x00)); + oa.write_character(value ? to_char_type(0x01) : to_char_type(0x00)); } /*! @@ -18125,9 +20287,12 @@ class binary_writer write_bson_entry_header(name, 0x02); write_number(to_bson_length(value.size() + 1ul), true); - oa->write_characters( - reinterpret_cast(value.c_str()), - value.size() + 1); + oa.write_characters( + reinterpret_cast(value.data()), + value.size()); + // the terminating null byte is written explicitly rather than taken + // from the buffer, so that string_t::data() need not be null-terminated + oa.write_character(to_char_type(0x00)); } /*! @@ -18218,7 +20383,11 @@ class binary_writer const std::size_t embedded_document_size = std::accumulate(std::begin(value), std::end(value), static_cast(0), [&array_index](std::size_t result, const typename BasicJsonType::array_t::value_type & el) { - return result + calc_bson_element_size(std::to_string(array_index++), el); + // the index is built as a std::string, while calc_bson_element_size + // takes a string_t; convert explicitly, as the two are only + // implicitly convertible for some string types + const auto key = std::to_string(array_index++); + return result + calc_bson_element_size(string_t(key.data(), key.size()), el); }); return sizeof(std::int32_t) + embedded_document_size + 1ul; @@ -18245,10 +20414,14 @@ class binary_writer for (const auto& el : value) { - write_bson_element(std::to_string(array_index++), el); + // the index is built as a std::string, while write_bson_element takes + // a string_t; convert explicitly, as the two are only implicitly + // convertible for some string types + const auto key = std::to_string(array_index++); + write_bson_element(string_t(key.data(), key.size()), el); } - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0x00)); } /*! @@ -18260,9 +20433,15 @@ class binary_writer write_bson_entry_header(name, 0x05); write_number(to_bson_length(value.size()), true); + + if (value.has_subtype() && JSON_HEDLEY_UNLIKELY(value.subtype() > (std::numeric_limits::max)())) + { + JSON_THROW(out_of_range::create(415, concat("subtype ", std::to_string(value.subtype()), " is too large for the BSON binary subtype (max 255)"), nullptr)); + } + write_number(value.has_subtype() ? static_cast(value.subtype()) : static_cast(0x00)); - oa->write_characters(reinterpret_cast(value.data()), value.size()); + oa.write_characters(reinterpret_cast(value.data()), value.size()); } /*! @@ -18388,7 +20567,7 @@ class binary_writer write_bson_element(el.first, el.second); } - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0x00)); } ////////// @@ -18432,7 +20611,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(get_ubjson_float_prefix(n)); + oa.write_character(get_ubjson_float_prefix(n)); } write_number(n, use_bjdata); } @@ -18448,7 +20627,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('i')); // int8 + oa.write_character(to_char_type('i')); // int8 } write_number(static_cast(n), use_bjdata); } @@ -18456,7 +20635,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('U')); // uint8 + oa.write_character(to_char_type('U')); // uint8 } write_number(static_cast(n), use_bjdata); } @@ -18464,7 +20643,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('I')); // int16 + oa.write_character(to_char_type('I')); // int16 } write_number(static_cast(n), use_bjdata); } @@ -18472,7 +20651,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('u')); // uint16 - bjdata only + oa.write_character(to_char_type('u')); // uint16 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -18480,7 +20659,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('l')); // int32 + oa.write_character(to_char_type('l')); // int32 } write_number(static_cast(n), use_bjdata); } @@ -18488,7 +20667,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('m')); // uint32 - bjdata only + oa.write_character(to_char_type('m')); // uint32 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -18496,7 +20675,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('L')); // int64 + oa.write_character(to_char_type('L')); // int64 } write_number(static_cast(n), use_bjdata); } @@ -18504,7 +20683,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('M')); // uint64 - bjdata only + oa.write_character(to_char_type('M')); // uint64 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -18512,14 +20691,14 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('H')); // high-precision number + oa.write_character(to_char_type('H')); // high-precision number } const auto number = BasicJsonType(n).dump(); write_number_with_ubjson_prefix(number.size(), true, use_bjdata); for (std::size_t i = 0; i < number.size(); ++i) { - oa->write_character(to_char_type(static_cast(number[i]))); + oa.write_character(to_char_type(static_cast(number[i]))); } } } @@ -18536,7 +20715,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('i')); // int8 + oa.write_character(to_char_type('i')); // int8 } write_number(static_cast(n), use_bjdata); } @@ -18544,7 +20723,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('U')); // uint8 + oa.write_character(to_char_type('U')); // uint8 } write_number(static_cast(n), use_bjdata); } @@ -18552,7 +20731,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('I')); // int16 + oa.write_character(to_char_type('I')); // int16 } write_number(static_cast(n), use_bjdata); } @@ -18560,7 +20739,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('u')); // uint16 - bjdata only + oa.write_character(to_char_type('u')); // uint16 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -18568,7 +20747,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('l')); // int32 + oa.write_character(to_char_type('l')); // int32 } write_number(static_cast(n), use_bjdata); } @@ -18576,7 +20755,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('m')); // uint32 - bjdata only + oa.write_character(to_char_type('m')); // uint32 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -18584,7 +20763,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('L')); // int64 + oa.write_character(to_char_type('L')); // int64 } write_number(static_cast(n), use_bjdata); } @@ -18593,14 +20772,14 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('H')); // high-precision number + oa.write_character(to_char_type('H')); // high-precision number } const auto number = BasicJsonType(n).dump(); write_number_with_ubjson_prefix(number.size(), true, use_bjdata); for (std::size_t i = 0; i < number.size(); ++i) { - oa->write_character(to_char_type(static_cast(number[i]))); + oa.write_character(to_char_type(static_cast(number[i]))); } } // LCOV_EXCL_STOP @@ -18710,6 +20889,21 @@ class binary_writer } } + /*! + @brief whether BJData forbids @a marker as the type of an optimized array + or object + + Containers, strings, high-precision numbers, booleans and null cannot be + declared as the single type of an optimized container in BJData; such a + container is written unoptimized. The reader rejects them with the same + list (binary_reader::bjd_optimized_type_markers). + */ + static constexpr bool is_bjdata_excluded_type_marker(const CharType marker) noexcept + { + return marker == '[' || marker == '{' || marker == 'S' || marker == 'H' + || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; + } + static constexpr CharType get_ubjson_float_prefix(float /*unused*/) { return 'd'; // float 32 @@ -18720,6 +20914,20 @@ class binary_writer return 'D'; // float 64 } + /*! + @brief checks whether a JSON number fits into @a TargetType + @param[in] el a JSON number of either the signed or unsigned integer kind + @return whether @a el's value can be represented by @a TargetType without + wrapping, regardless of which of the two kinds it is stored as + */ + template + static bool bjdata_ndarray_value_in_range(const BasicJsonType& el) + { + return el.is_number_unsigned() + ? value_in_range_of(el.template get()) + : value_in_range_of(el.template get()); + } + /*! @return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid */ @@ -18731,6 +20939,16 @@ class binary_writer }; string_t key = "_ArrayType_"; + // the type name is looked up as a string below; a non-string + // annotation (e.g. a number, null, or an array) cannot name a known + // dtype, so it is treated the same as an unrecognized type name and + // falls back to a plain object encoding instead of throwing + // type_error.302 out of get() + if (!value.at(key).is_string()) + { + return true; + } + // use get() instead of static_cast to avoid an // ambiguous conversion under explicit instantiation on C++17 (see #4825) auto it = bjdtype.find(value.at(key).template get()); @@ -18740,9 +20958,39 @@ class binary_writer } CharType dtype = it->second; + // the 'B' (byte) marker is only defined from BJData Draft 3 onward; + // emitting it under an earlier draft would produce a stream that an + // earlier-draft reader rejects, so such an object falls back to a + // plain object encoding instead (see the "Binary values" section of + // the BJData documentation) + if (dtype == 'B' && bjdata_version < bjdata_version_t::draft3) + { + return true; + } + key = "_ArraySize_"; - std::size_t len = (value.at(key).empty() ? 0 : 1); - for (const auto& el : value.at(key)) + // the dimensions are written verbatim as the header length below, so a + // value that is not an array cannot produce a valid one: null emits 'Z' + // and an object emits '{', neither of which a reader accepts after '#'. + // Such an object is not a valid ndarray and falls back to a plain object. + if (!value.at(key).is_array()) + { + return true; + } + + // the reader only restores an annotated object from an ND-array header + // with at least two dimensions: an empty dimension vector, a single + // dimension, or a 1xN row vector is read back as a plain array, which + // would silently drop the annotation, so such an object falls back to + // a plain object encoding instead + const auto& dims = value.at(key); + if (dims.size() < 2 || (dims.size() == 2 && dims.at(0).is_number_integer() && dims.at(0).template get() == 1)) + { + return true; + } + + std::size_t len = 1; + for (const auto& el : dims) { // a dimension is read as an unsigned value below, so anything that // is not a non-negative integer is rejected: a non-integer entry @@ -18764,15 +21012,26 @@ class binary_writer return true; } const auto dim_size = static_cast(dim); - if (dim_size != 0 && len > (std::numeric_limits::max)() / dim_size) + + // the reader turns an ND-array with any zero dimension into an + // empty plain array, dropping the annotation, so keep the object + if (dim_size == 0) + { + return true; + } + if (len > (std::numeric_limits::max)() / dim_size) { return true; } len *= dim_size; } + // the elements are written from _ArrayData_ as a flat list, so it has + // to be an array: size() is 0 for null and 1 for any other scalar, and + // iterating an object visits its values, so any of these could match + // the dimensions by accident and be encoded as an unrelated ND-array key = "_ArrayData_"; - if (value.at(key).size() != len) + if (!value.at(key).is_array() || value.at(key).size() != len) { return true; } @@ -18795,10 +21054,64 @@ class binary_writer } } - oa->write_character('['); - oa->write_character('$'); - oa->write_character(dtype); - oa->write_character('#'); + // every element is cast to the (possibly narrower) C++ type matching + // dtype below; a value that does not fit that type would silently + // wrap (integers) or overflow to infinity (the "single" precision + // float) instead of being reported, so such an object falls back to + // a plain object encoding as well + for (const auto& el : value.at(key)) + { + bool in_range = true; + switch (dtype) + { + case 'U': + case 'C': + case 'B': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'i': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'u': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'I': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'm': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'l': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'M': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'L': + in_range = bjdata_ndarray_value_in_range(el); + break; + case 'd': + { + const auto dval = el.template get(); + in_range = !std::isfinite(dval) || + (dval >= static_cast(std::numeric_limits::lowest()) && + dval <= static_cast((std::numeric_limits::max)())); + break; + } + default: + // 'D' (double) already spans the full range of number_float_t + break; + } + if (!in_range) + { + return true; + } + } + + oa.write_character('['); + oa.write_character('$'); + oa.write_character(dtype); + oa.write_character('#'); key = "_ArraySize_"; write_ubjson(value.at(key), use_count, use_type, true, true, bjdata_version); @@ -18894,6 +21207,87 @@ class binary_writer On the other hand, BSON and BJData use little endian and should reorder on big endian systems. */ + // single-instruction byte swaps (compilers lower these to bswap/rev/movbe); + // used to emit big-endian numbers without a per-byte std::reverse loop + static std::uint16_t byte_swap(std::uint16_t x) noexcept + { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_bswap16(x); +#elif defined(_MSC_VER) + return _byteswap_ushort(x); +#else + return static_cast((x >> 8) | (x << 8)); +#endif + } + + static std::uint32_t byte_swap(std::uint32_t x) noexcept + { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_bswap32(x); +#elif defined(_MSC_VER) + return _byteswap_ulong(x); +#else + return ((x & 0x000000FFu) << 24) | ((x & 0x0000FF00u) << 8) + | ((x & 0x00FF0000u) >> 8) | ((x & 0xFF000000u) >> 24); +#endif + } + + static std::uint64_t byte_swap(std::uint64_t x) noexcept + { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_bswap64(x); +#elif defined(_MSC_VER) + return _byteswap_uint64(x); +#else + x = ((x & 0x00000000FFFFFFFFull) << 32) | ((x & 0xFFFFFFFF00000000ull) >> 32); + x = ((x & 0x0000FFFF0000FFFFull) << 16) | ((x & 0xFFFF0000FFFF0000ull) >> 16); + x = ((x & 0x00FF00FF00FF00FFull) << 8) | ((x & 0xFF00FF00FF00FF00ull) >> 8); + return x; +#endif + } + + /*! + @brief reverse the bytes of a buffer by byte-swapping it as UIntType + + Loading the buffer into an unsigned integer of the same width and swapping + that is what lets the compiler emit a single bswap/rev/movbe; reversing the + buffer element by element does not reliably get there (clang keeps a scalar + shuffle). The two memcpy calls are the only portable way to reinterpret the + bytes and are folded away by every optimizer. + */ + template + static void byte_swap_buffer(std::array& a) noexcept + { + static_assert(sizeof(UIntType) == N, "swap width must match the buffer size"); + UIntType v{}; + std::memcpy(&v, a.data(), sizeof(v)); + v = byte_swap(v); + std::memcpy(a.data(), &v, sizeof(v)); + } + + // reverse the bytes of a fixed-size buffer; a single byte_swap() for the + // common 2/4/8-byte number payloads, std::reverse for any other size + static void reverse_bytes(std::array& a) noexcept + { + byte_swap_buffer(a); + } + + static void reverse_bytes(std::array& a) noexcept + { + byte_swap_buffer(a); + } + + static void reverse_bytes(std::array& a) noexcept + { + byte_swap_buffer(a); + } + + template + static void reverse_bytes(std::array& a) noexcept + { + std::reverse(a.begin(), a.end()); + } + template void write_number(const NumberType n, const bool OutputIsLittleEndian = false) { @@ -18905,36 +21299,45 @@ class binary_writer if (is_little_endian != OutputIsLittleEndian) { // reverse byte order prior to conversion if necessary - std::reverse(vec.begin(), vec.end()); + reverse_bytes(vec); } - oa->write_characters(vec.data(), sizeof(NumberType)); + oa.write_characters(vec.data(), sizeof(NumberType)); } void write_compact_float(const number_float_t n, detail::input_format_t format) { #ifdef __GNUC__ -#pragma GCC diagnostic push -#pragma GCC diagnostic ignored "-Wfloat-equal" + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") +#endif + // When number_float_t is float, static_cast(n) is the identity and + // both branches below are intentionally identical (the "compact" float + // representation is the value itself). Only GCC diagnoses this, and only + // when the sink calls are inlined; clang has no such warning. + // (-Wduplicated-branches only exists from GCC 7 on; naming it on an older + // GCC would itself warn under -Wpragmas) +#if defined(__GNUC__) && !defined(__clang__) && (__GNUC__ >= 7) + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wduplicated-branches") #endif if (!std::isfinite(n) || ((static_cast(n) >= static_cast(std::numeric_limits::lowest()) && static_cast(n) <= static_cast((std::numeric_limits::max)()) && static_cast(static_cast(n)) == static_cast(n)))) { - oa->write_character(format == detail::input_format_t::cbor - ? get_cbor_float_prefix(static_cast(n)) - : get_msgpack_float_prefix(static_cast(n))); + oa.write_character(format == detail::input_format_t::cbor + ? get_cbor_float_prefix(static_cast(n)) + : get_msgpack_float_prefix(static_cast(n))); write_number(static_cast(n)); } else { - oa->write_character(format == detail::input_format_t::cbor - ? get_cbor_float_prefix(n) - : get_msgpack_float_prefix(n)); + oa.write_character(format == detail::input_format_t::cbor + ? get_cbor_float_prefix(n) + : get_msgpack_float_prefix(n)); write_number(n); } #ifdef __GNUC__ -#pragma GCC diagnostic pop + JSON_HEDLEY_DIAGNOSTIC_POP #endif } @@ -18997,7 +21400,7 @@ class binary_writer const bool is_little_endian = little_endianness(); /// the output - output_adapter_t oa = nullptr; + OutputSinkType oa; }; } // namespace detail @@ -19017,18 +21420,19 @@ NLOHMANN_JSON_NAMESPACE_END -#include // reverse, remove, fill, find, none_of +#include // reverse, remove, fill, find, none_of, min #include // array #include // localeconv, lconv #include // labs, isfinite, isnan, signbit #include // size_t, ptrdiff_t #include // uint8_t #include // snprintf +#include // memcpy, memset #include // numeric_limits #include // string, char_traits -#include // setfill, setw #include // is_same #include // move +#include // vector // #include // __ _____ _____ _____ @@ -20109,8 +22513,8 @@ char* to_chars(char* first, const char* last, FloatType value) } #ifdef __GNUC__ -#pragma GCC diagnostic push -#pragma GCC diagnostic ignored "-Wfloat-equal" + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif if (value == 0) // +-0 { @@ -20121,7 +22525,7 @@ char* to_chars(char* first, const char* last, FloatType value) return first; } #ifdef __GNUC__ -#pragma GCC diagnostic pop + JSON_HEDLEY_DIAGNOSTIC_POP #endif JSON_ASSERT(last - first >= std::numeric_limits::max_digits10); @@ -20153,6 +22557,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -20161,6 +22567,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -20195,18 +22603,32 @@ class serializer public: /*! - @param[in] s output stream to serialize to + @param[in] s output adapter to serialize to; not owned by the serializer, + so it must outlive it (it lives at the call site) @param[in] ichar indentation character to use + @param[in] pretty_print_ whether the output shall be pretty-printed + @param[in] ensure_ascii_ If @a ensure_ascii_ is true, all non-ASCII + characters in the output are escaped with `\uXXXX` sequences, and the + result consists of ASCII characters only. + @param[in] indent_step_ the indent level @param[in] error_handler_ how to react on decoding errors + + None of @a pretty_print_, @a ensure_ascii_ and @a indent_step_ change over + the life of the serializer, so they are captured once here instead of + being threaded through every call to @ref dump, @ref dump_internal and + @ref dump_iteratively. */ - serializer(output_adapter_t s, const char ichar, + serializer(output_adapter_protocol& s, const char ichar, + const bool pretty_print_ = false, + const bool ensure_ascii_ = false, + const std::size_t indent_step_ = 0, error_handler_t error_handler_ = error_handler_t::strict) - : o(std::move(s)) - , loc(std::localeconv()) - , thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits::to_char_type(* (loc->thousands_sep))) - , decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits::to_char_type(* (loc->decimal_point))) + : o(&s) + , locale(std::localeconv()) , indent_char(ichar) - , indent_string(512, indent_char) + , pretty_print(pretty_print_) + , ensure_ascii(ensure_ascii_) + , indent_step(indent_step_) , error_handler(error_handler_) {} @@ -20222,8 +22644,8 @@ class serializer This function is called by the public member function dump and organizes the serialization internally. The indentation level is propagated as - additional parameter. In case of arrays and objects, the function is - called recursively. + additional parameter. Arrays and objects are serialized without recursion, + however deeply they are nested. - strings and object keys are escaped using `escape_string()` - integer numbers are converted implicitly via `operator<<` @@ -20232,89 +22654,109 @@ class serializer byte array @param[in] val value to serialize - @param[in] pretty_print whether the output shall be pretty-printed - @param[in] ensure_ascii If @a ensure_ascii is true, all non-ASCII characters - in the output are escaped with `\uXXXX` sequences, and the result consists - of ASCII characters only. - @param[in] indent_step the indent level @param[in] current_indent the current indent level (only used internally) */ void dump(const BasicJsonType& val, - const bool pretty_print, - const bool ensure_ascii, - const unsigned int indent_step, - const unsigned int current_indent = 0) + const std::size_t current_indent = 0) + { + dump_internal(val, current_indent); + flush(); + } + + JSON_PRIVATE_UNLESS_TESTED: + /*! + @brief worker for @ref dump + + Identical in behavior to the historical @ref dump, but writes into the + serializer's internal @ref write_buffer instead of issuing a virtual call + per token. The public @ref dump wraps this and flushes the buffer once the + top-level value has been serialized. + + Serializing a container descends into its elements, so a value nested deeply + enough used to exhaust the call stack and terminate the process with no + exception to catch. The descent is bounded here: once @ref recursion_depth_limit + levels have been entered, @ref dump_iteratively writes out what is left + without the call stack. A value nested less deeply than that - all but a + vanishing minority - is written by exactly the code that always wrote it. + + @sa https://github.com/nlohmann/json/issues/5387 + */ + void dump_internal(const BasicJsonType& val, + const std::size_t current_indent = 0, + const std::size_t depth = 0) { switch (val.m_data.m_type) { case value_t::object: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + dump_iteratively(val, current_indent); + return; + } + if (val.m_data.m_value.object->empty()) { - o->write_characters("{}", 2); + put_literal("{}"); return; } if (pretty_print) { - o->write_characters("{\n", 2); + put_literal("{\n"); // variable to hold indentation for recursive calls - const auto new_indent = current_indent + indent_step; - if (JSON_HEDLEY_UNLIKELY(indent_string.size() < new_indent)) - { - indent_string.resize(indent_string.size() * 2, ' '); - } + const auto new_indent = next_indent(current_indent, indent_step); // first n-1 elements auto i = val.m_data.m_value.object->cbegin(); for (std::size_t cnt = 0; cnt < val.m_data.m_value.object->size() - 1; ++cnt, ++i) { - o->write_characters(indent_string.c_str(), new_indent); - o->write_character('\"'); - dump_escaped(i->first, ensure_ascii); - o->write_characters("\": ", 3); - dump(i->second, true, ensure_ascii, indent_step, new_indent); - o->write_characters(",\n", 2); + put_indent(new_indent); + put_char('"'); + dump_escaped(i->first); + put_literal("\": "); + dump_internal(i->second, new_indent, depth + 1); + put_literal(",\n"); } // last element JSON_ASSERT(i != val.m_data.m_value.object->cend()); JSON_ASSERT(std::next(i) == val.m_data.m_value.object->cend()); - o->write_characters(indent_string.c_str(), new_indent); - o->write_character('\"'); - dump_escaped(i->first, ensure_ascii); - o->write_characters("\": ", 3); - dump(i->second, true, ensure_ascii, indent_step, new_indent); + put_indent(new_indent); + put_char('"'); + dump_escaped(i->first); + put_literal("\": "); + dump_internal(i->second, new_indent, depth + 1); - o->write_character('\n'); - o->write_characters(indent_string.c_str(), current_indent); - o->write_character('}'); + put_char('\n'); + put_indent(current_indent); + put_char('}'); } else { - o->write_character('{'); + put_char('{'); // first n-1 elements auto i = val.m_data.m_value.object->cbegin(); for (std::size_t cnt = 0; cnt < val.m_data.m_value.object->size() - 1; ++cnt, ++i) { - o->write_character('\"'); - dump_escaped(i->first, ensure_ascii); - o->write_characters("\":", 2); - dump(i->second, false, ensure_ascii, indent_step, current_indent); - o->write_character(','); + put_char('"'); + dump_escaped(i->first); + put_literal("\":"); + dump_internal(i->second, current_indent, depth + 1); + put_char(','); } // last element JSON_ASSERT(i != val.m_data.m_value.object->cend()); JSON_ASSERT(std::next(i) == val.m_data.m_value.object->cend()); - o->write_character('\"'); - dump_escaped(i->first, ensure_ascii); - o->write_characters("\":", 2); - dump(i->second, false, ensure_ascii, indent_step, current_indent); + put_char('"'); + dump_escaped(i->first); + put_literal("\":"); + dump_internal(i->second, current_indent, depth + 1); - o->write_character('}'); + put_char('}'); } return; @@ -20322,58 +22764,60 @@ class serializer case value_t::array: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + dump_iteratively(val, current_indent); + return; + } + if (val.m_data.m_value.array->empty()) { - o->write_characters("[]", 2); + put_literal("[]"); return; } if (pretty_print) { - o->write_characters("[\n", 2); + put_literal("[\n"); // variable to hold indentation for recursive calls - const auto new_indent = current_indent + indent_step; - if (JSON_HEDLEY_UNLIKELY(indent_string.size() < new_indent)) - { - indent_string.resize(indent_string.size() * 2, ' '); - } + const auto new_indent = next_indent(current_indent, indent_step); // first n-1 elements for (auto i = val.m_data.m_value.array->cbegin(); i != val.m_data.m_value.array->cend() - 1; ++i) { - o->write_characters(indent_string.c_str(), new_indent); - dump(*i, true, ensure_ascii, indent_step, new_indent); - o->write_characters(",\n", 2); + put_indent(new_indent); + dump_internal(*i, new_indent, depth + 1); + put_literal(",\n"); } // last element JSON_ASSERT(!val.m_data.m_value.array->empty()); - o->write_characters(indent_string.c_str(), new_indent); - dump(val.m_data.m_value.array->back(), true, ensure_ascii, indent_step, new_indent); + put_indent(new_indent); + dump_internal(val.m_data.m_value.array->back(), new_indent, depth + 1); - o->write_character('\n'); - o->write_characters(indent_string.c_str(), current_indent); - o->write_character(']'); + put_char('\n'); + put_indent(current_indent); + put_char(']'); } else { - o->write_character('['); + put_char('['); // first n-1 elements for (auto i = val.m_data.m_value.array->cbegin(); i != val.m_data.m_value.array->cend() - 1; ++i) { - dump(*i, false, ensure_ascii, indent_step, current_indent); - o->write_character(','); + dump_internal(*i, current_indent, depth + 1); + put_char(','); } // last element JSON_ASSERT(!val.m_data.m_value.array->empty()); - dump(val.m_data.m_value.array->back(), false, ensure_ascii, indent_step, current_indent); + dump_internal(val.m_data.m_value.array->back(), current_indent, depth + 1); - o->write_character(']'); + put_char(']'); } return; @@ -20381,9 +22825,9 @@ class serializer case value_t::string: { - o->write_character('\"'); - dump_escaped(*val.m_data.m_value.string, ensure_ascii); - o->write_character('\"'); + put_char('"'); + dump_escaped(*val.m_data.m_value.string); + put_char('"'); return; } @@ -20391,70 +22835,66 @@ class serializer { if (pretty_print) { - o->write_characters("{\n", 2); + put_literal("{\n"); // variable to hold indentation for recursive calls - const auto new_indent = current_indent + indent_step; - if (JSON_HEDLEY_UNLIKELY(indent_string.size() < new_indent)) - { - indent_string.resize(indent_string.size() * 2, ' '); - } + const auto new_indent = next_indent(current_indent, indent_step); - o->write_characters(indent_string.c_str(), new_indent); + put_indent(new_indent); - o->write_characters("\"bytes\": [", 10); + put_literal("\"bytes\": ["); if (!val.m_data.m_value.binary->empty()) { for (auto i = val.m_data.m_value.binary->cbegin(); i != val.m_data.m_value.binary->cend() - 1; ++i) { - dump_integer(*i); - o->write_characters(", ", 2); + dump_byte(*i); + put_literal(", "); } - dump_integer(val.m_data.m_value.binary->back()); + dump_byte(val.m_data.m_value.binary->back()); } - o->write_characters("],\n", 3); - o->write_characters(indent_string.c_str(), new_indent); + put_literal("],\n"); + put_indent(new_indent); - o->write_characters("\"subtype\": ", 11); + put_literal("\"subtype\": "); if (val.m_data.m_value.binary->has_subtype()) { dump_integer(val.m_data.m_value.binary->subtype()); } else { - o->write_characters("null", 4); + put_literal("null"); } - o->write_character('\n'); - o->write_characters(indent_string.c_str(), current_indent); - o->write_character('}'); + put_char('\n'); + put_indent(current_indent); + put_char('}'); } else { - o->write_characters("{\"bytes\":[", 10); + put_literal("{\"bytes\":["); if (!val.m_data.m_value.binary->empty()) { for (auto i = val.m_data.m_value.binary->cbegin(); i != val.m_data.m_value.binary->cend() - 1; ++i) { - dump_integer(*i); - o->write_character(','); + dump_byte(*i); + put_char(','); } - dump_integer(val.m_data.m_value.binary->back()); + dump_byte(val.m_data.m_value.binary->back()); } - o->write_characters("],\"subtype\":", 12); + put_literal("],\"subtype\":"); if (val.m_data.m_value.binary->has_subtype()) { dump_integer(val.m_data.m_value.binary->subtype()); - o->write_character('}'); + put_char('}'); } else { - o->write_characters("null}", 5); + put_literal("null}"); } } return; @@ -20464,11 +22904,11 @@ class serializer { if (val.m_data.m_value.boolean) { - o->write_characters("true", 4); + put_literal("true"); } else { - o->write_characters("false", 5); + put_literal("false"); } return; } @@ -20493,13 +22933,13 @@ class serializer case value_t::discarded: { - o->write_characters("", 11); + put_literal(""); return; } case value_t::null: { - o->write_characters("null", 4); + put_literal("null"); return; } @@ -20508,6 +22948,360 @@ class serializer } } + private: + /*! + @brief write out @a val and everything below it without the call stack + + Emits the same bytes as @ref dump_internal, keeping the containers it has + entered on an explicit stack instead of descending into them. Only reached + for values nested deeper than @ref recursion_depth_limit, which is why it is not + written for speed: walking every value this way measured up to 20% slower on + object-heavy documents than letting the compiler drive the descent. + */ + void dump_iteratively(const BasicJsonType& val, + const std::size_t current_indent = 0) + { + // Scalars, empty containers and binary values are written by dump_value + // alone, so nothing is allocated for them: only a container with + // elements is ever pushed. + std::vector stack; + + dump_value(val, current_indent, stack); + + while (!stack.empty()) + { + dump_frame& frame = stack.back(); + + if (frame.value->m_data.m_type == value_t::object) + { + const auto* object = frame.value->m_data.m_value.object; + + if (frame.object_it == object->cend()) + { + if (pretty_print) + { + put_char('\n'); + put_indent(frame.current_indent); + } + + put_char('}'); + stack.pop_back(); + continue; + } + + // the separator goes in front of every element but the first, + // which puts exactly one between each pair and none at the end + if (frame.object_it != object->cbegin()) + { + if (pretty_print) + { + put_literal(",\n"); + } + else + { + put_char(','); + } + } + + if (pretty_print) + { + put_indent(frame.child_indent); + } + + put_char('"'); + dump_escaped(frame.object_it->first); + + if (pretty_print) + { + put_literal("\": "); + } + else + { + put_literal("\":"); + } + + const BasicJsonType& element = frame.object_it->second; + ++frame.object_it; + + // read everything needed from the frame before this: entering a + // container pushes another one and can move them all + const std::size_t element_indent = frame.child_indent; + dump_value(element, element_indent, stack); + } + else + { + const auto* array = frame.value->m_data.m_value.array; + + if (frame.array_it == array->cend()) + { + if (pretty_print) + { + put_char('\n'); + put_indent(frame.current_indent); + } + + put_char(']'); + stack.pop_back(); + continue; + } + + if (frame.array_it != array->cbegin()) + { + if (pretty_print) + { + put_literal(",\n"); + } + else + { + put_char(','); + } + } + + if (pretty_print) + { + put_indent(frame.child_indent); + } + + const BasicJsonType& element = *frame.array_it; + ++frame.array_it; + + // see above + const std::size_t element_indent = frame.child_indent; + dump_value(element, element_indent, stack); + } + } + } + + private: + /// @brief a container that has been opened but not closed yet + struct dump_frame + { + dump_frame(const BasicJsonType* value_, const std::size_t current_indent_, + const std::size_t child_indent_) noexcept + : value(value_) + , current_indent(current_indent_) + , child_indent(child_indent_) + {} + + /// the object or array being serialized + const BasicJsonType* value; + /// the element to serialize next; which of the two is live follows from + /// the type of @a value. They are kept side by side rather than in a + /// union, which would need its special members written out by hand, see + /// detail/iterators/internal_iterator.hpp + typename BasicJsonType::object_t::const_iterator object_it{}; + typename BasicJsonType::array_t::const_iterator array_it{}; + /// the indentation of the container itself, used by its closing bracket + std::size_t current_indent; + /// the indentation of the container's elements + std::size_t child_indent; + }; + + /*! + @brief serialize the value @a val, but not the elements of a container + + An object or array with elements is opened and pushed onto @a stack for + @ref dump_internal to walk; everything else - including a binary value, + which looks like an object but has no elements to descend into - is written + out here in full. + */ + void dump_value(const BasicJsonType& val, + const std::size_t current_indent, + std::vector& stack) + { + switch (val.m_data.m_type) + { + case value_t::object: + { + if (val.m_data.m_value.object->empty()) + { + put_literal("{}"); + return; + } + + std::size_t child_indent = current_indent; + + if (pretty_print) + { + put_literal("{\n"); + child_indent = next_indent(current_indent, indent_step); + } + else + { + put_char('{'); + } + + stack.emplace_back(&val, current_indent, child_indent); + stack.back().object_it = val.m_data.m_value.object->cbegin(); + return; + } + + case value_t::array: + { + if (val.m_data.m_value.array->empty()) + { + put_literal("[]"); + return; + } + + std::size_t child_indent = current_indent; + + if (pretty_print) + { + put_literal("[\n"); + child_indent = next_indent(current_indent, indent_step); + } + else + { + put_char('['); + } + + stack.emplace_back(&val, current_indent, child_indent); + stack.back().array_it = val.m_data.m_value.array->cbegin(); + return; + } + + case value_t::string: + { + put_char('"'); + dump_escaped(*val.m_data.m_value.string); + put_char('"'); + return; + } + + case value_t::binary: + { + if (pretty_print) + { + put_literal("{\n"); + + // variable to hold indentation for the bytes + const auto new_indent = next_indent(current_indent, indent_step); + + put_indent(new_indent); + + put_literal("\"bytes\": ["); + + if (!val.m_data.m_value.binary->empty()) + { + for (auto i = val.m_data.m_value.binary->cbegin(); + i != val.m_data.m_value.binary->cend() - 1; ++i) + { + dump_byte(*i); + put_literal(", "); + } + dump_byte(val.m_data.m_value.binary->back()); + } + + put_literal("],\n"); + put_indent(new_indent); + + put_literal("\"subtype\": "); + if (val.m_data.m_value.binary->has_subtype()) + { + dump_integer(val.m_data.m_value.binary->subtype()); + } + else + { + put_literal("null"); + } + put_char('\n'); + put_indent(current_indent); + put_char('}'); + } + else + { + put_literal("{\"bytes\":["); + + if (!val.m_data.m_value.binary->empty()) + { + for (auto i = val.m_data.m_value.binary->cbegin(); + i != val.m_data.m_value.binary->cend() - 1; ++i) + { + dump_byte(*i); + put_char(','); + } + dump_byte(val.m_data.m_value.binary->back()); + } + + put_literal("],\"subtype\":"); + if (val.m_data.m_value.binary->has_subtype()) + { + dump_integer(val.m_data.m_value.binary->subtype()); + put_char('}'); + } + else + { + put_literal("null}"); + } + } + return; + } + + case value_t::boolean: + { + if (val.m_data.m_value.boolean) + { + put_literal("true"); + } + else + { + put_literal("false"); + } + return; + } + + case value_t::number_integer: + { + dump_integer(val.m_data.m_value.number_integer); + return; + } + + case value_t::number_unsigned: + { + dump_integer(val.m_data.m_value.number_unsigned); + return; + } + + case value_t::number_float: + { + dump_float(val.m_data.m_value.number_float); + return; + } + + case value_t::discarded: + { + put_literal(""); + return; + } + + case value_t::null: + { + put_literal("null"); + return; + } + + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + } + } + + + /*! + @brief the indentation level to use for the children of the current value + + A very large @a indent_step can wrap the unsigned accumulation on deep + nesting, which would silently truncate the indentation. Far harder to reach + now that the accumulator is a std::size_t, but still reachable where that is + 32 bits wide. + */ + static std::size_t next_indent(const std::size_t current_indent, const std::size_t indent_step) + { + const std::size_t new_indent = current_indent + indent_step; + JSON_ASSERT(new_indent >= current_indent); + return new_indent; + } + JSON_PRIVATE_UNLESS_TESTED: /*! @brief dump escaped string @@ -20518,12 +23312,32 @@ class serializer representation. The escaped string is written to output stream @a o. @param[in] s the string to escape - @param[in] ensure_ascii whether to escape non-ASCII characters with - \uXXXX sequences @complexity Linear in the length of string @a s. */ - void dump_escaped(const string_t& s, const bool ensure_ascii) + void dump_escaped(const string_t& s) + { + // dispatch once here rather than test the flag inside the loop: it does + // not change while a string is written, and folding it lets each of the + // two scanners be inlined into a loop of its own + if (ensure_ascii) + { + dump_escaped_impl(s); + } + else + { + dump_escaped_impl(s); + } + } + + /*! + @brief worker for @ref dump_escaped + + @a ensure_ascii is a template parameter here so that the branch on it is + resolved once, outside the loop; see @ref dump_escaped. + */ + template + void dump_escaped_impl(const string_t& s) { std::uint32_t codepoint{}; std::uint8_t state = UTF8_ACCEPT; @@ -20535,6 +23349,56 @@ class serializer for (std::size_t i = 0; i < s.size(); ++i) { + // Fast path: at a character boundary (state == UTF8_ACCEPT), + // bulk-copy the longest run of bytes that need no escaping using a + // SWAR scanner shared with the lexer's contiguous path. The scanner + // stops exactly at the first byte dump_escaped would handle + // individually, so that byte is left to the byte-at-a-time path + // below, keeping escaping output and error diagnostics unchanged. + // + // - EnsureAscii == false: string_bulk_run() copies ordinary bytes + // and complete well-formed UTF-8, stopping at a quote, backslash, + // control character (< 0x20), or ill-formed/truncated sequence. + // - EnsureAscii == true: only printable ASCII may be copied + // verbatim; find_ascii_copyable_run() additionally stops at 0x7F + // and every non-ASCII byte (>= 0x80), which must be \u-escaped. + if (state == UTF8_ACCEPT) + { + const auto* const data = reinterpret_cast(s.data()); + // A run can only be non-empty when the very first byte is one + // the scanner may copy, so test that single byte before paying + // for the scan. Without it, text whose characters all have to be + // escaped - CJK under ensure_ascii, where every byte is >= 0x80 - + // runs the scanner once per character only to be told zero. + std::size_t run = 0; + if (!EnsureAscii) + { + run = string_bulk_run(data + i, s.size() - i); + } + else if (is_ascii_copyable(data[i])) + { + run = find_ascii_copyable_run(data + i, s.size() - i); + } + if (run != 0) + { + // emit any bytes still pending in string_buffer first to + // preserve output order, then write the run directly + if (bytes != 0) + { + put_buffer(string_buffer, bytes); + bytes = 0; + } + put_string(s, i, i + run); + bytes_after_last_accept = 0; + undumped_chars = 0; + i += run; + if (i >= s.size()) + { + break; + } + } + } + const auto byte = static_cast(s[i]); switch (decode(state, codepoint, byte)) @@ -20581,7 +23445,7 @@ class serializer case 0x22: // quotation mark { string_buffer[bytes++] = '\\'; - string_buffer[bytes++] = '\"'; + string_buffer[bytes++] = '"'; break; } @@ -20595,8 +23459,8 @@ class serializer default: { // escape control characters (0x00..0x1F) or, if - // ensure_ascii parameter is used, non-ASCII characters - if ((codepoint <= 0x1F) || (ensure_ascii && (codepoint >= 0x7F))) + // EnsureAscii parameter is used, non-ASCII characters + if ((codepoint <= 0x1F) || (EnsureAscii && (codepoint >= 0x7F))) { if (codepoint <= 0xFFFF) { @@ -20623,7 +23487,7 @@ class serializer // written ("\uxxxx\uxxxx\0") for one code point if (string_buffer.size() - bytes < 13) { - o->write_characters(string_buffer.data(), bytes); + put_buffer(string_buffer, bytes); bytes = 0; } @@ -20661,7 +23525,7 @@ class serializer if (error_handler == error_handler_t::replace) { // add a replacement character - if (ensure_ascii) + if (EnsureAscii) { string_buffer[bytes++] = '\\'; string_buffer[bytes++] = 'u'; @@ -20682,7 +23546,7 @@ class serializer // written ("\uxxxx\uxxxx\0") for one code point if (string_buffer.size() - bytes < 13) { - o->write_characters(string_buffer.data(), bytes); + put_buffer(string_buffer, bytes); bytes = 0; } @@ -20704,7 +23568,7 @@ class serializer default: // decode found yet incomplete multibyte code point { - if (!ensure_ascii) + if (!EnsureAscii) { // code point will not be escaped - copy byte to buffer string_buffer[bytes++] = s[i]; @@ -20721,7 +23585,7 @@ class serializer // write buffer if (bytes > 0) { - o->write_characters(string_buffer.data(), bytes); + put_buffer(string_buffer, bytes); } } else @@ -20731,28 +23595,28 @@ class serializer { case error_handler_t::strict: { - JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast(s.back() | 0))), nullptr)); + JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast(s[s.size() - 1] | 0))), nullptr)); } case error_handler_t::ignore: { // write all accepted bytes - o->write_characters(string_buffer.data(), bytes_after_last_accept); + put_buffer(string_buffer, bytes_after_last_accept); break; } case error_handler_t::replace: { // write all accepted bytes - o->write_characters(string_buffer.data(), bytes_after_last_accept); + put_buffer(string_buffer, bytes_after_last_accept); // add a replacement character - if (ensure_ascii) + if (EnsureAscii) { - o->write_characters("\\ufffd", 6); + put_literal("\\ufffd"); } else { - o->write_characters("\xEF\xBF\xBD", 3); + put_literal("\xEF\xBF\xBD"); } break; } @@ -20763,6 +23627,160 @@ class serializer } } + private: + /*! + @brief append a single character to the write buffer + + Structural characters ('{', '"', ',', ...) previously went straight to the + output adapter, one virtual call each. Buffering them and flushing in bulk + turns those many indirect calls into a single memcpy plus an occasional + flush, which dominates the cost of serializing object/array-heavy values. + */ + void put_char(char c) + { + if (JSON_HEDLEY_UNLIKELY(write_buffer_pos == write_buffer.size())) + { + flush(); + } + write_buffer[write_buffer_pos++] = c; + } + + /*! + @brief append @a indent indentation characters to the write buffer + + Writes the indentation straight into the buffer instead of copying it out of + a pre-grown indentation string, so no auxiliary string has to be sized, + resized, or kept in sync with the deepest nesting level reached. + + An indentation wider than the buffer is emitted by filling the buffer with + the indentation character once and flushing that same content repeatedly: + flushing does not disturb what the buffer holds, so re-filling it between + flushes would be redundant work. + */ + void put_indent(std::size_t indent) + { + // closing braces at the outermost level ask for no indentation at all + if (indent == 0) + { + return; + } + + const std::size_t capacity = write_buffer.size(); + + // fill whatever room is left in the buffer; this is the whole job + // whenever the indentation is narrower than the buffer, which is the + // case for every sane indent_step + const std::size_t head = (std::min)(indent, capacity - write_buffer_pos); + std::memset(write_buffer.data() + write_buffer_pos, indent_char, head); + write_buffer_pos += head; + indent -= head; + + if (JSON_HEDLEY_LIKELY(indent == 0)) + { + return; + } + + // the buffer is full and the remainder spans whole buffer-fulls: flush + // what is pending, then fill the buffer with the indentation character + // exactly once and hand the same bytes to the adapter as often as needed + flush(); + std::memset(write_buffer.data(), indent_char, capacity); + + while (indent >= capacity) + { + write_buffer_pos = capacity; + flush(); + indent -= capacity; + } + + // the buffer still holds indentation characters throughout, so the tail + // only has to be claimed, not written again + write_buffer_pos = indent; + } + + /*! + @brief append a string literal to the write buffer + + The length comes from the array bound rather than a hand-written count, so + it cannot drift out of sync with the literal. A literal always fits into the + buffer (checked at compile time), so unlike @ref put_string this needs no + write-through path for oversized runs. + */ + template + void put_literal(const char (&s)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + { + static_assert(N >= 2, "put_literal expects a non-empty string literal"); + // the array bound counts the terminating NUL, which is not written + constexpr std::size_t length = N - 1; + static_assert(length < write_buffer_size, "string literal must fit into the write buffer"); + + if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + length > write_buffer.size())) + { + flush(); + } + std::memcpy(write_buffer.data() + write_buffer_pos, s, length); + write_buffer_pos += length; + } + + /*! + @brief append the characters of @a str in [@a start, @a end) + + The only way to append a run of characters: @a str carries its own bound, + so the range can be checked against it, which a bare pointer plus a count + could not do. Runs that do not fit the buffer are written straight through + the output adapter (after flushing what is pending), so large string and + number payloads are not copied an extra time. + */ + template + void put_string(const StringType& str, std::size_t start, std::size_t end) + { + JSON_ASSERT(start <= end); + JSON_ASSERT(end <= str.size()); + + const char* const s = str.data() + start; + const std::size_t length = end - start; + + if (JSON_HEDLEY_UNLIKELY(length >= write_buffer.size())) + { + flush(); + o->write_characters(s, length); + return; + } + if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + length > write_buffer.size())) + { + flush(); + } + std::memcpy(write_buffer.data() + write_buffer_pos, s, length); + write_buffer_pos += length; + } + + /*! + @brief append the first @a length characters of a fixed-size buffer + */ + template + void put_buffer(const std::array& buffer, std::size_t length) + { + put_string(buffer, 0, length); + } + + JSON_PRIVATE_UNLESS_TESTED: + /*! + @brief flush the write buffer to the output adapter + + Writing zero characters is a well-defined no-op for every output adapter, so + the buffered length is passed through unconditionally (no empty-guard branch + to leave uncovered). + + @note dump_escaped() and dump_integer()/dump_float() write into the internal + write buffer; callers that invoke them directly (rather than through the + public dump()) must call flush() before inspecting the output. + */ + void flush() + { + o->write_characters(write_buffer.data(), write_buffer_pos); + write_buffer_pos = 0; + } + private: /*! @brief count digits @@ -20838,6 +23856,19 @@ class serializer pos += 6; } + /*! + @brief convert a single element of a binary value to its byte value + + The elements of a binary value are dumped as the numbers 0..255, regardless + of the value type of the configured BinaryType: that type may be signed + (`char`), unsigned (`std::uint8_t`), or not an integer at all + (`std::byte`), none of which @ref dump_integer can handle uniformly. + */ + static std::uint8_t to_byte_value(binary_char_t x) noexcept + { + return static_cast(x); + } + // templates to avoid warnings about useless casts template ::value, int> = 0> bool is_negative_number(NumberType x) @@ -20851,6 +23882,64 @@ class serializer return false; } + /*! + @brief write the decimal representation of the byte @a value + + A binary value's bytes are always in [0, 255], so writing one needs neither + the digit counting nor the 64-bit arithmetic that @ref dump_integer does for + an arbitrary number, and the three digits it takes at most are written + straight into the write buffer. + + Any byte type that is not a plain unsigned byte is converted to its + @ref to_byte_value "byte value" and left to @ref dump_integer, so a signed + or non-integral BinaryType::value_type (`char`, `std::byte`, ...) still + dumps as 0..255. + */ + template + void dump_byte(const ByteType value) + { + dump_byte(value, std::integral_constant < bool, + std::is_unsigned::value && sizeof(ByteType) == 1 + && !std::is_same::value > {}); + } + + template + void dump_byte(const ByteType value, std::false_type /*is_plain_byte*/) + { + dump_integer(to_byte_value(value)); + } + + template + void dump_byte(const ByteType value, std::true_type /*is_plain_byte*/) + { + if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + 3 > write_buffer.size())) + { + flush(); + } + + const auto byte = static_cast(value); + // Accumulate the offset in a local and store it back once. Writing + // through write_buffer[] is a char write, which may alias any object, + // so with the member updated in place the compiler has to reload and + // store it around every digit - measured 2.4x slower on a dump of a + // multi-megabyte binary value. + std::size_t pos = write_buffer_pos; + + if (byte >= 100) + { + write_buffer[pos++] = static_cast('0' + (byte / 100)); + write_buffer[pos++] = static_cast('0' + ((byte / 10) % 10)); + } + else if (byte >= 10) + { + write_buffer[pos++] = static_cast('0' + (byte / 10)); + } + + write_buffer[pos++] = static_cast('0' + (byte % 10)); + + write_buffer_pos = pos; + } + /*! @brief dump an integer @@ -20863,8 +23952,7 @@ class serializer template < typename NumberType, detail::enable_if_t < std::is_integral::value || std::is_same::value || - std::is_same::value || - std::is_same::value, + std::is_same::value, int > = 0 > void dump_integer(NumberType x) { @@ -20887,7 +23975,7 @@ class serializer // special case for "0" if (x == 0) { - o->write_character('0'); + put_char('0'); return; } @@ -20940,7 +24028,7 @@ class serializer *(--buffer_ptr) = static_cast('0' + abs_value); } - o->write_characters(number_buffer.data(), n_chars); + put_buffer(number_buffer, n_chars); } /*! @@ -20956,7 +24044,7 @@ class serializer // NaN / inf if (!std::isfinite(x)) { - o->write_characters("null", 4); + put_literal("null"); return; } @@ -20977,7 +24065,7 @@ class serializer auto* begin = number_buffer.data(); auto* end = ::nlohmann::detail::to_chars(begin, begin + number_buffer.size(), x); - o->write_characters(begin, static_cast(end - begin)); + put_buffer(number_buffer, static_cast(end - begin)); } JSON_HEDLEY_NON_NULL(1) @@ -21008,27 +24096,27 @@ class serializer JSON_ASSERT(static_cast(len) < number_buffer.size()); // erase thousands separators - if (thousands_sep != '\0') + if (locale.thousands_sep != '\0') { // NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::remove returns an iterator, see https://github.com/nlohmann/json/issues/3081 - const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, thousands_sep); + const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, locale.thousands_sep); std::fill(end, number_buffer.end(), '\0'); JSON_ASSERT((end - number_buffer.begin()) <= len); len = (end - number_buffer.begin()); } // convert decimal point to '.' - if (decimal_point != '\0' && decimal_point != '.') + if (locale.decimal_point != '\0' && locale.decimal_point != '.') { // NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::find returns an iterator, see https://github.com/nlohmann/json/issues/3081 - const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), decimal_point); + const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), locale.decimal_point); if (dec_pos != number_buffer.end()) { *dec_pos = '.'; } } - o->write_characters(number_buffer.data(), static_cast(len)); + put_buffer(number_buffer, static_cast(len)); // determine if we need to append ".0" const bool value_is_int_like = @@ -21040,7 +24128,7 @@ class serializer if (value_is_int_like) { - o->write_characters(".0", 2); + put_literal(".0"); } } @@ -21127,34 +24215,60 @@ class serializer } private: - /// the output of the serializer - output_adapter_t o = nullptr; + /// the locale's thousand separator and decimal point characters + struct locale_chars + { + explicit locale_chars(const std::lconv* loc) noexcept + : thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits::to_char_type(* (loc->thousands_sep))) + , decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits::to_char_type(* (loc->decimal_point))) + {} + + const char thousands_sep; + const char decimal_point; + }; + + /// the output of the serializer (non-owning; the adapter lives at the call site) + output_adapter_protocol* o = nullptr; /// a (hopefully) large enough character buffer std::array number_buffer{{}}; - /// the locale - const std::lconv* loc = nullptr; - /// the locale's thousand separator character - const char thousands_sep = '\0'; - /// the locale's decimal point character - const char decimal_point = '\0'; + /// computed once from std::localeconv() at construction; @ref + /// locale_chars keeps std::localeconv()'s pointer from having to be held + /// past the constructor, while still letting these stay const + const locale_chars locale; /// string buffer std::array string_buffer{{}}; /// the indentation character const char indent_char; - /// the indentation string - string_t indent_string; + + /// whether to pretty-print the output + const bool pretty_print; + + /// whether to escape non-ASCII characters with \uXXXX sequences + const bool ensure_ascii; + + /// the indent level + const std::size_t indent_step; /// error_handler how to react on decoding errors const error_handler_t error_handler; + + /// buffer collecting output before it is flushed to the output adapter, so + /// that the many small structural writes become few bulk writes + static constexpr std::size_t write_buffer_size = 1024; + std::array write_buffer{{}}; + /// number of valid bytes currently held in @ref write_buffer + std::size_t write_buffer_pos = 0; }; } // namespace detail NLOHMANN_JSON_NAMESPACE_END +// #include + // #include // #include @@ -21624,7 +24738,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec friend ::nlohmann::detail::serializer; template friend class ::nlohmann::detail::iter_impl; - template + template friend class ::nlohmann::detail::binary_writer; template friend class ::nlohmann::detail::binary_reader; @@ -21648,11 +24762,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec detail::parser_callback_tcb = nullptr, const bool allow_exceptions = true, const bool ignore_comments = false, - const bool ignore_trailing_commas = false + const bool ignore_trailing_commas = false, + const bool discard_number_values = false ) { return ::nlohmann::detail::parser(std::move(adapter), - std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas); + std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas, discard_number_values); } private: @@ -21671,6 +24786,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec template using binary_reader = ::nlohmann::detail::binary_reader; template using binary_writer = ::nlohmann::detail::binary_writer; + // binary_writer over a concrete (non-virtual) sink appending into a std::vector, + // used by the vector-returning to_* overloads + template using vector_binary_writer = + ::nlohmann::detail::binary_writer>; + template static vector_binary_writer vector_writer(std::vector& v) + { + return vector_binary_writer(::nlohmann::detail::output_vector_sink(v)); + } JSON_PRIVATE_UNLESS_TESTED: using serializer = ::nlohmann::detail::serializer; @@ -21887,6 +25010,18 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @} + // Two template parameter requirements that would otherwise be silently + // violated: neither produces a diagnostic of its own, and both corrupt + // values rather than failing. + + static_assert(sizeof(typename BinaryType::value_type) == 1, + "BinaryType::value_type must be exactly one byte wide, " + "because the binary readers and writers reinterpret the container's storage as raw bytes"); + + static_assert(sizeof(NumberUnsignedType) >= sizeof(NumberIntegerType), + "NumberUnsignedType must be at least as wide as NumberIntegerType, " + "because it has to hold the absolute value of every NumberIntegerType value"); + private: /// helper for exception-safe object creation @@ -22267,21 +25402,76 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return it; } - reference set_parent(reference j, std::size_t old_capacity = detail::unknown_size()) + /// @brief erase an element from the object and return the following one + /// Not every map returns an iterator from erase(iterator): some containers + /// (e.g., Abseil's hash maps) return void to avoid computing a successor + /// the caller may not need. Compute it before erasing for those. + template < typename It, detail::enable_if_t < + !detail::erase_returns_void::value, int > = 0 > + typename object_t::iterator erase_from_object(It pos) + { + return m_data.m_value.object->erase(pos); + } + + template < typename It, detail::enable_if_t < + detail::erase_returns_void::value, int > = 0 > + typename object_t::iterator erase_from_object(It pos) + { + auto next = std::next(pos); + m_data.m_value.object->erase(pos); + return next; + } + + /// @brief the capacity of the stored array, or unknown_size() + /// Only JSON_DIAGNOSTICS uses the value, to detect a reallocation that + /// would invalidate the parent pointers. Array types that do not have a + /// capacity() member function report unknown_size(), which is treated as + /// "the elements may have moved". +#if JSON_DIAGNOSTICS + template < typename A = array_t, detail::enable_if_t < detail::has_capacity::value, int > = 0 > + std::size_t array_capacity() const noexcept + { + return m_data.m_value.array->capacity(); + } + + template < typename A = array_t, detail::enable_if_t < !detail::has_capacity::value, int > = 0 > + std::size_t array_capacity() const noexcept + { + return detail::unknown_size(); + } +#else + static constexpr std::size_t array_capacity() noexcept + { + return detail::unknown_size(); + } +#endif + + /// @brief set the parent of a value that has just been added to an array + /// @param j the added value + /// @param old_capacity the value @ref array_capacity() returned before the + /// insertion + reference set_parent_after_array_insert(reference j, std::size_t old_capacity) { #if JSON_DIAGNOSTICS - if (old_capacity != detail::unknown_size()) + // see https://github.com/nlohmann/json/issues/2838 + JSON_ASSERT(type() == value_t::array); + if (JSON_HEDLEY_UNLIKELY(old_capacity == detail::unknown_size() + || array_capacity() != old_capacity)) { - // see https://github.com/nlohmann/json/issues/2838 - JSON_ASSERT(type() == value_t::array); - if (JSON_HEDLEY_UNLIKELY(m_data.m_value.array->capacity() != old_capacity)) - { - // capacity has changed: update all parents - set_parents(); - return j; - } + // the capacity has changed, or the array type does not let us tell: + // the elements may have moved, so update all parents + set_parents(); + return j; } +#else + static_cast(old_capacity); +#endif + return set_parent(j); + } + reference set_parent(reference j) + { +#if JSON_DIAGNOSTICS // ordered_json uses a vector internally, so pointers could have // been invalidated; see https://github.com/nlohmann/json/issues/2962 #ifdef JSON_HEDLEY_MSVC_VERSION @@ -22300,11 +25490,377 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec j.m_parent = this; #else static_cast(j); - static_cast(old_capacity); #endif return j; } +#ifndef JSON_NO_THREAD_LOCAL + /// the number of levels an operation descends into before it finishes the + /// value below it without the call stack + static constexpr std::uint8_t nesting_depth_limit() + { + return 128; + } + + /*! + @brief how many levels the operation going on in this thread has descended into + + Copying a value and comparing two values share this count. The library never + nests one inside the other - copying a value does not compare one, and + comparing two values does not copy them - and where user code nests them + anyway, sharing the count only ends a descent sooner than it had to, which + costs a little speed and is never wrong. + + A byte is enough: the count never exceeds the limit by more than the single + level that notices the limit has been reached. + */ + static std::uint8_t& nesting_depth() noexcept + { + static thread_local std::uint8_t depth = 0; // NOLINT(misc-use-internal-linkage) + return depth; + } +#endif + + /*! + @brief counts one level of a bounded descent for as long as it runs, and + reports whether the descent was still within the limit when it began + + Looks the count up and tests it against the limit itself, rather than + leaving that to the caller: either way it is reached exactly once, so + there is nothing to be gained by making the caller do it. + + Does nothing and is never @ref okay without thread-local storage, where no + descent can be bounded at all: a caller that only descends while this says + it may always ends up finishing without the call stack, exactly as if every + value were nested past the limit. + */ + class nesting_depth_guard + { + public: + nesting_depth_guard() noexcept +#ifdef JSON_NO_THREAD_LOCAL + : m_okay(false) +#else + : m_okay(nesting_depth() < nesting_depth_limit()) +#endif + { +#ifndef JSON_NO_THREAD_LOCAL + ++nesting_depth(); +#endif + } + + ~nesting_depth_guard() + { +#ifndef JSON_NO_THREAD_LOCAL + --nesting_depth(); +#endif + } + + nesting_depth_guard(const nesting_depth_guard&) = delete; + nesting_depth_guard& operator=(const nesting_depth_guard&) = delete; + nesting_depth_guard(nesting_depth_guard&&) = delete; + nesting_depth_guard& operator=(nesting_depth_guard&&) = delete; + + bool okay() const noexcept + { + return m_okay; + } + + private: + bool m_okay; + }; + + /// an entry of the iterative deep copy's worklist: a structured value and + /// the value that is to become its copy + using copy_worklist_t = std::vector>; + + /// scratch space to build the key skeleton of an object copy in one go + using copy_scratch_t = std::vector>; + + /// @brief copy everything of @a src into @a dst but its type and value + static void copy_metadata(const basic_json& src, basic_json& dst) + { + // a custom base class is only required to be copy-constructible and + // move-assignable, so the copy has to go through a temporary + static_cast(dst) = json_base_class_t(static_cast(src)); + +#if JSON_DIAGNOSTIC_POSITIONS + dst.start_position = src.start_position; + dst.end_position = src.end_position; +#endif + } + + /*! + @brief copy the value of @a src into @a dst, which must not be structured + + Objects and arrays are left alone: creating those is the one thing the copy + constructor and @ref copy_shallow do differently from one another, and it is + the reason copying a value can descend at all. + */ + /// @note inlined on purpose: both callers have already told an object or an + /// array apart from the rest, and letting the compiler fold that test + /// into this switch is worth a few percent when copying a value made + /// mostly of numbers + JSON_HEDLEY_ALWAYS_INLINE + static void copy_leaf_value(const basic_json& src, basic_json& dst) + { + switch (src.m_data.m_type) + { + case value_t::string: + { + dst.m_data.m_value = *src.m_data.m_value.string; + break; + } + + case value_t::binary: + { + dst.m_data.m_value = *src.m_data.m_value.binary; + break; + } + + case value_t::boolean: + { + dst.m_data.m_value = src.m_data.m_value.boolean; + break; + } + + case value_t::number_integer: + { + dst.m_data.m_value = src.m_data.m_value.number_integer; + break; + } + + case value_t::number_unsigned: + { + dst.m_data.m_value = src.m_data.m_value.number_unsigned; + break; + } + + case value_t::number_float: + { + dst.m_data.m_value = src.m_data.m_value.number_float; + break; + } + + case value_t::object: + case value_t::array: + case value_t::null: + case value_t::discarded: + default: + break; + } + } + + /*! + @brief copy everything of @a src into the null value @a dst but the children + + Objects and arrays are not copied here; they are appended to @a worklist to + be created later by @ref copy_iteratively. Until that happens, @a dst remains + a null value, so that a partially built copy can be destroyed at any point + without ever violating the class invariants. + */ + static void copy_shallow(const basic_json& src, basic_json& dst, copy_worklist_t& worklist) + { + copy_metadata(src, dst); + + if (src.m_data.m_type == value_t::object || src.m_data.m_type == value_t::array) + { + // defer: dst stays a null value until its container exists + worklist.emplace_back(&src, &dst); + return; + } + + copy_leaf_value(src, dst); + + // only now that the value exists may the type be set: had the creation + // of the value thrown, dst would have been left as a valid null value + dst.m_data.m_type = src.m_data.m_type; + } + + /// @brief create the copy of the array @a src in @a dst + /// @note structured elements are appended to @a worklist instead + static void copy_array_level(const basic_json& src, basic_json& dst, copy_worklist_t& worklist) + { + const array_t& src_array = *src.m_data.m_value.array; + + // create all elements up front: growing the array afterwards could + // invalidate the pointers that are handed to the worklist; resize() + // rather than the fill constructor, because not every array type + // provides the latter (e.g., ones without a matching allocator-aware + // fill constructor) + dst.m_data.m_value.array = create(); + dst.m_data.m_value.array->resize(src_array.size()); + + auto dst_it = dst.m_data.m_value.array->begin(); + for (auto src_it = src_array.cbegin(); src_it != src_array.cend(); ++src_it, ++dst_it) + { + copy_shallow(*src_it, *dst_it, worklist); + } + } + + /// @brief create the copy of the object @a src in @a dst + /// @note structured values are appended to @a worklist instead + static void copy_object_level(const basic_json& src, basic_json& dst, + copy_worklist_t& worklist, copy_scratch_t& scratch) + { + const object_t& src_object = *src.m_data.m_value.object; + + // build the complete key skeleton and hand it to the object's range + // constructor: adding the keys one by one would be quadratic for object + // types that are backed by a vector, such as nlohmann::ordered_map + scratch.clear(); + scratch.reserve(src_object.size()); + for (const auto& element : src_object) + { + scratch.emplace_back(element.first, basic_json()); + } + + dst.m_data.m_value.object = create(std::make_move_iterator(scratch.begin()), + std::make_move_iterator(scratch.end())); + scratch.clear(); + + // pair every value of the copy with its counterpart in the original; + // both are enumerated in the same order for every object type with a + // deterministic order, so the lookup is only needed for exotic ones + auto src_it = src_object.cbegin(); + for (auto& element : *dst.m_data.m_value.object) + { + if (JSON_HEDLEY_LIKELY(src_it != src_object.cend() && src_it->first == element.first)) + { + copy_shallow(src_it->second, element.second, worklist); + ++src_it; + } + else + { + const auto found = src_object.find(element.first); + JSON_ASSERT(found != src_object.cend()); + copy_shallow(found->second, element.second, worklist); + } + } + } + + /*! + @brief deep-copy the object or array @a src into this value without recursing + + The values whose copy has not been created yet are kept on an explicit + worklist rather than on the call stack. This is only reached for values + nested deeper than @ref nesting_depth_limit levels, which is why it copies + every container by hand instead of letting the container do it: the fast + ways of doing so would descend into the elements and defeat the purpose. + */ + void copy_iteratively(const basic_json& src) + { + copy_worklist_t worklist; + copy_scratch_t scratch; + + const basic_json* src_value = &src; + basic_json* dst_value = this; + + for (;;) + { + if (src_value->m_data.m_type == value_t::array) + { + copy_array_level(*src_value, *dst_value, worklist); + } + else + { + copy_object_level(*src_value, *dst_value, worklist, scratch); + } + + // the container is complete and will not be modified again + dst_value->set_parents(); + + if (worklist.empty()) + { + break; + } + + const auto& next = worklist.back(); + src_value = next.first; + dst_value = next.second; + worklist.pop_back(); + + // the value stops being a null value exactly here + dst_value->m_data.m_type = src_value->m_data.m_type; + } + } + + /*! + @brief copy one level of the object or array @a src into this value + + The container copies its own elements, which is the fastest way to fill it. + Every element that is structured itself comes back to @ref copy_structured. + */ + void copy_level(const basic_json& src) + { + if (m_data.m_type == value_t::object) + { + m_data.m_value = *src.m_data.m_value.object; + } + else + { + m_data.m_value = *src.m_data.m_value.array; + } + + set_parents(); + } + + /*! + @brief deep-copy the object or array @a src into this value + + Copying a container copies its elements, so a value nested deeply enough + used to exhaust the call stack. The descent is bounded here: the first + @ref nesting_depth_limit levels are copied by the containers themselves, just + as they always were, and anything below that is copied without the call + stack by @ref copy_iteratively. Copying a value can therefore no longer + exhaust the stack, however deeply it is nested, just like destroying one + cannot since #1436. + + Nothing has to be scanned or built by hand to reach that: a value that is + not nested deeper than the limit - all but a vanishing minority - is copied + exactly as it was before, and this whole detour costs it one counter. + + @sa https://github.com/nlohmann/json/issues/5387 + */ + void copy_structured(const basic_json& src) + { + const nesting_depth_guard guard; + + if (JSON_HEDLEY_LIKELY(guard.okay())) + { + copy_level(src); + return; + } + + // Finish this value without descending any further. It is completed + // before this returns, so a copy made by a custom base class - or by + // anything else that runs while a copy is going on - is unaffected by + // the copy it is nested in. + copy_iteratively(src); + } + + + /// @brief restore the parent pointers after erasing from an object + /// ordered_json keeps its members in a vector, and erasing a member + /// re-constructs every member after it in place, which resets their + /// parent pointers + void set_parents_after_object_erase() + { +#if JSON_DIAGNOSTICS +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning(push ) +#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr +#endif + if (detail::is_ordered_map::value) + { + set_parents(); + } +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning( pop ) +#endif +#endif + } + public: ////////////////////////// // JSON parser callback // @@ -22684,60 +26240,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // check of passed value is valid other.assert_invariant(); - switch (m_data.m_type) + if (m_data.m_type == value_t::object || m_data.m_type == value_t::array) { - case value_t::object: - { - m_data.m_value = *other.m_data.m_value.object; - break; - } - - case value_t::array: - { - m_data.m_value = *other.m_data.m_value.array; - break; - } - - case value_t::string: - { - m_data.m_value = *other.m_data.m_value.string; - break; - } - - case value_t::boolean: - { - m_data.m_value = other.m_data.m_value.boolean; - break; - } - - case value_t::number_integer: - { - m_data.m_value = other.m_data.m_value.number_integer; - break; - } - - case value_t::number_unsigned: - { - m_data.m_value = other.m_data.m_value.number_unsigned; - break; - } - - case value_t::number_float: - { - m_data.m_value = other.m_data.m_value.number_float; - break; - } - - case value_t::binary: - { - m_data.m_value = *other.m_data.m_value.binary; - break; - } - - case value_t::null: - case value_t::discarded: - default: - break; + // copying the container directly would call this constructor again + // for every element, once per nesting level + copy_structured(other); + } + else + { + copy_leaf_value(other, *this); } set_parents(); @@ -22819,21 +26330,26 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief serialization /// @sa https://json.nlohmann.me/api/basic_json/dump/ + JSON_HEDLEY_WARN_UNUSED_RESULT string_t dump(const int indent = -1, const char indent_char = ' ', const bool ensure_ascii = false, const error_handler_t error_handler = error_handler_t::strict) const { string_t result; - serializer s(detail::output_adapter(result), indent_char, error_handler); + detail::output_string_adapter string_adapter(result); if (indent >= 0) { - s.dump(*this, true, ensure_ascii, static_cast(indent)); + serializer s(string_adapter, indent_char, + true, ensure_ascii, static_cast(indent), error_handler); + s.dump(*this); } else { - s.dump(*this, false, ensure_ascii, 0); + serializer s(string_adapter, indent_char, + false, ensure_ascii, 0, error_handler); + s.dump(*this); } return result; @@ -22841,6 +26357,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return the type of the JSON value (explicit) /// @sa https://json.nlohmann.me/api/basic_json/type/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr value_t type() const noexcept { return m_data.m_type; @@ -22848,6 +26365,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether type is primitive /// @sa https://json.nlohmann.me/api/basic_json/is_primitive/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_primitive() const noexcept { return is_null() || is_string() || is_boolean() || is_number() || is_binary(); @@ -22855,6 +26373,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether type is structured /// @sa https://json.nlohmann.me/api/basic_json/is_structured/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_structured() const noexcept { return is_array() || is_object(); @@ -22862,6 +26381,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is null /// @sa https://json.nlohmann.me/api/basic_json/is_null/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_null() const noexcept { return m_data.m_type == value_t::null; @@ -22869,6 +26389,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is a boolean /// @sa https://json.nlohmann.me/api/basic_json/is_boolean/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_boolean() const noexcept { return m_data.m_type == value_t::boolean; @@ -22876,6 +26397,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is a number /// @sa https://json.nlohmann.me/api/basic_json/is_number/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_number() const noexcept { return is_number_integer() || is_number_float(); @@ -22883,6 +26405,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is an integer number /// @sa https://json.nlohmann.me/api/basic_json/is_number_integer/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_number_integer() const noexcept { return m_data.m_type == value_t::number_integer || m_data.m_type == value_t::number_unsigned; @@ -22890,6 +26413,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is an unsigned integer number /// @sa https://json.nlohmann.me/api/basic_json/is_number_unsigned/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_number_unsigned() const noexcept { return m_data.m_type == value_t::number_unsigned; @@ -22897,6 +26421,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is a floating-point number /// @sa https://json.nlohmann.me/api/basic_json/is_number_float/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_number_float() const noexcept { return m_data.m_type == value_t::number_float; @@ -22904,6 +26429,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is an object /// @sa https://json.nlohmann.me/api/basic_json/is_object/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_object() const noexcept { return m_data.m_type == value_t::object; @@ -22911,6 +26437,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is an array /// @sa https://json.nlohmann.me/api/basic_json/is_array/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_array() const noexcept { return m_data.m_type == value_t::array; @@ -22918,6 +26445,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is a string /// @sa https://json.nlohmann.me/api/basic_json/is_string/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_string() const noexcept { return m_data.m_type == value_t::string; @@ -22925,6 +26453,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is a binary array /// @sa https://json.nlohmann.me/api/basic_json/is_binary/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_binary() const noexcept { return m_data.m_type == value_t::binary; @@ -22932,6 +26461,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return whether value is discarded /// @sa https://json.nlohmann.me/api/basic_json/is_discarded/ + JSON_HEDLEY_WARN_UNUSED_RESULT constexpr bool is_discarded() const noexcept { return m_data.m_type == value_t::discarded; @@ -23493,22 +27023,17 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec reference at(size_type idx) { // at only works for arrays - if (JSON_HEDLEY_LIKELY(is_array())) - { - JSON_TRY - { - return set_parent(m_data.m_value.array->at(idx)); - } - JSON_CATCH (std::out_of_range&) - { - // create a better exception explanation - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } // cppcheck-suppress[missingReturn] - } - else + if (JSON_HEDLEY_UNLIKELY(!is_array())) { JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); } + + if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) + { + JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); + } + + return set_parent((*m_data.m_value.array)[idx]); } /// @brief access specified array element with bounds checking @@ -23516,22 +27041,17 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const_reference at(size_type idx) const { // at only works for arrays - if (JSON_HEDLEY_LIKELY(is_array())) - { - JSON_TRY - { - return m_data.m_value.array->at(idx); - } - JSON_CATCH (std::out_of_range&) - { - // create a better exception explanation - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } // cppcheck-suppress[missingReturn] - } - else + if (JSON_HEDLEY_UNLIKELY(!is_array())) { JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); } + + if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) + { + JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); + } + + return (*m_data.m_value.array)[idx]; } /// @brief access specified object element with bounds checking @@ -23631,12 +27151,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec #if JSON_DIAGNOSTICS // remember array size & capacity before resizing const auto old_size = m_data.m_value.array->size(); - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); #endif m_data.m_value.array->resize(idx + 1); #if JSON_DIAGNOSTICS - if (JSON_HEDLEY_UNLIKELY(m_data.m_value.array->capacity() != old_capacity)) + if (JSON_HEDLEY_UNLIKELY(old_capacity == detail::unknown_size() + || array_capacity() != old_capacity)) { // capacity has changed: update all parents set_parents(); @@ -24027,7 +27548,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { - result.m_it.object_iterator = m_data.m_value.object->erase(pos.m_it.object_iterator); + result.m_it.object_iterator = erase_from_object(pos.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -24100,6 +27622,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { result.m_it.object_iterator = m_data.m_value.object->erase(first.m_it.object_iterator, last.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -24130,7 +27653,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - return m_data.m_value.object->erase(std::forward(key)); + const auto erased = m_data.m_value.object->erase(std::forward(key)); + set_parents_after_object_erase(); + return erased; } template < typename KeyType, detail::enable_if_t < @@ -24147,6 +27672,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); + set_parents_after_object_erase(); return 1; } return 0; @@ -24263,6 +27789,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief returns the number of occurrences of a key in a JSON object /// @sa https://json.nlohmann.me/api/basic_json/count/ + JSON_HEDLEY_WARN_UNUSED_RESULT size_type count(const typename object_t::key_type& key) const { // return 0 for all nonobject types @@ -24273,6 +27800,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/count/ template::value, int> = 0> + JSON_HEDLEY_WARN_UNUSED_RESULT size_type count(KeyType && key) const { // return 0 for all nonobject types @@ -24281,6 +27809,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief check the existence of an element in a JSON object /// @sa https://json.nlohmann.me/api/basic_json/contains/ + JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(const typename object_t::key_type& key) const { return is_object() && m_data.m_value.object->find(key) != m_data.m_value.object->end(); @@ -24290,6 +27819,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/contains/ template::value, int> = 0> + JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(KeyType && key) const { return is_object() && m_data.m_value.object->find(std::forward(key)) != m_data.m_value.object->end(); @@ -24297,12 +27827,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief check the existence of an element in a JSON object given a JSON pointer /// @sa https://json.nlohmann.me/api/basic_json/contains/ + JSON_HEDLEY_WARN_UNUSED_RESULT bool contains(const json_pointer& ptr) const { return ptr.contains(this); } template::value, int> = 0> + JSON_HEDLEY_WARN_UNUSED_RESULT JSON_HEDLEY_DEPRECATED_FOR(3.11.0, basic_json::json_pointer or nlohmann::json_pointer) // NOLINT(readability/alt_tokens) bool contains(const typename ::nlohmann::json_pointer& ptr) const { @@ -24458,6 +27990,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief checks whether the container is empty. /// @sa https://json.nlohmann.me/api/basic_json/empty/ + JSON_HEDLEY_WARN_UNUSED_RESULT bool empty() const noexcept { switch (m_data.m_type) @@ -24497,6 +28030,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief returns the number of elements /// @sa https://json.nlohmann.me/api/basic_json/size/ + JSON_HEDLEY_WARN_UNUSED_RESULT size_type size() const noexcept { switch (m_data.m_type) @@ -24536,6 +28070,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief returns the maximum possible number of elements /// @sa https://json.nlohmann.me/api/basic_json/max_size/ + JSON_HEDLEY_WARN_UNUSED_RESULT size_type max_size() const noexcept { switch (m_data.m_type) @@ -24657,9 +28192,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // add the element to the array (move semantics) - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); m_data.m_value.array->push_back(std::move(val)); - set_parent(m_data.m_value.array->back(), old_capacity); + set_parent_after_array_insert(m_data.m_value.array->back(), old_capacity); // if val is moved from, basic_json move constructor marks it null, so we do not call the destructor } @@ -24690,9 +28225,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // add the element to the array - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); m_data.m_value.array->push_back(val); - set_parent(m_data.m_value.array->back(), old_capacity); + set_parent_after_array_insert(m_data.m_value.array->back(), old_capacity); } /// @brief add an object to an array @@ -24778,9 +28313,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // add the element to the array (perfect forwarding) - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); m_data.m_value.array->emplace_back(std::forward(args)...); - return set_parent(m_data.m_value.array->back(), old_capacity); + return set_parent_after_array_insert(m_data.m_value.array->back(), old_capacity); } /// @brief add an object to an object if key does not exist @@ -24859,7 +28394,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/insert/ iterator insert(const_iterator pos, basic_json&& val) // NOLINT(performance-unnecessary-value-param) { - return insert(pos, val); + return insert(std::move(pos), val); } /// @brief inserts copies of element into array @@ -24909,6 +28444,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(invalid_iterator::create(211, "passed iterators may not belong to container", this)); } + // passed iterators must belong to arrays + if (JSON_HEDLEY_UNLIKELY(!first.m_object->is_array())) + { + JSON_THROW(invalid_iterator::create(202, "iterators first and last must point to arrays", this)); + } + // insert to array and return iterator return insert_iterator(pos, first.m_it.array_iterator, last.m_it.array_iterator); } @@ -24995,27 +28536,117 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(312, detail::concat("cannot use update() with ", first.m_object->type_name()), first.m_object)); } + update_members(first, last, merge_objects, 0); + } + + private: + /// @brief an object @ref update_members_iteratively or @ref + /// merge_patch_iteratively is merging into, and the members still to merge + struct merge_frame + { + merge_frame(basic_json* target_, const_iterator position_, const_iterator last_) noexcept + : target(target_), position(std::move(position_)), last(std::move(last_)) + {} + + basic_json* target; + const_iterator position; + const_iterator last; + }; + + /*! + @brief the members loop of @ref update, for this object and range + + Merging a nested object calls this function again, once per nesting + level, so a value nested deeply enough used to exhaust the call stack and + terminate the process. The descent is bounded here: once @ref + detail::recursion_depth_limit levels have been entered, @ref + update_members_iteratively merges what is left without the call stack. + + @param[in] depth nesting level of this object, counted from the object + @ref update was called on + */ + void update_members(const const_iterator& first, const const_iterator& last, const bool merge_objects, const std::size_t depth) + { + if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit())) + { + update_members_iteratively(first, last); + return; + } + for (auto it = first; it != last; ++it) { if (merge_objects && it.value().is_object()) { - auto it2 = m_data.m_value.object->find(it.key()); - if (it2 != m_data.m_value.object->end()) + const auto it2 = m_data.m_value.object->find(it.key()); + // Only recurse when the existing value is itself an object. + // Otherwise overwrite, matching the documented "all other values + // are overwritten as usual" behavior (see #5402). + if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { - it2->second.update(it.value(), true); -#if JSON_DIAGNOSTICS - it2->second.set_parents(); -#endif + it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); continue; } } - m_data.m_value.object->operator[](it.key()) = it.value(); -#if JSON_DIAGNOSTICS - m_data.m_value.object->operator[](it.key()).m_parent = this; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + set_parent(m_data.m_value.object->operator[](it.key()) = it.value()); } } + /*! + @brief merge @a first to @a last into this object without the call stack + + Does the same as @ref update_members with `merge_objects` set, keeping the + objects whose merge was interrupted by a nested one on an explicit stack + instead of descending into them. A nested object is still merged + completely before the next member, in the same order as the recursive + version. Only reached for values nested deeper than @ref + detail::recursion_depth_limit. + */ + void update_members_iteratively(const_iterator first, const_iterator last) + { + std::vector stack; + + basic_json* target = this; + while (true) + { + if (first == last) + { + if (stack.empty()) + { + break; + } + + // a nested object is merged: continue with its parent + target = stack.back().target; + first = stack.back().position; + last = stack.back().last; + stack.pop_back(); + continue; + } + + if (first.value().is_object()) + { + const auto it2 = target->m_data.m_value.object->find(first.key()); + if (it2 != target->m_data.m_value.object->end() && it2->second.is_object()) + { + const basic_json& source = first.value(); + ++first; + stack.emplace_back(target, first, last); + target = &it2->second; + first = source.cbegin(); + last = source.cend(); + continue; + } + } + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + target->set_parent(target->m_data.m_value.object->operator[](first.key()) = first.value()); + ++first; + } + } + + public: /// @brief exchanges the values /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(reference other) noexcept ( @@ -25028,6 +28659,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec std::swap(m_data.m_type, other.m_data.m_type); std::swap(m_data.m_value, other.m_data.m_value); +#if JSON_DIAGNOSTIC_POSITIONS + std::swap(start_position, other.start_position); + std::swap(end_position, other.end_position); +#endif + set_parents(); other.set_parents(); assert_invariant(); @@ -25054,6 +28690,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { using std::swap; swap(*(m_data.m_value.array), other); + set_parents(); } else { @@ -25070,6 +28707,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { using std::swap; swap(*(m_data.m_value.object), other); + set_parents(); } else { @@ -25184,19 +28822,19 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } \ else if (lhs_type == value_t::number_integer && rhs_type == value_t::number_float) \ { \ - return static_cast(lhs.m_data.m_value.number_integer) op rhs.m_data.m_value.number_float; \ + return (detail::compare_integer_with_float(lhs.m_data.m_value.number_integer, rhs.m_data.m_value.number_float)) op (static_cast(0)); \ } \ else if (lhs_type == value_t::number_float && rhs_type == value_t::number_integer) \ { \ - return lhs.m_data.m_value.number_float op static_cast(rhs.m_data.m_value.number_integer); \ + return (static_cast(0)) op (detail::compare_integer_with_float(rhs.m_data.m_value.number_integer, lhs.m_data.m_value.number_float)); \ } \ else if (lhs_type == value_t::number_unsigned && rhs_type == value_t::number_float) \ { \ - return static_cast(lhs.m_data.m_value.number_unsigned) op rhs.m_data.m_value.number_float; \ + return (detail::compare_integer_with_float(lhs.m_data.m_value.number_unsigned, rhs.m_data.m_value.number_float)) op (static_cast(0)); \ } \ else if (lhs_type == value_t::number_float && rhs_type == value_t::number_unsigned) \ { \ - return lhs.m_data.m_value.number_float op static_cast(rhs.m_data.m_value.number_unsigned); \ + return (static_cast(0)) op (detail::compare_integer_with_float(rhs.m_data.m_value.number_unsigned, lhs.m_data.m_value.number_float)); \ } \ else if (lhs_type == value_t::number_unsigned && rhs_type == value_t::number_integer) \ { \ @@ -25251,13 +28889,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec bool operator==(const_reference rhs) const noexcept { #ifdef __GNUC__ -#pragma GCC diagnostic push -#pragma GCC diagnostic ignored "-Wfloat-equal" + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif const_reference lhs = *this; JSON_IMPLEMENT_OPERATOR( ==, true, false, false) #ifdef __GNUC__ -#pragma GCC diagnostic pop + JSON_HEDLEY_DIAGNOSTIC_POP #endif } @@ -25344,12 +28982,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec friend bool operator==(const_reference lhs, const_reference rhs) noexcept { #ifdef __GNUC__ -#pragma GCC diagnostic push -#pragma GCC diagnostic ignored "-Wfloat-equal" + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif JSON_IMPLEMENT_OPERATOR( ==, true, false, false) #ifdef __GNUC__ -#pragma GCC diagnostic pop + JSON_HEDLEY_DIAGNOSTIC_POP #endif } @@ -25536,8 +29174,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec o.width(0); // do the actual serialization - serializer s(detail::output_adapter(o), o.fill()); - s.dump(j, pretty_print, false, static_cast(indentation)); + detail::output_stream_adapter stream_adapter(o); + serializer s(stream_adapter, o.fill(), + pretty_print, false, static_cast(indentation)); + s.dump(j); return o; } @@ -25610,22 +29250,24 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief check if the input is valid JSON /// @sa https://json.nlohmann.me/api/basic_json/accept/ template + JSON_HEDLEY_WARN_UNUSED_RESULT static bool accept(InputType&& i, const bool ignore_comments = false, const bool ignore_trailing_commas = false) { - return parser(detail::input_adapter(std::forward(i)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true); + return parser(detail::input_adapter(std::forward(i)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true); } /// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support) /// @sa https://json.nlohmann.me/api/basic_json/accept/ template::value, int> = 0> + JSON_HEDLEY_WARN_UNUSED_RESULT static bool accept(IteratorType first, SentinelType last, const bool ignore_comments = false, const bool ignore_trailing_commas = false) { - return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true); + return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true); } JSON_HEDLEY_WARN_UNUSED_RESULT @@ -25634,7 +29276,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool ignore_comments = false, const bool ignore_trailing_commas = false) { - return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true); + return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true); } /// @brief generate SAX events @@ -25720,6 +29362,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief return the type as string /// @sa https://json.nlohmann.me/api/basic_json/type_name/ + JSON_HEDLEY_WARN_UNUSED_RESULT JSON_HEDLEY_RETURNS_NON_NULL const char* type_name() const noexcept { @@ -25821,7 +29464,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_cbor(const basic_json& j) { std::vector result; - to_cbor(j, result); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_cbor(j); return result; } @@ -25844,7 +29488,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_msgpack(const basic_json& j) { std::vector result; - to_msgpack(j, result); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_msgpack(j); return result; } @@ -25869,7 +29514,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool use_type = false) { std::vector result; - to_ubjson(j, result, use_size, use_type); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_ubjson(j, use_size, use_type); return result; } @@ -25897,7 +29543,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bjdata_version_t version = bjdata_version_t::draft2) { std::vector result; - to_bjdata(j, result, use_size, use_type, version); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_ubjson(j, use_size, use_type, true, true, version); return result; } @@ -25924,7 +29571,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_bson(const basic_json& j) { std::vector result; - to_bson(j, result); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_bson(j); return result; } @@ -25954,8 +29602,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in CBOR format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -25971,8 +29622,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } template @@ -25997,8 +29651,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - const bool res = binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in MessagePack format @@ -26012,8 +29669,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in MessagePack format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -26028,8 +29688,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } template @@ -26052,8 +29715,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - const bool res = binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in UBJSON format @@ -26067,8 +29733,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in UBJSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -26083,8 +29752,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } template @@ -26107,8 +29779,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - const bool res = binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in BJData format @@ -26122,8 +29797,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in BJData format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -26138,8 +29816,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in BSON format @@ -26153,8 +29834,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @brief create a JSON value from an input in BSON format (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -26169,8 +29853,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - const bool res = binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } template @@ -26193,8 +29880,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - const bool res = binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict); // cppcheck-suppress[accessMoved] - return res ? result : basic_json(value_t::discarded); + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; } /// @} @@ -26420,6 +30110,36 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // note erase performs range check parent.erase(json_pointer::template array_index(last_path)); } + else + { + // the parent of a "remove" target must be an object or array + // (see #5396) + JSON_THROW(out_of_range::create(413, detail::concat("cannot remove value: the JSON Patch 'remove' target's parent is of type ", parent.type_name(), ", but must be an object or array"), &parent)); + } + }; + + // RFC 6902 (section 4.4) forbids "from" from being a proper prefix + // of "path" for a "move" operation: a location cannot be moved into + // one of its own children. Compares reference tokens (already + // unescaped by json_pointer's parser) rather than the raw pointer + // strings, since a token may itself contain an escaped '/' or '~' + // that would defeat a naive string-prefix comparison. "from" equal + // to "path" is *not* a proper prefix and must return false. + const auto is_proper_prefix = [](const json_pointer & from, const json_pointer & to) + { + const auto from_size = from.reference_tokens.size(); + if (from_size >= to.reference_tokens.size()) + { + return false; + } + for (std::size_t i = 0; i < from_size; ++i) + { + if (!(from.reference_tokens[i] == to.reference_tokens[i])) + { + return false; + } + } + return true; }; // type check: top level value must be an array @@ -26497,6 +30217,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const auto from_path = get_value("move", "from", true).template get(); json_pointer from_ptr(from_path); + if (JSON_HEDLEY_UNLIKELY(is_proper_prefix(from_ptr, ptr))) + { + JSON_THROW(out_of_range::create(414, detail::concat("cannot move value: 'from' path '", from_path, "' is a proper prefix of 'path' '", path, "'"), &result)); + } + // the "from" location must exist - use at() basic_json const v = result.at(from_ptr); @@ -26609,19 +30334,17 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // We now reached the end of at least one array // in a second pass, traverse the remaining elements - // remove my remaining elements - const auto end_index = static_cast(result.size()); - while (i < source.size()) + // remove my remaining elements, highest index first; appending + // in that order avoids the quadratic reinsertion done before + for (std::size_t j = source.size(); j > i; --j) { - // add operations in reverse order to avoid invalid - // indices - result.insert(result.begin() + end_index, object( + result.push_back(object( { {"op", "remove"}, - {"path", detail::concat(path, '/', detail::to_string(i))} + {"path", detail::concat(path, '/', detail::to_string(j - 1))} })); - ++i; } + i = source.size(); // add other remaining elements while (i < target.size()) @@ -26640,34 +30363,139 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { - // first pass: traverse this object's elements + // first pass: record, for every source key, whether it is + // common to both objects (in source's iteration order) or + // was deleted (i.e., in source but not in target) -- this is + // a by-product of the target.find() call already needed to + // tell the two cases apart, so it adds no extra lookups. The + // "remove" ops themselves are emitted later, interleaved + // with the recursive per-key diffs in the fast path below, + // to match source's original iteration order (as the + // original, pre-reordering-aware implementation did) instead + // of grouping all removes before all recursive diffs. + std::vector common_keys_source_order; for (auto it = source.cbegin(); it != source.cend(); ++it) { - // escape the key name to be used in a JSON patch - const auto path_key = detail::concat(path, '/', detail::escape(it.key())); - if (target.find(it.key()) != target.end()) { - // recursive call to compare object values at key it - auto temp_diff = diff(it.value(), target[it.key()], path_key); - result.insert(result.end(), temp_diff.begin(), temp_diff.end()); + common_keys_source_order.push_back(it.key()); + } + } + + // second pass: find keys that were added (i.e., in target but + // not in source), and record the keys common to both, in + // target's iteration order -- again a by-product of the + // source.find() call already needed to detect added keys. At + // the same time, determine whether every added key comes + // after every common key in target's order (a precondition + // for the fast path below, which only ever appends new keys + // at the very end): for an object_t whose iteration order is + // a pure function of the key set (e.g. the default std::map, + // which always iterates in sorted key order), the order + // check further below is always true and this whole + // mechanism is effectively a no-op; it only matters for a + // reorderable object_t such as the one backing `ordered_json`. + // patch ops for keys that were added (i.e., in target but not + // in source); built here so the fast path below can reuse + // them without a second source.find() per target key. Only + // used by the fast path -- the slow (reordering) path + // rebuilds "add" ops for every key itself. + std::vector common_keys_target_order; + basic_json added_ops(value_t::array); + bool new_keys_form_suffix = true; + bool seen_new_key = false; + for (auto it = target.cbegin(); it != target.cend(); ++it) + { + if (source.find(it.key()) == source.end()) + { + seen_new_key = true; + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); + added_ops.push_back( + { + {"op", "add"}, {"path", path_key}, + {"value", it.value()} + }); } else { - // found a key that is not in o -> remove it + common_keys_target_order.push_back(it.key()); + if (seen_new_key) + { + new_keys_form_suffix = false; + } + } + } + + if (common_keys_source_order == common_keys_target_order && new_keys_form_suffix) + { + // fast path: order of common keys already matches (or the + // object_t's iteration order does not depend on + // insertion history), so a plain per-key recursive diff + // is correct and minimal, as before. common_keys_source_order + // is, by construction, the subsequence of source's keys + // that are common to both objects, in source's iteration + // order -- so it can be walked in lockstep with `source` + // using a cheap key comparison instead of another lookup. + // Deleted keys (those source keys not in common_keys_source_order) + // are interleaved here too, in source's original order, to + // match the historical (pre-reordering-aware) output order. + auto common_it = common_keys_source_order.cbegin(); + for (auto it = source.cbegin(); it != source.cend(); ++it) + { + if (common_it != common_keys_source_order.cend() && it.key() == *common_it) + { + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); + auto temp_diff = diff(it.value(), target[it.key()], path_key); + result.insert(result.end(), temp_diff.begin(), temp_diff.end()); + ++common_it; + } + else + { + // found a key that is not in target -> remove it + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); + result.push_back(object( + { + {"op", "remove"}, {"path", path_key} + })); + } + } + + // append the "add" ops for brand-new keys collected above + // during the pass over target -- no second source.find() + // per target key needed + result.insert(result.end(), added_ops.begin(), added_ops.end()); + } + else + { + // slow path: the common keys are in a different relative + // order in source and target (only possible for a + // reorderable object_t like ordered_map). Building a + // minimal reordering patch is a nontrivial (LCS-like) + // problem; instead, remove every source key -- both + // deleted keys (which must be removed regardless) and + // common keys (removed so they can be re-added in + // target's order) -- and re-add every key that should + // remain, with its final target value, in target's + // order. basic_json::patch()'s "add" operation on an + // object uses operator[], which appends at the end for a + // vector-backed insertion-ordered map when the key does + // not already exist -- so removing a key and then adding + // it moves it to the end, fixing its position. + for (auto it = source.cbegin(); it != source.cend(); ++it) + { + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); result.push_back(object( { {"op", "remove"}, {"path", path_key} })); } - } - // second pass: traverse other object's elements - for (auto it = target.cbegin(); it != target.cend(); ++it) - { - if (source.find(it.key()) == source.end()) + // add every key that is either common (just removed + // above) or brand new, in target's iteration order, so + // that the final order after applying the patch matches + // target exactly + for (auto it = target.cbegin(); it != target.cend(); ++it) { - // found a key that is not in this -> add it const auto path_key = detail::concat(path, '/', detail::escape(it.key())); result.push_back( { @@ -26713,9 +30541,30 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief applies a JSON Merge Patch /// @sa https://json.nlohmann.me/api/basic_json/merge_patch/ void merge_patch(const basic_json& apply_patch) + { + apply_merge_patch(apply_patch, 0); + } + + private: + /*! + @brief @ref merge_patch, for a patch at nesting level @a depth + + Applying a nested object calls this function again, once per nesting + level, so a patch nested deeply enough used to exhaust the call stack and + terminate the process. The descent is bounded here: once @ref + detail::recursion_depth_limit levels have been entered, @ref + merge_patch_iteratively applies what is left without the call stack. + */ + void apply_merge_patch(const basic_json& apply_patch, const std::size_t depth) { if (apply_patch.is_object()) { + if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit())) + { + merge_patch_iteratively(apply_patch); + return; + } + if (!is_object()) { *this = object(); @@ -26728,7 +30577,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } else { - operator[](it.key()).merge_patch(it.value()); + operator[](it.key()).apply_merge_patch(it.value(), depth + 1); } } } @@ -26738,6 +30587,62 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } } + /*! + @brief apply @a apply_patch to this value without the call stack + + Does the same as @ref merge_patch, keeping the objects being patched on an + explicit stack instead of descending into them. A nested object is still + patched completely before the next member, in the same order as the + recursive version. Only reached for patches nested deeper than @ref + detail::recursion_depth_limit. + */ + void merge_patch_iteratively(const basic_json& apply_patch) + { + std::vector stack; + + // patch `target` with `patch`, or start patching it member by member + const auto apply = [&stack](basic_json & target, const basic_json & patch) + { + if (patch.is_object()) + { + if (!target.is_object()) + { + target = basic_json::object(); + } + stack.emplace_back(&target, patch.cbegin(), patch.cend()); + } + else + { + target = patch; + } + }; + + apply(*this, apply_patch); + while (!stack.empty()) + { + // a copy, as applying a member below can reallocate the stack; + // the frame itself is only changed through stack.back() + const merge_frame frame = stack.back(); + if (frame.position == frame.last) + { + stack.pop_back(); + continue; + } + + const const_iterator member = frame.position; + ++stack.back().position; + if (member.value().is_null()) + { + frame.target->erase(member.key()); + } + else + { + apply(frame.target->operator[](member.key()), member.value()); + } + } + } + + public: /// @} }; @@ -26980,7 +30885,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION #undef JSON_USE_GLOBAL_UDLS -#undef JSON_BRACE_INIT_COPY_SEMANTICS +#undef JSON_STRICT_NUL_HANDLING #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH @@ -26998,6 +30903,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_HAS_STD_FORMAT #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON + #undef JSON_BRACE_INIT_COPY_SEMANTICS #endif // #include @@ -27020,7 +30926,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_HEDLEY_CLANG_HAS_ATTRIBUTE #undef JSON_HEDLEY_CLANG_HAS_BUILTIN #undef JSON_HEDLEY_CLANG_HAS_CPP_ATTRIBUTE -#undef JSON_HEDLEY_CLANG_HAS_DECLSPEC_DECLSPEC_ATTRIBUTE +#undef JSON_HEDLEY_CLANG_HAS_DECLSPEC_ATTRIBUTE #undef JSON_HEDLEY_CLANG_HAS_EXTENSION #undef JSON_HEDLEY_CLANG_HAS_FEATURE #undef JSON_HEDLEY_CLANG_HAS_WARNING @@ -27111,7 +31017,10 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_HEDLEY_PELLES_VERSION_CHECK #undef JSON_HEDLEY_PGI_VERSION #undef JSON_HEDLEY_PGI_VERSION_CHECK +#undef JSON_HEDLEY_PRAGMA #undef JSON_HEDLEY_PREDICT +#undef JSON_HEDLEY_PREDICT_FALSE +#undef JSON_HEDLEY_PREDICT_TRUE #undef JSON_HEDLEY_PRINTF_FORMAT #undef JSON_HEDLEY_PRIVATE #undef JSON_HEDLEY_PUBLIC diff --git a/single_include/nlohmann/json_fwd.hpp b/single_include/nlohmann/json_fwd.hpp index 525e65b64..281c05efa 100644 --- a/single_include/nlohmann/json_fwd.hpp +++ b/single_include/nlohmann/json_fwd.hpp @@ -52,6 +52,10 @@ #define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0 #endif +#ifndef JSON_BRACE_INIT_COPY_SEMANTICS + #define JSON_BRACE_INIT_COPY_SEMANTICS 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -70,20 +74,27 @@ #define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS _bics +#else + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) json_abi ## a ## b ## c -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ - NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS) + NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 4383b582c..7a9bc5471 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -2,6 +2,9 @@ cmake_minimum_required(VERSION 3.13...4.0) option(JSON_Valgrind "Execute test suite with Valgrind." OFF) option(JSON_FastTests "Skip expensive/slow tests." OFF) +option(JSON_TestSimdutf "Build the unit tests against the simdutf UTF-8 validation backend." OFF) + +set(JSON_SIMDUTF_VERSION 9.1.0 CACHE STRING "The simdutf version used by JSON_TestSimdutf.") set(JSON_32bitTest AUTO CACHE STRING "Enable the 32bit unit test (ON/OFF/AUTO/ONLY).") set(JSON_TestStandards "" CACHE STRING "The list of standards to test explicitly.") @@ -72,7 +75,12 @@ target_compile_options(test_main PUBLIC # is annotated JSON_HEDLEY_NO_RETURN (it always throws), which # makes MSVC flag the code following its call in binary_reader.hpp # as unreachable for that instantiation, in both Debug and Release - $<$:/W4;/wd4566;/wd4996;/wd4702> + # Disable warning C4503: decorated name length exceeded, name was truncated; the deep + # copy support added for #5387 pushes the mangled name of + # std::allocator_traits<...>::construct for the custom-base-class + # test's map type past VS2015's limit. The name is only used for + # debug info, so truncation does not affect the build. + $<$:/W4;/wd4566;/wd4996;/wd4702;/wd4503> # https://github.com/nlohmann/json/issues/1114 $<$:/bigobj> $<$:-Wa,-mbig-obj> @@ -125,6 +133,51 @@ json_test_set_test_options(test-unicode4 TEST_PROPERTIES TIMEOUT 3000) # add unit tests ############################################################################# +# Generate the leak checks for every JSON_HEDLEY_* macro defined in +# hedley.hpp; tests/src/unit-no-macro-leak.cpp #include-s the result after +# nlohmann/json.hpp (see issue #5408). Using the shared +# cmake/scripts/gen_hedley_undef_check.cmake script (also used by `make +# update_hedley_undef`) instead of a hand-maintained list of macro names +# means this test can never go stale after a future `make update_hedley`. +set(hedley_hpp "${PROJECT_SOURCE_DIR}/include/nlohmann/thirdparty/hedley/hedley.hpp") +set(hedley_undef_check_script "${PROJECT_SOURCE_DIR}/cmake/scripts/gen_hedley_undef_check.cmake") +set(hedley_undef_checks "${PROJECT_BINARY_DIR}/include/hedley_undef_checks.inc") + +# Reconfigure whenever the vendored header or the generator script changes, +# so a `cmake --build` after `make update_hedley` does not silently keep a +# stale generated file around. +set_property(DIRECTORY APPEND PROPERTY CMAKE_CONFIGURE_DEPENDS + "${hedley_hpp}" + "${hedley_undef_check_script}") + +# Generate once at configure time, so the very first build (before any +# custom-command build step has run) already has an up-to-date file. +execute_process( + COMMAND ${CMAKE_COMMAND} + "-DHEDLEY_HPP=${hedley_hpp}" + "-DOUTPUT=${hedley_undef_checks}" + -DMODE=checks + -P "${hedley_undef_check_script}" + RESULT_VARIABLE hedley_undef_check_result +) +if(NOT hedley_undef_check_result EQUAL 0) + message(FATAL_ERROR "Failed to generate ${hedley_undef_checks}") +endif() + +# Also (re)generate as a build step, so an incremental build after editing +# hedley.hpp without a full reconfigure still picks up the change. +add_custom_command( + OUTPUT "${hedley_undef_checks}" + COMMAND ${CMAKE_COMMAND} + "-DHEDLEY_HPP=${hedley_hpp}" + "-DOUTPUT=${hedley_undef_checks}" + -DMODE=checks + -P "${hedley_undef_check_script}" + DEPENDS "${hedley_hpp}" "${hedley_undef_check_script}" + COMMENT "Generating Hedley undef leak checks" + VERBATIM) +add_custom_target(generate_hedley_undef_checks DEPENDS "${hedley_undef_checks}") + if("${JSON_TestStandards}" STREQUAL "") set(test_cxx_standards 11 14 17 20 23) unset(test_force) @@ -149,6 +202,71 @@ if(test_force) endif() message(STATUS "${msg}") +############################################################################# +# optionally validate UTF-8 with simdutf (JSON_USE_SIMDUTF) +############################################################################# + +# The simdutf backend is opt-in and not vendored, so it is fetched here rather +# than being a checked-in dependency. Everything below hangs off test_main, +# whose usage requirements every test target inherits; the library target and +# the installed CMake package are deliberately left untouched. +if (JSON_TestSimdutf) + # simdutf requires C++17, both to compile itself and to be reachable from + # the library, which keeps its scalar validator below that. Find a tested + # standard that satisfies it. + set(simdutf_standard "") + foreach(cxx_standard ${test_cxx_standards}) + if(NOT cxx_standard LESS 17 AND compiler_supports_cpp_${cxx_standard}) + set(simdutf_standard ${cxx_standard}) + break() + endif() + endforeach() + + if("${simdutf_standard}" STREQUAL "") + # Building simdutf would fail outright without a C++17 compiler, and + # even with one it would go unused if no C++17-or-later standard is + # tested. Say so and fall back to the scalar validator rather than + # failing the build. + if(NOT compiler_supports_cpp_17) + set(simdutf_reason "the compiler does not support C++17") + else() + set(simdutf_reason "no tested standard is C++17 or later (testing ${msg_standards})") + endif() + message(WARNING + "JSON_TestSimdutf is enabled, but ${simdutf_reason}. simdutf requires C++17, so it " + "is not fetched and JSON_USE_SIMDUTF is not defined: the tests run against the " + "built-in scalar UTF-8 validator instead. Set JSON_TestStandards to include 17 or " + "later, or build with a compiler that supports C++17.") + else() + if (CMAKE_VERSION VERSION_LESS 3.18) + message(FATAL_ERROR "JSON_TestSimdutf requires CMake 3.18 or later (simdutf's minimum).") + endif() + + include(FetchContent) + + # simdutf builds its tests and tools by default, and its tests pull + # further dependencies of their own; only the library is needed here + set(SIMDUTF_TESTS OFF CACHE BOOL "" FORCE) + set(SIMDUTF_TOOLS OFF CACHE BOOL "" FORCE) + set(SIMDUTF_BENCHMARKS OFF CACHE BOOL "" FORCE) + set(SIMDUTF_ICONV OFF CACHE BOOL "" FORCE) + + FetchContent_Declare(simdutf + URL https://github.com/simdutf/simdutf/archive/refs/tags/v${JSON_SIMDUTF_VERSION}.tar.gz + DOWNLOAD_EXTRACT_TIMESTAMP TRUE + ) + FetchContent_MakeAvailable(simdutf) + + target_compile_definitions(test_main PUBLIC JSON_USE_SIMDUTF) + target_link_libraries(test_main PUBLIC simdutf::simdutf) + + # simdutf.h requires C++17; below that the library keeps its scalar + # validator, so any C++11/14 test targets exercise the fallback and the + # C++17-and-later ones exercise simdutf. Both must agree. + message(STATUS "UTF-8 validation delegated to simdutf ${JSON_SIMDUTF_VERSION} for C++17 and later (JSON_USE_SIMDUTF)") + endif() +endif() + # *DO* use json_test_set_test_options() above this line json_test_should_build_32bit_test(json_32bit_test json_32bit_test_only "${JSON_32bitTest}") @@ -163,6 +281,14 @@ foreach(file ${files}) json_test_add_test_for(${file} MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}) endforeach() +# tests/src/unit-no-macro-leak.cpp #include-s the generated leak-check file, +# so its test targets must be built after generate_hedley_undef_checks. +foreach(cxx_standard ${test_cxx_standards}) + if(TARGET test-no-macro-leak_cpp${cxx_standard}) + add_dependencies(test-no-macro-leak_cpp${cxx_standard} generate_hedley_undef_checks) + endif() +endforeach() + if(json_32bit_test_only) # Skip all other tests in this file return() @@ -177,6 +303,24 @@ json_test_add_test_for(src/unit-comparison.cpp MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force} ) +# test the parser again with JSON_DIAGNOSTIC_POSITIONS enabled +json_test_set_test_options(test-class_parser_diagnostic_positions + COMPILE_DEFINITIONS JSON_DIAGNOSTIC_POSITIONS=1 +) +json_test_add_test_for(src/unit-class_parser.cpp + NAME test-class_parser_diagnostic_positions + MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force} +) + +# test diagnostic positions again without regular diagnostics (JSON pointer paths) +json_test_set_test_options(test-diagnostic-positions_only + COMPILE_DEFINITIONS JSON_DIAGNOSTICS=0 +) +json_test_add_test_for(src/unit-diagnostic-positions.cpp + NAME test-diagnostic-positions_only + MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force} +) + # *DO NOT* use json_test_set_test_options() below this line ############################################################################# diff --git a/tests/abi/config/CMakeLists.txt b/tests/abi/config/CMakeLists.txt index 3a8367690..52941dc33 100644 --- a/tests/abi/config/CMakeLists.txt +++ b/tests/abi/config/CMakeLists.txt @@ -14,6 +14,20 @@ add_test( NAME test-abi_config_noversion COMMAND abi_config_noversion ${DOCTEST_TEST_FILTER}) +# test default and no version namespace with all ABI tags enabled, so the +# expected tag order is checked regardless of the JSON_* CMake options +foreach(test default noversion) + add_executable(abi_config_${test}_all_tags ${test}.cpp) + target_compile_definitions(abi_config_${test}_all_tags PRIVATE + JSON_DIAGNOSTICS=1 + JSON_DIAGNOSTIC_POSITIONS=1 + JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1) + target_link_libraries(abi_config_${test}_all_tags PRIVATE abi_compat_main) + add_test( + NAME test-abi_config_${test}_all_tags + COMMAND abi_config_${test}_all_tags ${DOCTEST_TEST_FILTER}) +endforeach() + # test custom namespace add_executable(abi_config_custom custom.cpp) target_link_libraries(abi_config_custom PRIVATE abi_compat_main) diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index 0edc12e62..d0b4ba54b 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -24,12 +24,16 @@ TEST_CASE("default namespace") expected += "_diag"; #endif +#if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON + expected += "_ldvcmp"; +#endif + #if JSON_DIAGNOSTIC_POSITIONS expected += "_dp"; #endif -#if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON - expected += "_ldvcmp"; +#if JSON_BRACE_INIT_COPY_SEMANTICS + expected += "_bics"; #endif expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index 2ae5cf5ac..789107181 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -25,12 +25,16 @@ TEST_CASE("default namespace without version component") expected += "_diag"; #endif +#if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON + expected += "_ldvcmp"; +#endif + #if JSON_DIAGNOSTIC_POSITIONS expected += "_dp"; #endif -#if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON - expected += "_ldvcmp"; +#if JSON_BRACE_INIT_COPY_SEMANTICS + expected += "_bics"; #endif expected += "::basic_json"; diff --git a/tests/benchmarks/src/benchmarks.cpp b/tests/benchmarks/src/benchmarks.cpp index 4de238272..f949f3f12 100644 --- a/tests/benchmarks/src/benchmarks.cpp +++ b/tests/benchmarks/src/benchmarks.cpp @@ -81,6 +81,44 @@ BENCHMARK_CAPTURE(ParseString, signed_ints, TEST_DATA_DIRECTORY "/regressi BENCHMARK_CAPTURE(ParseString, unsigned_ints, TEST_DATA_DIRECTORY "/regression/unsigned_ints.json"); BENCHMARK_CAPTURE(ParseString, small_signed_ints, TEST_DATA_DIRECTORY "/regression/small_signed_ints.json"); +////////////////////////////////////////////////////////////////////////////// +// parse pretty-printed JSON from string +// +// Every file in the corpus above is minified or only lightly spaced, so none of +// them exercise the lexer's whitespace handling. Real-world JSON is frequently +// indented - configuration files, pretty-printed API responses, anything kept +// under version control - where insignificant whitespace can outweigh the data. +// Re-serializing a document with an indentation and parsing that keeps the +// content identical to the ParseString row above, so the pair isolates the cost +// of the whitespace alone. +////////////////////////////////////////////////////////////////////////////// + +static void ParseIndented(benchmark::State& state, const char* filename, int indent) +{ + std::ifstream f(filename); + std::string str((std::istreambuf_iterator(f)), std::istreambuf_iterator()); + const std::string indented = json::parse(str).dump(indent); + + while (state.KeepRunning()) + { + state.PauseTiming(); + auto* j = new json(); + state.ResumeTiming(); + + *j = json::parse(indented); + + state.PauseTiming(); + delete j; + state.ResumeTiming(); + } + + state.SetBytesProcessed(state.iterations() * indented.size()); +} +BENCHMARK_CAPTURE(ParseIndented, jeopardy / 4, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", 4); +BENCHMARK_CAPTURE(ParseIndented, canada / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", 4); +BENCHMARK_CAPTURE(ParseIndented, citm_catalog / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", 4); +BENCHMARK_CAPTURE(ParseIndented, twitter / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", 4); + ////////////////////////////////////////////////////////////////////////////// // serialize JSON ////////////////////////////////////////////////////////////////////////////// @@ -214,4 +252,323 @@ static void BinaryToCbor(benchmark::State& state) } BENCHMARK(BinaryToCbor)->RangeMultiplier(2)->Range(8, 8 << 12); +////////////////////////////////////////////////////////////////////////////// +// parse binary formats +////////////////////////////////////////////////////////////////////////////// + +// Only MessagePack had a read benchmark (FromMsgpack above, left untouched so +// its numbers stay comparable across releases). The benchmarks below cover the +// other formats, and read from a contiguous buffer as well as from a FILE*: +// most callers pass a container, and the two adapters compile to different +// code. The test data repository ships JSON only, so the input for each is +// derived at setup time by serializing a parsed test file. + +/// binary format to benchmark; the _optimized variants add UBJSON/BJData size +/// and type annotations, which the readers handle in a separate code path +enum class binary_format +{ + cbor, + msgpack, + ubjson, + ubjson_optimized, + bjdata, + bjdata_optimized, + bson +}; + +static std::vector to_binary(const json& j, const binary_format format) +{ + switch (format) + { + case binary_format::cbor: + return json::to_cbor(j); + case binary_format::msgpack: + return json::to_msgpack(j); + case binary_format::ubjson: + return json::to_ubjson(j); + case binary_format::ubjson_optimized: + return json::to_ubjson(j, true, true); + case binary_format::bjdata: + return json::to_bjdata(j); + case binary_format::bjdata_optimized: + return json::to_bjdata(j, true, true); + case binary_format::bson: + default: + return json::to_bson(j); + } +} + +static json from_binary(const std::vector& bytes, const binary_format format) +{ + switch (format) + { + case binary_format::cbor: + return json::from_cbor(bytes); + case binary_format::msgpack: + return json::from_msgpack(bytes); + case binary_format::ubjson: + case binary_format::ubjson_optimized: + return json::from_ubjson(bytes); + case binary_format::bjdata: + case binary_format::bjdata_optimized: + return json::from_bjdata(bytes); + case binary_format::bson: + default: + return json::from_bson(bytes); + } +} + +static json from_binary(std::FILE* file, const binary_format format) +{ + switch (format) + { + case binary_format::cbor: + return json::from_cbor(file); + case binary_format::msgpack: + return json::from_msgpack(file); + case binary_format::ubjson: + case binary_format::ubjson_optimized: + return json::from_ubjson(file); + case binary_format::bjdata: + case binary_format::bjdata_optimized: + return json::from_bjdata(file); + case binary_format::bson: + default: + return json::from_bson(file); + } +} + +/*! +@brief serialize a parsed test file to @a format + +Returns an empty vector and marks the benchmark as skipped if the file cannot +be represented in the format, rather than letting the exception escape: BSON +requires an object at the top level, and several test files are arrays. +*/ +static std::vector binary_input(benchmark::State& state, const char* filename, const binary_format format) +{ + std::ifstream f(filename); + std::string const str((std::istreambuf_iterator(f)), std::istreambuf_iterator()); + const json j = json::parse(str); + + if (format == binary_format::bson && !j.is_object()) + { + state.SkipWithError("BSON requires an object at the top level"); + return {}; + } + + return to_binary(j, format); +} + +static void FromBinaryBuffer(benchmark::State& state, const char* filename, const binary_format format) +{ + const std::vector bytes = binary_input(state, filename, format); + if (bytes.empty()) + { + return; + } + + for (auto _ : state) + { + // the value is destroyed outside the timed section, because destroying + // a large DOM is not what this benchmark measures + state.PauseTiming(); + auto* j = new json(); + state.ResumeTiming(); + + *j = from_binary(bytes, format); + + state.PauseTiming(); + delete j; + state.ResumeTiming(); + } + + state.SetBytesProcessed(state.iterations() * bytes.size()); +} + +BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / floats, TEST_DATA_DIRECTORY "/regression/floats.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / signed_ints, TEST_DATA_DIRECTORY "/regression/signed_ints.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson_optimized / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson_optimized); +BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson_optimized / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson_optimized); +BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bjdata); +BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata); +BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata_optimized / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bjdata_optimized); +BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata_optimized / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata_optimized); +// BSON requires an object at the top level, so the array-rooted test files +// (jeopardy and the regression files) cannot be captured here +BENCHMARK_CAPTURE(FromBinaryBuffer, bson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bson); +BENCHMARK_CAPTURE(FromBinaryBuffer, bson / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::bson); +BENCHMARK_CAPTURE(FromBinaryBuffer, bson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bson); + +static void FromBinaryFile(benchmark::State& state, const char* filename, const binary_format format) +{ + const std::vector bytes = binary_input(state, filename, format); + if (bytes.empty()) + { + return; + } + + const char* tmp = "benchmark_input.bin"; + std::ofstream o(tmp, std::ios::binary); + o.write(reinterpret_cast(bytes.data()), static_cast(bytes.size())); + o.flush(); + o.close(); + + for (auto _ : state) + { + state.PauseTiming(); + auto* j = new json(); + auto* file = std::fopen(tmp, "rb"); + state.ResumeTiming(); + + *j = from_binary(file, format); + + state.PauseTiming(); + std::fclose(file); + delete j; + state.ResumeTiming(); + } + + state.SetBytesProcessed(state.iterations() * bytes.size()); +} + +BENCHMARK_CAPTURE(FromBinaryFile, cbor / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryFile, cbor / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryFile, ubjson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryFile, ubjson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryFile, bjdata / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata); +BENCHMARK_CAPTURE(FromBinaryFile, bson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bson); + +////////////////////////////////////////////////////////////////////////////// +// parse binary formats: value shapes +////////////////////////////////////////////////////////////////////////////// + +// The test files above are wide and shallow, but the readers' cost is per +// container, so these cover the shapes that stress the container handling +// itself. Every shape is wrapped in an object so that BSON, which requires an +// object at the top level, measures the same value as the other formats. + +/// deeply nested arrays: one container per level, no other work +static json make_nested() +{ + json nested = json::array(); + json* p = &nested; + for (std::size_t i = 1; i < 1000; ++i) + { + p->push_back(json::array()); + p = &p->operator[](0); + } + + json j = json::object(); + j["data"] = std::move(nested); + return j; +} + +/// many sibling containers: maximum container churn, minimum nesting +static json make_containers() +{ + json data = json::array(); + for (std::size_t i = 0; i < 100000; ++i) + { + data.push_back(json::array({1, 2})); + } + + json j = json::object(); + j["data"] = std::move(data); + return j; +} + +/// one flat array of numbers: the scalar decoding path, which must not move +static json make_scalars() +{ + json data = json::array(); + for (std::size_t i = 0; i < 1000000; ++i) + { + data.push_back(i); + } + + json j = json::object(); + j["data"] = std::move(data); + return j; +} + +static void FromBinaryShape(benchmark::State& state, json (*build)(), const binary_format format) +{ + const std::vector bytes = to_binary(build(), format); + + for (auto _ : state) + { + state.PauseTiming(); + auto* j = new json(); + state.ResumeTiming(); + + *j = from_binary(bytes, format); + + state.PauseTiming(); + delete j; + state.ResumeTiming(); + } + + state.SetBytesProcessed(state.iterations() * bytes.size()); +} + +BENCHMARK_CAPTURE(FromBinaryShape, nested / cbor, make_nested, binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryShape, nested / msgpack, make_nested, binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryShape, nested / ubjson, make_nested, binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryShape, nested / bjdata, make_nested, binary_format::bjdata); +BENCHMARK_CAPTURE(FromBinaryShape, nested / bson, make_nested, binary_format::bson); +BENCHMARK_CAPTURE(FromBinaryShape, containers / cbor, make_containers, binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryShape, containers / msgpack, make_containers, binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryShape, containers / ubjson, make_containers, binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryShape, containers / ubjson_optimized, make_containers, binary_format::ubjson_optimized); +BENCHMARK_CAPTURE(FromBinaryShape, containers / bjdata, make_containers, binary_format::bjdata); +BENCHMARK_CAPTURE(FromBinaryShape, containers / bson, make_containers, binary_format::bson); +// BSON names every array element, so a large array measures key generation +// rather than scalar decoding and is left out here +BENCHMARK_CAPTURE(FromBinaryShape, scalars / cbor, make_scalars, binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryShape, scalars / msgpack, make_scalars, binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryShape, scalars / ubjson, make_scalars, binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryShape, scalars / bjdata, make_scalars, binary_format::bjdata); + +/*! +@brief parse an indefinite-length CBOR string + +The writer never emits this form, so the input is assembled by hand: 0x7F +opens the string, each chunk is a one-character string, and 0xFF closes it. +*/ +static void FromCborChunkedString(benchmark::State& state, const std::size_t chunks) +{ + std::vector bytes; + bytes.reserve(2 * chunks + 2); + bytes.push_back(0x7F); + for (std::size_t i = 0; i < chunks; ++i) + { + bytes.push_back(0x61); // string of length 1 + bytes.push_back(0x61); // 'a' + } + bytes.push_back(0xFF); + + for (auto _ : state) + { + json j = json::from_cbor(bytes); + benchmark::DoNotOptimize(j); + } + + state.SetBytesProcessed(state.iterations() * bytes.size()); +} + +BENCHMARK_CAPTURE(FromCborChunkedString, 10000 chunks, 10000); + BENCHMARK_MAIN(); diff --git a/tests/fuzzing.md b/tests/fuzzing.md index cfbf4f249..b3bf90d5d 100644 --- a/tests/fuzzing.md +++ b/tests/fuzzing.md @@ -79,3 +79,26 @@ the same `fuzzers` target as above and also relies on the `FUZZER_ENGINE` variab [build script](https://github.com/google/oss-fuzz/blob/master/projects/json/build.sh) for more information. In case the build at OSS-Fuzz fails, an issue will be created automatically. + +### Handling OSS-Fuzz reports + +OSS-Fuzz files the crashes it finds in its own [issue tracker](https://issues.oss-fuzz.com), not on GitHub. So that +each report can be traced to the change that fixed it, and each fix to the report it answers, fixes follow these +conventions: + +- **Reference the OSS-Fuzz issue in the pull request**, next to any GitHub issue it closes, as `OSS-Fuzz: ` (for + example, `OSS-Fuzz: 563659413`), and in the commit message. The ID alone does not disclose the crash. If the report + was triaged into a GitHub issue, link the OSS-Fuzz issue there too. +- **Turn the reproducer into a unit test.** Download the testcase from the OSS-Fuzz report, reduce it if possible, and + add it as a regression test to the unit test of the affected format (e.g., `tests/src/unit-bjdata.cpp`), with a + comment naming the OSS-Fuzz issue. This way the input is checked by every CI run rather than only by OSS-Fuzz, and + it stays covered even if OSS-Fuzz later closes the report as not reproducible. +- **Keep the fuzzer drivers and the unit tests in sync.** The round-trip checks of the UBJSON and BJData drivers are + also run on a fixed corpus in the unit tests (see `tests/src/round_trip_corpus.hpp` and the "round-trip invariants" + test cases), so a regression shows up in CI first. When a driver's checks change, change the unit tests with them. +- **Record in the report whether the bug shipped.** OSS-Fuzz asks whether a crash was a short-lived regression or + affects a released version; answer it when the fix is merged, as it decides whether the fix needs a release note or + a security advisory (see the [security policy](../.github/SECURITY.md)). + +After the fix is merged, OSS-Fuzz re-runs the reproducer on its next build and marks the report as verified and +closed. If it does not, the fix is incomplete. diff --git a/tests/src/fuzzer-parse_bjdata.cpp b/tests/src/fuzzer-parse_bjdata.cpp index 1d1d56a5c..d3c9e7a33 100644 --- a/tests/src/fuzzer-parse_bjdata.cpp +++ b/tests/src/fuzzer-parse_bjdata.cpp @@ -21,16 +21,53 @@ array data, it performs the following steps: - j4 = from_bjdata(vec3) - assert(j1 == j4) +Re-serializing j2/j3/j4 with the same use_size/use_type settings is checked +for value-stability rather than byte-exact stability: from_bjdata(to_bjdata(j2)) +must equal j2 (and likewise for j3, j4). Byte-exact stability does not hold in +general, because a BJData value can lose type fidelity across a round trip +(e.g. a binary_t value serialized without the optimized "$U#" array header is +parsed back as a plain array of numbers, see #5398 and the discussion on +PR #5494) - the numeric value is preserved, but the writer's smallest-type +selection for the now-plain numbers may legitimately pick a different, but +equally valid, single-byte type marker than the dedicated binary-data writer +would have. Both encodings are valid BJData and both decode to the same +value, so this is not treated as a round-trip failure here. + +"Value-stable" is checked by comparing dump()s rather than with operator== +directly: a BJData/UBJSON payload can decode to a non-finite double (NaN or ++-Infinity), and IEEE 754 NaN is never equal to itself, so operator== would +report two structurally-identical trees as different whenever a NaN is +involved -- not a round-trip bug, just NaN's ordinary (non-)reflexivity. +dump() serializes any non-finite double the same deterministic way (as JSON +`null`, since JSON itself cannot represent NaN/Infinity), so comparing +dumps is stable under exactly the same values that break operator==. + +The unit tests run the same checks on a fixed corpus (see the "BJData round-trip +invariants" test case), so keep both in sync. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; +// value-stable comparison for the round-trip checks below; see the note +// above on why this compares dump()s rather than the json values directly +static bool is_value_stable(const json& lhs, const json& rhs) +{ + return lhs.dump() == rhs.dump(); +} + // see http://llvm.org/docs/LibFuzzer.html extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { @@ -56,10 +93,12 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) json const j3 = json::from_bjdata(vec3); json const j4 = json::from_bjdata(vec4); - // serializations must match - assert(json::to_bjdata(j2, false, false) == vec2); - assert(json::to_bjdata(j3, true, false) == vec3); - assert(json::to_bjdata(j4, true, true) == vec4); + // re-serializing must be value-stable (see the notes above on + // why byte-exact stability is not guaranteed in general, and + // why this compares dump()s rather than the values directly) + assert(is_value_stable(json::from_bjdata(json::to_bjdata(j2, false, false)), j2)); + assert(is_value_stable(json::from_bjdata(json::to_bjdata(j3, true, false)), j3)); + assert(is_value_stable(json::from_bjdata(json::to_bjdata(j4, true, true)), j4)); } catch (const json::parse_error&) { diff --git a/tests/src/fuzzer-parse_bson.cpp b/tests/src/fuzzer-parse_bson.cpp index c5f74c7cc..16f36445b 100644 --- a/tests/src/fuzzer-parse_bson.cpp +++ b/tests/src/fuzzer-parse_bson.cpp @@ -19,10 +19,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/fuzzer-parse_cbor.cpp b/tests/src/fuzzer-parse_cbor.cpp index b38e3c1e5..7d599abe2 100644 --- a/tests/src/fuzzer-parse_cbor.cpp +++ b/tests/src/fuzzer-parse_cbor.cpp @@ -19,10 +19,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/fuzzer-parse_json.cpp b/tests/src/fuzzer-parse_json.cpp index 59a278c7b..217d9e0fd 100644 --- a/tests/src/fuzzer-parse_json.cpp +++ b/tests/src/fuzzer-parse_json.cpp @@ -20,10 +20,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/fuzzer-parse_msgpack.cpp b/tests/src/fuzzer-parse_msgpack.cpp index 0b4ab0af7..df961b8d7 100644 --- a/tests/src/fuzzer-parse_msgpack.cpp +++ b/tests/src/fuzzer-parse_msgpack.cpp @@ -19,10 +19,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/fuzzer-parse_ubjson.cpp b/tests/src/fuzzer-parse_ubjson.cpp index 463656c71..ebf775b59 100644 --- a/tests/src/fuzzer-parse_ubjson.cpp +++ b/tests/src/fuzzer-parse_ubjson.cpp @@ -21,14 +21,23 @@ array data, it performs the following steps: - j4 = from_ubjson(vec3) - assert(j1 == j4) +The unit tests run the same checks on a fixed corpus (see the "UBJSON round-trip +invariants" test case), so keep both in sync. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/round_trip_corpus.hpp b/tests/src/round_trip_corpus.hpp new file mode 100644 index 000000000..41cdac3e6 --- /dev/null +++ b/tests/src/round_trip_corpus.hpp @@ -0,0 +1,213 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // nan +#include // size_t +#include // int32_t, int64_t, uint32_t, uint64_t +#include // numeric_limits +#include // mt19937 +#include // string, to_string +#include // move +#include // vector + +#include + +// Values for the round-trip property tests of the UBJSON and BJData writers. +// +// The fuzzer drivers (tests/src/fuzzer-parse_ubjson.cpp and +// fuzzer-parse_bjdata.cpp) check that anything the library parses can be +// serialized, parsed back, and serialized again without loss. Those checks +// only run at OSS-Fuzz, so a regression used to surface days later as an +// external report. The unit tests run the same checks on this corpus in CI. +// +// The corpus is deterministic: std::mt19937's output sequence is fixed by +// the standard, and it is used directly rather than through a distribution +// (whose results are implementation-defined). +namespace utils +{ + +class round_trip_corpus +{ + public: + using json = nlohmann::json; + + static std::vector values() + { + round_trip_corpus corpus; + return corpus.build(); + } + + // whether a value contains a binary value, which a BJData or UBJSON round + // trip may turn into an array of integers + static bool contains_binary(const json& j) + { + if (j.is_binary()) + { + return true; + } + if (j.is_structured()) + { + for (const auto& element : j) + { + if (contains_binary(element)) + { + return true; + } + } + } + return false; + } + + private: + std::vector atoms; + // a fixed seed is the point: the corpus must be the same in every run + std::mt19937 generator{42}; // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) + + round_trip_corpus() + : atoms + { + nullptr, true, false, + // integers at the boundaries of every UBJSON/BJData integer type + 0, 1, -1, 127, 128, 255, 256, -128, -129, + 32767, 32768, 65535, 65536, -32768, -32769, + (std::numeric_limits::min)(), (std::numeric_limits::max)(), + (std::numeric_limits::max)(), + (std::numeric_limits::min)(), (std::numeric_limits::max)(), + static_cast((std::numeric_limits::max)()) + 1u, + (std::numeric_limits::max)(), + // floating-point numbers, including non-finite ones + 0.0, -0.0, 1.5, -2.25, 3.4e38, (std::numeric_limits::max)(), + std::nan(""), std::numeric_limits::infinity(), -std::numeric_limits::infinity(), + // strings, including a non-ASCII one and one longer than 255 bytes + "", "a", "\xC3\xA4", std::string(300, 'x'), + // binary values with and without subtype + json::binary({}), json::binary({1, 2, 255}), json::binary({0x80, 0x7F}, 42), json::binary({1}, 0) + } + {} + + std::vector build() + { + std::vector result = atoms; + + // each atom inside containers, including homogeneous ones that the + // writers encode as optimized (typed) containers + result.emplace_back(json::array()); + result.emplace_back(json::object()); + for (const auto& atom : atoms) + { + result.push_back(json::array({atom})); + result.push_back(json::array({atom, atom, atom})); + result.push_back(json::array({json::array({atom})})); + result.push_back(json::object({{"key", atom}})); + } + result.push_back(json::array({1, 1.5})); + result.push_back(json::array({-1, 255})); + result.push_back(json::array({"a", "b"})); + + // deep, but well below any recursion or depth limit + json nested_array = 1; + json nested_object = 1; + for (int i = 0; i < 300; ++i) + { + nested_array = json::array({nested_array}); + nested_object = json::object({{"key", nested_object}}); + } + result.push_back(nested_array); + result.push_back(nested_object); + + add_annotated_arrays(result); + add_random_values(result); + return result; + } + + // objects in the JData annotated array format, which the BJData writer + // encodes as ND-arrays when the annotation describes a packed array, and + // as plain objects otherwise (see #5398, #5399, #5403, #5404, and #5542) + static void add_annotated_arrays(std::vector& result) + { + const std::vector types = + { + "uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", + "single", "double", "char", "byte", "bool", "unknown", 5, nullptr + }; + const std::vector sizes = + { + json::array(), {3}, {1, 3}, {3, 1}, {2, 3}, {2, 0}, {0, 2}, {2, 2, 2}, {-1, 2}, {2, 1.5}, + "3", 3, nullptr, json::binary({}) + }; + const std::vector data = + { + nullptr, 5, "s", json::object({{"a", 1}}), json::array(), + {1, 2, 3}, {1, 2, 3, 4, 5, 6}, {1, 2, 3, 4, 5, 6, 7, 8}, + {1.5, 2.5, 3.5, 4.5, 5.5, 6.5}, {300, -300, 70000, -70000, 1, 2}, + {"a", "b", "c", "d", "e", "f"}, {json::array({1, 2, 3}), json::array({4, 5, 6})} + }; + + for (const auto& type : types) + { + for (const auto& size : sizes) + { + for (const auto& d : data) + { + result.push_back({{"_ArrayType_", type}, {"_ArraySize_", size}, {"_ArrayData_", d}}); + } + } + } + + // incomplete annotations and annotations with an extra key + result.push_back({{"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}, {"extra", 1}}); + } + + // random containers of atoms, both homogeneous and mixed + void add_random_values(std::vector& result) + { + for (int i = 0; i < 1000; ++i) + { + result.push_back(random_value(0)); + } + } + + std::size_t random_below(std::size_t bound) + { + return generator() % bound; + } + + json random_value(int depth) + { + const auto kind = random_below(10); + if (depth > 3 || kind < 5) + { + return atoms[random_below(atoms.size())]; + } + + json result = kind < 8 ? json::array() : json::object(); + const auto count = random_below(5); + const bool homogeneous = random_below(2) == 0; + const json fixed = atoms[random_below(atoms.size())]; + for (std::size_t i = 0; i < count; ++i) + { + json element = homogeneous ? fixed : random_value(depth + 1); + if (result.is_array()) + { + result.push_back(std::move(element)); + } + else + { + result[std::to_string(i)] = std::move(element); + } + } + return result; + } +}; + +} // namespace utils diff --git a/tests/src/skip_library_version_check.cpp b/tests/src/skip_library_version_check.cpp new file mode 100644 index 000000000..ddaa4415c --- /dev/null +++ b/tests/src/skip_library_version_check.cpp @@ -0,0 +1,61 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// Standalone compile-and-run check for the JSON_SKIP_LIBRARY_VERSION_CHECK +// configuration macro, which (per #5423) was never exercised anywhere in the +// test matrix. +// +// include/nlohmann/detail/abi_macros.hpp normally emits a #warning if +// NLOHMANN_JSON_VERSION_MAJOR/MINOR/PATCH are already defined (as they would +// be by an earlier inclusion of a different version of the library) with +// values that mismatch the version about to be defined -- unless +// JSON_SKIP_LIBRARY_VERSION_CHECK is defined, in which case the check (and +// that #warning) is skipped. +// +// This file deliberately is not named tests/src/unit-*.cpp: it is compiled +// directly (with a modest, non-strict warning set) by the dedicated +// ci_test_skiplibraryversioncheck target in cmake/ci.cmake, rather than being +// folded into the library's own -Weverything/-Werror unit test matrix. That +// is because the scenario simulated here -- mixing two different, already +// differently-versioned inclusions of the library in one translation unit -- +// unavoidably also triggers the *compiler's own* "macro redefined" warning, +// independent of (and unaffected by) JSON_SKIP_LIBRARY_VERSION_CHECK, which +// only ever silences the library's own #warning. Building this file under +// -Weverything -Werror would therefore fail for a reason unrelated to the +// macro under test. +#define NLOHMANN_JSON_VERSION_MAJOR 0 +#define NLOHMANN_JSON_VERSION_MINOR 0 +#define NLOHMANN_JSON_VERSION_PATCH 0 + +#define JSON_SKIP_LIBRARY_VERSION_CHECK 1 + +#include + +int main() +{ + // reaching this point at all already proves that the mismatched, + // pre-defined version macros above did not stop compilation -- which is + // exactly what JSON_SKIP_LIBRARY_VERSION_CHECK is for. The library must + // also still be fully usable. + const nlohmann::json j = {{"a", 1}, {"b", {1, 2, 3}}}; + if (j.dump() != "{\"a\":1,\"b\":[1,2,3]}") + { + return 1; + } + + // include/nlohmann/detail/abi_macros.hpp unconditionally (re)defines the + // version macros to the library's real, current version right after the + // (here, skipped) mismatch check, regardless of the deliberately wrong + // stand-in values defined above. + if (NLOHMANN_JSON_VERSION_MAJOR == 0 && NLOHMANN_JSON_VERSION_MINOR == 0 && NLOHMANN_JSON_VERSION_PATCH == 0) + { + return 1; + } + + return 0; +} diff --git a/tests/src/test_utils.hpp b/tests/src/test_utils.hpp index baa802f71..4c81a8ef4 100644 --- a/tests/src/test_utils.hpp +++ b/tests/src/test_utils.hpp @@ -15,6 +15,15 @@ namespace utils { +// Some tests intentionally discard the [[nodiscard]]/JSON_HEDLEY_WARN_UNUSED_RESULT +// return value of a call they only make to exercise its side effects (e.g. checking +// that it does not throw). A plain (void) cast on the call expression does not +// suppress GCC's warning for functions using the GNU __attribute__((warn_unused_result)) +// form (as opposed to the C++17 [[nodiscard]] attribute) -- passing the value into an +// ordinary function call does. +template +inline void ignore_return_value(T&& /*unused*/) noexcept {} + inline std::vector read_binary_file(const std::string& filename) { std::ifstream file(filename, std::ios::binary); diff --git a/tests/src/unit-allocator.cpp b/tests/src/unit-allocator.cpp index 2dbb746b0..5c7b4230f 100644 --- a/tests/src/unit-allocator.cpp +++ b/tests/src/unit-allocator.cpp @@ -216,6 +216,57 @@ TEST_CASE("controlled bad_alloc") CHECK_THROWS_AS(my_json(s), std::bad_alloc&); next_construct_fails = false; } + + SECTION("basic_json(const basic_json&) of a deeply nested value (#5387)") + { + // Copying a value nested deeper than the descent bound builds the + // copy from the top down: every value whose own copy has not been + // made yet stays a null value until it is. Failing an allocation + // part-way through is what proves such a half-built copy can still + // be destroyed. + // + // Which path the failure lands in depends on the build: the first + // allocation of a copy belongs to the outermost level, so here it + // is the descending one. Built with JSON_NO_THREAD_LOCAL - as the + // ci_test_no_thread_local target builds the whole suite - no + // descent is made at all and the very same failure lands in the + // iterative path instead, part-way through its worklist. + const auto check_deep_copy = [](bool objects) + { + CAPTURE(objects); + + next_construct_fails = false; + + // deeper than the 128 levels the copy constructor descends into + const std::size_t depth = 300; + + my_json j = 1; + for (std::size_t i = 0; i < depth; ++i) + { + if (objects) + { + my_json wrapper = my_json::object(); + wrapper["a"] = std::move(j); + j = std::move(wrapper); + } + else + { + j = my_json::array({std::move(j)}); + } + } + + // NOLINTNEXTLINE(performance-unnecessary-copy-initialization): the copy is what is tested + CHECK_NOTHROW(my_json(j)); + + next_construct_fails = true; + // NOLINTNEXTLINE(performance-unnecessary-copy-initialization): the copy is what is tested + CHECK_THROWS_AS(my_json(j), std::bad_alloc&); + next_construct_fails = false; + }; + + check_deep_copy(false); + check_deep_copy(true); + } } } diff --git a/tests/src/unit-alt-string.cpp b/tests/src/unit-alt-string.cpp index 46e062c6e..ec2ac146b 100644 --- a/tests/src/unit-alt-string.cpp +++ b/tests/src/unit-alt-string.cpp @@ -11,8 +11,10 @@ #include +#include #include #include +#include /* forward declarations */ class alt_string; @@ -22,6 +24,10 @@ void int_to_string(alt_string& target, std::size_t value); // NOLINT(misc-use-in /* * This is virtually a string class. * It covers std::string under the hood. + * + * It deliberately does not provide c_str(), back(), find(str, pos), replace(), + * or substr(): the library must not rely on them. Do not add members here + * without checking that the library actually needs them. */ class alt_string { @@ -106,11 +112,6 @@ class alt_string return str_impl < op.str_impl; } - const char* c_str() const - { - return str_impl.c_str(); - } - char& operator[](std::size_t index) { return str_impl[index]; @@ -121,16 +122,6 @@ class alt_string return str_impl[index]; } - char& back() - { - return str_impl.back(); - } - - const char& back() const - { - return str_impl.back(); - } - void clear() { str_impl.clear(); @@ -146,28 +137,11 @@ class alt_string return str_impl.empty(); } - std::size_t find(const alt_string& str, std::size_t pos = 0) const - { - return str_impl.find(str.str_impl, pos); - } - std::size_t find_first_of(char c, std::size_t pos = 0) const { return str_impl.find_first_of(c, pos); } - alt_string substr(std::size_t pos = 0, std::size_t count = npos) const - { - const std::string s = str_impl.substr(pos, count); - return {s.data(), s.size()}; - } - - alt_string& replace(std::size_t pos, std::size_t count, const alt_string& str) - { - str_impl.replace(pos, count, str.str_impl); - return *this; - } - void reserve( std::size_t new_cap = 0 ) { str_impl.reserve(new_cap); @@ -202,6 +176,31 @@ bool operator<(const char* op1, const alt_string& op2) noexcept TEST_CASE("alternative string type") { + SECTION("binary formats") + { + alt_json doc; + doc["pi"] = 3.141; + doc["happy"] = true; + doc["list"] = {1, 2, 3}; + + CHECK(alt_json::from_cbor(alt_json::to_cbor(doc)) == doc); + CHECK(alt_json::from_msgpack(alt_json::to_msgpack(doc)) == doc); + // BSON is not covered: it additionally needs string_t::find(value_type), + // which alt_string does not provide + CHECK(alt_json::from_ubjson(alt_json::to_ubjson(doc)) == doc); + + // a UBJSON high-precision number is parsed into a std::string that the + // reader has to hand to the SAX interface as an alt_string + const std::vector high_precision = + { + 'H', 'i', 0x16, '3', '.', '1', '4', '1', '5', '9', '2', '6', '5', '3', + '5', '8', '9', '7', '9', '3', '2', '3', '8', '4', '6' + }; + const auto number = alt_json::from_ubjson(high_precision); + CHECK(number.is_number_float()); + CHECK(number.get() == doctest::Approx(3.14159265358979323846)); + } + SECTION("dump") { { @@ -332,6 +331,15 @@ TEST_CASE("alternative string type") CHECK(j.at(alt_json::json_pointer("/foo/0")) == j["foo"][0]); CHECK(j.at(alt_json::json_pointer("/foo/1")) == j["foo"][1]); + + // RFC 6901 escaping works without string_t::find(str, pos), replace(), + // and substr() + auto j2 = alt_json::parse(R"({"a/b": 1, "m~n": 2, "~/~~//": 3})"); + CHECK(j2.at(alt_json::json_pointer("/a~1b")) == 1); + CHECK(j2.at(alt_json::json_pointer("/m~0n")) == 2); + CHECK(j2.at(alt_json::json_pointer("/~0~1~0~0~1~1")) == 3); + CHECK(alt_json::json_pointer("/~0~1~0~0~1~1").to_string() == alt_string("/~0~1~0~0~1~1")); + CHECK(j2.flatten().unflatten() == j2); } SECTION("patch") diff --git a/tests/src/unit-binary_writer_sinks.cpp b/tests/src/unit-binary_writer_sinks.cpp new file mode 100644 index 000000000..f60e1bf51 --- /dev/null +++ b/tests/src/unit-binary_writer_sinks.cpp @@ -0,0 +1,198 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include +using nlohmann::json; + +#include +#include +#include + +namespace +{ + +// a spread of values exercising every writer path: scalars of each width, the +// float paths, strings, binary, and containers big enough to reallocate +std::vector test_values() +{ + json big_array = json::array(); + for (int i = 0; i < 5000; ++i) + { + big_array.push_back(i); + } + + json big_object = json::object(); + for (int i = 0; i < 1000; ++i) + { + big_object[std::to_string(i)] = i; + } + + return + { + json(nullptr), json(true), json(false), + json(0), json(-1), json(255), json(-129), json(65535), json(-32769), + json(4294967295U), json(-2147483649LL), json(18446744073709551615ULL), + json(0.0), json(-0.5), json(3.1415926535897932), + json(""), json("hello"), json(std::string(1000, 'x')), + json::binary({0x00, 0x01, 0x02}, 42), + json::array(), json::object(), + json::array({1, 2, 3}), json({{"a", 1}, {"b", nullptr}}), + json({{"nested", {{"deep", json::array({1, "two", 3.0, nullptr})}}}}), + big_array, big_object + }; +} + +// values to_bson() accepts: the document must be an object +std::vector bson_values() +{ + json big_object = json::object(); + for (int i = 0; i < 1000; ++i) + { + big_object[std::to_string(i)] = i; + } + + return + { + json::object(), + json({{"a", 1}, {"b", nullptr}, {"c", true}, {"d", 2.5}, {"e", "text"}}), + json({{"arr", json::array({1, 2, 3})}, {"obj", {{"k", "v"}}}}), + big_object + }; +} + +} // namespace + +// The vector-returning to_*(j) overloads write through the non-virtual +// output_vector_sink, while to_*(j, adapter) goes through output_adapter_sink. +// The two are separate code paths that must stay byte-for-byte identical; these +// checks fail if either overload is ever changed without the other. +TEST_CASE("binary writer output sinks") +{ + SECTION("vector sink and adapter sink agree") + { + // note: no SUBCASE inside these loops - doctest keys subcases by + // name/file/line, so a subcase in a loop body would only ever run for + // the first iteration + for (const auto& j : test_values()) + { + CAPTURE(j.dump(-1, ' ', false, json::error_handler_t::replace)); + + std::vector cbor; + json::to_cbor(j, cbor); + CHECK(json::to_cbor(j) == cbor); + + std::vector msgpack; + json::to_msgpack(j, msgpack); + CHECK(json::to_msgpack(j) == msgpack); + + for (const bool use_size : + { + false, true + }) + { + for (const bool use_type : + { + false, true + }) + { + if (use_type && !use_size) + { + continue; // not a supported combination + } + CAPTURE(use_size); + CAPTURE(use_type); + std::vector ubjson; + json::to_ubjson(j, ubjson, use_size, use_type); + CHECK(json::to_ubjson(j, use_size, use_type) == ubjson); + } + } + + for (const auto version : + { + json::bjdata_version_t::draft2, json::bjdata_version_t::draft3 + }) + { + std::vector bjdata; + json::to_bjdata(j, bjdata, false, false, version); + CHECK(json::to_bjdata(j, false, false, version) == bjdata); + } + } + + for (const auto& j : bson_values()) + { + CAPTURE(j.dump()); + std::vector bson; + json::to_bson(j, bson); + CHECK(json::to_bson(j) == bson); + } + } + + SECTION("the char adapter produces the same bytes") + { + for (const auto& j : test_values()) + { + CAPTURE(j.dump(-1, ' ', false, json::error_handler_t::replace)); + + const std::vector expected = json::to_cbor(j); + std::vector as_char; + json::to_cbor(j, as_char); + + REQUIRE(as_char.size() == expected.size()); + std::vector as_bytes; + as_bytes.reserve(as_char.size()); + for (const char c : as_char) + { + as_bytes.push_back(static_cast(c)); + } + CHECK(as_bytes == expected); + } + } +} + +// binary_reserve_hint() is documented as a *lower* bound on the serialized size, +// so that reserving it up front can never leave the returned vector holding +// capacity beyond what the value actually needs. +TEST_CASE("binary_reserve_hint never over-reserves") +{ + for (const auto& j : test_values()) + { + CAPTURE(j.dump(-1, ' ', false, json::error_handler_t::replace)); + + const std::size_t hint = nlohmann::detail::binary_reserve_hint(j); + + CHECK(hint <= json::to_cbor(j).size()); + CHECK(hint <= json::to_msgpack(j).size()); + CHECK(hint <= json::to_ubjson(j).size()); + CHECK(hint <= json::to_ubjson(j, true, true).size()); + CHECK(hint <= json::to_bjdata(j).size()); + } + + for (const auto& j : bson_values()) + { + CAPTURE(j.dump()); + CHECK(nlohmann::detail::binary_reserve_hint(j) <= json::to_bson(j).size()); + } + + SECTION("scalars get no hint") + { + CHECK(nlohmann::detail::binary_reserve_hint(json(nullptr)) == 0); + CHECK(nlohmann::detail::binary_reserve_hint(json(42)) == 0); + CHECK(nlohmann::detail::binary_reserve_hint(json("a string")) == 0); + CHECK(nlohmann::detail::binary_reserve_hint(json::binary({0x01})) == 0); + } + + SECTION("containers are hinted from their element count") + { + CHECK(nlohmann::detail::binary_reserve_hint(json::array()) == 1); + CHECK(nlohmann::detail::binary_reserve_hint(json::array({1, 2, 3})) == 4); + CHECK(nlohmann::detail::binary_reserve_hint(json::object()) == 1); + CHECK(nlohmann::detail::binary_reserve_hint(json({{"a", 1}, {"b", 2}})) == 5); + } +} diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index d5f9c3acd..ebbbbfaf6 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -19,6 +19,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "round_trip_corpus.hpp" #include "test_utils.hpp" namespace @@ -2586,7 +2587,12 @@ TEST_CASE("BJData") CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d); CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D); CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C); - CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true) == v_B); + // v_B uses the Draft-3-only 'B' marker, so it round-trips only when + // Draft 3 is explicitly selected (see GitHub issue #5404); the + // default Draft 2 falls back to a plain object instead, covered by + // the "ndarray with _ArrayType_ "byte" is gated by the BJData draft + // version" section below + CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true, json::bjdata_version_t::draft3) == v_B); } SECTION("ndarray with data not matching _ArrayType_ is written as an object") @@ -2599,25 +2605,25 @@ TEST_CASE("BJData") // that still round-trips. // string data declared as a uint64 array - json const j_str = json({{"_ArrayType_", "uint64"}, {"_ArraySize_", {1}}, {"_ArrayData_", {"pointer"}}}); + json const j_str = json({{"_ArrayType_", "uint64"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {"pointer", "value"}}}); const auto out_str = json::to_bjdata(j_str); CHECK(out_str.at(0) == '{'); CHECK(json::from_bjdata(out_str) == j_str); // integer data declared as a double array - json const j_float = json({{"_ArrayType_", "double"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + json const j_float = json({{"_ArrayType_", "double"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, 2}}}); const auto out_float = json::to_bjdata(j_float); CHECK(out_float.at(0) == '{'); CHECK(json::from_bjdata(out_float) == j_float); // a non-integer shape entry is likewise not treated as an ndarray - json const j_size = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {"x"}}, {"_ArrayData_", {1}}}); + json const j_size = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {"x", 1}}, {"_ArrayData_", {1}}}); const auto out_size = json::to_bjdata(j_size); CHECK(out_size.at(0) == '{'); CHECK(json::from_bjdata(out_size) == j_size); // a negative shape entry is not a usable dimension either - json const j_neg = json::parse(R"({"_ArrayType_":"uint8","_ArraySize_":[-1],"_ArrayData_":[1]})"); + json const j_neg = json::parse(R"({"_ArrayType_":"uint8","_ArraySize_":[-1,1],"_ArrayData_":[1]})"); const auto out_neg = json::to_bjdata(j_neg); CHECK(out_neg.at(0) == '{'); CHECK(json::from_bjdata(out_neg) == j_neg); @@ -2629,8 +2635,10 @@ TEST_CASE("BJData") // the C++ API stores an int literal as number_integer, so _ArrayType_ // names the wire type rather than the storage. Both storages have to // produce the same typed array for every type. + // "byte" is checked separately below since it additionally requires + // BJData Draft 3 to be selected explicitly (see GitHub issue #5404). for (const char* type : - {"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char", "byte" + {"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char" }) { CAPTURE(type); @@ -2641,15 +2649,23 @@ TEST_CASE("BJData") CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}))); } + { + const std::string text = R"({"_ArrayType_":"byte","_ArraySize_":[2,3],"_ArrayData_":[1,2,3,4,5,6]})"; + const auto from_text = json::to_bjdata(json::parse(text), true, true, json::bjdata_version_t::draft3); + CHECK(from_text.at(0) == '['); + CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}), + true, true, json::bjdata_version_t::draft3)); + } + // negative values under a signed type behave the same way - const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})")); + const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2,1],"_ArrayData_":[-5,7]})")); CHECK(from_neg.at(0) == '['); - CHECK(from_neg == json::to_bjdata(json({{"_ArrayType_", "int32"}, {"_ArraySize_", {2}}, {"_ArrayData_", {-5, 7}}}))); + CHECK(from_neg == json::to_bjdata(json({{"_ArrayType_", "int32"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {-5, 7}}}))); // and so do the floating point types - const auto from_float = json::to_bjdata(json::parse(R"({"_ArrayType_":"double","_ArraySize_":[2],"_ArrayData_":[1.5,2.5]})")); + const auto from_float = json::to_bjdata(json::parse(R"({"_ArrayType_":"double","_ArraySize_":[2,1],"_ArrayData_":[1.5,2.5]})")); CHECK(from_float.at(0) == '['); - CHECK(from_float == json::to_bjdata(json({{"_ArrayType_", "double"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1.5, 2.5}}}))); + CHECK(from_float == json::to_bjdata(json({{"_ArrayType_", "double"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 2.5}}}))); } SECTION("optimized ndarray (type and vector-size as 1D array)") @@ -2731,6 +2747,83 @@ TEST_CASE("BJData") CHECK(json::from_bjdata(json::to_bjdata(j_size), true, true) == j_size); } + SECTION("ndarray whose _ArrayType_ is not a string stays as object") + { + // the type name is looked up as a string below the annotation + // check; a non-string _ArrayType_ cannot name a known dtype, + // so calling get() on it would throw type_error.302 + // instead of falling back like an unrecognized type name + // already does (see GitHub issue #5398) + json const j_number = json({{"_ArrayType_", 1}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_number = json::to_bjdata(j_number); + CHECK(out_number.at(0) == '{'); + CHECK(json::from_bjdata(out_number) == j_number); + + json const j_null = json({{"_ArrayType_", nullptr}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_null = json::to_bjdata(j_null); + CHECK(out_null.at(0) == '{'); + CHECK(json::from_bjdata(out_null) == j_null); + + json const j_bool = json({{"_ArrayType_", true}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_bool = json::to_bjdata(j_bool); + CHECK(out_bool.at(0) == '{'); + CHECK(json::from_bjdata(out_bool) == j_bool); + + json const j_array = json({{"_ArrayType_", {"uint8"}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_array = json::to_bjdata(j_array); + CHECK(out_array.at(0) == '{'); + CHECK(json::from_bjdata(out_array) == j_array); + + json const j_object = json({{"_ArrayType_", {{"a", 1}}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_object = json::to_bjdata(j_object); + CHECK(out_object.at(0) == '{'); + CHECK(json::from_bjdata(out_object) == j_object); + } + + SECTION("re-serializing a value containing a plain-array-of-bytes is value-stable but not byte-stable") + { + // OSS-Fuzz found this input (an array whose first element is a + // binary_t byte, followed by an object whose _ArrayType_ is + // not a string) while exercising the fix for #5398 above: once + // the fix stops to_bjdata() from throwing type_error.302 for + // the third element, serialization proceeds far enough to + // reach a pre-existing, unrelated round-trip quirk in how a + // single-byte binary_t value is re-encoded. + std::vector const input + { + 0x5b, 0x5b, 0x24, 0x42, 0x23, 0x5b, 0x69, 0x01, 0x5d, 0x5b, 0x5b, 0x5d, 0x7b, 0x55, 0x0b, + 0x5f, 0x41, 0x72, 0x72, 0x61, 0x79, 0x44, 0x61, 0x74, 0x61, 0x5f, 0x54, 0x55, 0x0b, 0x5f, + 0x41, 0x72, 0x72, 0x61, 0x79, 0x53, 0x69, 0x7a, 0x65, 0x5f, 0x5a, 0x55, 0x0b, 0x5f, 0x41, + 0x72, 0x72, 0x61, 0x79, 0x54, 0x79, 0x70, 0x65, 0x5f, 0x54, 0x7d, 0x5d + }; + json const j1 = json::from_bjdata(input); + + // to_bjdata() must not throw (this is what #5398 fixes) + std::vector vec2; + CHECK_NOTHROW(vec2 = json::to_bjdata(j1, false, false)); + + // parsing back a plain (non-optimized) array of bytes cannot + // recover that it used to be a binary_t: from_bjdata() has no + // way to distinguish "array of uint8 numbers" from "array of + // bytes" unless the compact "$U#" array header is used, so + // the binary_t collapses into a plain JSON array + json const j2 = json::from_bjdata(vec2); + CHECK(j1 != j2); + CHECK(j2 == json({{91}, json::array(), {{"_ArrayData_", true}, {"_ArraySize_", nullptr}, {"_ArrayType_", true}}})); + + // re-serializing j2 no longer goes through the dedicated + // binary_t writer (which always uses the 'U' marker for raw + // bytes); the now-plain number 91 goes through the generic + // smallest-type writer instead, which - like the rest of the + // UBJSON/BJData writer, and unchanged by this fix - prefers + // the 'i' (int8) marker over 'U' (uint8) for values that fit + // both. Both markers are valid BJData and both decode back to + // 91, so this is not byte-for-byte identical to vec2, but it + // is value-stable: parsing it again reproduces j2 exactly. + std::vector const vec3 = json::to_bjdata(j2, false, false); + CHECK(json::from_bjdata(vec3) == j2); + } + SECTION("ndarray whose dimensions overflow stays as object") { // the product of the dimensions wraps around std::size_t to 0 @@ -2743,7 +2836,7 @@ TEST_CASE("BJData") // a single dimension that does not fit into std::size_t is // rejected for the same reason (only observable where // std::size_t is narrower than 64 bit) - json j_huge = json({{"_ArrayData_", json::array()}, {"_ArraySize_", {18446744073709551615ull}}, {"_ArrayType_", "uint8"}}); + json j_huge = json({{"_ArrayData_", json::array()}, {"_ArraySize_", {18446744073709551615ull, 2}}, {"_ArrayType_", "uint8"}}); CHECK(json::from_bjdata(json::to_bjdata(j_huge), true, true) == j_huge); // a well-formed ndarray is still encoded as one @@ -2751,6 +2844,197 @@ TEST_CASE("BJData") CHECK(json::to_bjdata(j_ok) == std::vector({'[', '$', 'U', '#', '[', 'i', 2, 'i', 3, ']', 1, 2, 3, 4, 5, 6})); CHECK(json::from_bjdata(json::to_bjdata(j_ok), true, true) == j_ok); } + + SECTION("ndarray whose _ArraySize_ is not an array stays as object") + { + // the shape is written verbatim as the header length, so a + // value that is not an array cannot produce a valid one: null + // would emit 'Z' and an object '{', neither of which a reader + // accepts after '#'. Both have to stay plain objects. + json const j_null = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", nullptr}, {"_ArrayData_", json::array()}}); + const auto out_null = json::to_bjdata(j_null); + CHECK(out_null.at(0) == '{'); + CHECK(json::from_bjdata(out_null) == j_null); + + // an object shape passes the per-entry check by iterating its + // values rather than dimensions, so it needs rejecting too + json const j_obj = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {{"a", 1}}}, {"_ArrayData_", {1}}}); + const auto out_obj = json::to_bjdata(j_obj); + CHECK(out_obj.at(0) == '{'); + CHECK(json::from_bjdata(out_obj) == j_obj); + + // a scalar shape is not a dimension list either + json const j_num = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", 1}, {"_ArrayData_", {1}}}); + const auto out_num = json::to_bjdata(j_num); + CHECK(out_num.at(0) == '{'); + CHECK(json::from_bjdata(out_num) == j_num); + + // OSS-Fuzz issue 474400817: an empty object _ArraySize_ was + // written as the ND-array header length, which from_bjdata() + // could not read back + const std::vector input = + { + '[', '{', 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'D', 'a', 't', 'a', '_', 'Z', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'T', 'y', 'p', 'e', '_', 'S', 'i', 5, 'i', 'n', 't', '1', '6', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'S', 'i', 'z', 'e', '_', '{', '}', '}', ']' + }; + const json j1 = json::from_bjdata(input); + CHECK(j1 == json::parse(R"([{"_ArrayType_":"int16","_ArraySize_":{},"_ArrayData_":null}])")); + json j2; + CHECK_NOTHROW(j2 = json::from_bjdata(json::to_bjdata(j1, false, false))); + CHECK(j2 == j1); + } + + SECTION("ndarray with out-of-range _ArrayData_ elements stays as object") + { + // each element is cast to the (possibly narrower) C++ type + // named by _ArrayType_ before being written; a value that + // does not fit that type would silently wrap instead of + // being reported, so such an object falls back to a plain + // object encoding that still round-trips (see GitHub issue #5403) + + // an unsigned element that does not fit uint8 + json const j_uint8 = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, 256}}}); + const auto out_uint8 = json::to_bjdata(j_uint8); + CHECK(out_uint8.at(0) == '{'); + CHECK(json::from_bjdata(out_uint8) == j_uint8); + + // a signed element that does not fit int8 + json const j_int8 = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, 200}}}); + const auto out_int8 = json::to_bjdata(j_int8); + CHECK(out_int8.at(0) == '{'); + CHECK(json::from_bjdata(out_int8) == j_int8); + + // a negative element is likewise out of range for an + // unsigned _ArrayType_ + json const j_uint16_neg = json({{"_ArrayType_", "uint16"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, -1}}}); + const auto out_uint16_neg = json::to_bjdata(j_uint16_neg); + CHECK(out_uint16_neg.at(0) == '{'); + CHECK(json::from_bjdata(out_uint16_neg) == j_uint16_neg); + + // a double element that overflows to infinity when narrowed + // to the "single" (float) precision named by _ArrayType_ + json const j_single = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 1e40}}}); + const auto out_single = json::to_bjdata(j_single); + CHECK(out_single.at(0) == '{'); + CHECK(json::from_bjdata(out_single) == j_single); + + // in-range boundary values still use the compact ndarray encoding + json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {0, 255}}}); + CHECK(json::to_bjdata(j_uint8_ok) == std::vector({'[', '$', 'U', '#', '[', 'i', 2, 'i', 1, ']', 0, 255})); + + json const j_int8_ok = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {-128, 127}}}); + CHECK(json::to_bjdata(j_int8_ok) == std::vector({'[', '$', 'i', '#', '[', 'i', 2, 'i', 1, ']', 0x80, 0x7F})); + + json const j_single_ok = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, -1.5}}}); + const auto out_single_ok = json::to_bjdata(j_single_ok); + CHECK(out_single_ok.at(0) == '['); + CHECK(json::from_bjdata(out_single_ok) == json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5f, -1.5f}}})); + } + + SECTION("ndarray that would not be read back as an annotated object stays as object") + { + // the reader only restores an annotated object from an ND-array + // with at least two non-zero dimensions that is not a 1xN row + // vector; any other shape is read back as a plain array. Writing + // such an object as an ND-array would drop its annotation, so it + // falls back to a plain object encoding that round-trips. + for (const char* text : + { + R"({"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":[]})", + R"({"_ArrayType_":"int16","_ArraySize_":[2],"_ArrayData_":[1,2]})", + R"({"_ArrayType_":"int16","_ArraySize_":[1,2],"_ArrayData_":[1,2]})", + R"({"_ArrayType_":"int16","_ArraySize_":[0],"_ArrayData_":[]})", + R"({"_ArrayType_":"int16","_ArraySize_":[2,0],"_ArrayData_":[]})", + R"({"_ArrayType_":"int16","_ArraySize_":[0,2],"_ArrayData_":[]})" + }) + { + CAPTURE(text); + const json j = json::parse(text); + for (const bool use_size : + { + false, true + }) + { + const auto out = json::to_bjdata(j, use_size, use_size); + CHECK(out.at(0) == '{'); + CHECK(json::from_bjdata(out) == j); + } + } + + // a genuine ND-array still uses the compact encoding and round-trips + const json j_2d = json::parse(R"({"_ArrayType_":"int16","_ArraySize_":[2,1],"_ArrayData_":[1,2]})"); + const auto out_2d = json::to_bjdata(j_2d); + CHECK(out_2d.at(0) == '['); + CHECK(json::from_bjdata(out_2d) == j_2d); + } + + SECTION("ndarray with non-array _ArrayData_ stays as object") + { + // the elements are written from _ArrayData_ as a flat list, so it + // has to be an array: null has size 0, any other scalar has size 1, + // and iterating an object visits its values, so each of these could + // match the dimensions and be encoded as an unrelated ND-array + for (const char* text : + { + R"({"_ArrayType_":"int16","_ArraySize_":[2,1],"_ArrayData_":null})", + R"({"_ArrayType_":"int16","_ArraySize_":[2,1],"_ArrayData_":{"a":1,"b":2}})", + R"({"_ArrayType_":"int16","_ArraySize_":[1],"_ArrayData_":5})", + R"({"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":null})" + }) + { + CAPTURE(text); + const json j = json::parse(text); + const auto out = json::to_bjdata(j); + CHECK(out.at(0) == '{'); + CHECK(json::from_bjdata(out) == j); + } + + // OSS-Fuzz issue 563659413: an empty binary _ArraySize_ is written + // as a plain object and read back as an empty array, after which + // the object with a null _ArrayData_ was encoded as an empty + // ND-array and re-read as [], so a second round trip lost the value + const std::vector input = + { + '{', 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'D', 'a', 't', 'a', '_', 'Z', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'T', 'y', 'p', 'e', '_', 'S', 'i', 5, 'i', 'n', 't', '1', '6', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'S', 'i', 'z', 'e', '_', '[', '$', 'B', '#', '[', ']', '}' + }; + const json j1 = json::from_bjdata(input); + const json j2 = json::from_bjdata(json::to_bjdata(j1, false, false)); + CHECK(j2 == json::parse(R"({"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":null})")); + CHECK(json::from_bjdata(json::to_bjdata(j2, false, false)) == j2); + } + + SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version") + { + // the 'B' (byte) marker used by _ArrayType_ "byte" is only defined + // by BJData Draft 3; Draft 2 (the default) has no such marker, so + // emitting it unconditionally produced a stream that a Draft 2 + // reader could not parse as intended (see GitHub issue #5404). + // Two dimensions are used so that a successfully written ndarray + // round-trips back into the annotated object (a single dimension + // is, by the BJData ndarray convention, read back as a plain + // binary value rather than the annotated object, same as every + // other single-dimension ndarray of a non-"byte" type is read + // back as a plain array instead of the annotated object). + json const j_byte = json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + + // default (Draft 2): falls back to a plain object and round-trips + const auto out_draft2 = json::to_bjdata(j_byte); + CHECK(out_draft2.at(0) == '{'); + CHECK(json::from_bjdata(out_draft2) == j_byte); + + // explicit Draft 2: same as the default + const auto out_draft2_explicit = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft2); + CHECK(out_draft2_explicit.at(0) == '{'); + CHECK(json::from_bjdata(out_draft2_explicit) == j_byte); + + // Draft 3 explicitly selected: still uses the compact 'B' ndarray encoding + const auto out_draft3 = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft3); + CHECK(out_draft3 == std::vector({'[', '$', 'B', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6})); + CHECK(json::from_bjdata(out_draft3) == j_byte); + } } } @@ -3263,8 +3547,10 @@ TEST_CASE("BJData") CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR1), "[json.exception.parse_error.113] parse error at byte 6: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&); CHECK(json::from_bjdata(vR1, true, false).is_discarded()); + // a dimension vector that opens another one is rejected where the + // nested '[' is read, rather than after it has been descended into std::vector const vR2 = {'[', '$', 'i', '#', '[', '#', '[', 'i', 1, ']', ']', 1}; - CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR2), "[json.exception.parse_error.113] parse error at byte 11: syntax error while parsing BJData size: expected length type specification (U, i, u, I, m, l, M, L) after '#'; last byte: 0x5D", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR2), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&); CHECK(json::from_bjdata(vR2, true, false).is_discarded()); std::vector const vR3 = {'[', '#', '[', 'i', '2', 'i', 2, ']'}; @@ -3272,7 +3558,7 @@ TEST_CASE("BJData") CHECK(json::from_bjdata(vR3, true, false).is_discarded()); std::vector const vR4 = {'[', '$', 'i', '#', '[', '$', 'i', '#', '[', 'i', 1, ']', 1}; - CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR4), "[json.exception.parse_error.110] parse error at byte 14: syntax error while parsing BJData number: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR4), "[json.exception.parse_error.113] parse error at byte 9: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&); CHECK(json::from_bjdata(vR4, true, false).is_discarded()); std::vector const vR5 = {'[', '$', 'i', '#', '[', '[', '[', ']', ']', ']'}; @@ -3280,12 +3566,25 @@ TEST_CASE("BJData") CHECK(json::from_bjdata(vR5, true, false).is_discarded()); std::vector const vR6 = {'[', '$', 'i', '#', '[', '$', 'i', '#', '[', 'i', '2', 'i', 2, ']'}; - CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR6), "[json.exception.parse_error.112] parse error at byte 14: syntax error while parsing BJData size: ndarray can not be recursive", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vR6), "[json.exception.parse_error.113] parse error at byte 9: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&); CHECK(json::from_bjdata(vR6, true, false).is_discarded()); std::vector const vH = {'[', 'H', '[', '#', '[', '$', 'i', '#', '[', 'i', '2', 'i', 2, ']'}; CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vH), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&); CHECK(json::from_bjdata(vH, true, false).is_discarded()); + + // Every "#[" of this chain used to open another dimension vector + // and cost several stack frames before anything was rejected, so a + // long enough chain crashed the process (see #5104). The nested + // vector is refused where it is read, so the length is irrelevant. + std::vector vRdeep = {'['}; + for (std::size_t i = 0; i < 100000; ++i) + { + vRdeep.push_back('#'); + vRdeep.push_back('['); + } + CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vRdeep), "[json.exception.parse_error.113] parse error at byte 5: syntax error while parsing BJData size: ndarray dimensional vector is not allowed", json::parse_error&); + CHECK(json::from_bjdata(vRdeep, true, false).is_discarded()); } SECTION("objects") @@ -3464,6 +3763,111 @@ TEST_CASE("BJData") } } +TEST_CASE("issue #5405 - array reserve for definite-length BJData arrays") +{ +#if !defined(JSON_NOEXCEPTION) + // this SECTION relies on catching a thrown exception to distinguish + // which of two acceptable, bounded rejections a hostile header took; + // under JSON_NOEXCEPTION, JSON_THROW never produces a catchable C++ + // exception (it aborts instead), so this cannot be tested that way here + SECTION("a huge claimed length with no element data must not over-allocate") + { + // optimized form [$type#count: type 'i' (int8), count as a four-byte + // little-endian 'l' (int32) of 0x7FFFFFFF (2147483647), but no + // element data at all. max_size() for a std::vector is far larger + // than this count, so it does not reject the header outright; the + // (capped) reservation must not attempt to allocate space for + // billions of elements before the missing data is detected. + json _; + const std::vector input = {'[', '$', 'i', '#', 'l', 0xFF, 0xFF, 0xFF, 0x7F}; + // On a platform where std::vector::max_size() is smaller than + // the claimed count (e.g. 32-bit, where max_size() is bounded by a + // 32-bit SIZE_MAX divided by sizeof(json)), the SAX consumer's own + // check rejects the header outright (out_of_range.408, with the + // claimed count in the message) instead of accepting it and only + // finding it short of data once the (capped) reservation looks for + // element bytes that were never provided (parse_error.110). Either + // is an acceptable, bounded rejection of the hostile header -- the + // property under test is that no path attempts to allocate space + // for billions of elements. + bool threw = false; + try + { + _ = json::from_bjdata(input); + } + catch (const json::parse_error& e) + { + threw = true; + CHECK(e.id == 110); + CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 10: syntax error while parsing BJData number: unexpected end of input"); + } + catch (const json::out_of_range& e) + { + threw = true; + CHECK(e.id == 408); + CHECK(std::string(e.what()).find("excessive array size") != std::string::npos); + } + CHECK(threw); + + // json_sax_dom_parser::start_array()'s max_size() check (unlike the + // scanner's own parse_error path) throws unconditionally via + // JSON_THROW rather than going through sax->parse_error(), so it is + // not gated by allow_exceptions=false on a platform where this + // header hits that check (e.g. 32-bit, see above) -- allow either + // a discarded result or the same out_of_range it throws with + // exceptions enabled. + try + { + CHECK(json::from_bjdata(input, true, false).is_discarded()); + } + catch (const json::out_of_range& e) + { + CHECK(e.id == 408); + } + } +#endif + + SECTION("arrays of various sizes decode to the same value as before the reserve optimization") + { + for (const auto size : + { + std::size_t{0}, std::size_t{1}, std::size_t{5}, // small + std::size_t{16384}, // exactly at the reserve cap + std::size_t{20000} // above the reserve cap + }) + { + CAPTURE(size) + json j = json::array(); + for (std::size_t i = 0; i < size; ++i) + { + j.push_back(static_cast(i % 1000)); + } + + // exercise both the plain and the optimized [$type#count encoding + const auto packed_plain = json::to_bjdata(j); + CHECK(json::from_bjdata(packed_plain) == j); + + const auto packed_optimized = json::to_bjdata(j, true, true); + CHECK(json::from_bjdata(packed_optimized) == j); + } + } + + SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization") + { + // the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser; + // a custom SAX consumer that does not touch a DOM array sees identical events + json j = json::array(); + for (int i = 0; i < 100; ++i) + { + j.push_back(i); + } + const auto packed = json::to_bjdata(j, true, true); + + SaxCountdown scp(1000000); // large enough to never trigger an abort + CHECK(json::sax_parse(packed, &scp, json::input_format_t::bjdata)); + } +} + TEST_CASE("Universal Binary JSON Specification Examples 1") { SECTION("Null Value") @@ -3843,6 +4247,135 @@ TEST_CASE("all BJData first bytes") } #endif +TEST_CASE("BJData use_type requires use_size") +{ + SECTION("non-empty object throws other_error.502") + { + const json j = {{"a", 1}, {"b", 2}}; + CHECK_THROWS_WITH_AS(json::to_bjdata(j, false, true), + "[json.exception.other_error.502] use_type requires use_size = true", + json::other_error&); + } + + SECTION("non-empty array throws other_error.502") + { + const json j = {1, 2, 3}; + CHECK_THROWS_WITH_AS(json::to_bjdata(j, false, true), + "[json.exception.other_error.502] use_type requires use_size = true", + json::other_error&); + } + + SECTION("scalars do not throw with use_type=true, use_count=false") + { + CHECK_NOTHROW(json::to_bjdata(42, false, true)); + CHECK_NOTHROW(json::to_bjdata(3.14, false, true)); + CHECK_NOTHROW(json::to_bjdata("hello", false, true)); + CHECK_NOTHROW(json::to_bjdata(true, false, true)); + CHECK_NOTHROW(json::to_bjdata(nullptr, false, true)); + } + + SECTION("empty containers do not throw with use_type=true, use_count=false") + { + CHECK_NOTHROW(json::to_bjdata(json::array(), false, true)); + CHECK_NOTHROW(json::to_bjdata(json::object(), false, true)); + } + + SECTION("valid combinations on non-empty containers") + { + const json j = {{"a", 1}, {"b", 2}}; + CHECK_NOTHROW(json::to_bjdata(j, false, false)); + CHECK_NOTHROW(json::to_bjdata(j, true, false)); + CHECK_NOTHROW(json::to_bjdata(j, true, true)); + } +} + +TEST_CASE("BJData round-trip invariants") +{ + // This checks what the parse_bjdata_fuzzer driver checks (see + // tests/src/fuzzer-parse_bjdata.cpp), so that a regression shows up in CI + // rather than as an OSS-Fuzz report: every value from_bjdata() returns + // (j1) can be serialized with any combination of options, the result can + // be parsed back (j2), and serializing j2 again with the same options + // yields a value-equal result. + // + // Beyond the driver, this also checks that j2 equals j1 and that + // serializing j2 reproduces the exact bytes, both except for values that + // contain a binary value: a binary value is only written as a binary + // value with Draft 3's optimized binary array, and otherwise read back as + // an array of integers, for which the writer may choose different (but + // equally valid) type markers when it is serialized again (see #5494). + // + // Values are compared with dump() rather than operator==, because a NaN + // never compares equal to itself. + struct options + { + bool use_size; + bool use_type; + json::bjdata_version_t version; + }; + const std::vector all_options = + { + {false, false, json::bjdata_version_t::draft2}, + {true, false, json::bjdata_version_t::draft2}, + {true, true, json::bjdata_version_t::draft2}, + {false, false, json::bjdata_version_t::draft3}, + {true, false, json::bjdata_version_t::draft3}, + {true, true, json::bjdata_version_t::draft3}, + }; + + for (const auto& j0 : utils::round_trip_corpus::values()) + { + // turn the corpus value into a value as from_bjdata() returns it + for (const auto& initial : all_options) + { + const json j1 = json::from_bjdata(json::to_bjdata(j0, initial.use_size, initial.use_type, initial.version)); + const bool has_binary = utils::round_trip_corpus::contains_binary(j1); + + for (const auto& o : all_options) + { + INFO("j1 = " << j1.dump() << ", use_size = " << o.use_size << ", use_type = " << o.use_type + << ", draft3 = " << (o.version == json::bjdata_version_t::draft3)); + + const std::vector vec = json::to_bjdata(j1, o.use_size, o.use_type, o.version); + json j2; + // anything the library writes must be parsable by the library + REQUIRE_NOTHROW(j2 = json::from_bjdata(vec)); + const std::vector vec2 = json::to_bjdata(j2, o.use_size, o.use_type, o.version); + CHECK(json::from_bjdata(vec2).dump() == j2.dump()); + + if (!has_binary) + { + CHECK(j2.dump() == j1.dump()); + CHECK(vec2 == vec); + } + } + } + } +} + +TEST_CASE("BJData round trip of a binary value is value-stable, not byte-stable") +{ + // OSS-Fuzz issue 474480402: a Draft 3 optimized binary array is read as a + // binary value, which to_bjdata() writes in the default Draft 2 mode as a + // plain array of uint8 numbers. That is read back as an array of numbers, + // for which the writer then picks the smallest type marker, int8 ('i'), + // so re-serializing changes the bytes, but not the value. This is the + // exception described in the "Round trips" note of the BJData + // documentation, and why the fuzzer checks value stability (see #5494). + const std::vector input = {'[', '$', 'B', '#', 'U', 1, 0x20}; + const json j1 = json::from_bjdata(input); + CHECK(j1 == json::binary({0x20})); + + const std::vector vec = json::to_bjdata(j1, false, false); + CHECK(vec == std::vector({'[', 'U', 0x20, ']'})); + const json j2 = json::from_bjdata(vec); + CHECK(j2 == json::array({0x20})); + + const std::vector vec2 = json::to_bjdata(j2, false, false); + CHECK(vec2 == std::vector({'[', 'i', 0x20, ']'})); + CHECK(json::from_bjdata(vec2) == j2); +} + TEST_CASE("BJData roundtrips" * doctest::skip()) { SECTION("input from self-generated BJData files") diff --git a/tests/src/unit-brace-init-copy-semantics.cpp b/tests/src/unit-brace-init-copy-semantics.cpp new file mode 100644 index 000000000..1ee0c6607 --- /dev/null +++ b/tests/src/unit-brace-init-copy-semantics.cpp @@ -0,0 +1,167 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// This file tests the opt-in JSON_BRACE_INIT_COPY_SEMANTICS, so it defines the +// macro itself rather than relying on a -D flag, and runs in every build. +#ifdef JSON_BRACE_INIT_COPY_SEMANTICS + #undef JSON_BRACE_INIT_COPY_SEMANTICS +#endif + +#define JSON_BRACE_INIT_COPY_SEMANTICS 1 + +#include +using nlohmann::json; + +#include +#include +#include +#include +#include +#include +#include + +#define STRINGIZE_EX(x) #x +#define STRINGIZE(x) STRINGIZE_EX(x) + +TEST_CASE("JSON_BRACE_INIT_COPY_SEMANTICS") +{ + SECTION("the macro is part of the ABI tag") + { + const std::string ns = STRINGIZE(NLOHMANN_JSON_NAMESPACE); + // other tags may come before it, e.g. json_abi_ldvcmp_bics + CHECK(ns.find("_bics") != std::string::npos); + } + + SECTION("single-element brace initialization copies the element (#5074)") + { + json const j_obj = {{"key", "value"}, {"num", 42}}; + json const j_arr = {1, 2, 3}; + + // object: brace init copies instead of wrapping + json const j1{j_obj}; + CHECK(j1.is_object()); + CHECK(j1 == j_obj); + + // array: brace init copies instead of wrapping + json const j2{j_arr}; + CHECK(j2.is_array()); + CHECK(j2.size() == 3); + CHECK(j2 == j_arr); + + // this applies to any single element, not only to JSON values + json const j3{true}; + CHECK(j3.is_boolean()); + + json const j4{42}; + CHECK(j4.is_number_integer()); + + json const j5 = {1}; + CHECK(j5 == 1); + + json const j6 = {"text"}; + CHECK(j6 == "text"); + + json const j7 = {{1, 2}}; + CHECK(j7 == json::array({1, 2})); + } + + SECTION("what the macro does not change") + { + // lists with more than one element are unaffected + json const j1 = {1, 2}; + CHECK(j1.is_array()); + CHECK(j1.size() == 2); + + // a single [string, value] pair still describes an object + json const j2 = {{"key", "value"}}; + CHECK(j2.is_object()); + CHECK(j2["key"] == "value"); + + // json::array() always creates an array + json const j3 = json::array({1}); + CHECK(j3.is_array()); + CHECK(j3.size() == 1); + CHECK(j3[0] == 1); + + json const j_obj = {{"key", "value"}}; + json const j4 = json::array({j_obj}); + CHECK(j4.is_array()); + CHECK(j4.size() == 1); + CHECK(j4[0] == j_obj); + } + + SECTION("conversions build the same values as without the macro") + { + SECTION("one-element std::tuple") + { + json const j1 = std::tuple {5}; + CHECK(j1.dump() == "[5]"); + CHECK(std::get<0>(j1.get>()) == 5); + + json const j2 = std::tuple {"text"}; + CHECK(j2.dump() == "[\"text\"]"); + CHECK(std::get<0>(j2.get>()) == "text"); + + json const j3 = std::tuple {json::array({1, 2})}; + CHECK(j3.dump() == "[[1,2]]"); + + // as without the macro, a [string, value] pair becomes an object + // member (see the known limitation documented for std::pair) + json const j4 = std::tuple> {{"a", 1}}; + CHECK(j4.dump() == "{\"a\":1}"); + } + + SECTION("tuples with more elements") + { + json const j1 = std::tuple {1, "a"}; + CHECK(j1.dump() == "[1,\"a\"]"); + + json const j2 = std::tuple<> {}; + CHECK(j2.dump() == "[]"); + } + + SECTION("one-element containers") + { + json const j1 = std::vector {1}; + CHECK(j1.dump() == "[1]"); + CHECK(j1.get>() == std::vector {1}); + + std::array const arr = {{1}}; + json const j2 = arr; + CHECK(j2.dump() == "[1]"); + + json const j3 = std::list {"a"}; + CHECK(j3.dump() == "[\"a\"]"); + + json const j4 = std::map {{"a", 1}}; + CHECK(j4.dump() == "{\"a\":1}"); + + json const j5 = std::map {{1, 2}}; + CHECK(j5.dump() == "[[1,2]]"); + } + + SECTION("std::pair") + { + json const j = std::pair {1, 2}; + CHECK(j.dump() == "[1,2]"); + CHECK((j.get>() == std::pair {1, 2})); + } + + SECTION("items()") + { + json j_obj = {{"key", 1}}; + for (const auto& el : j_obj.items()) + { + json const j = el; + CHECK(j.dump() == "{\"key\":1}"); + } + } + } +} diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index 7896b9f18..669a4bfe1 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -38,6 +38,54 @@ class huge_binary_t : public std::vector using huge_binary_json = nlohmann::basic_json < std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, double, std::allocator, nlohmann::adl_serializer, huge_binary_t, void >; + +// a string type that can be made to report a size beyond INT32_MAX without +// allocating that much memory, so BSON length overflow can be tested for +// strings and (embedded) documents as well, following the same idea as +// huge_binary_t. +// +// Unlike huge_binary_t (which is only ever used as the BSON *value* type), +// this type doubles as basic_json's StringType and is therefore also used +// for *object keys* (e.g. "s" or "nested" below). Only the designated test +// value is meant to lie about its size - if every huge_string_t (including +// keys) reported a huge size, the running totals computed while walking the +// BSON document (see calc_bson_object_size & friends in binary_writer.hpp) +// would need more than 32 bits, and on platforms where std::size_t is only +// 32 bits wide that arithmetic would silently wrap around, producing wrong +// (or even unguarded) lengths. The fake size is therefore opt-in via +// as_huge(), and plain strings - in particular object keys - keep reporting +// their real, small size. +class huge_string_t : public std::string +{ + public: + using std::string::string; + huge_string_t(const std::string& s) : std::string(s) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions) + + // returns a copy of @a s whose size() pretends to be huge + static huge_string_t as_huge(const std::string& s) + { + huge_string_t result(s); + result.pretend_huge = true; + return result; + } + + size_type size() const noexcept + { + if (pretend_huge) + { + // one byte more than the BSON length field can represent + return static_cast((std::numeric_limits::max)()) + 1; + } + return std::string::size(); + } + + private: + bool pretend_huge = false; +}; + +using huge_string_json = nlohmann::basic_json < + std::map, std::vector, huge_string_t, bool, std::int64_t, std::uint64_t, + double, std::allocator, nlohmann::adl_serializer, std::vector, void >; } // namespace TEST_CASE("BSON") @@ -105,10 +153,36 @@ TEST_CASE("BSON") SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON") { - huge_binary_json j; - j["b"] = huge_binary_json::binary(huge_binary_t{}); + // out_of_range.412 is thrown from a single shared helper + // (to_bson_length) that guards the BSON length fields of binary + // values, strings, and (embedded) documents alike + SECTION("binary") + { + huge_binary_json j; + j["b"] = huge_binary_json::binary(huge_binary_t{}); - CHECK_THROWS_WITH_AS(huge_binary_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483661 exceeds maximum of 2147483647", huge_binary_json::out_of_range&); + CHECK_THROWS_WITH_AS(huge_binary_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483661 exceeds maximum of 2147483647", huge_binary_json::out_of_range&); + } + + SECTION("string") + { + huge_string_json j; + j["s"] = huge_string_t::as_huge("value"); + + CHECK_THROWS_WITH_AS(huge_string_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483661 exceeds maximum of 2147483647", huge_string_json::out_of_range&); + } + + SECTION("document") + { + // an oversized string nested one level deep makes the + // *embedded* document's own length exceed INT32_MAX as well + huge_string_json nested; + nested["s"] = huge_string_t::as_huge("value"); + huge_string_json j; + j["nested"] = nested; + + CHECK_THROWS_WITH_AS(huge_string_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483674 exceeds maximum of 2147483647", huge_string_json::out_of_range&); + } } SECTION("string length must be at least 1") @@ -193,6 +267,23 @@ TEST_CASE("BSON") CHECK(json::from_bson(result, true, false) == j); } + SECTION("non-empty object with bool from a non-0/1 byte (lenient parsing)") + { + // documented lenient behavior (see gh-5333): any non-zero byte + // is accepted as `true`, not just 0x01 + std::vector const input = + { + 0x0D, 0x00, 0x00, 0x00, // size (little endian) + 0x08, // entry: boolean + 'e', 'n', 't', 'r', 'y', '\x00', + 0x02, // value = 0x02 (neither 0x00 nor 0x01) + 0x00 // end marker + }; + + const json expected = { { "entry", true } }; + CHECK(json::from_bson(input) == expected); + } + SECTION("non-empty object with double") { json const j = @@ -499,6 +590,29 @@ TEST_CASE("BSON") CHECK(json::from_bson(result, true, false) == j); } + SECTION("array elements with non-conforming keys (lenient parsing)") + { + // documented lenient behavior (see gh-5333): BSON array element + // keys are not checked against the required decimal sequence + // "0", "1", "2", ... - elements are taken in encoded order + std::vector const input = + { + 0x26, 0x00, 0x00, 0x00, // size (little endian) + 0x04, 'e', 'n', 't', 'r', 'y', '\x00', // entry: embedded array + + 0x1A, 0x00, 0x00, 0x00, // size (little endian) + 0x10, '5', 0x00, 0x0A, 0x00, 0x00, 0x00, // key "5" (bogus) -> 10 + 0x10, 'x', 0x00, 0x14, 0x00, 0x00, 0x00, // key "x" (non-numeric) -> 20 + 0x10, '1', 0x00, 0x1E, 0x00, 0x00, 0x00, // key "1" (out of order) -> 30 + 0x00, // end marker (embedded array) + + 0x00 // end marker + }; + + const json expected = { { "entry", json::array({10, 20, 30}) } }; + CHECK(json::from_bson(input) == expected); + } + SECTION("non-empty object with binary member") { const size_t N = 10; @@ -594,6 +708,31 @@ TEST_CASE("BSON") CHECK(json::from_bson(result, true, false) == j); } + SECTION("binary member with subtype 0x02 (old binary) keeps its inner length prefix (lenient parsing)") + { + // documented lenient behavior (see gh-5333): the payload for + // binary subtype 0x02 ("old binary") is returned as-is, + // including its own inner 4-byte length prefix; it is not + // stripped or reinterpreted + std::vector const input = + { + 0x17, 0x00, 0x00, 0x00, // size (little endian) + 0x05, 'e', 'n', 't', 'r', 'y', '\x00', // entry: binary + + 0x06, 0x00, 0x00, 0x00, // size of binary (little endian) + 0x02, // "old binary" subtype + 0x02, 0x00, 0x00, 0x00, // inner length prefix (part of the old-binary payload) + 0x68, 0x69, // payload ('h', 'i') + + 0x00 // end marker + }; + + // the inner length prefix is part of the (unmodified) payload + const std::vector expected_payload = {0x02, 0x00, 0x00, 0x00, 0x68, 0x69}; + const json expected = { { "entry", json::binary(expected_payload, 0x02) } }; + CHECK(json::from_bson(input) == expected); + } + SECTION("Some more complex document") { json const j = @@ -652,6 +791,15 @@ TEST_CASE("BSON") } } +TEST_CASE("regression test - BSON binary subtype rejects a value that doesn't fit a single byte") +{ + json const doc255 = {{"b", json::binary({1, 2}, 255)}}; + CHECK(json::from_bson(json::to_bson(doc255))["b"].get_binary().subtype() == 255); + + CHECK_THROWS_AS(json::to_bson(json{{"b", json::binary({1, 2}, 256)}}), json::out_of_range); + CHECK_THROWS_WITH_AS(json::to_bson(json{{"b", json::binary({1, 2}, 300)}}), "[json.exception.out_of_range.415] subtype 300 is too large for the BSON binary subtype (max 255)", json::out_of_range); +} + TEST_CASE("BSON input/output_adapters") { const json json_representation = @@ -1011,6 +1159,91 @@ TEST_CASE("BSON document size mismatch") } } +TEST_CASE("BSON nesting does not consume the call stack") +{ + // An embedded document or array used to be read by calling back into the + // document reader, so the native call stack grew with the nesting depth of + // the input (#5104). The open documents are kept on a heap stack now. + // + // Deeply nested values must not be compared, copied or dumped here: those + // operations are still recursive and would reintroduce the crash. + + // A document nested deeply enough to have crashed. The bytes are built + // here rather than with to_bson(), because the writer still recurses once + // per level and would overflow the stack before the reader is ever + // reached. Every level is + // 0x03 'a' 0x00 0x00 + // so a level is eight bytes larger than the one it holds, and the sizes + // can be filled in from the outside in. + const std::size_t depth = 30000; + std::vector input; + input.reserve(5 + (8 * depth)); + for (std::size_t i = 0; i < depth; ++i) + { + const auto size = static_cast(5 + (8 * (depth - i))); + input.push_back(static_cast(size & 0xFF)); + input.push_back(static_cast((size >> 8) & 0xFF)); + input.push_back(static_cast((size >> 16) & 0xFF)); + input.push_back(static_cast((size >> 24) & 0xFF)); + input.push_back(0x03); // embedded document + input.push_back('a'); + input.push_back(0x00); + } + // the innermost document is empty, then one terminator closes each level + input.insert(input.end(), {0x05, 0x00, 0x00, 0x00, 0x00}); + input.insert(input.end(), depth, 0x00); + + SECTION("a well-formed deep document is read through the SAX interface") + { + SaxCountdown accept_all(1000000); + CHECK(json::sax_parse(input, &accept_all, json::input_format_t::bson)); + } + + SECTION("a well-formed deep document is read into a value") + { + json j = json::from_bson(input); + + // walked rather than compared: comparing, copying or dumping a value + // this deep is still recursive + std::size_t measured = 0; + const json* q = &j; + while (q->is_object() && !q->empty()) + { + q = &q->begin().value(); + ++measured; + } + CHECK(measured == depth); + } + + SECTION("embedded documents and arrays are still read the same way") + { + const json values = {{"a", {{"b", {{"c", 1}}}}}}; + CHECK(json::from_bson(json::to_bson(values)) == values); + + const json array = {{"a", {1, 2, 3}}}; + CHECK(json::from_bson(json::to_bson(array)) == array); + + const json mixed = {{"a", {json{{"x", 1}}, json{{"y", 2}}}}}; + CHECK(json::from_bson(json::to_bson(mixed)) == mixed); + + CHECK(json::from_bson(json::to_bson(json::object())) == json::object()); + } + + SECTION("a size that does not match is still reported per document") + { + // the embedded document claims one byte too many + std::vector const bad = + { + 0x15, 0x00, 0x00, 0x00, 0x03, 'a', 0x00, + 0x0D, 0x00, 0x00, 0x00, 0x08, 'b', 0x00, 0x01, 0x00, + 0x00 + }; + json _; + CHECK_THROWS_AS(_ = json::from_bson(bad), json::parse_error&); + CHECK(json::from_bson(bad, true, false).is_discarded()); + } +} + TEST_CASE("BSON numerical data") { SECTION("number") diff --git a/tests/src/unit-byte_container_with_subtype.cpp b/tests/src/unit-byte_container_with_subtype.cpp index 3983ba95f..2e448ac7d 100644 --- a/tests/src/unit-byte_container_with_subtype.cpp +++ b/tests/src/unit-byte_container_with_subtype.cpp @@ -42,6 +42,39 @@ TEST_CASE("byte_container_with_subtype") CHECK(container.subtype() == static_cast(-1)); } + SECTION("move semantics") + { + // the rvalue-reference constructor (without a subtype) must actually move + // the passed-in container rather than copy it; comparing the buffer address + // before and after is a stronger check than just observing the source is + // empty afterward, since a copy-then-clear could also leave it empty + { + std::vector bytes = {{0xCA, 0xFE, 0xBA, 0xBE}}; + const auto* const data_ptr = bytes.data(); + + nlohmann::byte_container_with_subtype> container(std::move(bytes)); + + CHECK(container.size() == 4); + CHECK(container.data() == data_ptr); + CHECK(!container.has_subtype()); + CHECK(bytes.empty()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved) + } + + // same check for the rvalue-reference constructor that also takes a subtype + { + std::vector bytes = {{0xCA, 0xFE, 0xBA, 0xBE}}; + const auto* const data_ptr = bytes.data(); + + nlohmann::byte_container_with_subtype> container(std::move(bytes), 42); + + CHECK(container.size() == 4); + CHECK(container.data() == data_ptr); + CHECK(container.has_subtype()); + CHECK(container.subtype() == 42); + CHECK(bytes.empty()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved) + } + } + SECTION("comparisons") { std::vector const bytes = {{0xCA, 0xFE, 0xBA, 0xBE}}; diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index 2a6bd41d7..4c9107517 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -2035,6 +2035,231 @@ TEST_CASE("CBOR definite length equal to the indefinite-length sentinel") } } +TEST_CASE("CBOR nesting does not consume the call stack") +{ + // Containers used to be read by calling back into the value reader once + // per element, and a tag by calling it for the tagged value, so the native + // call stack grew with the nesting depth of the input. Each of the three + // costs a single byte to encode -- 0x9F, 0x81 and 0xC2 -- so a payload of + // repeated bytes crashed the process (#5104). The containers are kept on a + // heap stack now, and a tag is read in a loop. + // + // Deeply nested values must not be compared, copied or dumped here: those + // operations are still recursive and would reintroduce the crash. + json _; + + SECTION("indefinite-length containers") + { + const std::vector input(500000, 0x9F); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.110] parse error at byte 500001: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&); + CHECK(json::from_cbor(input, true, false).is_discarded()); + } + + SECTION("definite-length containers") + { + const std::vector input(500000, 0x81); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.110] parse error at byte 500001: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&); + CHECK(json::from_cbor(input, true, false).is_discarded()); + } + + SECTION("tags") + { + // a tag is not a value of its own, so a chain of them used to recurse + const std::vector input(500000, 0xC2); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(input, true, true, json::cbor_tag_handler_t::ignore), "[json.exception.parse_error.110] parse error at byte 500001: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&); + CHECK(json::from_cbor(input, true, false, json::cbor_tag_handler_t::ignore).is_discarded()); + } + + SECTION("a well-formed deep value is read through the SAX interface") + { + std::vector input(200000, 0x9F); + input.insert(input.end(), 200000, 0xFF); + + SaxCountdown accept_all(1000000); + CHECK(json::sax_parse(input, &accept_all, json::input_format_t::cbor)); + } + + SECTION("a well-formed deep value is read into a value") + { + const std::size_t depth = 10000; + std::vector input(depth, 0x81); + input.push_back(0x00); + + json j = json::from_cbor(input); + + std::size_t measured = 0; + const json* p = &j; + while (p->is_array() && !p->empty()) + { + p = &p->front(); + ++measured; + } + CHECK(measured == depth); + CHECK(p->is_number()); + } + + SECTION("containers are still read the same way") + { + CHECK(json::from_cbor(std::vector({0x80})) == json::array()); + CHECK(json::from_cbor(std::vector({0xA0})) == json::object()); + CHECK(json::from_cbor(std::vector({0x9F, 0xFF})) == json::array()); + CHECK(json::from_cbor(std::vector({0xBF, 0xFF})) == json::object()); + CHECK(json::from_cbor(std::vector({0x9F, 0x01, 0x02, 0xFF})) == json({1, 2})); + CHECK(json::from_cbor(std::vector({0xBF, 0x61, 'a', 0x01, 0xFF})) == json({{"a", 1}})); + // definite and indefinite forms nested inside each other + CHECK(json::from_cbor(std::vector({0x9F, 0x82, 0x01, 0x02, 0xA1, 0x61, 'k', 0xBF, 0xFF, 0xFF})) == json({{1, 2}, {{"k", json::object()}}})); + } + + SECTION("tagged values are still read the same way") + { + const auto ignore = json::cbor_tag_handler_t::ignore; + CHECK(json::from_cbor(std::vector({0xC2, 0x01}), true, true, ignore) == json(1)); + // a chain of tags resolves to the value that follows it + CHECK(json::from_cbor(std::vector({0xC2, 0xC2, 0xC2, 0x01}), true, true, ignore) == json(1)); + // a tag inside a container, and one in front of a container + CHECK(json::from_cbor(std::vector({0x82, 0xC2, 0x01, 0x02}), true, true, ignore) == json({1, 2})); + CHECK(json::from_cbor(std::vector({0xC2, 0x82, 0x01, 0x02}), true, true, ignore) == json({1, 2})); + } +} + +TEST_CASE("CBOR indefinite-length strings do not recurse per chunk") +{ + // Reading an indefinite-length string or byte array used to call itself + // once per chunk, so a payload of repeated 0x7F (or 0x5F) bytes exhausted + // the call stack before any of the input was rejected. The open levels are + // counted now, and the levels below prove the reader still reads the same + // values and reports the same errors at the same byte offsets. + json _; + + SECTION("many open levels are reported, not crashed on") + { + const std::vector input(200000, 0x7F); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.110] parse error at byte 200001: syntax error while parsing CBOR string: unexpected end of input", json::parse_error&); + CHECK(json::from_cbor(input, true, false).is_discarded()); + } + + SECTION("many open levels are reported, not crashed on (binary)") + { + const std::vector input(200000, 0x5F); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.110] parse error at byte 200001: syntax error while parsing CBOR binary: unexpected end of input", json::parse_error&); + CHECK(json::from_cbor(input, true, false).is_discarded()); + } + + SECTION("chunks are still concatenated") + { + CHECK(json::from_cbor(std::vector({0x7F, 0xFF})) == json("")); + CHECK(json::from_cbor(std::vector({0x7F, 0x61, 0x61, 0xFF})) == json("a")); + // nested indefinite-length strings are concatenated across levels + CHECK(json::from_cbor(std::vector({0x7F, 0x7F, 0x61, 0x61, 0xFF, 0x61, 0x62, 0xFF})) == json("ab")); + CHECK(json::from_cbor(std::vector({0x7F, 0x7F, 0x7F, 0x61, 0x7A, 0xFF, 0xFF, 0xFF})) == json("z")); + CHECK(json::from_cbor(std::vector({0xA1, 0x7F, 0x61, 0x61, 0xFF, 0x01})) == json({{"a", 1}})); + } + + SECTION("chunks are still concatenated (binary)") + { + CHECK(json::from_cbor(std::vector({0x5F, 0x41, 0x61, 0xFF})) == json::binary({0x61})); + CHECK(json::from_cbor(std::vector({0x5F, 0x5F, 0x41, 0x61, 0xFF, 0x41, 0x62, 0xFF})) == json::binary({0x61, 0x62})); + } + + SECTION("a chunk that is not a string is still rejected") + { + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x7F, 0x7F, 0x00})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x00", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x5F, 0x5F, 0x00})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR binary: expected length specification (0x40-0x5B) or indefinite binary array type (0x5F); last byte: 0x00", json::parse_error&); + } + + SECTION("a break marker outside an indefinite-length string is not a string") + { + // 0xFF only closes a string that was opened; on its own it is not one + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0xA1, 0xFF, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xFF", json::parse_error&); + } +} + +TEST_CASE("issue #5405 - array reserve for definite-length CBOR arrays") +{ +#if !defined(JSON_NOEXCEPTION) + // this SECTION relies on catching a thrown exception to distinguish + // which of two acceptable, bounded rejections a hostile header took; + // under JSON_NOEXCEPTION, JSON_THROW never produces a catchable C++ + // exception (it aborts instead), so this cannot be tested that way here + SECTION("a huge claimed length with no element data must not over-allocate") + { + // 0x9A: array with a four-byte length; claims 0xFFFFFFFF (4294967295) + // elements but provides none. max_size() for a std::vector is far + // larger than this count, so it does not reject the header outright; + // the (capped) reservation must not attempt to allocate space for + // billions of elements before the missing data is detected. + json _; + const std::vector input = {0x9A, 0xFF, 0xFF, 0xFF, 0xFF}; + // On a platform where std::size_t is narrower than 64 bits (e.g. + // 32-bit), the claimed count 0xFFFFFFFF coincides with that + // platform's detail::unknown_size() sentinel (SIZE_MAX), so the + // format-level size check rejects it outright (out_of_range.408, + // "excessive ... size") before the SAX consumer's own max_size() + // check would even run; on a 64-bit platform it passes both of + // those checks and is only found short of data once the (capped) + // reservation looks for element bytes that were never provided + // (parse_error.110). Either is an acceptable, bounded rejection of + // the hostile header -- the property under test is that no path + // attempts to allocate space for billions of elements. + bool threw = false; + try + { + _ = json::from_cbor(input); + } + catch (const json::parse_error& e) + { + threw = true; + CHECK(e.id == 110); + CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 6: syntax error while parsing CBOR value: unexpected end of input"); + } + catch (const json::out_of_range& e) + { + threw = true; + CHECK(e.id == 408); + CHECK(std::string(e.what()).find("excessive") != std::string::npos); + } + CHECK(threw); + CHECK(json::from_cbor(input, true, false).is_discarded()); + } +#endif + + SECTION("arrays of various sizes decode to the same value as before the reserve optimization") + { + for (const auto size : + { + std::size_t{0}, std::size_t{1}, std::size_t{5}, // small + std::size_t{16384}, // exactly at the reserve cap + std::size_t{20000} // above the reserve cap + }) + { + CAPTURE(size) + json j = json::array(); + for (std::size_t i = 0; i < size; ++i) + { + j.push_back(static_cast(i % 1000)); + } + + const auto packed = json::to_cbor(j); + CHECK(json::from_cbor(packed) == j); + } + } + + SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization") + { + // the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser; + // a custom SAX consumer that does not touch a DOM array sees identical events + json j = json::array(); + for (int i = 0; i < 100; ++i) + { + j.push_back(i); + } + const auto packed = json::to_cbor(j); + + SaxCountdown scp(1000000); // large enough to never trigger an abort + CHECK(json::sax_parse(packed, &scp, json::input_format_t::cbor)); + } +} + TEST_CASE("CBOR roundtrips" * doctest::skip()) { SECTION("input from flynn") diff --git a/tests/src/unit-class_lexer.cpp b/tests/src/unit-class_lexer.cpp index 64baf3da6..e89497738 100644 --- a/tests/src/unit-class_lexer.cpp +++ b/tests/src/unit-class_lexer.cpp @@ -12,6 +12,11 @@ #include using nlohmann::json; +#include // strtod +#include // stringstream +#include // string +#include // vector + namespace { // shortcut to scan a string literal @@ -224,3 +229,431 @@ TEST_CASE("lexer class") CHECK((scan_string("/**//**//**/", true) == json::lexer::token_type::end_of_input)); } } + +TEST_CASE("lexer number fast path") +{ + // The contiguous fast path (used for pointer/string input) must agree with + // the streaming byte path (used for std::istream) on token type, numeric + // value, and round-trip text for every well-formed number, and reject the + // same malformed numbers with the same message. + SECTION("contiguous vs streaming parity") + { + const std::vector numbers = + { + "0", "-0", "1", "-1", "42", "-42", "10", "100", "1234567890", + "0.0", "-0.0", "3.14", "-3.14", "0.5", "-0.001", "123.456789", + "1e0", "1E0", "1e10", "1e-10", "1e+10", "1.5e3", "-2.5E-4", + "9223372036854775807", // INT64_MAX -> unsigned + "9223372036854775808", // INT64_MAX + 1 -> unsigned + "18446744073709551615", // UINT64_MAX -> unsigned + "18446744073709551616", // UINT64_MAX + 1 -> float + "-9223372036854775808", // INT64_MIN -> integer + "-9223372036854775809", // INT64_MIN - 1 -> float + "123456789012345678901234567890", // huge -> float + "0.30000000000000004", "2.2250738585072014e-308", "1e308", + // high-precision / wide-exponent values that exercise the + // std::from_chars (Eisel-Lemire) path beyond the Clinger subset + "1.7976931348623157e308", "1.2345678901234567e-250", + "9007199254740993", "5e-324", "1e-320" + }; + + for (const auto& n : numbers) + { + const std::string doc = "[" + n + "]"; + + // contiguous fast path + const json a = json::parse(doc); + // streaming byte path + std::stringstream ss(doc); + const json b = json::parse(ss); + + CAPTURE(n); + CHECK(a == b); + CHECK(a.dump() == b.dump()); + CHECK(a[0].type() == b[0].type()); + } + } + + SECTION("significant-digit gate for the Clinger fast path") + { + // Clinger's fast path needs a significand below 2^53, so it cannot + // succeed once the mantissa has 17 or more significant digits (the + // significand would be at least 10^16). The lexer skips the attempt + // there. That is only allowed to save work: every value must still come + // out bit-exactly, and both scanners must agree. In particular the gate + // must not fire for tokens whose leading zeros merely look like extra + // digits - "0.1234567890123456" has 16 significant digits, not 17. + const std::vector numbers = + { + "1234567890123456", // 16 significant digits + "12345678901234567", // 17 -> attempt skipped + "123456789012345678", // 18 -> attempt skipped + "0.1234567890123456", // 16: the leading "0" is not significant + "0.12345678901234567", // 17 + "0.00000000000000001", // 1, in a long token + "0.000000000000000012345678901234", // 14, in a long token + "-0.0000000000000000000001", // 1, negative + "1.0000000000000000", // 17: trailing zeros are significant here + "10000000000000000", // 17 + "9007199254740992", // 2^53 + "9007199254740993", // 2^53 + 1 + "-65.613616999999977", // canada.json shape + "1.2345678901234567e-250", // 17 with an exponent + "1.234567890123456e-250", // 16 with an exponent + "1e10", "0.0", "-0.0", "0e0", "0.000123" + }; + + for (const auto& n : numbers) + { + CAPTURE(n); + const std::string doc = "[" + n + "]"; + + const json a = json::parse(doc); // contiguous fast path + std::stringstream ss(doc); + const json b = json::parse(ss); // streaming byte path + + CHECK(a[0].type() == b[0].type()); + CHECK(a == b); + + if (a[0].is_number_float()) + { + const double expected = std::strtod(n.c_str(), nullptr); + CHECK(a[0].get() == expected); + CHECK(b[0].get() == expected); + } + } + } + + SECTION("token type classification") + { + CHECK((scan_string("0") == json::lexer::token_type::value_unsigned)); + CHECK((scan_string("-1") == json::lexer::token_type::value_integer)); + CHECK((scan_string("1.5") == json::lexer::token_type::value_float)); + CHECK((scan_string("1e5") == json::lexer::token_type::value_float)); + CHECK((scan_string("18446744073709551615") == json::lexer::token_type::value_unsigned)); + CHECK((scan_string("18446744073709551616") == json::lexer::token_type::value_float)); + CHECK((scan_string("-9223372036854775808") == json::lexer::token_type::value_integer)); + CHECK((scan_string("-9223372036854775809") == json::lexer::token_type::value_float)); + } + + SECTION("malformed numbers are rejected identically") + { + for (const char* bad : + {"-", "1.", "1e", "1e+", "1.2e", "01", "-01", "1..2", "1.2.3" + }) + { + CAPTURE(bad); + // the contiguous fast path must decline and let the byte path report + const std::string doc = std::string("[") + bad + "]"; + CHECK_FALSE(json::accept(doc)); + std::stringstream ss(doc); + CHECK_FALSE(json::accept(ss)); + } + } + +#if !defined(JSON_NOEXCEPTION) + // these sections parse invalid input, which aborts when exceptions are off + SECTION("exhaustive grammar parity with the streaming path") + { + // The JSON number grammar is encoded twice: once as the scan_number() + // state machine and once as the contiguous fast path. Enumerate every + // short string over the number alphabet and require the two encodings to + // agree exactly - on acceptance, on the reported error, and on the parsed + // value - so they cannot drift apart. + const std::string alphabet = "01.eE+-"; + + // full outcome of parsing @a doc, so a mismatch in type, value, or error + // message is caught, not just a mismatch in acceptance + const auto outcome = [](const std::string & doc, bool streaming) -> std::string + { + try + { + if (streaming) + { + std::stringstream ss(doc); + const json j = json::parse(ss); + return std::string(j[0].type_name()) + '|' + j.dump(); + } + const json j = json::parse(doc); + return std::string(j[0].type_name()) + '|' + j.dump(); + } + catch (const json::parse_error& e) + { + return {e.what()}; + } + }; + + std::vector mismatches; + std::vector tokens{""}; + for (std::size_t length = 1; length <= 4; ++length) + { + std::vector next; + next.reserve(tokens.size() * alphabet.size()); + for (const auto& prefix : tokens) + { + for (const char c : alphabet) + { + next.push_back(prefix + c); + } + } + tokens = next; + + for (const auto& token : tokens) + { + const std::string doc = "[" + token + "]"; + if (outcome(doc, false) != outcome(doc, true)) + { + mismatches.push_back(doc); + } + } + } + + // 7 + 49 + 343 + 2401 tokens + CHECK(tokens.size() == 2401); + CAPTURE(mismatches); + CHECK(mismatches.empty()); + } + + SECTION("error positions match the streaming path") + { + // Rejecting identically is not enough: the fast path must also report the + // error at the same position as the byte path. A number directly followed + // by a newline is the interesting case, because the byte path reaches the + // newline (which resets the column) and then ungets it. + // returns the parse_error message, or "" if the document parsed + const auto contiguous_error = [](const std::string & doc) -> std::string + { + try + { + const json j = json::parse(doc); + static_cast(j); + } + catch (const json::parse_error& e) + { + return {e.what()}; + } + return {}; + }; + const auto streaming_error = [](const std::string & doc) -> std::string + { + try + { + std::stringstream ss(doc); + const json j = json::parse(ss); + static_cast(j); + } + catch (const json::parse_error& e) + { + return {e.what()}; + } + return {}; + }; + + for (const char* bad : + {"[01\n]", "[00\n]", "[-01\n]", "{1\n}", "[1\n2]", "[1.2.3\n]", + "[1 \n2]", "[\n1\n2]", "1\n2", "[01\r\n]", "[1e\n]", "[-\n]" + }) + { + CAPTURE(bad); + const std::string doc = bad; + const std::string contiguous_what = contiguous_error(doc); + + CHECK_FALSE(contiguous_what.empty()); + CHECK(contiguous_what == streaming_error(doc)); + } + + // A number terminated by a newline must report the same position as the + // same number terminated by anything else: scan_number() reads the + // terminator and ungets it, so the reported column is the one reached + // after the number's last character - not the 0 that an unget() across + // the newline used to leave behind. + CHECK(contiguous_error("[01\n]") == contiguous_error("[01 ]")); + CHECK(contiguous_error("[01\n]") == + "[json.exception.parse_error.101] parse error at line 1, column 3: " + "syntax error while parsing array - unexpected number literal; expected ']'"); + + // the same for a multi-character token, where the column of the last + // character (the '3' of "-2.5e3") differs from the column it starts at + CHECK(contiguous_error("null -2.5e3\nfalse") == contiguous_error("null -2.5e3 false")); + CHECK(contiguous_error("null -2.5e3\nfalse") == + "[json.exception.parse_error.101] parse error at line 1, column 11: " + "syntax error while parsing value - unexpected number literal; expected end of input"); + } +#endif +} + +TEST_CASE("lexer string fast path") +{ + // Build a byte string from explicit values: a hex escape in a string + // literal swallows every following hex digit, which makes sequences like + // "\xC3\xA9b" mean something other than they look like. + const auto bytes = [](std::initializer_list values) + { + std::string result; + for (const int value : values) + { + result.push_back(static_cast(value)); + } + return result; + }; + +#if !defined(JSON_NOEXCEPTION) + // the full outcome of parsing @a doc: the parsed value, or the exact error + // message, so a mismatch in either is caught. Only usable with exceptions + // on: parsing invalid input aborts when they are off. + const auto outcome = [](const std::string & doc, bool streaming) -> std::string + { + try + { + if (streaming) + { + std::stringstream ss(doc); + const json j = json::parse(ss); + return j.dump(); + } + const json j = json::parse(doc); + return j.dump(); + } + // not just parse_error: if a bulk scanner ever let ill-formed UTF-8 + // through, dump() would throw type_error.316, and that has to surface + // as a reported mismatch rather than as an uncaught exception + catch (const json::exception& e) + { + return {e.what()}; + } + }; +#endif + + // once at the start of the string, once past the first 8-byte SWAR word, so + // the bulk scanner sees each case with and without a run behind it + const std::vector offsets{0, 9}; + +#if !defined(JSON_NOEXCEPTION) + SECTION("exhaustive contiguous vs streaming parity") + { + // ordinary ASCII, both specials, a control byte, characters that make + // the preceding backslash a valid escape, a UTF-8 lead byte of each + // length, a continuation byte, and a byte that is never valid + const std::vector alphabet = + { + "a", "\"", "\\", "n", "u", "0", bytes({0x01}), + bytes({0xC3}), bytes({0xA9}), bytes({0xE4}), bytes({0xF0}), + bytes({0x80}), bytes({0xFF}) + }; + + std::vector mismatches; + std::vector tokens{""}; + for (std::size_t length = 1; length <= 3; ++length) + { + std::vector next; + next.reserve(tokens.size() * alphabet.size()); + for (const auto& prefix : tokens) + { + for (const auto& symbol : alphabet) + { + next.push_back(prefix + symbol); + } + } + tokens = next; + + for (const auto& token : tokens) + { + for (const std::size_t offset : offsets) + { + const std::string doc = "[\"" + std::string(offset, 'a') + token + "\"]"; + if (outcome(doc, false) != outcome(doc, true)) + { + mismatches.push_back(doc); + } + } + } + } + + // 13 + 169 + 2197 tokens, each at two offsets + CHECK(tokens.size() == 2197); + CAPTURE(mismatches); + CHECK(mismatches.empty()); + } + + SECTION("special bytes at every offset of the SWAR stride") + { + // The bulk scanner consumes 8 bytes at a time and then a tail; place + // every kind of byte that ends a run at each offset across two words, + // so multibyte sequences also straddle the word boundary. + const std::vector specials = + { + "\"", "\\", bytes({0x01}), bytes({0x1F}), bytes({0x7F}), + bytes({0xC3, 0xA9}), bytes({0xE4, 0xB8, 0xAD}), bytes({0xF0, 0x9F, 0x98, 0x80}), + bytes({0xFF}), bytes({0xC3}), bytes({0xE4, 0xB8}) + }; + + std::vector mismatches; + for (std::size_t offset = 0; offset <= 17; ++offset) + { + for (const auto& special : specials) + { + const std::string doc = "[\"" + std::string(offset, 'a') + special + "\"]"; + if (outcome(doc, false) != outcome(doc, true)) + { + mismatches.push_back(doc); + } + } + } + CAPTURE(mismatches); + CHECK(mismatches.empty()); + } +#endif + + // json::accept() never throws, so the ranges stay covered without exceptions + SECTION("UTF-8 ranges are accepted and rejected as documented") + { + // The bulk validator must accept exactly what the byte-at-a-time + // scanner accepts, so pin the boundaries of every range it recognizes. + // aggregate, only ever brace-initialized below; default member + // initializers would stop it being an aggregate in C++11 + struct utf8_case // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init) + { + std::string sequence; + bool valid; + const char* description; + }; + const std::vector cases = + { + {bytes({0xC2, 0x80}), true, "U+0080, shortest two-byte"}, + {bytes({0xDF, 0xBF}), true, "U+07FF, longest two-byte"}, + {bytes({0xC1, 0xBF}), false, "overlong two-byte"}, + {bytes({0xC2, 0x7F}), false, "two-byte with bad continuation"}, + {bytes({0xE0, 0xA0, 0x80}), true, "U+0800, shortest three-byte"}, + {bytes({0xE0, 0x9F, 0xBF}), false, "overlong three-byte"}, + {bytes({0xED, 0x9F, 0xBF}), true, "U+D7FF, just below the surrogates"}, + {bytes({0xED, 0xA0, 0x80}), false, "surrogate U+D800"}, + {bytes({0xED, 0xBF, 0xBF}), false, "surrogate U+DFFF"}, + {bytes({0xEE, 0x80, 0x80}), true, "U+E000, just above the surrogates"}, + {bytes({0xEF, 0xBF, 0xBF}), true, "U+FFFF"}, + {bytes({0xF0, 0x90, 0x80, 0x80}), true, "U+10000, shortest four-byte"}, + {bytes({0xF0, 0x8F, 0xBF, 0xBF}), false, "overlong four-byte"}, + {bytes({0xF4, 0x8F, 0xBF, 0xBF}), true, "U+10FFFF, highest code point"}, + {bytes({0xF4, 0x90, 0x80, 0x80}), false, "above U+10FFFF"}, + {bytes({0xF5, 0x80, 0x80, 0x80}), false, "lead byte out of range"}, + {bytes({0x80}), false, "bare continuation byte"}, + {bytes({0xFF}), false, "byte that never appears in UTF-8"}, + {bytes({0xC3}), false, "truncated two-byte"}, + {bytes({0xE4, 0xB8}), false, "truncated three-byte"}, + {bytes({0xF0, 0x9F, 0x98}), false, "truncated four-byte"} + }; + + for (const auto& test_case : cases) + { + CAPTURE(test_case.description); + for (const std::size_t offset : offsets) + { + CAPTURE(offset); + const std::string doc = "[\"" + std::string(offset, 'a') + test_case.sequence + "\"]"; + CHECK(json::accept(doc) == test_case.valid); +#if !defined(JSON_NOEXCEPTION) + CHECK(outcome(doc, false) == outcome(doc, true)); +#endif + } + } + } +} diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index 8b3ea660e..e532992f4 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -8,6 +8,14 @@ #include "doctest_compatibility.h" +// capture whether JSON_STRICT_NUL_HANDLING was enabled on the command line +// (e.g. -DJSON_STRICT_NUL_HANDLING=1) *before* including json.hpp, since the +// library #undefs JSON_STRICT_NUL_HANDLING itself once the header has been +// fully processed (see include/nlohmann/detail/macro_unscope.hpp) +#if defined(JSON_STRICT_NUL_HANDLING) && (JSON_STRICT_NUL_HANDLING == 1) + #define JSON_TEST_STRICT_NUL_HANDLING_ENABLED 1 +#endif + #define JSON_TESTS_PRIVATE #include using nlohmann::json; @@ -17,12 +25,16 @@ using nlohmann::json; #include #include +#include +#include #include #include #include #include #include +#include "test_utils.hpp" + namespace { class SaxEventLogger @@ -344,6 +356,50 @@ void trailing_comma_helper(const std::string& s) } } +#if JSON_DIAGNOSTIC_POSITIONS +/** + * Validates that the generated JSON object is the same as expected + * Validates that the start position and end position match the start and end of the string + * + * This check assumes that there is no whitespace around the json object in the original string. + */ +void validate_generated_json_and_start_end_pos_helper(const std::string& original_string, const json& j, const json& check) +{ + CHECK(j == check); + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == original_string.size()); +} + +/** + * Parses the root object from the given root string and validates that the start and end positions for the nested object are correct. + * + * This checks that whitespace around the nested object is included in the start and end positions of the root object. + */ +void validate_start_end_pos_for_nested_obj_helper(const std::string& nested_type_json_str, const std::string& root_type_json_str, const json& expected_json, const json::parser_callback_t& cb = nullptr) +{ + json j; + + // 1. If callback is provided, use callback version of parse() + if (cb) + { + j = json::parse(root_type_json_str, cb); + } + else + { + j = json::parse(root_type_json_str); + } + + // 2. Check if the generated JSON is as expected + // Assumptions: The root_type_json_str does not have any whitespace around the json object + validate_generated_json_and_start_end_pos_helper(root_type_json_str, j, expected_json); + + // 3. Get the nested object + const auto& nested = j["nested"]; + // 4. Check if the start and end positions are generated correctly for nested objects and arrays + CHECK(nested_type_json_str == root_type_json_str.substr(nested.start_pos(), nested.end_pos() - nested.start_pos())); +} +#endif + } // namespace TEST_CASE("parser class") @@ -497,6 +553,88 @@ TEST_CASE("parser class") } } + SECTION("NUL byte handling (issue #5530, JSON_STRICT_NUL_HANDLING)") + { + // by default, a NUL byte anywhere in the input (not inside a quoted + // string, which is covered above) is silently treated the same as + // real end of input; JSON_STRICT_NUL_HANDLING (off by default, see + // docs/mkdocs/docs/api/macros/json_strict_nul_handling.md) makes a + // NUL byte an error like any other unexpected byte instead. + // + // The two sections below are mutually exclusive: this whole test + // binary is compiled once, with JSON_STRICT_NUL_HANDLING either + // left at its default or forced to 1 (e.g. by the dedicated + // ci_test_strict_nul_handling CI target), so only the section + // matching the actual, compiled-in behavior can pass. +#if !defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED) + SECTION("default behavior (macro not enabled)") + { + // a NUL byte after a complete value silently truncates the input + std::string s = "123"; + s.push_back('\0'); + s += "4"; + CHECK(json::parse(s) == json(123)); + CHECK(json::accept(s)); + + // parsing from a string literal is unaffected either way + CHECK(json::parse("123") == json(123)); + } +#endif + +#if defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED) + SECTION("opt-in strict behavior (JSON_STRICT_NUL_HANDLING == 1)") + { + // a NUL byte after a complete value is now a parse error, + // instead of silently truncating the input + { + std::string s = "123"; + s.push_back('\0'); + json _; // NOLINT(readability-identifier-naming) + CHECK_THROWS_WITH_AS(_ = json::parse(s), + "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '123'; expected end of input", + json::parse_error&); + CHECK_FALSE(json::accept(s)); + } + + // a NUL byte where a value is expected is now a parse error, + // instead of being treated the same as an empty input + { + const std::string s(1, '\0'); + json _; // NOLINT(readability-identifier-naming) + CHECK_THROWS_WITH_AS(_ = json::parse(s), + "[json.exception.parse_error.101] parse error at line 1, column 1: syntax error while parsing value - invalid literal; last read: ''", + json::parse_error&); + CHECK_FALSE(json::accept(s)); + } + + // a NUL byte inside a // comment no longer stops the comment + // scan early; scanning continues correctly past it + { + std::string s = "1 // a"; + s.push_back('\0'); + s += "b\n"; + CHECK(json::parse(s, nullptr, true, true) == json(1)); + CHECK(json::accept(s, true, true)); + } + + // a NUL byte inside a /* */ comment no longer stops the + // comment scan early either + { + std::string s = "1 /* a"; + s.push_back('\0'); + s += "b */ "; + CHECK(json::parse(s, nullptr, true, true) == json(1)); + CHECK(json::accept(s, true, true)); + } + + // regression guard: parsing from a string literal (which + // carries a compiler-appended trailing '\0') still works, + // even though a NUL byte is now rejected everywhere else + CHECK(json::parse("123") == json(123)); + } +#endif + } + SECTION("number") { SECTION("integers") @@ -624,7 +762,8 @@ TEST_CASE("parser class") SECTION("overflow") { // overflows during parsing yield an exception - CHECK_THROWS_WITH_AS(parser_helper("1.18973e+4932").empty(), "[json.exception.out_of_range.406] number overflow parsing '1.18973e+4932'", json::out_of_range&); + // empty() is nodiscard; the exception is thrown by parser_helper() itself, before empty() would run + CHECK_THROWS_WITH_AS(utils::ignore_return_value(parser_helper("1.18973e+4932").empty()), "[json.exception.out_of_range.406] number overflow parsing '1.18973e+4932'", json::out_of_range&); } SECTION("invalid numbers") @@ -930,6 +1069,98 @@ TEST_CASE("parser class") CHECK(accept_helper("+1") == false); CHECK(accept_helper("+0") == false); } + + SECTION("issue #5411 - skip conversion when accept() does not need the numeric value") + { + // lexer::scan_number() may skip strtoull()/strtoll() for + // value_unsigned/value_integer tokens when the caller (e.g. + // json::accept()) does not need the converted value, as long + // as the digit count alone guarantees no 64-bit overflow (see + // the "safe_digit_count" fast path in scan_number()). This + // differential test checks that json::accept() (which enables + // the fast path) and json::parse() (which never does) always + // agree, over a corpus that exercises both the fast path + // (<=18 digits) and the untouched, exact fallback path (>=19 + // digits) -- including reclassification of huge digit-only + // integers to a (possibly non-finite) floating-point value. + const std::vector> cases = + { + // normal small/large integers, both signs + {"0", true}, {"1", true}, {"-1", true}, {"42", true}, {"-42", true}, + {"123456789", true}, {"-123456789", true}, + + // digit-count boundary around the 18-digit safe cutoff (both signs) + {std::string(17, '9'), true}, + {std::string(18, '9'), true}, + {std::string(19, '9'), true}, + {std::string(20, '9'), true}, + {"-" + std::string(17, '9'), true}, + {"-" + std::string(18, '9'), true}, + {"-" + std::string(19, '9'), true}, + {"-" + std::string(20, '9'), true}, + + // 64-bit boundaries + {"9223372036854775807", true}, // INT64_MAX + {"-9223372036854775808", true}, // INT64_MIN + {"18446744073709551615", true}, // UINT64_MAX + {"18446744073709551616", true}, // UINT64_MAX + 1 (overflows uint64_t, finite double) + + // the 28-digit example from the issue: overflows uint64_t + // but is finite as a double, so the scanner reclassifies + // it to value_float and it is accepted + {"9999999999999999999999999999", true}, + + // huge digit-only integers that overflow even a double -> rejected + {std::string(309, '9'), false}, + {std::string(400, '9'), false}, + {"1" + std::string(400, '0'), false}, + + // 1e999 / 1e400 style overflow -> rejected + {"1e999", false}, + {"1e400", false}, + {"-1e999", false}, + {"1E999", false}, + + // values straddling DBL_MAX + {"1.7976931348623157e308", true}, // <= DBL_MAX, finite + {"1.7976931348623159e308", false}, // > DBL_MAX, overflows to inf + + // a mix of other valid/invalid numeric syntax + {"3.14159", true}, + {"-0.0", true}, + {"1.0e10", true}, + {"01", false}, + {"-", false}, + {"1.", false}, + {"1e", false}, + {"+1", false}, + }; + + for (const auto& c : cases) + { + const std::string& number = c.first; + const bool expected = c.second; + CAPTURE(number) + CAPTURE(expected) + + // accept() takes the fast path (skips conversion when possible) + CHECK(json::accept(number) == expected); + + // parse() always performs the full conversion; it must agree + json j; + CHECK_NOTHROW(json::parser(nlohmann::detail::input_adapter(number), nullptr, false).parse(true, j)); + CHECK(!j.is_discarded() == expected); + + // wrap in an array so get_token() is exercised beyond the + // very first (constructor-time) scan as well + std::string wrapped = "["; + wrapped += number; + wrapped += ","; + wrapped += number; + wrapped += "]"; + CHECK(json::accept(wrapped) == expected); + } + } } } @@ -1394,6 +1625,71 @@ TEST_CASE("parser class") CHECK(accept_helper("\"\\uD80C\\uFFFF\"") == false); } +#if !defined(JSON_NOEXCEPTION) + SECTION("issue #5412 - whitespace skipping bookkeeping (compact vs. pretty-printed)") + { + // lexer::skip_whitespace() reads its first character with get() (to + // honor a possibly pending unget() from the previous token) and every + // further whitespace character with get_ignoring_pending_unget() (a + // get() variant that skips the then-always-false next_unget check). + // This must not change the reported byte offset, line, or column of + // a syntax error, even when a long run of whitespace containing + // multiple newlines is skipped beforehand (as with pretty-printed + // input). The expected values below were captured from the + // unmodified do-while(get()) loop, so any regression that miscounts + // characters or newlines while skipping whitespace changes them. + const auto check_error = [](const std::string & input, std::size_t expected_byte, + const std::string & expected_what) + { + CAPTURE(input) + try + { + json _ = json::parse(input); + FAIL_CHECK("expected a parse_error, but parsing succeeded"); + } + catch (const json::parse_error& e) + { + CHECK(e.byte == expected_byte); + CHECK(std::string(e.what()) == expected_what); + } + }; + + // a nested document, serialized both compactly and pretty-printed + // (dump(4)), each truncated right before the final closing '}' so + // that the parser hits EOF after skipping all of the (in the + // pretty-printed case, substantial) indentation whitespace + const json doc = + { + {"a", 1}, + {"b", json::array({true, false, nullptr, "x"})}, + {"c", json::object({{"d", 3.14}, {"e", json::array({1, 2, 3})}})} + }; + + const std::string compact = doc.dump(); + const std::string pretty = doc.dump(4); + + check_error(compact.substr(0, compact.size() - 1), 60, + "[json.exception.parse_error.101] parse error at line 1, column 60: syntax error while parsing object - unexpected end of input; expected '}'"); + check_error(pretty.substr(0, pretty.size() - 1), 193, + "[json.exception.parse_error.101] parse error at line 17, column 1: syntax error while parsing object - unexpected end of input; expected '}'"); + + // an invalid token appearing after several indented, multi-line + // whitespace runs vs. the same document without any of that + // whitespace + check_error(R"({ + "a": 1, + "b": [ + true, + false + ], + "c": @ +})", 70, + "[json.exception.parse_error.101] parse error at line 7, column 10: syntax error while parsing value - invalid literal; last read: '\"c\": @'"); + check_error(R"({"a":1,"b":[true,false],"c":@})", 29, + "[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing value - invalid literal; last read: '\"c\":@'"); + } +#endif + SECTION("tests found by mutate++") { // test case to make sure no comma precedes the first key @@ -1564,6 +1860,58 @@ TEST_CASE("parser class") CHECK (j_filtered2 == json({{"foo", {1, 2}}})); } + SECTION("filter many members of one container") + { + // Rejecting a value makes the parser remove the placeholder its key + // event stored. Locating that placeholder used to be a scan of the + // whole parent, which made filtering a large container quadratic: + // 128k members took ~25 s. These cases keep many members alive + // while discarding many others, so the removal cost is the whole + // point; they run in milliseconds when the placeholder is erased + // directly. + constexpr int count = 20000; + + std::string s = "{"; + for (int i = 0; i < count; ++i) + { + // "a" is kept, "z" is discarded + s += "\"a" + std::to_string(i) + "\":" + std::to_string(i) + ","; + s += "\"z" + std::to_string(i) + "\":-1,"; + } + s.back() = '}'; + + const json j_values = json::parse(s, [](int /*unused*/, json::parse_event_t e, const json & parsed) noexcept + { + return !(e == json::parse_event_t::value && parsed == json(-1)); + }); + + CHECK(j_values.size() == count); + CHECK(j_values.at("a0") == json(0)); + CHECK(j_values.at("a" + std::to_string(count - 1)) == json(count - 1)); + CHECK_FALSE(j_values.contains("z0")); + CHECK_FALSE(j_values.contains("z" + std::to_string(count - 1))); + + // the same, but discarding whole containers rather than values, + // which takes the end_object()/end_array() removal path + std::string s_nested = "{"; + for (int i = 0; i < count; ++i) + { + s_nested += "\"a" + std::to_string(i) + "\":" + std::to_string(i) + ","; + s_nested += "\"z" + std::to_string(i) + "\":[1,2],"; + } + s_nested.back() = '}'; + + const json j_arrays = json::parse(s_nested, [](int /*unused*/, json::parse_event_t e, const json& /*unused*/) noexcept + { + return e != json::parse_event_t::array_end; + }); + + CHECK(j_arrays.size() == count); + CHECK(j_arrays.at("a0") == json(0)); + CHECK_FALSE(j_arrays.contains("z0")); + CHECK_FALSE(j_arrays.contains("z" + std::to_string(count - 1))); + } + SECTION("filter specific events") { SECTION("first closing event") @@ -1636,7 +1984,13 @@ TEST_CASE("parser class") SECTION("from std::array") { - std::array v { {'t', 'r', 'u', 'e'} }; + // NOTE: this array is sized to exactly the length of "true" (unlike + // the trailing-NUL-tolerant default behavior elsewhere in this file, + // see the "NUL byte handling" section above); a size of 5 here would + // leave a value-initialized trailing 0x00 element that is only + // silently accepted as end-of-input by default and would fail under + // JSON_STRICT_NUL_HANDLING + std::array v { {'t', 'r', 'u', 'e'} }; json j; json::parser(nlohmann::detail::input_adapter(std::begin(v), std::end(v))).parse(true, j); CHECK(j == json(true)); @@ -1777,8 +2131,240 @@ TEST_CASE("parser class") { json _; CHECK_THROWS_WITH_AS(_ = json::parse("/a", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid comment; expecting '/' or '*' after '/'; last read: '/a'", json::parse_error); + // "/*" is a string literal, so it carries a compiler-appended trailing + // '\0'; by default that NUL is read like any other byte and shows up + // in "last read", but JSON_STRICT_NUL_HANDLING trims exactly that one + // trailing byte from a char array (see + // docs/mkdocs/docs/api/macros/json_strict_nul_handling.md), so it no + // longer appears in the message in that state +#if defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED) + CHECK_THROWS_WITH_AS(_ = json::parse("/*", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid comment; missing closing '*/'; last read: '/*'", json::parse_error); +#else CHECK_THROWS_WITH_AS(_ = json::parse("/*", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid comment; missing closing '*/'; last read: '/*'", json::parse_error); +#endif } + +#if JSON_DIAGNOSTIC_POSITIONS + // Macro for all test cases for start_pos and end_pos +#define SETUP_TESTCASES() \ + SECTION("with callback") \ + { \ + SECTION("filter nothing") \ + { \ + json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t /*unused*/, json& /*unused*/) noexcept \ + { \ + return true; \ + }; \ + validate_start_end_pos_for_nested_obj_helper(nested_type_json_str, root_type_json_str, expected, cb); \ + } \ + SECTION("filter element") \ + { \ + json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t event, json& j) noexcept \ + { \ + return (event != json::parse_event_t::key && event != json::parse_event_t::value) || j != json("a"); \ + }; \ + validate_start_end_pos_for_nested_obj_helper(nested_type_json_str, root_type_json_str, filteredExpected, cb); \ + } \ + } \ + SECTION("without callback") \ + { \ + validate_start_end_pos_for_nested_obj_helper(nested_type_json_str, root_type_json_str, expected); \ + } + + SECTION("retrieve start position and end position") + { + SECTION("for object") + { + // Create an object with spaces to test the start and end positions. Spaces will not be included in the + // JSON object, however, the start and end positions should include the spaces from the input JSON string. + const std::string nested_type_json_str = R"({ "a": 1,"b" : "test1"})"; + const std::string root_type_json_str = R"({ "nested": )" + nested_type_json_str + R"(, "anotherValue": "test2"})"; + auto expected = json({{"nested", {{"a", 1}, {"b", "test1"}}}, {"anotherValue", "test2"}}); + auto filteredExpected = expected; + filteredExpected["nested"].erase("a"); + + SETUP_TESTCASES() + } + + SECTION("for array") + { + const std::string nested_type_json_str = R"(["a", "test", 45])"; + const std::string root_type_json_str = R"({ "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; + auto expected = json({{"nested", {"a", "test", 45}}, {"anotherValue", "test"}}); + auto filteredExpected = expected; + filteredExpected["nested"] = json({"test", 45}); + SETUP_TESTCASES() + } + + SECTION("for array with objects") + { + const std::string nested_type_json_str = R"([{"a": 1, "b": "test"}, {"c": 2, "d": "test2"}])"; + const std::string root_type_json_str = R"({ "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; + auto expected = json({{"nested", {{{"a", 1}, {"b", "test"}}, {{"c", 2}, {"d", "test2"}}}}, {"anotherValue", "test"}}); + auto filteredExpected = expected; + filteredExpected["nested"][0].erase("a"); + SETUP_TESTCASES() + + auto j = json::parse(root_type_json_str); + auto nested_array = j["nested"]; + const auto& nested_obj = nested_array[0]; + CHECK(nested_type_json_str.substr(1, 21) == root_type_json_str.substr(nested_obj.start_pos(), nested_obj.end_pos() - nested_obj.start_pos())); + CHECK(nested_type_json_str.substr(24, 22) == root_type_json_str.substr(nested_array[1].start_pos(), nested_array[1].end_pos() - nested_array[1].start_pos())); + } + + SECTION("for two levels of nesting objects") + { + const std::string nested_type_json_str = R"({"nested2": {"b": "test"}})"; + const std::string root_type_json_str = R"({ "a": 2, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; + auto expected = json({{"a", 2}, {"nested", {{"nested2", {{"b", "test"}}}}}, {"anotherValue", "test"}}); + auto filteredExpected = expected; + filteredExpected.erase("a"); + SETUP_TESTCASES() + + auto j = json::parse(root_type_json_str); + auto nested_obj = j["nested"]["nested2"]; + CHECK(nested_type_json_str.substr(12, 13) == root_type_json_str.substr(nested_obj.start_pos(), nested_obj.end_pos() - nested_obj.start_pos())); + } + + SECTION("for simple types") + { + SECTION("no nested") + { + SECTION("with callback") + { + json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t /*unused*/, json& /*unused*/) noexcept + { + return true; + }; + + // 1. string type + std::string json_str = R"("test")"; + auto j = json::parse(json_str, cb); + validate_generated_json_and_start_end_pos_helper(json_str, j, "test"); + + // 2. number type + json_str = R"(1)"; + j = json::parse(json_str, cb); + validate_generated_json_and_start_end_pos_helper(json_str, j, 1); + + // 3. boolean type + json_str = R"(true)"; + j = json::parse(json_str, cb); + validate_generated_json_and_start_end_pos_helper(json_str, j, true); + + // 4. null type + json_str = R"(null)"; + j = json::parse(json_str, cb); + validate_generated_json_and_start_end_pos_helper(json_str, j, nullptr); + } + + SECTION("without callback") + { + // 1. string type + std::string json_str = R"("test")"; + auto j = json::parse(json_str); + validate_generated_json_and_start_end_pos_helper(json_str, j, "test"); + + // 2. number type + json_str = R"(1)"; + j = json::parse(json_str); + validate_generated_json_and_start_end_pos_helper(json_str, j, 1); + + json_str = R"(1.001239923)"; + j = json::parse(json_str); + validate_generated_json_and_start_end_pos_helper(json_str, j, 1.001239923); + + json_str = R"(1.123812389000000)"; + j = json::parse(json_str); + validate_generated_json_and_start_end_pos_helper(json_str, j, 1.123812389); + + // 3. boolean type + json_str = R"(true)"; + j = json::parse(json_str); + validate_generated_json_and_start_end_pos_helper(json_str, j, true); + + json_str = R"(false)"; + j = json::parse(json_str); + validate_generated_json_and_start_end_pos_helper(json_str, j, false); + + // 4. null type + json_str = R"(null)"; + j = json::parse(json_str); + validate_generated_json_and_start_end_pos_helper(json_str, j, nullptr); + } + } + + SECTION("string type") + { + const std::string nested_type_json_str = R"("test")"; + const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; + auto expected = json({{"nested", "test"}, {"anotherValue", "test"}, {"a", 1}}); + auto filteredExpected = expected; + filteredExpected.erase("a"); + SETUP_TESTCASES() + } + + SECTION("number type") + { + const std::string nested_type_json_str = R"(2)"; + const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; + auto expected = json({{"nested", 2}, {"anotherValue", "test"}, {"a", 1}}); + auto filteredExpected = expected; + filteredExpected.erase("a"); + SETUP_TESTCASES() + } + + SECTION("boolean type") + { + const std::string nested_type_json_str = R"(true)"; + const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; + auto expected = json({{"nested", true}, {"anotherValue", "test"}, {"a", 1}}); + auto filteredExpected = expected; + filteredExpected.erase("a"); + SETUP_TESTCASES() + } + + SECTION("null type") + { + const std::string nested_type_json_str = R"(null)"; + const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; + auto expected = json({{"nested", nullptr}, {"anotherValue", "test"}, {"a", 1}}); + auto filteredExpected = expected; + filteredExpected.erase("a"); + SETUP_TESTCASES() + } + } + SECTION("with leading whitespace and newlines around root JSON") + { + const std::string initial_whitespace = R"( + + )"; + const std::string nested_type_json_str = R"({ + "a": 1, + "nested": { + "b": "test" + }, + "anotherValue": "test" + })"; + const std::string end_whitespace = R"( + + )"; + const std::string root_type_json_str = initial_whitespace + nested_type_json_str + end_whitespace; + + auto expected = json({{"a", 1}, {"nested", {{"b", "test"}}}, {"anotherValue", "test"}}); + + auto j = json::parse(root_type_json_str); + + // 2. Check if the generated JSON is as expected + CHECK(j == expected); + + // 3. Check if the start and end positions do not include the surrounding whitespace + CHECK(j.start_pos() == initial_whitespace.size()); + CHECK(j.end_pos() == root_type_json_str.size() - end_whitespace.size()); + } + } +#undef SETUP_TESTCASES +#endif } // this test relies on parse errors being thrown, so it is skipped when @@ -1887,3 +2473,329 @@ TEST_CASE("last-read diagnostics are identical across input adapters") } } #endif // !defined(JSON_NOEXCEPTION) + +// this test characterizes the current (documented-by-example, not otherwise +// specified) behavior of JSON_DIAGNOSTIC_POSITIONS positions with respect to +// value lifetime (copy/move/swap/mutation), the various input adapters, and +// user-driven SAX usage. It is regression protection, not a behavior +// specification: if any of these checks fail after a change to json.hpp, +// that change deliberately altered observable behavior and the test (and +// this comment) should be updated accordingly, rather than "fixed" blindly. +#if JSON_DIAGNOSTIC_POSITIONS +TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") +{ + SECTION("value lifetime") + { + SECTION("copy constructor copies positions, recursively") + { + // basic_json(const basic_json&) (json.hpp, around line 1192) copies + // start_position/end_position for the value itself; nested values + // are copied via their own copy constructor (through the copied + // object/array container), so positions are preserved throughout + // the whole tree. + const std::string s = R"({"a":1,"b":[1,2,3]})"; + const json a = json::parse(s); + const json b = a; // NOLINT(performance-unnecessary-copy-initialization) + + CHECK(b.start_pos() == a.start_pos()); + CHECK(b.end_pos() == a.end_pos()); + CHECK(b["b"].start_pos() == a["b"].start_pos()); + CHECK(b["b"].end_pos() == a["b"].end_pos()); + CHECK(b["b"][0].start_pos() == a["b"][0].start_pos()); + CHECK(b["b"][0].end_pos() == a["b"][0].end_pos()); + + // sanity: the positions are meaningful (not all npos) + CHECK(b.start_pos() == 0); + CHECK(b.end_pos() == s.size()); + } + + SECTION("move constructor resets the moved-from value to npos") + { + // basic_json(basic_json&&) (json.hpp, around line 1265) copies + // other's start_position/end_position into *this and then resets + // other's to npos (see the cppcheck-suppress[accessForwarded] + // annotation there, which flags this reset as worth a second + // look). Only the top-level moved-from value is affected; its + // (moved-away) children are gone along with it. + const std::string s = R"({"a":1,"b":[1,2,3]})"; + json a = json::parse(s); + const auto a_start = a.start_pos(); + const auto a_end = a.end_pos(); + const auto nested_start = a["b"].start_pos(); + const auto nested_end = a["b"].end_pos(); + + const json b(std::move(a)); + + // the destination retains the original positions, recursively + CHECK(b.start_pos() == a_start); + CHECK(b.end_pos() == a_end); + CHECK(b["b"].start_pos() == nested_start); + CHECK(b["b"].end_pos() == nested_end); + + // the moved-from value is reset to a null and reports npos + CHECK(a.is_null()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) + CHECK(a.start_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) + CHECK(a.end_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) + } + + SECTION("swap() exchanges positions along with values") + { + // basic_json::swap() (json.hpp, around line 3626, and the friend + // swap() that forwards to it) swaps start_position/end_position + // together with m_data.m_type and m_data.m_value, so after + // swap(a, b) each variable's position describes its own new + // content, consistent with copy-assignment's + // operator=(basic_json) (json.hpp, around line 1291), which also + // swaps positions as part of its copy-and-swap implementation. + json a = json::parse(R"({"a":1})"); + json b = json::parse(R"([1,2,3,4,5])"); + const auto a_start = a.start_pos(); + const auto a_end = a.end_pos(); + const auto b_start = b.start_pos(); + const auto b_end = b.end_pos(); + // both start at 0 (root values start right away), but their + // lengths (and thus end positions) differ, which is enough to + // tell after the swap whether positions actually moved with + // the values + CHECK(a_end != b_end); + + using std::swap; + swap(a, b); + + // values were exchanged as expected ... + CHECK(a == json::parse(R"([1,2,3,4,5])")); + CHECK(b == json::parse(R"({"a":1})")); + + // ... and so were positions: each variable now carries the + // other's original position, describing its own new content + CHECK(a.start_pos() == b_start); + CHECK(a.end_pos() == b_end); + CHECK(b.start_pos() == a_start); + CHECK(b.end_pos() == a_end); + + // member swap() behaves the same as the free function + json c = json::parse(R"({"a":1})"); + json d = json::parse(R"([1,2,3,4,5])"); + const auto c_start = c.start_pos(); + const auto c_end = c.end_pos(); + const auto d_start = d.start_pos(); + const auto d_end = d.end_pos(); + + c.swap(d); + + CHECK(c.start_pos() == d_start); + CHECK(c.end_pos() == d_end); + CHECK(d.start_pos() == c_start); + CHECK(d.end_pos() == c_end); + } + + SECTION("mutating a parsed document leaves positions of unrelated values untouched") + { + // Positions are recorded once, during parsing, and are not + // recomputed on mutation. As a consequence, after a mutation the + // parent's own recorded span may no longer describe its current + // (serialized) content -- it still describes what was originally + // parsed. This is characterized here as current behavior, not + // asserted to be desirable or specified. + SECTION("operator[] adding a new object key") + { + const std::string s = R"({"a":1})"; + json j = json::parse(s); + const auto root_start = j.start_pos(); + const auto root_end = j.end_pos(); + const auto a_start = j["a"].start_pos(); + const auto a_end = j["a"].end_pos(); + + j["c"] = 42; + + // the newly-added value was never parsed, so it has no position + CHECK(j["c"].start_pos() == std::string::npos); + CHECK(j["c"].end_pos() == std::string::npos); + + // the existing sibling's position is unaffected + CHECK(j["a"].start_pos() == a_start); + CHECK(j["a"].end_pos() == a_end); + + // the parent's own recorded span is left as-is (now stale: + // it still reflects the original, shorter `{"a":1}` string) + CHECK(j.start_pos() == root_start); + CHECK(j.end_pos() == root_end); + } + + SECTION("push_back on a parsed array") + { + const std::string s = R"([1,2,3])"; + json j = json::parse(s); + const auto root_start = j.start_pos(); + const auto root_end = j.end_pos(); + const auto first_start = j[0].start_pos(); + + j.push_back(4); + + CHECK(j.back().start_pos() == std::string::npos); + CHECK(j.back().end_pos() == std::string::npos); + CHECK(j[0].start_pos() == first_start); + CHECK(j.start_pos() == root_start); + CHECK(j.end_pos() == root_end); + } + + SECTION("erase on a parsed array shifts elements but keeps their own positions") + { + const std::string s = R"([1,2,3])"; + json j = json::parse(s); + const auto second_start = j[1].start_pos(); + const auto third_start = j[2].start_pos(); + const auto root_start = j.start_pos(); + const auto root_end = j.end_pos(); + + j.erase(0); + + // remaining elements moved down an index, but each one still + // reports the position it had *before* the erase (i.e. its + // position in the original source string, not a + // recalculated one) + CHECK(j[0].start_pos() == second_start); + CHECK(j[1].start_pos() == third_start); + + // the parent's own recorded span is again left as-is + CHECK(j.start_pos() == root_start); + CHECK(j.end_pos() == root_end); + } + } + } + + SECTION("input adapters") + { + SECTION("wide string input: positions count transcoded UTF-8 bytes, not wide characters") + { + // 'é' (U+00E9) is a single code unit in a wchar_t/UTF-16 string, but + // transcodes to 2 bytes in UTF-8; the lexer only ever sees the + // transcoded UTF-8 byte stream, so reported positions are byte + // offsets into that UTF-8 stream, not indices into the original + // std::wstring. + // é (rather than a literal 'é' byte sequence in this source + // file) so the wide-string literal's meaning does not depend on + // the compiler's assumed source character set (MSVC, without + // /utf-8, would otherwise decode the raw UTF-8 bytes using the + // system code page instead of as UTF-8) + const std::wstring ws = L"{\"a\":\"\u00e9\u00e9\"}"; + CHECK(ws.size() == 10); // 10 wide characters + + const json j = json::parse(ws); + CHECK(j.start_pos() == 0); + // the transcoded UTF-8 form is 2 bytes longer than the wide string, + // because each of the two 'é' characters becomes 2 UTF-8 bytes + CHECK(j.end_pos() == 12); + CHECK(j.end_pos() != ws.size()); + + const json& a = j["a"]; + CHECK(a.start_pos() == 5); + CHECK(a.end_pos() == 11); + } + + SECTION("BOM-prefixed input: start_pos() reflects the skipped 3-byte BOM") + { + const std::string s = "\xEF\xBB\xBF{\"a\":1}"; + const json j = json::parse(s); + + // the lexer silently skips the BOM before parsing the value, so + // the root value's recorded span starts right after it + CHECK(j.start_pos() == 3); + CHECK(j.end_pos() == s.size()); + } + + SECTION("std::istringstream: positions are consistent, not npos") + { + const std::string s = R"({"a":1,"b":2})"; + std::istringstream ss(s); + const json j = json::parse(ss); + + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == s.size()); + CHECK(j["a"].start_pos() == 5); + } + + SECTION("std::ifstream: positions are consistent, not npos") + { + const std::string s = R"({"a":1,"b":2})"; + { + std::ofstream file("unit-class_parser_diagnostic_positions.tmp"); + file << s; + } + + { + std::ifstream f("unit-class_parser_diagnostic_positions.tmp"); + const json j = json::parse(f); + + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == s.size()); + CHECK(j["a"].start_pos() == 5); + } + + static_cast(std::remove("unit-class_parser_diagnostic_positions.tmp")); + } + + SECTION("iterator-pair input: positions are consistent, not npos") + { + const std::string s = R"({"a":1,"b":2})"; + const json j = json::parse(s.begin(), s.end()); + + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == s.size()); + CHECK(j["a"].start_pos() == 5); + } + + SECTION("binary formats have no text positions") + { + // binary formats (CBOR, MessagePack, UBJSON, BSON, BJData) are + // parsed via detail::binary_reader, which never sets + // start_position/end_position on the values it produces (they + // have no notion of a text offset), so every value's position + // stays at its default of npos. + const json src = json::parse(R"({"a":1,"b":[1,2]})"); + + const json from_cbor = json::from_cbor(json::to_cbor(src)); + CHECK(from_cbor.start_pos() == std::string::npos); + CHECK(from_cbor.end_pos() == std::string::npos); + CHECK(from_cbor["a"].start_pos() == std::string::npos); + CHECK(from_cbor["b"][0].start_pos() == std::string::npos); + + const json from_msgpack = json::from_msgpack(json::to_msgpack(src)); + CHECK(from_msgpack.start_pos() == std::string::npos); + CHECK(from_msgpack.end_pos() == std::string::npos); + + const json from_ubjson = json::from_ubjson(json::to_ubjson(src)); + CHECK(from_ubjson.start_pos() == std::string::npos); + CHECK(from_ubjson.end_pos() == std::string::npos); + + const json from_bson_val = json::from_bson(json::to_bson(src)); + CHECK(from_bson_val.start_pos() == std::string::npos); + CHECK(from_bson_val.end_pos() == std::string::npos); + } + } + + SECTION("user-driven SAX consumers with no lexer report npos") + { + // json::parse() internally wires up its json_sax_dom_parser with a + // pointer to its own lexer (see parser.hpp), which is how positions + // get set at all. A user who constructs a json_sax_dom_parser + // directly (e.g. to drive it via json::sax_parse()) and does not + // supply a lexer pointer gets a consumer with m_lexer_ref == nullptr; + // every "if (m_lexer_ref)" guard in json_sax.hpp is then skipped, so + // every value it produces keeps its default, unset position (npos). + // This was previously true but silently unasserted (operator== + // ignores positions), see #5420. + json result; + nlohmann::detail::json_sax_dom_parser sdp(result); + const std::string s = R"({"a":1,"b":[1,2,3]})"; + CHECK(json::sax_parse(s, &sdp)); + + CHECK(result.start_pos() == std::string::npos); + CHECK(result.end_pos() == std::string::npos); + CHECK(result["a"].start_pos() == std::string::npos); + CHECK(result["a"].end_pos() == std::string::npos); + CHECK(result["b"][0].start_pos() == std::string::npos); + CHECK(result["b"][0].end_pos() == std::string::npos); + } +} +#endif diff --git a/tests/src/unit-class_parser_diagnostic_positions.cpp b/tests/src/unit-class_parser_diagnostic_positions.cpp deleted file mode 100644 index 2697ecf8a..000000000 --- a/tests/src/unit-class_parser_diagnostic_positions.cpp +++ /dev/null @@ -1,1957 +0,0 @@ -// __ _____ _____ _____ -// __| | __| | | | JSON for Modern C++ (supporting code) -// | | |__ | | | | | | version 3.12.0 -// |_____|_____|_____|_|___| https://github.com/nlohmann/json -// -// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann -// SPDX-License-Identifier: MIT - -#include "doctest_compatibility.h" -#define JSON_TESTS_PRIVATE -#ifdef JSON_DIAGNOSTIC_POSITIONS - #undef JSON_DIAGNOSTIC_POSITIONS -#endif - -#define JSON_DIAGNOSTIC_POSITIONS 1 -#include -using nlohmann::json; - -#ifdef JSON_TEST_NO_GLOBAL_UDLS - using namespace nlohmann::literals; // NOLINT(google-build-using-namespace) -#endif - -#include - -namespace -{ -class SaxEventLogger -{ - public: - bool null() - { - events.emplace_back("null()"); - return true; - } - - bool boolean(bool val) - { - events.emplace_back(val ? "boolean(true)" : "boolean(false)"); - return true; - } - - bool number_integer(json::number_integer_t val) - { - events.push_back("number_integer(" + std::to_string(val) + ")"); - return true; - } - - bool number_unsigned(json::number_unsigned_t val) - { - events.push_back("number_unsigned(" + std::to_string(val) + ")"); - return true; - } - - bool number_float(json::number_float_t /*unused*/, const std::string& s) - { - events.push_back("number_float(" + s + ")"); - return true; - } - - bool string(std::string& val) - { - events.push_back("string(" + val + ")"); - return true; - } - - bool binary(json::binary_t& val) - { - std::string binary_contents = "binary("; - std::string comma_space; - for (auto b : val) - { - binary_contents.append(comma_space); - binary_contents.append(std::to_string(static_cast(b))); - comma_space = ", "; - } - binary_contents.append(")"); - events.push_back(binary_contents); - return true; - } - - bool start_object(std::size_t elements) - { - if (elements == (std::numeric_limits::max)()) - { - events.emplace_back("start_object()"); - } - else - { - events.push_back("start_object(" + std::to_string(elements) + ")"); - } - return true; - } - - bool key(std::string& val) - { - events.push_back("key(" + val + ")"); - return true; - } - - bool end_object() - { - events.emplace_back("end_object()"); - return true; - } - - bool start_array(std::size_t elements) - { - if (elements == (std::numeric_limits::max)()) - { - events.emplace_back("start_array()"); - } - else - { - events.push_back("start_array(" + std::to_string(elements) + ")"); - } - return true; - } - - bool end_array() - { - events.emplace_back("end_array()"); - return true; - } - - bool parse_error(std::size_t position, const std::string& /*unused*/, const json::exception& /*unused*/) - { - errored = true; - events.push_back("parse_error(" + std::to_string(position) + ")"); - return false; - } - - std::vector events {}; // NOLINT(readability-redundant-member-init) - bool errored = false; -}; - -class SaxCountdown : public nlohmann::json::json_sax_t -{ - public: - explicit SaxCountdown(const int count) : events_left(count) - {} - - bool null() override - { - return events_left-- > 0; - } - - bool boolean(bool /*val*/) override - { - return events_left-- > 0; - } - - bool number_integer(json::number_integer_t /*val*/) override - { - return events_left-- > 0; - } - - bool number_unsigned(json::number_unsigned_t /*val*/) override - { - return events_left-- > 0; - } - - bool number_float(json::number_float_t /*val*/, const std::string& /*s*/) override - { - return events_left-- > 0; - } - - bool string(std::string& /*val*/) override - { - return events_left-- > 0; - } - - bool binary(json::binary_t& /*val*/) override - { - return events_left-- > 0; - } - - bool start_object(std::size_t /*elements*/) override - { - return events_left-- > 0; - } - - bool key(std::string& /*val*/) override - { - return events_left-- > 0; - } - - bool end_object() override - { - return events_left-- > 0; - } - - bool start_array(std::size_t /*elements*/) override - { - return events_left-- > 0; - } - - bool end_array() override - { - return events_left-- > 0; - } - - bool parse_error(std::size_t /*position*/, const std::string& /*last_token*/, const json::exception& /*ex*/) override - { - return false; - } - - private: - int events_left = 0; -}; - -json parser_helper(const std::string& s); -bool accept_helper(const std::string& s); -void comments_helper(const std::string& s); - -json parser_helper(const std::string& s) -{ - json j; - json::parser(nlohmann::detail::input_adapter(s)).parse(true, j); - - // if this line was reached, no exception occurred - // -> check if result is the same without exceptions - json j_nothrow; - CHECK_NOTHROW(json::parser(nlohmann::detail::input_adapter(s), nullptr, false).parse(true, j_nothrow)); - CHECK(j_nothrow == j); - - json j_sax; - nlohmann::detail::json_sax_dom_parser sdp(j_sax); - json::sax_parse(s, &sdp); - CHECK(j_sax == j); - - comments_helper(s); - - return j; -} - -bool accept_helper(const std::string& s) -{ - CAPTURE(s) - - // 1. parse s without exceptions - json j; - CHECK_NOTHROW(json::parser(nlohmann::detail::input_adapter(s), nullptr, false).parse(true, j)); - const bool ok_noexcept = !j.is_discarded(); - - // 2. accept s - const bool ok_accept = json::parser(nlohmann::detail::input_adapter(s)).accept(true); - - // 3. check if both approaches come to the same result - CHECK(ok_noexcept == ok_accept); - - // 4. parse with SAX (compare with relaxed accept result) - SaxEventLogger el; - CHECK_NOTHROW(json::sax_parse(s, &el, json::input_format_t::json, false)); - CHECK(json::parser(nlohmann::detail::input_adapter(s)).accept(false) == !el.errored); - - // 5. parse with simple callback - json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t /*unused*/, json& /*unused*/) noexcept - { - return true; - }; - json const j_cb = json::parse(s, cb, false); - const bool ok_noexcept_cb = !j_cb.is_discarded(); - - // 6. check if this approach came to the same result - CHECK(ok_noexcept == ok_noexcept_cb); - - // 7. check if comments are properly ignored - if (ok_accept) - { - comments_helper(s); - } - - // 8. return result - return ok_accept; -} - -void comments_helper(const std::string& s) -{ - json _; - - // parse/accept with default parser - CHECK_NOTHROW(_ = json::parse(s)); - CHECK(json::accept(s)); - - // parse/accept while skipping comments - CHECK_NOTHROW(_ = json::parse(s, nullptr, false, true)); - CHECK(json::accept(s, true)); - - std::vector json_with_comments; - - // start with a comment - json_with_comments.push_back(std::string("// this is a comment\n") + s); - json_with_comments.push_back(std::string("/* this is a comment */") + s); - // end with a comment - json_with_comments.push_back(s + "// this is a comment"); - json_with_comments.push_back(s + "/* this is a comment */"); - - // check all strings - for (const auto& json_with_comment : json_with_comments) - { - CAPTURE(json_with_comment) - CHECK_THROWS_AS(_ = json::parse(json_with_comment), json::parse_error); - CHECK(!json::accept(json_with_comment)); - - CHECK_NOTHROW(_ = json::parse(json_with_comment, nullptr, true, true)); - CHECK(json::accept(json_with_comment, true)); - } -} - -/** - * Validates that the generated JSON object is the same as expected - * Validates that the start position and end position match the start and end of the string - * - * This check assumes that there is no whitespace around the json object in the original string. - */ -void validate_generated_json_and_start_end_pos_helper(const std::string& original_string, const json& j, const json& check) -{ - CHECK(j == check); - CHECK(j.start_pos() == 0); - CHECK(j.end_pos() == original_string.size()); -} - -/** - * Parses the root object from the given root string and validates that the start and end positions for the nested object are correct. - * - * This checks that whitespace around the nested object is included in the start and end positions of the root object. - */ -void validate_start_end_pos_for_nested_obj_helper(const std::string& nested_type_json_str, const std::string& root_type_json_str, const json& expected_json, const json::parser_callback_t& cb = nullptr) -{ - json j; - - // 1. If callback is provided, use callback version of parse() - if (cb) - { - j = json::parse(root_type_json_str, cb); - } - else - { - j = json::parse(root_type_json_str); - } - - // 2. Check if the generated JSON is as expected - // Assumptions: The root_type_json_str does not have any whitespace around the json object - validate_generated_json_and_start_end_pos_helper(root_type_json_str, j, expected_json); - - // 3. Get the nested object - const auto& nested = j["nested"]; - // 4. Check if the start and end positions are generated correctly for nested objects and arrays - CHECK(nested_type_json_str == root_type_json_str.substr(nested.start_pos(), nested.end_pos() - nested.start_pos())); -} - -} // namespace - -TEST_CASE("parser class") -{ - SECTION("parse") - { - SECTION("null") - { - CHECK(parser_helper("null") == json(nullptr)); - } - - SECTION("true") - { - CHECK(parser_helper("true") == json(true)); - } - - SECTION("false") - { - CHECK(parser_helper("false") == json(false)); - } - - SECTION("array") - { - SECTION("empty array") - { - CHECK(parser_helper("[]") == json(json::value_t::array)); - CHECK(parser_helper("[ ]") == json(json::value_t::array)); - } - - SECTION("nonempty array") - { - CHECK(parser_helper("[true, false, null]") == json({true, false, nullptr})); - } - } - - SECTION("object") - { - SECTION("empty object") - { - CHECK(parser_helper("{}") == json(json::value_t::object)); - CHECK(parser_helper("{ }") == json(json::value_t::object)); - } - - SECTION("nonempty object") - { - CHECK(parser_helper("{\"\": true, \"one\": 1, \"two\": null}") == json({{"", true}, {"one", 1}, {"two", nullptr}})); - } - } - - SECTION("string") - { - // empty string - CHECK(parser_helper("\"\"") == json(json::value_t::string)); - - SECTION("errors") - { - // error: tab in string - CHECK_THROWS_WITH_AS(parser_helper("\"\t\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0009 (HT) must be escaped to \\u0009 or \\t; last read: '\"'", json::parse_error&); - // error: newline in string - CHECK_THROWS_WITH_AS(parser_helper("\"\n\""), "[json.exception.parse_error.101] parse error at line 2, column 0: syntax error while parsing value - invalid string: control character U+000A (LF) must be escaped to \\u000A or \\n; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\r\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+000D (CR) must be escaped to \\u000D or \\r; last read: '\"'", json::parse_error&); - // error: backspace in string - CHECK_THROWS_WITH_AS(parser_helper("\"\b\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0008 (BS) must be escaped to \\u0008 or \\b; last read: '\"'", json::parse_error&); - // improve code coverage - CHECK_THROWS_AS(parser_helper("\uFF01"), json::parse_error&); - CHECK_THROWS_AS(parser_helper("[-4:1,]"), json::parse_error&); - // unescaped control characters - CHECK_THROWS_WITH_AS(parser_helper("\"\x00\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: missing closing quote; last read: '\"'", json::parse_error&); // NOLINT(bugprone-string-literal-with-embedded-nul) - CHECK_THROWS_WITH_AS(parser_helper("\"\x01\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0001 (SOH) must be escaped to \\u0001; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x02\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0002 (STX) must be escaped to \\u0002; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x03\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0003 (ETX) must be escaped to \\u0003; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x04\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0004 (EOT) must be escaped to \\u0004; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x05\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0005 (ENQ) must be escaped to \\u0005; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x06\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0006 (ACK) must be escaped to \\u0006; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x07\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0007 (BEL) must be escaped to \\u0007; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x08\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0008 (BS) must be escaped to \\u0008 or \\b; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x09\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0009 (HT) must be escaped to \\u0009 or \\t; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x0a\""), "[json.exception.parse_error.101] parse error at line 2, column 0: syntax error while parsing value - invalid string: control character U+000A (LF) must be escaped to \\u000A or \\n; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x0b\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+000B (VT) must be escaped to \\u000B; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x0c\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+000C (FF) must be escaped to \\u000C or \\f; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x0d\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+000D (CR) must be escaped to \\u000D or \\r; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x0e\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+000E (SO) must be escaped to \\u000E; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x0f\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+000F (SI) must be escaped to \\u000F; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x10\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0010 (DLE) must be escaped to \\u0010; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x11\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0011 (DC1) must be escaped to \\u0011; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x12\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0012 (DC2) must be escaped to \\u0012; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x13\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0013 (DC3) must be escaped to \\u0013; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x14\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0014 (DC4) must be escaped to \\u0014; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x15\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0015 (NAK) must be escaped to \\u0015; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x16\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0016 (SYN) must be escaped to \\u0016; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x17\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0017 (ETB) must be escaped to \\u0017; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x18\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0018 (CAN) must be escaped to \\u0018; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x19\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0019 (EM) must be escaped to \\u0019; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x1a\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+001A (SUB) must be escaped to \\u001A; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x1b\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+001B (ESC) must be escaped to \\u001B; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x1c\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+001C (FS) must be escaped to \\u001C; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x1d\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+001D (GS) must be escaped to \\u001D; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x1e\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+001E (RS) must be escaped to \\u001E; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\x1f\""), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+001F (US) must be escaped to \\u001F; last read: '\"'", json::parse_error&); - - SECTION("additional test for null byte") - { - // The test above for the null byte is wrong, because passing - // a string to the parser only reads int until it encounters - // a null byte. This test inserts the null byte later on and - // uses an iterator range. - std::string s = "\"1\""; - s[1] = '\0'; - json _; - CHECK_THROWS_WITH_AS(_ = json::parse(s.begin(), s.end()), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: control character U+0000 (NUL) must be escaped to \\u0000; last read: '\"'", json::parse_error&); - } - } - - SECTION("escaped") - { - // quotation mark "\"" - auto r1 = R"("\"")"_json; - CHECK(parser_helper("\"\\\"\"") == r1); - // reverse solidus "\\" - auto r2 = R"("\\")"_json; - CHECK(parser_helper("\"\\\\\"") == r2); - // solidus - CHECK(parser_helper("\"\\/\"") == R"("/")"_json); - // backspace - CHECK(parser_helper("\"\\b\"") == json("\b")); - // formfeed - CHECK(parser_helper("\"\\f\"") == json("\f")); - // newline - CHECK(parser_helper("\"\\n\"") == json("\n")); - // carriage return - CHECK(parser_helper("\"\\r\"") == json("\r")); - // horizontal tab - CHECK(parser_helper("\"\\t\"") == json("\t")); - - CHECK(parser_helper("\"\\u0001\"").get() == "\x01"); - CHECK(parser_helper("\"\\u000a\"").get() == "\n"); - CHECK(parser_helper("\"\\u00b0\"").get() == "°"); - CHECK(parser_helper("\"\\u0c00\"").get() == "ఀ"); - CHECK(parser_helper("\"\\ud000\"").get() == "퀀"); - CHECK(parser_helper("\"\\u000E\"").get() == "\x0E"); - CHECK(parser_helper("\"\\u00F0\"").get() == "ð"); - CHECK(parser_helper("\"\\u0100\"").get() == "Ā"); - CHECK(parser_helper("\"\\u2000\"").get() == " "); - CHECK(parser_helper("\"\\uFFFF\"").get() == "￿"); - CHECK(parser_helper("\"\\u20AC\"").get() == "€"); - CHECK(parser_helper("\"€\"").get() == "€"); - CHECK(parser_helper("\"🎈\"").get() == "🎈"); - - CHECK(parser_helper("\"\\ud80c\\udc60\"").get() == "\xf0\x93\x81\xa0"); - CHECK(parser_helper("\"\\ud83c\\udf1e\"").get() == "🌞"); - } - } - - SECTION("number") - { - SECTION("integers") - { - SECTION("without exponent") - { - CHECK(parser_helper("-128") == json(-128)); - CHECK(parser_helper("-0") == json(-0)); - CHECK(parser_helper("0") == json(0)); - CHECK(parser_helper("128") == json(128)); - } - - SECTION("with exponent") - { - CHECK(parser_helper("0e1") == json(0e1)); - CHECK(parser_helper("0E1") == json(0e1)); - - CHECK(parser_helper("10000E-4") == json(10000e-4)); - CHECK(parser_helper("10000E-3") == json(10000e-3)); - CHECK(parser_helper("10000E-2") == json(10000e-2)); - CHECK(parser_helper("10000E-1") == json(10000e-1)); - CHECK(parser_helper("10000E0") == json(10000e0)); - CHECK(parser_helper("10000E1") == json(10000e1)); - CHECK(parser_helper("10000E2") == json(10000e2)); - CHECK(parser_helper("10000E3") == json(10000e3)); - CHECK(parser_helper("10000E4") == json(10000e4)); - - CHECK(parser_helper("10000e-4") == json(10000e-4)); - CHECK(parser_helper("10000e-3") == json(10000e-3)); - CHECK(parser_helper("10000e-2") == json(10000e-2)); - CHECK(parser_helper("10000e-1") == json(10000e-1)); - CHECK(parser_helper("10000e0") == json(10000e0)); - CHECK(parser_helper("10000e1") == json(10000e1)); - CHECK(parser_helper("10000e2") == json(10000e2)); - CHECK(parser_helper("10000e3") == json(10000e3)); - CHECK(parser_helper("10000e4") == json(10000e4)); - - CHECK(parser_helper("-0e1") == json(-0e1)); - CHECK(parser_helper("-0E1") == json(-0e1)); - CHECK(parser_helper("-0E123") == json(-0e123)); - - // numbers after exponent - CHECK(parser_helper("10E0") == json(10e0)); - CHECK(parser_helper("10E1") == json(10e1)); - CHECK(parser_helper("10E2") == json(10e2)); - CHECK(parser_helper("10E3") == json(10e3)); - CHECK(parser_helper("10E4") == json(10e4)); - CHECK(parser_helper("10E5") == json(10e5)); - CHECK(parser_helper("10E6") == json(10e6)); - CHECK(parser_helper("10E7") == json(10e7)); - CHECK(parser_helper("10E8") == json(10e8)); - CHECK(parser_helper("10E9") == json(10e9)); - CHECK(parser_helper("10E+0") == json(10e0)); - CHECK(parser_helper("10E+1") == json(10e1)); - CHECK(parser_helper("10E+2") == json(10e2)); - CHECK(parser_helper("10E+3") == json(10e3)); - CHECK(parser_helper("10E+4") == json(10e4)); - CHECK(parser_helper("10E+5") == json(10e5)); - CHECK(parser_helper("10E+6") == json(10e6)); - CHECK(parser_helper("10E+7") == json(10e7)); - CHECK(parser_helper("10E+8") == json(10e8)); - CHECK(parser_helper("10E+9") == json(10e9)); - CHECK(parser_helper("10E-1") == json(10e-1)); - CHECK(parser_helper("10E-2") == json(10e-2)); - CHECK(parser_helper("10E-3") == json(10e-3)); - CHECK(parser_helper("10E-4") == json(10e-4)); - CHECK(parser_helper("10E-5") == json(10e-5)); - CHECK(parser_helper("10E-6") == json(10e-6)); - CHECK(parser_helper("10E-7") == json(10e-7)); - CHECK(parser_helper("10E-8") == json(10e-8)); - CHECK(parser_helper("10E-9") == json(10e-9)); - } - - SECTION("edge cases") - { - // From RFC8259, Section 6: - // Note that when such software is used, numbers that are - // integers and are in the range [-(2**53)+1, (2**53)-1] - // are interoperable in the sense that implementations will - // agree exactly on their numeric values. - - // -(2**53)+1 - CHECK(parser_helper("-9007199254740991").get() == -9007199254740991); - // (2**53)-1 - CHECK(parser_helper("9007199254740991").get() == 9007199254740991); - } - - SECTION("over the edge cases") // issue #178 - Integer conversion to unsigned (incorrect handling of 64-bit integers) - { - // While RFC8259, Section 6 specifies a preference for support - // for ranges in range of IEEE 754-2008 binary64 (double precision) - // this does not accommodate 64-bit integers without loss of accuracy. - // As 64-bit integers are now widely used in software, it is desirable - // to expand support to the full 64 bit (signed and unsigned) range - // i.e. -(2**63) -> (2**64)-1. - - // -(2**63) ** Note: compilers see negative literals as negated positive numbers (hence the -1)) - CHECK(parser_helper("-9223372036854775808").get() == -9223372036854775807 - 1); - // (2**63)-1 - CHECK(parser_helper("9223372036854775807").get() == 9223372036854775807); - // (2**64)-1 - CHECK(parser_helper("18446744073709551615").get() == 18446744073709551615u); - } - } - - SECTION("floating-point") - { - SECTION("without exponent") - { - CHECK(parser_helper("-128.5") == json(-128.5)); - CHECK(parser_helper("0.999") == json(0.999)); - CHECK(parser_helper("128.5") == json(128.5)); - CHECK(parser_helper("-0.0") == json(-0.0)); - } - - SECTION("with exponent") - { - CHECK(parser_helper("-128.5E3") == json(-128.5E3)); - CHECK(parser_helper("-128.5E-3") == json(-128.5E-3)); - CHECK(parser_helper("-0.0e1") == json(-0.0e1)); - CHECK(parser_helper("-0.0E1") == json(-0.0e1)); - } - } - - SECTION("overflow") - { - // overflows during parsing yield an exception - CHECK_THROWS_WITH_AS(parser_helper("1.18973e+4932").empty(), "[json.exception.out_of_range.406] number overflow parsing '1.18973e+4932'", json::out_of_range&); - } - - SECTION("invalid numbers") - { - // numbers must not begin with "+" - CHECK_THROWS_AS(parser_helper("+1"), json::parse_error&); - CHECK_THROWS_AS(parser_helper("+0"), json::parse_error&); - - CHECK_THROWS_WITH_AS(parser_helper("01"), - "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - unexpected number literal; expected end of input", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-01"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - unexpected number literal; expected end of input", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("--1"), - "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid number; expected digit after '-'; last read: '--'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("1."), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid number; expected digit after '.'; last read: '1.'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("1E"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid number; expected '+', '-', or digit after exponent; last read: '1E'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("1E-"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid number; expected digit after exponent sign; last read: '1E-'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("1.E1"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid number; expected digit after '.'; last read: '1.E'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-1E"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid number; expected '+', '-', or digit after exponent; last read: '-1E'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-0E#"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid number; expected '+', '-', or digit after exponent; last read: '-0E#'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-0E-#"), - "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid number; expected digit after exponent sign; last read: '-0E-#'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-0#"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid literal; last read: '-0#'; expected end of input", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-0.0:"), - "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - unexpected ':'; expected end of input", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-0.0Z"), - "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: '-0.0Z'; expected end of input", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-0E123:"), - "[json.exception.parse_error.101] parse error at line 1, column 7: syntax error while parsing value - unexpected ':'; expected end of input", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-0e0-:"), - "[json.exception.parse_error.101] parse error at line 1, column 6: syntax error while parsing value - invalid number; expected digit after '-'; last read: '-:'; expected end of input", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-0e-:"), - "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid number; expected digit after exponent sign; last read: '-0e-:'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-0f"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '-0f'; expected end of input", json::parse_error&); - } - } - } - - SECTION("accept") - { - SECTION("null") - { - CHECK(accept_helper("null")); - } - - SECTION("true") - { - CHECK(accept_helper("true")); - } - - SECTION("false") - { - CHECK(accept_helper("false")); - } - - SECTION("array") - { - SECTION("empty array") - { - CHECK(accept_helper("[]")); - CHECK(accept_helper("[ ]")); - } - - SECTION("nonempty array") - { - CHECK(accept_helper("[true, false, null]")); - } - } - - SECTION("object") - { - SECTION("empty object") - { - CHECK(accept_helper("{}")); - CHECK(accept_helper("{ }")); - } - - SECTION("nonempty object") - { - CHECK(accept_helper("{\"\": true, \"one\": 1, \"two\": null}")); - } - } - - SECTION("string") - { - // empty string - CHECK(accept_helper("\"\"")); - - SECTION("errors") - { - // error: tab in string - CHECK(accept_helper("\"\t\"") == false); - // error: newline in string - CHECK(accept_helper("\"\n\"") == false); - CHECK(accept_helper("\"\r\"") == false); - // error: backspace in string - CHECK(accept_helper("\"\b\"") == false); - // improve code coverage - CHECK(accept_helper("\uFF01") == false); - CHECK(accept_helper("[-4:1,]") == false); - // unescaped control characters - CHECK(accept_helper("\"\x00\"") == false); // NOLINT(bugprone-string-literal-with-embedded-nul) - CHECK(accept_helper("\"\x01\"") == false); - CHECK(accept_helper("\"\x02\"") == false); - CHECK(accept_helper("\"\x03\"") == false); - CHECK(accept_helper("\"\x04\"") == false); - CHECK(accept_helper("\"\x05\"") == false); - CHECK(accept_helper("\"\x06\"") == false); - CHECK(accept_helper("\"\x07\"") == false); - CHECK(accept_helper("\"\x08\"") == false); - CHECK(accept_helper("\"\x09\"") == false); - CHECK(accept_helper("\"\x0a\"") == false); - CHECK(accept_helper("\"\x0b\"") == false); - CHECK(accept_helper("\"\x0c\"") == false); - CHECK(accept_helper("\"\x0d\"") == false); - CHECK(accept_helper("\"\x0e\"") == false); - CHECK(accept_helper("\"\x0f\"") == false); - CHECK(accept_helper("\"\x10\"") == false); - CHECK(accept_helper("\"\x11\"") == false); - CHECK(accept_helper("\"\x12\"") == false); - CHECK(accept_helper("\"\x13\"") == false); - CHECK(accept_helper("\"\x14\"") == false); - CHECK(accept_helper("\"\x15\"") == false); - CHECK(accept_helper("\"\x16\"") == false); - CHECK(accept_helper("\"\x17\"") == false); - CHECK(accept_helper("\"\x18\"") == false); - CHECK(accept_helper("\"\x19\"") == false); - CHECK(accept_helper("\"\x1a\"") == false); - CHECK(accept_helper("\"\x1b\"") == false); - CHECK(accept_helper("\"\x1c\"") == false); - CHECK(accept_helper("\"\x1d\"") == false); - CHECK(accept_helper("\"\x1e\"") == false); - CHECK(accept_helper("\"\x1f\"") == false); - } - - SECTION("escaped") - { - // quotation mark "\"" - auto r1 = R"("\"")"_json; - CHECK(accept_helper("\"\\\"\"")); - // reverse solidus "\\" - auto r2 = R"("\\")"_json; - CHECK(accept_helper("\"\\\\\"")); - // solidus - CHECK(accept_helper("\"\\/\"")); - // backspace - CHECK(accept_helper("\"\\b\"")); - // formfeed - CHECK(accept_helper("\"\\f\"")); - // newline - CHECK(accept_helper("\"\\n\"")); - // carriage return - CHECK(accept_helper("\"\\r\"")); - // horizontal tab - CHECK(accept_helper("\"\\t\"")); - - CHECK(accept_helper("\"\\u0001\"")); - CHECK(accept_helper("\"\\u000a\"")); - CHECK(accept_helper("\"\\u00b0\"")); - CHECK(accept_helper("\"\\u0c00\"")); - CHECK(accept_helper("\"\\ud000\"")); - CHECK(accept_helper("\"\\u000E\"")); - CHECK(accept_helper("\"\\u00F0\"")); - CHECK(accept_helper("\"\\u0100\"")); - CHECK(accept_helper("\"\\u2000\"")); - CHECK(accept_helper("\"\\uFFFF\"")); - CHECK(accept_helper("\"\\u20AC\"")); - CHECK(accept_helper("\"€\"")); - CHECK(accept_helper("\"🎈\"")); - - CHECK(accept_helper("\"\\ud80c\\udc60\"")); - CHECK(accept_helper("\"\\ud83c\\udf1e\"")); - } - } - - SECTION("number") - { - SECTION("integers") - { - SECTION("without exponent") - { - CHECK(accept_helper("-128")); - CHECK(accept_helper("-0")); - CHECK(accept_helper("0")); - CHECK(accept_helper("128")); - } - - SECTION("with exponent") - { - CHECK(accept_helper("0e1")); - CHECK(accept_helper("0E1")); - - CHECK(accept_helper("10000E-4")); - CHECK(accept_helper("10000E-3")); - CHECK(accept_helper("10000E-2")); - CHECK(accept_helper("10000E-1")); - CHECK(accept_helper("10000E0")); - CHECK(accept_helper("10000E1")); - CHECK(accept_helper("10000E2")); - CHECK(accept_helper("10000E3")); - CHECK(accept_helper("10000E4")); - - CHECK(accept_helper("10000e-4")); - CHECK(accept_helper("10000e-3")); - CHECK(accept_helper("10000e-2")); - CHECK(accept_helper("10000e-1")); - CHECK(accept_helper("10000e0")); - CHECK(accept_helper("10000e1")); - CHECK(accept_helper("10000e2")); - CHECK(accept_helper("10000e3")); - CHECK(accept_helper("10000e4")); - - CHECK(accept_helper("-0e1")); - CHECK(accept_helper("-0E1")); - CHECK(accept_helper("-0E123")); - } - - SECTION("edge cases") - { - // From RFC8259, Section 6: - // Note that when such software is used, numbers that are - // integers and are in the range [-(2**53)+1, (2**53)-1] - // are interoperable in the sense that implementations will - // agree exactly on their numeric values. - - // -(2**53)+1 - CHECK(accept_helper("-9007199254740991")); - // (2**53)-1 - CHECK(accept_helper("9007199254740991")); - } - - SECTION("over the edge cases") // issue #178 - Integer conversion to unsigned (incorrect handling of 64-bit integers) - { - // While RFC8259, Section 6 specifies a preference for support - // for ranges in range of IEEE 754-2008 binary64 (double precision) - // this does not accommodate 64 bit integers without loss of accuracy. - // As 64 bit integers are now widely used in software, it is desirable - // to expand support to the full 64 bit (signed and unsigned) range - // i.e. -(2**63) -> (2**64)-1. - - // -(2**63) ** Note: compilers see negative literals as negated positive numbers (hence the -1)) - CHECK(accept_helper("-9223372036854775808")); - // (2**63)-1 - CHECK(accept_helper("9223372036854775807")); - // (2**64)-1 - CHECK(accept_helper("18446744073709551615")); - } - } - - SECTION("floating-point") - { - SECTION("without exponent") - { - CHECK(accept_helper("-128.5")); - CHECK(accept_helper("0.999")); - CHECK(accept_helper("128.5")); - CHECK(accept_helper("-0.0")); - } - - SECTION("with exponent") - { - CHECK(accept_helper("-128.5E3")); - CHECK(accept_helper("-128.5E-3")); - CHECK(accept_helper("-0.0e1")); - CHECK(accept_helper("-0.0E1")); - } - } - - SECTION("overflow") - { - // overflows during parsing - CHECK(!accept_helper("1.18973e+4932")); - } - - SECTION("invalid numbers") - { - CHECK(accept_helper("01") == false); - CHECK(accept_helper("--1") == false); - CHECK(accept_helper("1.") == false); - CHECK(accept_helper("1E") == false); - CHECK(accept_helper("1E-") == false); - CHECK(accept_helper("1.E1") == false); - CHECK(accept_helper("-1E") == false); - CHECK(accept_helper("-0E#") == false); - CHECK(accept_helper("-0E-#") == false); - CHECK(accept_helper("-0#") == false); - CHECK(accept_helper("-0.0:") == false); - CHECK(accept_helper("-0.0Z") == false); - CHECK(accept_helper("-0E123:") == false); - CHECK(accept_helper("-0e0-:") == false); - CHECK(accept_helper("-0e-:") == false); - CHECK(accept_helper("-0f") == false); - - // numbers must not begin with "+" - CHECK(accept_helper("+1") == false); - CHECK(accept_helper("+0") == false); - } - } - } - - SECTION("parse errors") - { - // unexpected end of number - CHECK_THROWS_WITH_AS(parser_helper("0."), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid number; expected digit after '.'; last read: '0.'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-"), - "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid number; expected digit after '-'; last read: '-'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("--"), - "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid number; expected digit after '-'; last read: '--'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-0."), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid number; expected digit after '.'; last read: '-0.'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-."), - "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid number; expected digit after '-'; last read: '-.'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("-:"), - "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid number; expected digit after '-'; last read: '-:'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("0.:"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid number; expected digit after '.'; last read: '0.:'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("e."), - "[json.exception.parse_error.101] parse error at line 1, column 1: syntax error while parsing value - invalid literal; last read: 'e'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("1e."), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid number; expected '+', '-', or digit after exponent; last read: '1e.'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("1e/"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid number; expected '+', '-', or digit after exponent; last read: '1e/'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("1e:"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid number; expected '+', '-', or digit after exponent; last read: '1e:'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("1E."), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid number; expected '+', '-', or digit after exponent; last read: '1E.'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("1E/"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid number; expected '+', '-', or digit after exponent; last read: '1E/'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("1E:"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid number; expected '+', '-', or digit after exponent; last read: '1E:'", json::parse_error&); - - // unexpected end of null - CHECK_THROWS_WITH_AS(parser_helper("n"), - "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 'n'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("nu"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid literal; last read: 'nu'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("nul"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: 'nul'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("nulk"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: 'nulk'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("nulm"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: 'nulm'", json::parse_error&); - - // unexpected end of true - CHECK_THROWS_WITH_AS(parser_helper("t"), - "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("tr"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid literal; last read: 'tr'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("tru"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: 'tru'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("trud"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: 'trud'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("truf"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: 'truf'", json::parse_error&); - - // unexpected end of false - CHECK_THROWS_WITH_AS(parser_helper("f"), - "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 'f'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("fa"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid literal; last read: 'fa'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("fal"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: 'fal'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("fals"), - "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'fals'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("falsd"), - "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'falsd'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("falsf"), - "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'falsf'", json::parse_error&); - - // missing/unexpected end of array - CHECK_THROWS_WITH_AS(parser_helper("["), - "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - unexpected end of input; expected '[', '{', or a literal", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("[1"), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing array - unexpected end of input; expected ']'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("[1,"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - unexpected end of input; expected '[', '{', or a literal", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("[1,]"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - unexpected ']'; expected '[', '{', or a literal", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("]"), - "[json.exception.parse_error.101] parse error at line 1, column 1: syntax error while parsing value - unexpected ']'; expected '[', '{', or a literal", json::parse_error&); - - // missing/unexpected end of object - CHECK_THROWS_WITH_AS(parser_helper("{"), - "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing object key - unexpected end of input; expected string literal", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("{\"foo\""), - "[json.exception.parse_error.101] parse error at line 1, column 7: syntax error while parsing object separator - unexpected end of input; expected ':'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("{\"foo\":"), - "[json.exception.parse_error.101] parse error at line 1, column 8: syntax error while parsing value - unexpected end of input; expected '[', '{', or a literal", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("{\"foo\":}"), - "[json.exception.parse_error.101] parse error at line 1, column 8: syntax error while parsing value - unexpected '}'; expected '[', '{', or a literal", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("{\"foo\":1,}"), - "[json.exception.parse_error.101] parse error at line 1, column 10: syntax error while parsing object key - unexpected '}'; expected string literal", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("}"), - "[json.exception.parse_error.101] parse error at line 1, column 1: syntax error while parsing value - unexpected '}'; expected '[', '{', or a literal", json::parse_error&); - - // missing/unexpected end of string - CHECK_THROWS_WITH_AS(parser_helper("\""), - "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: missing closing quote; last read: '\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\\\""), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid string: missing closing quote; last read: '\"\\\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\\u\""), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '\"\\u\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\\u0\""), - "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '\"\\u0\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\\u01\""), - "[json.exception.parse_error.101] parse error at line 1, column 6: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '\"\\u01\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\\u012\""), - "[json.exception.parse_error.101] parse error at line 1, column 7: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '\"\\u012\"'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\\u"), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '\"\\u'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\\u0"), - "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '\"\\u0'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\\u01"), - "[json.exception.parse_error.101] parse error at line 1, column 6: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '\"\\u01'", json::parse_error&); - CHECK_THROWS_WITH_AS(parser_helper("\"\\u012"), - "[json.exception.parse_error.101] parse error at line 1, column 7: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '\"\\u012'", json::parse_error&); - - // invalid escapes - for (int c = 1; c < 128; ++c) - { - auto s = std::string("\"\\") + std::string(1, static_cast(c)) + "\""; - - switch (c) - { - // valid escapes - case ('"'): - case ('\\'): - case ('/'): - case ('b'): - case ('f'): - case ('n'): - case ('r'): - case ('t'): - { - CHECK_NOTHROW(parser_helper(s)); - break; - } - - // \u must be followed with four numbers, so we skip it here - case ('u'): - { - break; - } - - // any other combination of backslash and character is invalid - default: - { - CHECK_THROWS_AS(parser_helper(s), json::parse_error&); - // only check error message if c is not a control character - if (c > 0x1f) - { - CHECK_THROWS_WITH_STD_STR(parser_helper(s), - "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: forbidden character after backslash; last read: '\"\\" + std::string(1, static_cast(c)) + "'"); - } - break; - } - } - } - - // invalid \uxxxx escapes - { - // check whether character is a valid hex character - const auto valid = [](int c) - { - switch (c) - { - case ('0'): - case ('1'): - case ('2'): - case ('3'): - case ('4'): - case ('5'): - case ('6'): - case ('7'): - case ('8'): - case ('9'): - case ('a'): - case ('b'): - case ('c'): - case ('d'): - case ('e'): - case ('f'): - case ('A'): - case ('B'): - case ('C'): - case ('D'): - case ('E'): - case ('F'): - { - return true; - } - - default: - { - return false; - } - } - }; - - for (int c = 1; c < 128; ++c) - { - std::string const s = "\"\\u"; - - // create a string with the iterated character at each position - auto s1 = s + "000" + std::string(1, static_cast(c)) + "\""; - auto s2 = s + "00" + std::string(1, static_cast(c)) + "0\""; - auto s3 = s + "0" + std::string(1, static_cast(c)) + "00\""; - auto s4 = s + std::string(1, static_cast(c)) + "000\""; - - if (valid(c)) - { - CAPTURE(s1) - CHECK_NOTHROW(parser_helper(s1)); - CAPTURE(s2) - CHECK_NOTHROW(parser_helper(s2)); - CAPTURE(s3) - CHECK_NOTHROW(parser_helper(s3)); - CAPTURE(s4) - CHECK_NOTHROW(parser_helper(s4)); - } - else - { - CAPTURE(s1) - CHECK_THROWS_AS(parser_helper(s1), json::parse_error&); - // only check error message if c is not a control character - if (c > 0x1f) - { - CHECK_THROWS_WITH_STD_STR(parser_helper(s1), - "[json.exception.parse_error.101] parse error at line 1, column 7: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '" + s1.substr(0, 7) + "'"); - } - - CAPTURE(s2) - CHECK_THROWS_AS(parser_helper(s2), json::parse_error&); - // only check error message if c is not a control character - if (c > 0x1f) - { - CHECK_THROWS_WITH_STD_STR(parser_helper(s2), - "[json.exception.parse_error.101] parse error at line 1, column 6: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '" + s2.substr(0, 6) + "'"); - } - - CAPTURE(s3) - CHECK_THROWS_AS(parser_helper(s3), json::parse_error&); - // only check error message if c is not a control character - if (c > 0x1f) - { - CHECK_THROWS_WITH_STD_STR(parser_helper(s3), - "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '" + s3.substr(0, 5) + "'"); - } - - CAPTURE(s4) - CHECK_THROWS_AS(parser_helper(s4), json::parse_error&); - // only check error message if c is not a control character - if (c > 0x1f) - { - CHECK_THROWS_WITH_STD_STR(parser_helper(s4), - "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '" + s4.substr(0, 4) + "'"); - } - } - } - } - - json _; - - // missing part of a surrogate pair - CHECK_THROWS_WITH_AS(_ = json::parse("\"\\uD80C\""), "[json.exception.parse_error.101] parse error at line 1, column 8: syntax error while parsing value - invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF; last read: '\"\\uD80C\"'", json::parse_error&); - // invalid surrogate pair - CHECK_THROWS_WITH_AS(_ = json::parse("\"\\uD80C\\uD80C\""), - "[json.exception.parse_error.101] parse error at line 1, column 13: syntax error while parsing value - invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF; last read: '\"\\uD80C\\uD80C'", json::parse_error&); - CHECK_THROWS_WITH_AS(_ = json::parse("\"\\uD80C\\u0000\""), - "[json.exception.parse_error.101] parse error at line 1, column 13: syntax error while parsing value - invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF; last read: '\"\\uD80C\\u0000'", json::parse_error&); - CHECK_THROWS_WITH_AS(_ = json::parse("\"\\uD80C\\uFFFF\""), - "[json.exception.parse_error.101] parse error at line 1, column 13: syntax error while parsing value - invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF; last read: '\"\\uD80C\\uFFFF'", json::parse_error&); - } - - SECTION("parse errors (accept)") - { - // unexpected end of number - CHECK(accept_helper("0.") == false); - CHECK(accept_helper("-") == false); - CHECK(accept_helper("--") == false); - CHECK(accept_helper("-0.") == false); - CHECK(accept_helper("-.") == false); - CHECK(accept_helper("-:") == false); - CHECK(accept_helper("0.:") == false); - CHECK(accept_helper("e.") == false); - CHECK(accept_helper("1e.") == false); - CHECK(accept_helper("1e/") == false); - CHECK(accept_helper("1e:") == false); - CHECK(accept_helper("1E.") == false); - CHECK(accept_helper("1E/") == false); - CHECK(accept_helper("1E:") == false); - - // unexpected end of null - CHECK(accept_helper("n") == false); - CHECK(accept_helper("nu") == false); - CHECK(accept_helper("nul") == false); - - // unexpected end of true - CHECK(accept_helper("t") == false); - CHECK(accept_helper("tr") == false); - CHECK(accept_helper("tru") == false); - - // unexpected end of false - CHECK(accept_helper("f") == false); - CHECK(accept_helper("fa") == false); - CHECK(accept_helper("fal") == false); - CHECK(accept_helper("fals") == false); - - // missing/unexpected end of array - CHECK(accept_helper("[") == false); - CHECK(accept_helper("[1") == false); - CHECK(accept_helper("[1,") == false); - CHECK(accept_helper("[1,]") == false); - CHECK(accept_helper("]") == false); - - // missing/unexpected end of object - CHECK(accept_helper("{") == false); - CHECK(accept_helper("{\"foo\"") == false); - CHECK(accept_helper("{\"foo\":") == false); - CHECK(accept_helper("{\"foo\":}") == false); - CHECK(accept_helper("{\"foo\":1,}") == false); - CHECK(accept_helper("}") == false); - - // missing/unexpected end of string - CHECK(accept_helper("\"") == false); - CHECK(accept_helper("\"\\\"") == false); - CHECK(accept_helper("\"\\u\"") == false); - CHECK(accept_helper("\"\\u0\"") == false); - CHECK(accept_helper("\"\\u01\"") == false); - CHECK(accept_helper("\"\\u012\"") == false); - CHECK(accept_helper("\"\\u") == false); - CHECK(accept_helper("\"\\u0") == false); - CHECK(accept_helper("\"\\u01") == false); - CHECK(accept_helper("\"\\u012") == false); - - // unget of newline - CHECK(parser_helper("\n123\n") == 123); - - // invalid escapes - for (int c = 1; c < 128; ++c) - { - auto s = std::string("\"\\") + std::string(1, static_cast(c)) + "\""; - - switch (c) - { - // valid escapes - case ('"'): - case ('\\'): - case ('/'): - case ('b'): - case ('f'): - case ('n'): - case ('r'): - case ('t'): - { - CHECK(json::parser(nlohmann::detail::input_adapter(s)).accept()); - break; - } - - // \u must be followed with four numbers, so we skip it here - case ('u'): - { - break; - } - - // any other combination of backslash and character is invalid - default: - { - CHECK(json::parser(nlohmann::detail::input_adapter(s)).accept() == false); - break; - } - } - } - - // invalid \uxxxx escapes - { - // check whether character is a valid hex character - const auto valid = [](int c) - { - switch (c) - { - case ('0'): - case ('1'): - case ('2'): - case ('3'): - case ('4'): - case ('5'): - case ('6'): - case ('7'): - case ('8'): - case ('9'): - case ('a'): - case ('b'): - case ('c'): - case ('d'): - case ('e'): - case ('f'): - case ('A'): - case ('B'): - case ('C'): - case ('D'): - case ('E'): - case ('F'): - { - return true; - } - - default: - { - return false; - } - } - }; - - for (int c = 1; c < 128; ++c) - { - std::string const s = "\"\\u"; - - // create a string with the iterated character at each position - const auto s1 = s + "000" + std::string(1, static_cast(c)) + "\""; - const auto s2 = s + "00" + std::string(1, static_cast(c)) + "0\""; - const auto s3 = s + "0" + std::string(1, static_cast(c)) + "00\""; - const auto s4 = s + std::string(1, static_cast(c)) + "000\""; - - if (valid(c)) - { - CAPTURE(s1) - CHECK(json::parser(nlohmann::detail::input_adapter(s1)).accept()); - CAPTURE(s2) - CHECK(json::parser(nlohmann::detail::input_adapter(s2)).accept()); - CAPTURE(s3) - CHECK(json::parser(nlohmann::detail::input_adapter(s3)).accept()); - CAPTURE(s4) - CHECK(json::parser(nlohmann::detail::input_adapter(s4)).accept()); - } - else - { - CAPTURE(s1) - CHECK(json::parser(nlohmann::detail::input_adapter(s1)).accept() == false); - - CAPTURE(s2) - CHECK(json::parser(nlohmann::detail::input_adapter(s2)).accept() == false); - - CAPTURE(s3) - CHECK(json::parser(nlohmann::detail::input_adapter(s3)).accept() == false); - - CAPTURE(s4) - CHECK(json::parser(nlohmann::detail::input_adapter(s4)).accept() == false); - } - } - } - - // missing part of a surrogate pair - CHECK(accept_helper("\"\\uD80C\"") == false); - // invalid surrogate pair - CHECK(accept_helper("\"\\uD80C\\uD80C\"") == false); - CHECK(accept_helper("\"\\uD80C\\u0000\"") == false); - CHECK(accept_helper("\"\\uD80C\\uFFFF\"") == false); - } - - SECTION("tests found by mutate++") - { - // test case to make sure no comma precedes the first key - CHECK_THROWS_WITH_AS(parser_helper("{,\"key\": false}"), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing object key - unexpected ','; expected string literal", json::parse_error&); - // test case to make sure an object is properly closed - CHECK_THROWS_WITH_AS(parser_helper("[{\"key\": false true]"), "[json.exception.parse_error.101] parse error at line 1, column 19: syntax error while parsing object - unexpected true literal; expected '}'", json::parse_error&); - - // test case to make sure the callback is properly evaluated after reading a key - { - json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t event, json& /*unused*/) noexcept - { - return event != json::parse_event_t::key; - }; - - const json x = json::parse("{\"key\": false}", cb); - CHECK(x == json::object()); - } - } - - SECTION("callback function") - { - const auto* s_object = R"( - { - "foo": 2, - "bar": { - "baz": 1 - } - } - )"; - - const auto* s_array = R"( - [1,2,[3,4,5],4,5] - )"; - - const auto* structured_array = R"( - [ - 1, - { - "foo": "bar" - }, - { - "qux": "baz" - } - ] - )"; - - SECTION("filter nothing") - { - const json j_object = json::parse(s_object, [](int /*unused*/, json::parse_event_t /*unused*/, const json& /*unused*/) noexcept - { - return true; - }); - - CHECK (j_object == json({{"foo", 2}, {"bar", {{"baz", 1}}}})); - - const json j_array = json::parse(s_array, [](int /*unused*/, json::parse_event_t /*unused*/, const json& /*unused*/) noexcept - { - return true; - }); - - CHECK (j_array == json({1, 2, {3, 4, 5}, 4, 5})); - } - - SECTION("filter everything") - { - json const j_object = json::parse(s_object, [](int /*unused*/, json::parse_event_t /*unused*/, const json& /*unused*/) noexcept - { - return false; - }); - - // the top-level object will be discarded, leaving a null - CHECK (j_object.is_null()); - - json const j_array = json::parse(s_array, [](int /*unused*/, json::parse_event_t /*unused*/, const json& /*unused*/) noexcept - { - return false; - }); - - // the top-level array will be discarded, leaving a null - CHECK (j_array.is_null()); - } - - SECTION("filter specific element") - { - const json j_object = json::parse(s_object, [](int /*unused*/, json::parse_event_t event, const json & j) noexcept - { - // filter all number(2) elements - return event != json::parse_event_t::value || j != json(2); - }); - - CHECK (j_object == json({{"bar", {{"baz", 1}}}})); - - const json j_array = json::parse(s_array, [](int /*unused*/, json::parse_event_t event, const json & j) noexcept - { - return event != json::parse_event_t::value || j != json(2); - }); - - CHECK (j_array == json({1, {3, 4, 5}, 4, 5})); - } - - SECTION("filter object in array") - { - const json j_filtered1 = json::parse(structured_array, [](int /*unused*/, json::parse_event_t e, const json & parsed) - { - return !(e == json::parse_event_t::object_end && parsed.contains("foo")); - }); - - // the specified object will be discarded, and removed. - CHECK (j_filtered1.size() == 2); - CHECK (j_filtered1 == json({1, {{"qux", "baz"}}})); - - const json j_filtered2 = json::parse(structured_array, [](int /*unused*/, json::parse_event_t e, const json& /*parsed*/) noexcept - { - return e != json::parse_event_t::object_end; - }); - - // removed all objects in array. - CHECK (j_filtered2.size() == 1); - CHECK (j_filtered2 == json({1})); - } - - SECTION("filter specific events") - { - SECTION("first closing event") - { - { - const json j_object = json::parse(s_object, [](int /*unused*/, json::parse_event_t e, const json& /*unused*/) noexcept - { - static bool first = true; - if (e == json::parse_event_t::object_end && first) - { - first = false; - return false; - } - - return true; - }); - - // the first completed object will be discarded - CHECK (j_object == json({{"foo", 2}})); - } - - { - const json j_array = json::parse(s_array, [](int /*unused*/, json::parse_event_t e, const json& /*unused*/) noexcept - { - static bool first = true; - if (e == json::parse_event_t::array_end && first) - { - first = false; - return false; - } - - return true; - }); - - // the first completed array will be discarded - CHECK (j_array == json({1, 2, 4, 5})); - } - } - } - - SECTION("special cases") - { - // the following test cases cover the situation in which an empty - // object and array is discarded only after the closing character - // has been read - - const json j_empty_object = json::parse("{}", [](int /*unused*/, json::parse_event_t e, const json& /*unused*/) noexcept - { - return e != json::parse_event_t::object_end; - }); - CHECK(j_empty_object == json()); - - const json j_empty_array = json::parse("[]", [](int /*unused*/, json::parse_event_t e, const json& /*unused*/) noexcept - { - return e != json::parse_event_t::array_end; - }); - CHECK(j_empty_array == json()); - } - } - - SECTION("constructing from contiguous containers") - { - SECTION("from std::vector") - { - std::vector v = {'t', 'r', 'u', 'e'}; - json j; - json::parser(nlohmann::detail::input_adapter(std::begin(v), std::end(v))).parse(true, j); - CHECK(j == json(true)); - } - - SECTION("from std::array") - { - std::array v { {'t', 'r', 'u', 'e'} }; - json j; - json::parser(nlohmann::detail::input_adapter(std::begin(v), std::end(v))).parse(true, j); - CHECK(j == json(true)); - } - - SECTION("from array") - { - uint8_t v[] = {'t', 'r', 'u', 'e'}; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - json j; - json::parser(nlohmann::detail::input_adapter(std::begin(v), std::end(v))).parse(true, j); - CHECK(j == json(true)); - } - - SECTION("from char literal") - { - CHECK(parser_helper("true") == json(true)); - } - - SECTION("from std::string") - { - std::string v = {'t', 'r', 'u', 'e'}; - json j; - json::parser(nlohmann::detail::input_adapter(std::begin(v), std::end(v))).parse(true, j); - CHECK(j == json(true)); - } - - SECTION("from std::initializer_list") - { - std::initializer_list const v = {'t', 'r', 'u', 'e'}; - json j; - json::parser(nlohmann::detail::input_adapter(std::begin(v), std::end(v))).parse(true, j); - CHECK(j == json(true)); - } - - SECTION("from std::valarray") - { - std::valarray v = {'t', 'r', 'u', 'e'}; - json j; - json::parser(nlohmann::detail::input_adapter(std::begin(v), std::end(v))).parse(true, j); - CHECK(j == json(true)); - } - } - - SECTION("improve test coverage") - { - SECTION("parser with callback") - { - json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t /*unused*/, json& /*unused*/) noexcept - { - return true; - }; - - CHECK(json::parse("{\"foo\": true:", cb, false).is_discarded()); - - json _; - CHECK_THROWS_WITH_AS(_ = json::parse("{\"foo\": true:", cb), "[json.exception.parse_error.101] parse error at line 1, column 13: syntax error while parsing object - unexpected ':'; expected '}'", json::parse_error&); - - CHECK_THROWS_WITH_AS(_ = json::parse("1.18973e+4932", cb), "[json.exception.out_of_range.406] number overflow parsing '1.18973e+4932'", json::out_of_range&); - } - - SECTION("SAX parser") - { - SECTION("} without value") - { - SaxCountdown s(1); - CHECK(json::sax_parse("{}", &s) == false); - } - - SECTION("} with value") - { - SaxCountdown s(3); - CHECK(json::sax_parse("{\"k1\": true}", &s) == false); - } - - SECTION("second key") - { - SaxCountdown s(3); - CHECK(json::sax_parse("{\"k1\": true, \"k2\": false}", &s) == false); - } - - SECTION("] without value") - { - SaxCountdown s(1); - CHECK(json::sax_parse("[]", &s) == false); - } - - SECTION("] with value") - { - SaxCountdown s(2); - CHECK(json::sax_parse("[1]", &s) == false); - } - - SECTION("float") - { - SaxCountdown s(0); - CHECK(json::sax_parse("3.14", &s) == false); - } - - SECTION("false") - { - SaxCountdown s(0); - CHECK(json::sax_parse("false", &s) == false); - } - - SECTION("null") - { - SaxCountdown s(0); - CHECK(json::sax_parse("null", &s) == false); - } - - SECTION("true") - { - SaxCountdown s(0); - CHECK(json::sax_parse("true", &s) == false); - } - - SECTION("unsigned") - { - SaxCountdown s(0); - CHECK(json::sax_parse("12", &s) == false); - } - - SECTION("integer") - { - SaxCountdown s(0); - CHECK(json::sax_parse("-12", &s) == false); - } - - SECTION("string") - { - SaxCountdown s(0); - CHECK(json::sax_parse("\"foo\"", &s) == false); - } - } - } - - SECTION("error messages for comments") - { - json _; - CHECK_THROWS_WITH_AS(_ = json::parse("/a", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid comment; expecting '/' or '*' after '/'; last read: '/a'", json::parse_error); - CHECK_THROWS_WITH_AS(_ = json::parse("/*", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid comment; missing closing '*/'; last read: '/*'", json::parse_error); - } - - // Macro for all test cases for start_pos and end_pos -#define SETUP_TESTCASES() \ - SECTION("with callback") \ - { \ - SECTION("filter nothing") \ - { \ - json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t /*unused*/, json& /*unused*/) noexcept \ - { \ - return true; \ - }; \ - validate_start_end_pos_for_nested_obj_helper(nested_type_json_str, root_type_json_str, expected, cb); \ - } \ - SECTION("filter element") \ - { \ - json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t event, json& j) noexcept \ - { \ - return (event != json::parse_event_t::key && event != json::parse_event_t::value) || j != json("a"); \ - }; \ - validate_start_end_pos_for_nested_obj_helper(nested_type_json_str, root_type_json_str, filteredExpected, cb); \ - } \ - } \ - SECTION("without callback") \ - { \ - validate_start_end_pos_for_nested_obj_helper(nested_type_json_str, root_type_json_str, expected); \ - } - - SECTION("retrieve start position and end position") - { - SECTION("for object") - { - // Create an object with spaces to test the start and end positions. Spaces will not be included in the - // JSON object, however, the start and end positions should include the spaces from the input JSON string. - const std::string nested_type_json_str = R"({ "a": 1,"b" : "test1"})"; - const std::string root_type_json_str = R"({ "nested": )" + nested_type_json_str + R"(, "anotherValue": "test2"})"; - auto expected = json({{"nested", {{"a", 1}, {"b", "test1"}}}, {"anotherValue", "test2"}}); - auto filteredExpected = expected; - filteredExpected["nested"].erase("a"); - - SETUP_TESTCASES() - } - - SECTION("for array") - { - const std::string nested_type_json_str = R"(["a", "test", 45])"; - const std::string root_type_json_str = R"({ "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; - auto expected = json({{"nested", {"a", "test", 45}}, {"anotherValue", "test"}}); - auto filteredExpected = expected; - filteredExpected["nested"] = json({"test", 45}); - SETUP_TESTCASES() - } - - SECTION("for array with objects") - { - const std::string nested_type_json_str = R"([{"a": 1, "b": "test"}, {"c": 2, "d": "test2"}])"; - const std::string root_type_json_str = R"({ "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; - auto expected = json({{"nested", {{{"a", 1}, {"b", "test"}}, {{"c", 2}, {"d", "test2"}}}}, {"anotherValue", "test"}}); - auto filteredExpected = expected; - filteredExpected["nested"][0].erase("a"); - SETUP_TESTCASES() - - auto j = json::parse(root_type_json_str); - auto nested_array = j["nested"]; - const auto& nested_obj = nested_array[0]; - CHECK(nested_type_json_str.substr(1, 21) == root_type_json_str.substr(nested_obj.start_pos(), nested_obj.end_pos() - nested_obj.start_pos())); - CHECK(nested_type_json_str.substr(24, 22) == root_type_json_str.substr(nested_array[1].start_pos(), nested_array[1].end_pos() - nested_array[1].start_pos())); - } - - SECTION("for two levels of nesting objects") - { - const std::string nested_type_json_str = R"({"nested2": {"b": "test"}})"; - const std::string root_type_json_str = R"({ "a": 2, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; - auto expected = json({{"a", 2}, {"nested", {{"nested2", {{"b", "test"}}}}}, {"anotherValue", "test"}}); - auto filteredExpected = expected; - filteredExpected.erase("a"); - SETUP_TESTCASES() - - auto j = json::parse(root_type_json_str); - auto nested_obj = j["nested"]["nested2"]; - CHECK(nested_type_json_str.substr(12, 13) == root_type_json_str.substr(nested_obj.start_pos(), nested_obj.end_pos() - nested_obj.start_pos())); - } - - SECTION("for simple types") - { - SECTION("no nested") - { - SECTION("with callback") - { - json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t /*unused*/, json& /*unused*/) noexcept - { - return true; - }; - - // 1. string type - std::string json_str = R"("test")"; - auto j = json::parse(json_str, cb); - validate_generated_json_and_start_end_pos_helper(json_str, j, "test"); - - // 2. number type - json_str = R"(1)"; - j = json::parse(json_str, cb); - validate_generated_json_and_start_end_pos_helper(json_str, j, 1); - - // 3. boolean type - json_str = R"(true)"; - j = json::parse(json_str, cb); - validate_generated_json_and_start_end_pos_helper(json_str, j, true); - - // 4. null type - json_str = R"(null)"; - j = json::parse(json_str, cb); - validate_generated_json_and_start_end_pos_helper(json_str, j, nullptr); - } - - SECTION("without callback") - { - // 1. string type - std::string json_str = R"("test")"; - auto j = json::parse(json_str); - validate_generated_json_and_start_end_pos_helper(json_str, j, "test"); - - // 2. number type - json_str = R"(1)"; - j = json::parse(json_str); - validate_generated_json_and_start_end_pos_helper(json_str, j, 1); - - json_str = R"(1.001239923)"; - j = json::parse(json_str); - validate_generated_json_and_start_end_pos_helper(json_str, j, 1.001239923); - - json_str = R"(1.123812389000000)"; - j = json::parse(json_str); - validate_generated_json_and_start_end_pos_helper(json_str, j, 1.123812389); - - // 3. boolean type - json_str = R"(true)"; - j = json::parse(json_str); - validate_generated_json_and_start_end_pos_helper(json_str, j, true); - - json_str = R"(false)"; - j = json::parse(json_str); - validate_generated_json_and_start_end_pos_helper(json_str, j, false); - - // 4. null type - json_str = R"(null)"; - j = json::parse(json_str); - validate_generated_json_and_start_end_pos_helper(json_str, j, nullptr); - } - } - - SECTION("string type") - { - const std::string nested_type_json_str = R"("test")"; - const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; - auto expected = json({{"nested", "test"}, {"anotherValue", "test"}, {"a", 1}}); - auto filteredExpected = expected; - filteredExpected.erase("a"); - SETUP_TESTCASES() - } - - SECTION("number type") - { - const std::string nested_type_json_str = R"(2)"; - const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; - auto expected = json({{"nested", 2}, {"anotherValue", "test"}, {"a", 1}}); - auto filteredExpected = expected; - filteredExpected.erase("a"); - SETUP_TESTCASES() - } - - SECTION("boolean type") - { - const std::string nested_type_json_str = R"(true)"; - const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; - auto expected = json({{"nested", true}, {"anotherValue", "test"}, {"a", 1}}); - auto filteredExpected = expected; - filteredExpected.erase("a"); - SETUP_TESTCASES() - } - - SECTION("null type") - { - const std::string nested_type_json_str = R"(null)"; - const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })"; - auto expected = json({{"nested", nullptr}, {"anotherValue", "test"}, {"a", 1}}); - auto filteredExpected = expected; - filteredExpected.erase("a"); - SETUP_TESTCASES() - } - } - SECTION("with leading whitespace and newlines around root JSON") - { - const std::string initial_whitespace = R"( - - )"; - const std::string nested_type_json_str = R"({ - "a": 1, - "nested": { - "b": "test" - }, - "anotherValue": "test" - })"; - const std::string end_whitespace = R"( - - )"; - const std::string root_type_json_str = initial_whitespace + nested_type_json_str + end_whitespace; - - auto expected = json({{"a", 1}, {"nested", {{"b", "test"}}}, {"anotherValue", "test"}}); - - auto j = json::parse(root_type_json_str); - - // 2. Check if the generated JSON is as expected - CHECK(j == expected); - - // 3. Check if the start and end positions do not include the surrounding whitespace - CHECK(j.start_pos() == initial_whitespace.size()); - CHECK(j.end_pos() == root_type_json_str.size() - end_whitespace.size()); - } - } -} diff --git a/tests/src/unit-comparison.cpp b/tests/src/unit-comparison.cpp index 31fbdc57a..68d20aca9 100644 --- a/tests/src/unit-comparison.cpp +++ b/tests/src/unit-comparison.cpp @@ -326,6 +326,57 @@ TEST_CASE("lexicographical comparison operators") #endif } + SECTION("integer/float mixed comparison is exact") + { + // Widening the integer to a double loses precision past the + // mantissa, so 2^63-2 and 2^63-1 both used to compare equal to the + // double 2^63 while differing from each other. That makes equality + // intransitive and the ordering not a strict weak ordering. + const json below_two_63 = static_cast(9223372036854775806LL); + const json max_int64 = (std::numeric_limits::max)(); + const json two_63 = 9223372036854775808.0; + + CHECK_FALSE(below_two_63 == two_63); + CHECK_FALSE(max_int64 == two_63); + CHECK(below_two_63 != max_int64); + CHECK(below_two_63 < max_int64); + CHECK(below_two_63 < two_63); + CHECK(max_int64 < two_63); + CHECK(two_63 > max_int64); + CHECK_FALSE(two_63 < max_int64); + + // the same past the unsigned range + const json max_uint64 = (std::numeric_limits::max)(); + const json two_64 = 18446744073709551616.0; + CHECK_FALSE(max_uint64 == two_64); + CHECK(max_uint64 < two_64); + CHECK(two_64 > max_uint64); + + // values a double represents exactly still compare equal + CHECK(json(1) == json(1.0)); + CHECK(json(1u) == json(1.0)); + CHECK(json(-3) == json(-3.0)); + CHECK(json(1) < json(1.5)); + CHECK(json(1.5) < json(2)); + CHECK(json(2) > json(1.5)); + + // a NaN operand stays unordered against either integer kind + CHECK_FALSE(json(1) == json(nan)); + CHECK_FALSE(json(1) < json(nan)); + CHECK_FALSE(json(nan) < json(1)); + CHECK_FALSE(json(1u) == json(nan)); + +#if JSON_HAS_THREE_WAY_COMPARISON + // JSON_HAS_CPP_20 (do not remove; see note at top of file) + CHECK((max_int64 <=> two_63) == std::partial_ordering::less); // *NOPAD* + CHECK((two_63 <=> max_int64) == std::partial_ordering::greater); // *NOPAD* + CHECK((below_two_63 <=> max_int64) == std::partial_ordering::less); // *NOPAD* + CHECK((max_uint64 <=> two_64) == std::partial_ordering::less); // *NOPAD* + CHECK((json(1) <=> json(1.0)) == std::partial_ordering::equivalent); // *NOPAD* + CHECK((json(1) <=> json(nan)) == std::partial_ordering::unordered); // *NOPAD* +#endif + } + SECTION("compares unordered") { std::vector> expected = diff --git a/tests/src/unit-convenience.cpp b/tests/src/unit-convenience.cpp index 037a4e589..0ba57d846 100644 --- a/tests/src/unit-convenience.cpp +++ b/tests/src/unit-convenience.cpp @@ -98,8 +98,10 @@ void check_escaped(const char* original, const char* escaped = "", bool ensure_a void check_escaped(const char* original, const char* escaped, const bool ensure_ascii) { std::stringstream ss; - json::serializer s(nlohmann::detail::output_adapter(ss), ' '); - s.dump_escaped(original, ensure_ascii); + nlohmann::detail::output_stream_adapter adapter(ss); + json::serializer s(adapter, ' ', false, ensure_ascii); + s.dump_escaped(original); + s.flush(); // dump_escaped writes into the serializer's internal buffer CHECK(ss.str() == escaped); } } // namespace diff --git a/tests/src/unit-conversions.cpp b/tests/src/unit-conversions.cpp index 1937affbb..90d972f71 100644 --- a/tests/src/unit-conversions.cpp +++ b/tests/src/unit-conversions.cpp @@ -1389,6 +1389,37 @@ TEST_CASE("value conversion") // CHECK(m5["one"] == "eins"); } + SECTION("reserve is called on containers that support it (#5406)") + { + // build a larger object so that a missing/incorrect reserve() + // call would be more likely to corrupt or drop elements + json j_large; + for (int i = 0; i < 100; ++i) + { + j_large[std::to_string(i)] = i; + } + + SECTION("std::unordered_map (supports reserve)") + { + const auto m = j_large.get>(); + CHECK(m.size() == 100); + for (int i = 0; i < 100; ++i) + { + CHECK(m.at(std::to_string(i)) == i); + } + } + + SECTION("std::map (no reserve, fallback path)") + { + const auto m = j_large.get>(); + CHECK(m.size() == 100); + for (int i = 0; i < 100; ++i) + { + CHECK(m.at(std::to_string(i)) == i); + } + } + } + SECTION("std::multimap") { j1.get>(); @@ -1761,6 +1792,40 @@ TEST_CASE("std::filesystem::path") } #endif +// the ADL to_json overload for std::u8string only exists under the same guard +// as std::filesystem::path support (it is otherwise only reached indirectly, +// via std::filesystem::path::u8string()) -- mirror both #if conditions from +// include/nlohmann/detail/conversions/to_json.hpp exactly +#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM +#if defined(__cpp_lib_char8_t) +TEST_CASE("std::u8string") +{ + SECTION("ascii") + { + const std::u8string s = u8"Path"; + json const j = s; + + CHECK(j.template get() == "Path"); + } + + SECTION("utf-8") + { + // use \u universal-character-names (rather than raw \x byte escapes + // or literal non-ASCII source bytes) to compose the multi-byte UTF-8 + // encoding -- MSVC treats \x escapes used that way inside a u8 + // literal as a nonstandard extension (warning C5321), which some of + // our CI configs promote to an error; \u is portable and produces + // the exact same encoded bytes without depending on the source + // file's encoding + const std::u8string s = u8"P\u011B\u0161ina"; + json const j = s; + + CHECK(j.template get() == "P\xc4\x9b\xc5\xa1ina"); + } +} +#endif +#endif + TEST_CASE("std::optional") { SECTION("null") diff --git a/tests/src/unit-custom-array-type.cpp b/tests/src/unit-custom-array-type.cpp new file mode 100644 index 000000000..00606c6e0 --- /dev/null +++ b/tests/src/unit-custom-array-type.cpp @@ -0,0 +1,150 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include + +#include +#include +#include +#include +#include +#include + +namespace +{ + +// std::deque has no capacity() member function, which the library only needs +// to detect a reallocation for JSON_DIAGNOSTICS +using deque_json = nlohmann::basic_json; + +// a std::vector whose at() is hidden: the library performs its own bounds +// check and must not fall back to the container's checked accessor +template> +class vector_without_at : public std::vector +{ + public: + vector_without_at() = default; + + // the array of an initializer list is built from a range + template + vector_without_at(InputIt first, InputIt last) : std::vector(first, last) {} + + void at() = delete; +}; + +using no_at_json = nlohmann::basic_json; + +} // namespace + +TEST_CASE("array type without capacity()") +{ + SECTION("the iterators take their exception specification from the container") + { + // basic_json's iterators move exactly as the container iterators do: + // their move operations are defaulted without a declared noexcept, + // because an array or object type whose iterator is not nothrow move + // constructible would otherwise have them deleted (std::deque's is not + // with libstdc++ before 11, and neither are MSVC's debug iterators) + CHECK(std::is_nothrow_move_constructible::value == + (std::is_nothrow_move_constructible::value + && std::is_nothrow_move_constructible::value)); + CHECK(std::is_nothrow_move_assignable::value == + (std::is_nothrow_move_assignable::value + && std::is_nothrow_move_assignable::value)); + CHECK(std::is_nothrow_move_constructible::value == + (std::is_nothrow_move_constructible::value + && std::is_nothrow_move_constructible::value)); + + // and they are movable at all, which is what dropping the declared + // noexcept buys for a std::deque array + CHECK(std::is_move_constructible::value); + CHECK(std::is_move_assignable::value); + } + + SECTION("adding elements") + { + deque_json j = deque_json::array(); + j.push_back(1); + j.push_back("two"); + j.emplace_back(3); + j += 4; + + CHECK(j.size() == 4); + CHECK(j == deque_json({1, "two", 3, 4})); + CHECK(j.back() == 4); + CHECK(j.front() == 1); + } + + SECTION("accessing and modifying elements") + { + auto j = deque_json::parse(R"([1,2,3])"); + + CHECK(j[1] == 2); + CHECK(j.at(2) == 3); + + // growing through operator[] fills up with null values + j[5] = 6; + CHECK(j.size() == 6); + CHECK(j[4].is_null()); + CHECK(j[5] == 6); + + j.erase(0); + CHECK(j == deque_json({2, 3, nullptr, nullptr, 6})); + + auto it = j.erase(j.begin()); + CHECK(*it == 3); + + j.insert(j.begin(), 1); + CHECK(j.front() == 1); + } + + SECTION("serialization and deserialization") + { + const auto j = deque_json::parse(R"({"a":[1,[2,3]],"b":[]})"); + CHECK(j.dump() == R"({"a":[1,[2,3]],"b":[]})"); + CHECK(deque_json::parse(j.dump()) == j); + CHECK(deque_json::from_cbor(deque_json::to_cbor(j)) == j); + + // empty containers are flattened to null and cannot be restored + const auto nested = deque_json::parse(R"({"a":[1,[2,3]]})"); + CHECK(nested.flatten().unflatten() == nested); + } + + SECTION("references stay valid while the array grows") + { + deque_json j = deque_json::array(); + j.push_back(1); + auto& first = j[0]; + for (int i = 0; i < 100; ++i) + { + j.push_back(i); + } + CHECK(&first == &j[0]); + CHECK(first == 1); + } +} + +TEST_CASE("array type without at()") +{ + // built in memory rather than parsed, so that the exception message does + // not gain a byte range with JSON_DIAGNOSTIC_POSITIONS + no_at_json j = {1, 2, 3}; + const auto& jc = j; + + CHECK(j.at(0) == 1); + CHECK(j.at(2) == 3); + CHECK(jc.at(2) == 3); + + CHECK_THROWS_WITH_AS(j.at(3), "[json.exception.out_of_range.401] array index 3 is out of range", no_at_json::out_of_range); + CHECK_THROWS_WITH_AS(jc.at(3), "[json.exception.out_of_range.401] array index 3 is out of range", no_at_json::out_of_range); + + CHECK(j.at(no_at_json::json_pointer("/1")) == 2); + CHECK_THROWS_AS(j.at(no_at_json::json_pointer("/3")), no_at_json::out_of_range); +} diff --git a/tests/src/unit-custom-binary-type.cpp b/tests/src/unit-custom-binary-type.cpp new file mode 100644 index 000000000..d357ec9a3 --- /dev/null +++ b/tests/src/unit-custom-binary-type.cpp @@ -0,0 +1,79 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include + +#include +#include +#include +#include +#include +#include + +#ifdef JSON_HAS_CPP_17 + #include +#endif + +namespace +{ + +// a BinaryType whose value type is signed: the elements must still be +// processed as the numbers 0..255 +using char_binary_json = nlohmann::basic_json < + std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, + double, std::allocator, nlohmann::adl_serializer, std::vector, void >; + +#ifdef JSON_HAS_CPP_17 + // a BinaryType whose value type is not an integer type at all + using byte_binary_json = nlohmann::basic_json < + std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, + double, std::allocator, nlohmann::adl_serializer, std::vector, void >; +#endif + +} // namespace + +TEST_CASE("binary type whose value type is not std::uint8_t") +{ + SECTION("a signed value type does not dump negative numbers") + { + const std::vector chars{'\0', '\x01', '\xFF'}; + CHECK(char_binary_json::binary(chars).dump() == R"({"bytes":[0,1,255],"subtype":null})"); + CHECK(char_binary_json::binary(chars, 42).dump() == R"({"bytes":[0,1,255],"subtype":42})"); + CHECK(char_binary_json::binary({}).dump() == R"({"bytes":[],"subtype":null})"); + } + + SECTION("the default binary type is unchanged") + { + CHECK(nlohmann::json::binary({0, 1, 255}, 42).dump() == R"({"bytes":[0,1,255],"subtype":42})"); + } + +#ifdef JSON_HAS_CPP_17 + SECTION("dumping a value type that is not an integer") + { + const std::vector bytes{std::byte{0}, std::byte{1}, std::byte{0xFF}}; + CHECK(byte_binary_json::binary(bytes).dump() == R"({"bytes":[0,1,255],"subtype":null})"); + CHECK(byte_binary_json::binary(bytes, 42).dump() == R"({"bytes":[0,1,255],"subtype":42})"); + CHECK(byte_binary_json::binary({}).dump() == R"({"bytes":[],"subtype":null})"); + } + + SECTION("hashing and the binary formats") + { + const std::vector bytes{std::byte{0}, std::byte{1}, std::byte{0xFF}}; + const auto j = byte_binary_json::binary(bytes); + + CHECK(std::hash {}(j) == std::hash {}(j)); + CHECK(byte_binary_json::from_cbor(byte_binary_json::to_cbor(j)) == j); + CHECK(byte_binary_json::from_msgpack(byte_binary_json::to_msgpack(j)) == j); + + // UBJSON has no binary type, so binary values are written as an array + CHECK(byte_binary_json::from_ubjson(byte_binary_json::to_ubjson(j)) == byte_binary_json({0, 1, 255})); + } +#endif +} diff --git a/tests/src/unit-custom-object-type.cpp b/tests/src/unit-custom-object-type.cpp new file mode 100644 index 000000000..cb2cb5ff3 --- /dev/null +++ b/tests/src/unit-custom-object-type.cpp @@ -0,0 +1,323 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include + +#include +#include +#include +#include +#include +#include + + +namespace +{ + +// An ObjectType that does *not* define a key_compare member type, which is +// what every hash map looks like to the library. +// +// A hash map is deliberately not used here: object_t is probed for +// key_compare inside the definition of basic_json, that is, while basic_json +// is still an incomplete type, and whether a hash map can be instantiated +// with an incomplete mapped type depends on the standard library (libstdc++ 9 +// needs the size of the mapped type for its node type and rejects it). So the +// object type wraps a std::map instead of inheriting from it: an earlier +// version derived from std::map and shadowed the inherited key_compare type +// with a same-named member function, relying on ordinary member hiding to +// make key_compare unreachable as a type. MSVC 2017 (AppVeyor, /std:c++17) +// does not honor that hiding for a typename-qualified lookup performed from +// outside the class and still resolves key_compare to the base's comparator +// type, so the library's probe incorrectly found one. Composition sidesteps +// the question entirely: with no base class, there is no key_compare to find +// under any lookup rule. +template +class no_key_compare_map +{ + using map_t = std::map; + map_t data; + + public: + using key_type = typename map_t::key_type; + using mapped_type = typename map_t::mapped_type; + using value_type = typename map_t::value_type; + using size_type = typename map_t::size_type; + using allocator_type = typename map_t::allocator_type; + using iterator = typename map_t::iterator; + using const_iterator = typename map_t::const_iterator; + + // -Weffc++ asks for the member to be initialized in the member + // initialization list, which a defaulted constructor does not do; the + // exception specification a defaulted one would have carried has to be + // written out as well, or -Wnoexcept objects where the standard library + // takes noexcept(construct(...)) + no_key_compare_map() noexcept(std::is_nothrow_default_constructible::value) : data() {} + + // converting between two basic_json types builds the object from a range + template + no_key_compare_map(InputIt first, InputIt last) : data(first, last) {} + + iterator begin() noexcept + { + return data.begin(); + } + iterator end() noexcept + { + return data.end(); + } + const_iterator begin() const noexcept + { + return data.begin(); + } + const_iterator end() const noexcept + { + return data.end(); + } + const_iterator cbegin() const noexcept + { + return data.cbegin(); + } + const_iterator cend() const noexcept + { + return data.cend(); + } + + bool empty() const noexcept + { + return data.empty(); + } + size_type size() const noexcept + { + return data.size(); + } + size_type max_size() const noexcept + { + return data.max_size(); + } + void clear() noexcept + { + data.clear(); + } + + iterator find(const key_type& key) + { + return data.find(key); + } + const_iterator find(const key_type& key) const + { + return data.find(key); + } + size_type count(const key_type& key) const + { + return data.count(key); + } + + std::pair emplace(const key_type& key, const mapped_type& value) + { + return data.emplace(key, value); + } + + std::pair insert(const value_type& value) + { + return data.insert(value); + } + + template + void insert(InputIt first, InputIt last) + { + data.insert(first, last); + } + + mapped_type& operator[](const key_type& key) + { + return data[key]; + } + + mapped_type& at(const key_type& key) + { + return data.at(key); + } + const mapped_type& at(const key_type& key) const + { + return data.at(key); + } + + iterator erase(iterator pos) + { + return data.erase(pos); + } + iterator erase(iterator first, iterator last) + { + return data.erase(first, last); + } + size_type erase(const key_type& key) + { + return data.erase(key); + } + + void swap(no_key_compare_map& other) noexcept(noexcept(data.swap(other.data))) + { + data.swap(other.data); + } + + friend bool operator==(const no_key_compare_map& lhs, const no_key_compare_map& rhs) + { + return lhs.data == rhs.data; + } + friend bool operator<(const no_key_compare_map& lhs, const no_key_compare_map& rhs) + { + return lhs.data < rhs.data; + } +}; + +using no_key_compare_json = nlohmann::basic_json; + +// An ObjectType whose erase(iterator) returns void rather than the following +// iterator, as for instance Abseil's hash maps do +template +struct void_erase_map : std::map +{ + using base_t = std::map; + using iterator = typename base_t::iterator; + using base_t::erase; + + void erase(iterator pos) + { + base_t::erase(pos); + } +}; + +using void_erase_json = nlohmann::basic_json; + +} // namespace + +TEST_CASE("object type whose erase() returns void") +{ + SECTION("erasing every element through the returned iterator") + { + void_erase_json j; + for (int i = 0; i < 8; ++i) + { + j["k" + std::to_string(i)] = i; + } + + std::size_t erased = 0; + for (auto it = j.begin(); it != j.end(); ++erased) + { + it = j.erase(it); + } + CHECK(erased == 8); + CHECK(j.empty()); + } + + SECTION("erasing in the middle returns the following element") + { + void_erase_json j; + for (int i = 0; i < 4; ++i) + { + j["k" + std::to_string(i)] = i; + } + + auto it = j.begin(); + ++it; + const auto after = j.erase(it); + CHECK(j.size() == 3); + CHECK(after.key() == "k2"); + CHECK(after.value() == 2); + CHECK(!j.contains("k1")); + } + + SECTION("the other erase overloads are unaffected") + { + void_erase_json j; + j["a"] = 1; + j["b"] = 2; + j["c"] = 3; + + CHECK(j.erase("a") == 1); + CHECK(j.erase("nope") == 0); + j.erase(j.begin(), j.end()); + CHECK(j.empty()); + } +} + +TEST_CASE("object type without key_compare") +{ + SECTION("object_comparator_t falls back to default_object_comparator_t") + { + CHECK(std::is_same < no_key_compare_json::object_comparator_t, + no_key_compare_json::default_object_comparator_t >::value); + } + + SECTION("object types defining key_compare are unaffected") + { + CHECK(std::is_same::value); + CHECK(std::is_same::value); + } + + SECTION("creating and accessing values") + { + no_key_compare_json j; + j["one"] = 1; + j["two"] = "zwei"; + j["three"]["nested"] = true; + + CHECK(j.size() == 3); + CHECK(j.at("one") == 1); + CHECK(j["two"] == "zwei"); + CHECK(j["three"]["nested"] == true); + CHECK(j.contains("one")); + CHECK(!j.contains("four")); + CHECK(j.find("one") != j.end()); + CHECK(j.count("one") == 1); + CHECK(j.erase("one") == 1); + CHECK(j.size() == 2); + } + + SECTION("serialization and deserialization") + { + const auto j = no_key_compare_json::parse(R"({"a":[1,2,3],"b":{"c":null}})"); + CHECK(j["a"].size() == 3); + CHECK(j["a"][2] == 3); + CHECK(j["b"]["c"].is_null()); + CHECK(no_key_compare_json::parse(j.dump()) == j); + } + + SECTION("binary formats") + { + const auto j = no_key_compare_json::parse(R"({"a":[1,2,3],"b":"x"})"); + CHECK(no_key_compare_json::from_cbor(no_key_compare_json::to_cbor(j)) == j); + CHECK(no_key_compare_json::from_msgpack(no_key_compare_json::to_msgpack(j)) == j); + } + + SECTION("flatten and unflatten") + { + // "o" has a key that looks like an array index, so unflatten() must + // not turn it into an array + const auto j = no_key_compare_json::parse( + R"({"c":[1,2,3],"d":{"e":"s"},"n":[[0,1],[2]],"o":{"2":"x"}})"); + CHECK(j.flatten().unflatten() == j); + } + + SECTION("conversion to and from nlohmann::json") + { + const auto j = no_key_compare_json::parse(R"({"a":1,"b":[true,null]})"); + const nlohmann::json converted(j); + + CHECK(converted.is_object()); + CHECK(converted["a"] == 1); + CHECK(converted["b"][0] == true); + CHECK(converted["b"][1].is_null()); + CHECK(no_key_compare_json(converted) == j); + } +} + diff --git a/tests/src/unit-deserialization.cpp b/tests/src/unit-deserialization.cpp index 7879e5642..2d4439395 100644 --- a/tests/src/unit-deserialization.cpp +++ b/tests/src/unit-deserialization.cpp @@ -8,6 +8,14 @@ #include "doctest_compatibility.h" +// capture whether JSON_STRICT_NUL_HANDLING was enabled on the command line +// (e.g. -DJSON_STRICT_NUL_HANDLING=1) *before* including json.hpp, since the +// library #undefs JSON_STRICT_NUL_HANDLING itself once the header has been +// fully processed (see include/nlohmann/detail/macro_unscope.hpp) +#if defined(JSON_STRICT_NUL_HANDLING) && (JSON_STRICT_NUL_HANDLING == 1) + #define JSON_TEST_STRICT_NUL_HANDLING_ENABLED 1 +#endif + #include using nlohmann::json; #ifdef JSON_TEST_NO_GLOBAL_UDLS @@ -380,6 +388,23 @@ TEST_CASE("deserialization") CHECK(j == json({"foo", 1, 2, 3, false, {{"one", 1}}})); } + SECTION("operator>> with a NUL byte after the value (issue #5530)") + { + // operator>> parses non-strictly (it does not require the whole + // stream to be consumed), so a NUL byte following a complete + // value is simply left unread on the stream and never reaches + // the "expected end of input" check that JSON_STRICT_NUL_HANDLING + // affects; this holds regardless of the macro (verified below for + // the opt-in state as well) + std::string data = "123"; + data.push_back('\0'); + std::istringstream ss(data); + json j; + ss >> j; + CHECK(j == json(123)); + CHECK(ss.good()); + } + SECTION("user-defined string literal") { CHECK("[\"foo\",1,2,3,false,{\"one\":1}]"_json == json({"foo", 1, 2, 3, false, {{"one", 1}}})); @@ -462,6 +487,27 @@ TEST_CASE("deserialization") CHECK_THROWS_WITH_AS(ss >> j, "[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing array - unexpected end of input; expected ']'", json::parse_error&); } +#if defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED) + SECTION("operator>> with a NUL byte where a value is expected (JSON_STRICT_NUL_HANDLING == 1, issue #5530)") + { + // a trailing NUL byte *after* a complete value is unaffected by the + // macro (see the successful-deserialization "operator>> with a NUL + // byte after the value" section above): operator>> parses + // non-strictly and never reaches the "expected end of input" check + // that the macro changes. A NUL byte where a *value* is expected, + // however, goes through the same token dispatch as any other input + // and is affected: with the macro enabled it now raises + // parse_error.101 (like any other unrecognized byte) instead of + // being silently treated the same as an empty stream. + std::string const data(1, '\0'); + std::istringstream ss(data); + json j; + CHECK_THROWS_WITH_AS(ss >> j, + "[json.exception.parse_error.101] parse error at line 1, column 1: syntax error while parsing value - invalid literal; last read: ''", + json::parse_error&); + } +#endif + SECTION("user-defined string literal") { CHECK_THROWS_WITH_AS("[\"foo\",1,2,3,false,{\"one\":1}"_json, "[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing array - unexpected end of input; expected ']'", json::parse_error&); @@ -510,7 +556,11 @@ TEST_CASE("deserialization") SECTION("from std::array") { - std::array const v { {'t', 'r', 'u', 'e'} }; + // sized to exactly the length of "true": a size of 5 would leave + // a value-initialized trailing 0x00 element that is only + // silently accepted as end-of-input by default and would fail + // under JSON_STRICT_NUL_HANDLING + std::array const v { {'t', 'r', 'u', 'e'} }; CHECK(json::parse(v) == json(true)); CHECK(json::accept(v)); @@ -606,7 +656,9 @@ TEST_CASE("deserialization") SECTION("from std::array") { - std::array v { {'t', 'r', 'u', 'e'} }; + // sized to exactly the length of "true", see the analogous + // "from std::array" section above for why + std::array v { {'t', 'r', 'u', 'e'} }; CHECK(json::parse(std::begin(v), std::end(v)) == json(true)); CHECK(json::accept(std::begin(v), std::end(v))); diff --git a/tests/src/unit-diagnostic-positions-only.cpp b/tests/src/unit-diagnostic-positions-only.cpp deleted file mode 100644 index 735376514..000000000 --- a/tests/src/unit-diagnostic-positions-only.cpp +++ /dev/null @@ -1,44 +0,0 @@ -// __ _____ _____ _____ -// __| | __| | | | JSON for Modern C++ (supporting code) -// | | |__ | | | | | | version 3.12.0 -// |_____|_____|_____|_|___| https://github.com/nlohmann/json -// -// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann -// SPDX-License-Identifier: MIT - -#include "doctest_compatibility.h" - -#ifdef JSON_DIAGNOSTICS - #undef JSON_DIAGNOSTICS -#endif - -#define JSON_DIAGNOSTICS 0 -#define JSON_DIAGNOSTIC_POSITIONS 1 -#include - -using json = nlohmann::json; - -TEST_CASE("Better diagnostics with positions only") -{ - SECTION("invalid type") - { - const std::string json_invalid_string = R"( - { - "address": { - "street": "Fake Street", - "housenumber": "1" - } - } - )"; - json j = json::parse(json_invalid_string); - CHECK_THROWS_WITH_AS(j.at("address").at("housenumber").get(), - "[json.exception.type_error.302] (bytes 108-111) type must be number, but is string", json::type_error); - } - - SECTION("invalid type without positions") - { - const json j = "foo"; - CHECK_THROWS_WITH_AS(j.get(), - "[json.exception.type_error.302] type must be number, but is string", json::type_error); - } -} diff --git a/tests/src/unit-diagnostic-positions.cpp b/tests/src/unit-diagnostic-positions.cpp index ad9527540..d607e935c 100644 --- a/tests/src/unit-diagnostic-positions.cpp +++ b/tests/src/unit-diagnostic-positions.cpp @@ -8,7 +8,9 @@ #include "doctest_compatibility.h" -#define JSON_DIAGNOSTICS 1 +#ifndef JSON_DIAGNOSTICS + #define JSON_DIAGNOSTICS 1 +#endif #define JSON_DIAGNOSTIC_POSITIONS 1 #include @@ -27,8 +29,13 @@ TEST_CASE("Better diagnostics with positions") } )"; json j = json::parse(json_invalid_string); +#if JSON_DIAGNOSTICS CHECK_THROWS_WITH_AS(j.at("address").at("housenumber").get(), "[json.exception.type_error.302] (/address/housenumber) (bytes 108-111) type must be number, but is string", json::type_error); +#else + CHECK_THROWS_WITH_AS(j.at("address").at("housenumber").get(), + "[json.exception.type_error.302] (bytes 108-111) type must be number, but is string", json::type_error); +#endif } SECTION("invalid type without positions") @@ -68,13 +75,84 @@ TEST_CASE("Better diagnostics with positions") CHECK(j.end_pos() == root.size()); } + SECTION("copying keeps the positions of nested values (#5387)") + { + // Values nested deeper than the copy constructor's descent bound are + // copied without the call stack, on a path that has to carry the + // positions over itself; shallower ones copy their containers, which + // bring the positions along. Both sides of the bound are checked here. + const auto check_copy = [](std::size_t depth, bool objects) + { + CAPTURE(depth) + CAPTURE(objects) + + const std::string opening = objects ? R"({"a":)" : "["; + const std::string closing = objects ? "}" : "]"; + + std::string text; + for (std::size_t i = 0; i < depth; ++i) + { + text += opening; + } + text += "12"; + for (std::size_t i = 0; i < depth; ++i) + { + text += closing; + } + + const json original = json::parse(text); + const json copy(original); // NOLINT(performance-unnecessary-copy-initialization) + + const json* o = &original; + const json* c = © + for (std::size_t level = 0; level <= depth; ++level) + { + CAPTURE(level) + REQUIRE(c->start_pos() == o->start_pos()); + REQUIRE(c->end_pos() == o->end_pos()); + + if (level < depth) + { + o = objects ? &o->at("a") : &o->at(0); + c = objects ? &c->at("a") : &c->at(0); + } + } + }; + + const auto check_arrays = [&check_copy](std::size_t depth) + { + check_copy(depth, false); + }; + const auto check_objects = [&check_copy](std::size_t depth) + { + check_copy(depth, true); + }; + + check_arrays(1); + check_arrays(127); + check_arrays(128); + check_arrays(129); + check_arrays(300); + + check_objects(1); + check_objects(127); + check_objects(128); + check_objects(129); + check_objects(300); + } + SECTION("JSON patch add to primitive parent (#4292)") { // the JSON Patch "add" target /foo/bar/baz has a string parent // (/foo/bar); the position of that parent is reported in the message const json doc = json::parse(R"({"foo":{"bar":"a string"}})"); const json patch = json::parse(R"([{"op":"add","path":"/foo/bar/baz","value":1}])"); +#if JSON_DIAGNOSTICS CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.411] (/foo/bar) (bytes 14-24) cannot add value: the JSON Patch 'add' target's parent is of type string, but must be an object or array", json::out_of_range); +#else + CHECK_THROWS_WITH_AS(doc.patch(patch), + "[json.exception.out_of_range.411] (bytes 14-24) cannot add value: the JSON Patch 'add' target's parent is of type string, but must be an object or array", json::out_of_range); +#endif } } diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index 1e8ed6aa3..3ae649e5b 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -273,5 +273,242 @@ TEST_CASE("Regression tests for extended diagnostics") CHECK(j1["numbers"]["two"] == 2); CHECK(j1["string"] == "t"); } + + SECTION("Regression test for issue #5387 - copying keeps the parents of nested values") + { + // A value nested deeper than the copy constructor's descent bound is + // copied without the call stack. Every container that path creates has + // to have the parents of its children set, or the JSON Pointer in the + // diagnostic is cut short. + const std::size_t depth = 300; + + SECTION("objects") + { + json j = "not a number"; + std::string pointer; + for (std::size_t i = 0; i < depth; ++i) + { + j = json{{"a", j}}; + pointer += "/a"; + } + + json const copy(j); // NOLINT(performance-unnecessary-copy-initialization) + + const json* inner = © + for (std::size_t i = 0; i < depth; ++i) + { + inner = &inner->at("a"); + } + + std::string const expected = "[json.exception.type_error.302] (" + pointer + ") type must be number, but is string"; + int i = 0; + CHECK_THROWS_WITH_AS(i = inner->get(), expected.c_str(), json::type_error); + CHECK(i == 0); + } + + SECTION("arrays") + { + json j = "not a number"; + std::string pointer; + for (std::size_t i = 0; i < depth; ++i) + { + j = json::array({j}); + pointer += "/0"; + } + + json const copy(j); // NOLINT(performance-unnecessary-copy-initialization) + + const json* inner = © + for (std::size_t i = 0; i < depth; ++i) + { + inner = &inner->at(0); + } + + std::string const expected = "[json.exception.type_error.302] (" + pointer + ") type must be number, but is string"; + int i = 0; + CHECK_THROWS_WITH_AS(i = inner->get(), expected.c_str(), json::type_error); + CHECK(i == 0); + } + } + + SECTION("Regression test - swap(array_t&)/swap(object_t&) must update JSON_DIAGNOSTICS parent pointers") + { + // swap(array_t&) + { + json j = json::array(); + json::array_t arr = {json::array({1})}; + j.swap(arr); + + // parent pointers of the moved-in elements must point into j, not + // into the now-defunct free-standing array_t + CHECK_THROWS_WITH_AS(j[0][0].get(), "[json.exception.type_error.302] (/0/0) type must be string, but is number", json::type_error); + + // must not trigger assert_invariant() in a debug/assert-enabled build + json const k = j; + CHECK(k == j); + } + + // swap(object_t&) + { + json o = json::object(); + json::object_t obj = {{"a", json::array({1})}}; + o.swap(obj); + + CHECK_THROWS_WITH_AS(o["a"][0].get(), "[json.exception.type_error.302] (/a/0) type must be string, but is number", json::type_error); + + // must not trigger assert_invariant() in a debug/assert-enabled build + json const p = o; + CHECK(p == o); + } + } + + SECTION("Regression test - erase() and update() must keep JSON_DIAGNOSTICS parent pointers of ordered_json members") + { + // ordered_json keeps its members in a vector: erasing a member + // re-constructs all members after it in place, and adding a key may + // reallocate the vector; both reset the parent pointers of the members + // that were moved + using nlohmann::ordered_json; + + const auto check_parents = [](const ordered_json & j) + { + // const access, so operator[] cannot repair the parent pointers + CHECK_THROWS_WITH_AS(j["z"]["x"].at(0), "[json.exception.type_error.304] (/z/x) cannot use at() with number", ordered_json::type_error); + + // must not trigger assert_invariant() in a debug/assert-enabled build + ordered_json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + }; + + // erase(key) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + CHECK(j.erase("a") == 1); + check_parents(j); + } + + // erase(iterator) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + j.erase(j.begin()); + check_parents(j); + } + + // erase(iterator, iterator) + { + ordered_json j = {{"a", 1}, {"b", 2}, {"z", {{"x", 1}}}}; + j.erase(j.begin(), j.find("z")); + check_parents(j); + } + + // patch() removes via erase(iterator) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + j.patch_inplace(ordered_json::parse(R"([{"op": "remove", "path": "/a"}])")); + check_parents(j); + } + + // update(j) + { + ordered_json j = {{"z", {{"x", 1}}}}; + j.update({{"a", 1}, {"b", 2}}); + check_parents(j); + } + + // update(j, true), the outer and the nested vector both grow + { + ordered_json j = {{"z", {{"x", 1}}}}; + j.update({{"z", {{"y", 2}}}, {"a", 1}}, true); + check_parents(j); + } + + // update(j, true) around its descent bound, where the nested vectors + // grow while the objects are merged without recursing + for (const std::size_t depth : + { + nlohmann::detail::recursion_depth_limit() - 1, nlohmann::detail::recursion_depth_limit(), nlohmann::detail::recursion_depth_limit() + 2 + }) + { + ordered_json j = {{"z", {{"x", 1}}}}; + ordered_json patch = {{"a", 1}, {"b", 2}, {"c", {{"d", 3}}}}; + for (std::size_t i = 0; i < depth; ++i) + { + j = ordered_json{{"k", 0}, {"n", std::move(j)}}; + patch = ordered_json{{"n", std::move(patch)}, {"l", 1}, {"m", 2}}; + } + j.update(patch, true); + + // must not trigger assert_invariant() on any level in a + // debug/assert-enabled build + ordered_json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + } + + // merge_patch() inserts "c" and removes "d" at /a/c, then inserts "e" + // at /a, which copies /a/c + { + auto j = ordered_json::parse(R"({"a": {"c": {"d": {}}}})"); + j.merge_patch(ordered_json::parse(R"({"a": {"c": {"c": "s", "d": null}, "e": "s"}})")); + CHECK(j.dump() == R"({"a":{"c":{"c":"s"},"e":"s"}})"); + + auto const& constJ = j; +#if JSON_DIAGNOSTIC_POSITIONS + CHECK_THROWS_WITH_AS(constJ["a"]["c"]["c"].at(0), "[json.exception.type_error.304] (/a/c/c) (bytes 18-21) cannot use at() with string", ordered_json::type_error); +#else + CHECK_THROWS_WITH_AS(constJ["a"]["c"]["c"].at(0), "[json.exception.type_error.304] (/a/c/c) cannot use at() with string", ordered_json::type_error); +#endif + ordered_json const copy = j; + CHECK(copy == j); + } + } } +TEST_CASE("Better diagnostics past the descent bound of update() and merge_patch()") +{ + // Both merge objects nested more than detail::recursion_depth_limit() + // (128) levels deep without recursing; the values they add or replace + // there must still know their parents. + // The values are built rather than parsed, so that the expected messages + // carry no byte positions under JSON_DIAGNOSTIC_POSITIONS. + const std::size_t depth = 200; + json target = {{"x", 1}}; + json patch = {{"y", 2}}; + std::string path; + for (std::size_t i = 0; i < depth; ++i) + { + target = json{{"a", std::move(target)}}; + patch = json{{"a", std::move(patch)}}; + path += "/a"; + } + const std::string expected_x = "[json.exception.type_error.304] (" + path + "/x) cannot use at() with number"; + const std::string expected_y = "[json.exception.type_error.304] (" + path + "/y) cannot use at() with number"; + + SECTION("update()") + { + json j = target; + j.update(patch, true); + + // walk down through const references, which leave m_parent alone + const json* p = &j; + for (std::size_t i = 0; i < depth; ++i) + { + p = &p->at("a"); + } + CHECK_THROWS_WITH_AS(p->at("x").at(0), expected_x.c_str(), json::type_error); + CHECK_THROWS_WITH_AS(p->at("y").at(0), expected_y.c_str(), json::type_error); + } + + SECTION("merge_patch()") + { + json j = target; + j.merge_patch(patch); + + const json* p = &j; + for (std::size_t i = 0; i < depth; ++i) + { + p = &p->at("a"); + } + CHECK_THROWS_WITH_AS(p->at("x").at(0), expected_x.c_str(), json::type_error); + CHECK_THROWS_WITH_AS(p->at("y").at(0), expected_y.c_str(), json::type_error); + } +} diff --git a/tests/src/unit-hash.cpp b/tests/src/unit-hash.cpp index c161efa6e..eb843c291 100644 --- a/tests/src/unit-hash.cpp +++ b/tests/src/unit-hash.cpp @@ -13,6 +13,78 @@ using json = nlohmann::json; using ordered_json = nlohmann::ordered_json; #include +#include + +namespace +{ +// how detail::hash defines the hash of an array or object: the seeds of the +// elements, combined in order. Recursive, so only usable on values nested a +// few hundred levels deep - which is exactly what is needed to check that the +// iterative path taken below detail::recursion_depth_limit() computes the same. +template +std::size_t reference_hash(const BasicJsonType& j) +{ + using nlohmann::detail::combine; + using string_t = typename BasicJsonType::string_t; + + if (!j.is_structured()) + { + return std::hash {}(j); + } + + auto seed = combine(static_cast(j.type()), j.size()); + for (const auto& element : j.items()) + { + if (j.is_object()) + { + seed = combine(seed, std::hash {}(element.key())); + } + seed = combine(seed, reference_hash(element.value())); + } + return seed; +} + +// a value nested `depth` levels deep, with siblings on every level +template +BasicJsonType nested(const std::size_t depth, const bool objects) +{ + BasicJsonType value = "leaf"; + for (std::size_t i = 0; i < depth; ++i) + { + if (objects) + { + value = BasicJsonType{{"before", i}, {"nested", std::move(value)}, {"after", {i, "x"}}}; + } + else + { + value = BasicJsonType::array({i, std::move(value), BasicJsonType::object({{"k", i}})}); + } + } + return value; +} + +std::string nested_text(const std::size_t depth, const bool objects) +{ + std::string text; + if (objects) + { + text.reserve((6 * depth) + 1); + for (std::size_t i = 0; i < depth; ++i) + { + text += "{\"a\":"; + } + text += "1"; + text.append(depth, '}'); + } + else + { + text.assign(depth, '['); + text += "1"; + text.append(depth, ']'); + } + return text; +} +} // namespace TEST_CASE("hash") { @@ -111,3 +183,44 @@ TEST_CASE("hash") CHECK(hashes.size() == 21); } + +TEST_CASE("hash of deeply nested values") +{ + SECTION("hashing past the descent bound computes the same values") + { + // every depth on either side of where the iterative path takes over + for (std::size_t depth = 0; depth <= (2 * nlohmann::detail::recursion_depth_limit()) + 10; ++depth) + { + CAPTURE(depth); + const auto arrays = nested(depth, false); + const auto objects = nested(depth, true); + const auto ordered = nested(depth, true); + CHECK(std::hash {}(arrays) == reference_hash(arrays)); + CHECK(std::hash {}(objects) == reference_hash(objects)); + CHECK(std::hash {}(ordered) == reference_hash(ordered)); + } + } + + SECTION("values nested too deeply for the call stack (#5545)") + { + // recursing once per level used to exhaust the call stack here; the + // values are only parsed and hashed, never copied or compared, since + // those recurse as well + const std::size_t depth = 100000; + for (const bool objects : + { + false, true + }) + { + CAPTURE(objects); + const auto text = nested_text(depth, objects); + const auto a = json::parse(text); + const auto b = json::parse(text); + CHECK(std::hash {}(a) == std::hash {}(b)); + + const auto c = ordered_json::parse(text); + const auto d = ordered_json::parse(text); + CHECK(std::hash {}(c) == std::hash {}(d)); + } + } +} diff --git a/tests/src/unit-items-cpp17.cpp b/tests/src/unit-items-cpp17.cpp new file mode 100644 index 000000000..577dcea56 --- /dev/null +++ b/tests/src/unit-items-cpp17.cpp @@ -0,0 +1,42 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// This file contains the C++17-only part of unit-items.cpp (structured +// bindings support for json::items()). It is kept in a separate +// translation unit so the (much larger) unit-items.cpp does not need to +// be compiled a second time just for this one SECTION. + +#include "doctest_compatibility.h" + +#include +using nlohmann::json; + +#ifdef JSON_HAS_CPP_17 +#include +#include + +TEST_CASE("items()") +{ + SECTION("object") + { + SECTION("structured bindings") + { + json j = { {"A", 1}, {"B", 2} }; + + std::map m; + + for (auto const&[key, value] : j.items()) + { + m.emplace(key, value); + } + + CHECK(j.get() == m); + } + } +} +#endif diff --git a/tests/src/unit-items.cpp b/tests/src/unit-items.cpp index fa8948447..81959db8d 100644 --- a/tests/src/unit-items.cpp +++ b/tests/src/unit-items.cpp @@ -862,22 +862,6 @@ TEST_CASE("items()") CHECK(counter == 3); } - -#ifdef JSON_HAS_CPP_17 - SECTION("structured bindings") - { - json j = { {"A", 1}, {"B", 2} }; - - std::map m; - - for (auto const&[key, value] : j.items()) - { - m.emplace(key, value); - } - - CHECK(j.get() == m); - } -#endif } SECTION("const object") diff --git a/tests/src/unit-json_patch.cpp b/tests/src/unit-json_patch.cpp index 404dee43c..7731c7d92 100644 --- a/tests/src/unit-json_patch.cpp +++ b/tests/src/unit-json_patch.cpp @@ -672,6 +672,102 @@ TEST_CASE("JSON patch") } } + SECTION("patch_inplace") + { + SECTION("happy path: patch_inplace mirrors patch() on success") + { + // mirrors "A.5. Replacing a Value" above, but applies the patch with + // patch_inplace() to a mutable copy instead of using patch()'s + // returned copy + json doc = R"( + { + "baz": "qux", + "foo": "bar" + } + )"_json; + + json const patch = R"( + [ + { "op": "replace", "path": "/baz", "value": "boo" } + ] + )"_json; + + json const expected = R"( + { + "baz": "boo", + "foo": "bar" + } + )"_json; + + doc.patch_inplace(patch); + CHECK(doc == expected); + } + + // this test relies on the "test" operation actually throwing so the + // partial-application state can be observed right after the throw + // point; under JSON_NOEXCEPTION, JSON_THROW() calls std::abort() + // instead (there is no C++ exception to throw), and doctest's + // CHECK_THROWS_AS() is compiled out to a no-op that never even + // invokes the given expression (see doctest's "--no-throw" test + // filter, which ci_test_noexceptions passes) -- so patch()/ + // patch_inplace() would never be called at all and the follow-up + // state assertions below would fail against the untouched original +#if !defined(JSON_NOEXCEPTION) + SECTION("distinguishing contract vs patch(): partial application on failure") + { + // Unlike patch(), which is all-or-nothing because it applies the + // patch to an internal copy that is simply discarded when an + // exception is thrown (leaving the original untouched no matter + // what), patch_inplace() mutates the document it is called on + // directly and immediately, operation by operation. So if a JSON + // Patch fails partway through, whatever operations already + // succeeded remain applied -- the document is left in a partially + // patched state. This is empirically verified current behavior, + // not just documented intent, and is pinned here as such. + json const original = R"( + { + "baz": "qux", + "foo": "bar" + } + )"_json; + + // the first operation ("replace") succeeds; the second ("test") + // fails because the value at "/baz" no longer (and never did) + // equal "not boo" + json const patch = R"( + [ + { "op": "replace", "path": "/baz", "value": "boo" }, + { "op": "test", "path": "/baz", "value": "not boo" } + ] + )"_json; + + // patch() never modifies the object it is called on -- it always + // operates on (and returns) a separate copy, so the original is + // left completely untouched, regardless of success or failure. + // copy_for_patch is intentionally a real copy, not a reference + // to `original`: the whole point of this check is to catch a + // hypothetical future regression where patch() *does* mutate its + // receiver. Using a reference here would make the assertion + // below compare `original` to itself -- trivially true even if + // such a bug existed -- which is exactly what a static analyzer + // can't see when it suggests "this copy is never modified, use + // a reference instead". + json copy_for_patch = original; // NOLINT(performance-unnecessary-copy-initialization) + CHECK_THROWS_AS(copy_for_patch.patch(patch), json::other_error&); + CHECK(copy_for_patch == original); + + // patch_inplace(), in contrast, already applied the successful + // "replace" operation to the document before the "test" operation + // threw -- that change is not rolled back + json doc = original; + CHECK_THROWS_AS(doc.patch_inplace(patch), json::other_error&); + CHECK(doc != original); + CHECK(doc.at("baz") == "boo"); + CHECK(doc.at("foo") == "bar"); + } +#endif // !defined(JSON_NOEXCEPTION) + } + SECTION("errors") { SECTION("unknown operation") @@ -1388,3 +1484,270 @@ TEST_CASE("JSON patch - add to a primitive parent (regression #4292)") CHECK_THROWS_AS(doc.patch(patch), json::out_of_range&); } } + +TEST_CASE("JSON patch - remove with primitive or null parent (regression #5396)") +{ + // Regression test for https://github.com/nlohmann/json/issues/5396 + // + // RFC 6902 (§4.2) requires the target location of a "remove" operation + // to exist. When the target's parent resolves to a primitive value or + // null, the operation must fail. Previously operation_remove silently + // did nothing in this case (neither the "is_object" nor the "is_array" + // branch matched, and there was no final "else"), so the patch appeared + // to succeed without changing the document. It now throws + // out_of_range.413. + + SECTION("parent is a primitive (number)") + { + json const doc = {{"a", 1}}; + json const patch = {{{"op", "remove"}, {"path", "/a/b"}}}; +#if JSON_DIAGNOSTICS + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] (/a) cannot remove value: the JSON Patch 'remove' target's parent is of type number, but must be an object or array", json::out_of_range&); +#else + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] cannot remove value: the JSON Patch 'remove' target's parent is of type number, but must be an object or array", json::out_of_range&); +#endif + } + + SECTION("parent is a primitive (string)") + { + json const doc = {{"foo", {{"bar", "a string"}}}}; + json const patch = {{{"op", "remove"}, {"path", "/foo/bar/baz"}}}; +#if JSON_DIAGNOSTICS + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] (/foo/bar) cannot remove value: the JSON Patch 'remove' target's parent is of type string, but must be an object or array", json::out_of_range&); +#else + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] cannot remove value: the JSON Patch 'remove' target's parent is of type string, but must be an object or array", json::out_of_range&); +#endif + } + + SECTION("top-level document is null") + { + json const doc = nullptr; + json const patch = {{{"op", "remove"}, {"path", "/a"}}}; + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] cannot remove value: the JSON Patch 'remove' target's parent is of type null, but must be an object or array", json::out_of_range&); + } + + SECTION("legitimate removes still work") + { + // object member + json const doc1 = {{"a", 1}, {"b", 2}}; + json const patch1 = {{{"op", "remove"}, {"path", "/a"}}}; + CHECK(doc1.patch(patch1) == json({{"b", 2}})); + + // array element + json const doc2 = R"([1, 2, 3])"_json; + json const patch2 = {{{"op", "remove"}, {"path", "/1"}}}; + CHECK(doc2.patch(patch2) == R"([1, 3])"_json); + } +} + +TEST_CASE("JSON patch - move where 'from' is a proper prefix of 'path' (regression #5397)") +{ + // Regression test for https://github.com/nlohmann/json/issues/5397 + // + // RFC 6902 (§4.4) forbids "from" from being a proper prefix of "path" + // for a "move" operation: "a location cannot be moved into one of its + // children." "move" is implemented as remove-then-add; for an object + // target this happened to throw anyway as a side effect of the "add" + // step re-resolving through the now-removed parent, but for an array + // target the removal shifted subsequent indices, so "path" silently + // re-resolved to a different element and the operation "succeeded" + // with a corrupted result. It now throws out_of_range.414 for both + // object and array targets. + + SECTION("array target (from the issue)") + { + json const doc = R"([[1,2],[3]])"_json; + json const patch = {{{"op", "move"}, {"from", "/0"}, {"path", "/0/0"}}}; +#if JSON_DIAGNOSTIC_POSITIONS + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-11) cannot move value: 'from' path '/0' is a proper prefix of 'path' '/0/0'", json::out_of_range&); +#else + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '/0' is a proper prefix of 'path' '/0/0'", json::out_of_range&); +#endif + } + + SECTION("object target") + { + json const doc = R"({"a": {"b": 1}})"_json; + json const patch = {{{"op", "move"}, {"from", "/a"}, {"path", "/a/b"}}}; +#if JSON_DIAGNOSTIC_POSITIONS + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-15) cannot move value: 'from' path '/a' is a proper prefix of 'path' '/a/b'", json::out_of_range&); +#else + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '/a' is a proper prefix of 'path' '/a/b'", json::out_of_range&); +#endif + } + + SECTION("from == path is not a proper prefix and must not be rejected") + { + // "from" equal to "path" is a no-op move; it is not a *proper* + // prefix relationship, so this new check must not reject it. + json const doc = R"({"a": 1, "b": 2})"_json; + json const patch = {{{"op", "move"}, {"from", "/a"}, {"path", "/a"}}}; + CHECK(doc.patch(patch) == doc); + } + + SECTION("raw string prefix that is not a pointer-token prefix must be allowed") + { + // "/ab" is a string-prefix of "/abc/x" as raw text, but "ab" and + // "abc" are different reference tokens, so this is NOT a + // pointer-token prefix relationship and the move must succeed. + // This is the key case proving the check compares tokens, not + // raw pointer text (a naive std::string prefix/rfind check on + // the undecoded pointer would wrongly reject this). + json const doc = R"({"ab": 1, "abc": {"x": 2}})"_json; + json const patch = {{{"op", "move"}, {"from", "/ab"}, {"path", "/abc/x"}}}; + json const result = R"({"abc": {"x": 1}})"_json; + CHECK(doc.patch(patch) == result); + } + + SECTION("escaped reference tokens are compared unescaped") + { + // "from" is the single token "a/b" (escaped as "a~1b"); "path" + // addresses member "x" of that same value, so "from" is a + // proper (token-level) prefix of "path" and must be rejected. + json const doc = R"({"a/b": {"x": 1}})"_json; + json const patch = {{{"op", "move"}, {"from", "/a~1b"}, {"path", "/a~1b/x"}}}; +#if JSON_DIAGNOSTIC_POSITIONS + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-17) cannot move value: 'from' path '/a~1b' is a proper prefix of 'path' '/a~1b/x'", json::out_of_range&); +#else + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '/a~1b' is a proper prefix of 'path' '/a~1b/x'", json::out_of_range&); +#endif + } + + SECTION("ordinary valid moves still work") + { + // unrelated top-level members + json const doc1 = R"({"a": 1, "b": 2})"_json; + json const patch1 = {{{"op", "move"}, {"from", "/a"}, {"path", "/c"}}}; + CHECK(doc1.patch(patch1) == R"({"b": 2, "c": 1})"_json); + + // sibling paths that share a textual prefix but are unrelated + json const doc2 = R"({"a": {"x": 1}, "b": {"y": 2}})"_json; + json const patch2 = {{{"op", "move"}, {"from", "/a/x"}, {"path", "/b/z"}}}; + CHECK(doc2.patch(patch2) == R"({"a": {}, "b": {"y": 2, "z": 1}})"_json); + + // "path" is a proper prefix of "from" (the reverse relationship, + // which RFC 6902 does not forbid) + json const doc3 = R"({"a": {"b": 1}})"_json; + json const patch3 = {{{"op", "move"}, {"from", "/a/b"}, {"path", "/a"}}}; + CHECK(doc3.patch(patch3) == R"({"a": 1})"_json); + } + + SECTION("root 'from' is a proper prefix of every non-root 'path'") + { + // the whole document is a proper prefix of any location inside it + json const doc = R"({"a": 1})"_json; + json const patch = {{{"op", "move"}, {"from", ""}, {"path", "/a"}}}; +#if JSON_DIAGNOSTIC_POSITIONS + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-8) cannot move value: 'from' path '' is a proper prefix of 'path' '/a'", json::out_of_range&); +#else + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '' is a proper prefix of 'path' '/a'", json::out_of_range&); +#endif + } + + SECTION("root 'path' is never a proper prefix violation for a non-root 'from'") + { + // the reverse of the above: moving a non-root location to the root + // is the "path is a prefix of from" relationship, which RFC 6902 + // permits (already covered generally above; this pins the root + // case specifically, since root is the one path with no reference + // tokens at all) + json const doc = R"({"a": {"b": 1}})"_json; + json const patch = {{{"op", "move"}, {"from", "/a"}, {"path", ""}}}; + CHECK(doc.patch(patch) == R"({"b": 1})"_json); + } + + SECTION("the array-append token '-' is an ordinary child token") + { + // "-" (append-to-array) addresses a location *inside* the array, + // so "from" pointing at the array is still a proper prefix of + // "path" ending in "-" and must be rejected like any other child. + json const doc = R"({"a": [1, 2]})"_json; + json const patch = {{{"op", "move"}, {"from", "/a"}, {"path", "/a/-"}}}; +#if JSON_DIAGNOSTIC_POSITIONS + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-13) cannot move value: 'from' path '/a' is a proper prefix of 'path' '/a/-'", json::out_of_range&); +#else + CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '/a' is a proper prefix of 'path' '/a/-'", json::out_of_range&); +#endif + } +} + +TEST_CASE("JSON patch - diff emits array removals in descending index order") +{ + SECTION("array shrunk to empty") + { + json const source = {0, 1, 2, 3, 4}; + json const target = json::array(); + + json const patch = json::diff(source, target); + + json const expected = R"( + [ + {"op": "remove", "path": "/4"}, + {"op": "remove", "path": "/3"}, + {"op": "remove", "path": "/2"}, + {"op": "remove", "path": "/1"}, + {"op": "remove", "path": "/0"} + ] + )"_json; + + CHECK(patch == expected); + CHECK(source.patch(patch) == target); + } + + SECTION("array partially shrunk, after a replacement at a common index") + { + json const source = {0, 1, 2, 3, 4}; + json const target = {0, 9}; + + json const patch = json::diff(source, target); + + // the replacement comes first, then the removals, highest index first + json const expected = R"( + [ + {"op": "replace", "path": "/1", "value": 9}, + {"op": "remove", "path": "/4"}, + {"op": "remove", "path": "/3"}, + {"op": "remove", "path": "/2"} + ] + )"_json; + + CHECK(patch == expected); + CHECK(source.patch(patch) == target); + } + + SECTION("nested array shrunk") + { + json const source = {{"a", {0, 1, 2}}}; + json const target = {{"a", json::array()}}; + + json const patch = json::diff(source, target); + + json const expected = R"( + [ + {"op": "remove", "path": "/a/2"}, + {"op": "remove", "path": "/a/1"}, + {"op": "remove", "path": "/a/0"} + ] + )"_json; + + CHECK(patch == expected); + CHECK(source.patch(patch) == target); + } + + SECTION("many removals still round-trip") + { + json source = json::array(); + for (int i = 0; i < 1000; ++i) + { + source.push_back(i); + } + json const target = json::array(); + + json const patch = json::diff(source, target); + + CHECK(patch.size() == 1000); + CHECK(patch.front().at("path") == "/999"); + CHECK(patch.back().at("path") == "/0"); + CHECK(source.patch(patch) == target); + } +} diff --git a/tests/src/unit-json_pointer.cpp b/tests/src/unit-json_pointer.cpp index a8ed4a89e..b01df2921 100644 --- a/tests/src/unit-json_pointer.cpp +++ b/tests/src/unit-json_pointer.cpp @@ -319,6 +319,44 @@ TEST_CASE("JSON pointers") CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&); CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&); + + // #5395: contains() must not throw for a reference token that is a + // syntactically valid array index but numerically exceeds ULLONG_MAX + // (causing strtoull() to set errno to ERANGE) -- it should just report + // that the pointer does not resolve to an element + CHECK(!j.contains(jp)); + CHECK(!j_const.contains(jp)); + } + + { + // #5395: same as above, but using the exact reproduction from the issue + json::json_pointer const jp("/99999999999999999999"); + std::string const throw_msg = "[json.exception.out_of_range.404] unresolved reference token '99999999999999999999'"; + + CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&); + CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&); + CHECK_THROWS_WITH_AS(j.at(jp) = 1, throw_msg.c_str(), json::out_of_range&); + CHECK_THROWS_WITH_AS(j_const.at(jp) == 1, throw_msg.c_str(), json::out_of_range&); + + CHECK(!j.contains(jp)); + CHECK(!j_const.contains(jp)); + } + + { + // #5395: a reference token that is numerically representable in + // unsigned long long but exceeds size_type's max (e.g. ULLONG_MAX + // itself on typical 64-bit platforms, where size_type's max equals + // ULLONG_MAX) must not make contains() throw either + json::json_pointer const jp("/18446744073709551615"); + std::string const throw_msg = "[json.exception.out_of_range.410] array index 18446744073709551615 exceeds size_type"; + + CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&); + CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&); + CHECK_THROWS_WITH_AS(j.at(jp) = 1, throw_msg.c_str(), json::out_of_range&); + CHECK_THROWS_WITH_AS(j_const.at(jp) == 1, throw_msg.c_str(), json::out_of_range&); + + CHECK(!j.contains(jp)); + CHECK(!j_const.contains(jp)); } // on some machines, the check below is not constant @@ -334,6 +372,10 @@ TEST_CASE("JSON pointers") CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&); CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&); + + // #5395: contains() must not throw for a reference token exceeding size_type's max + CHECK(!j.contains(jp)); + CHECK(!j_const.contains(jp)); } DOCTEST_MSVC_SUPPRESS_WARNING_POP @@ -465,6 +507,16 @@ TEST_CASE("JSON pointers") // explicit roundtrip check CHECK(j.flatten().unflatten() == j); + // an object is only unflattened to an array if one of its keys is the + // reference token 0; this must not depend on which key is seen first + CHECK(json({{"/2", "x"}}).unflatten() == json({{"2", "x"}})); + CHECK(json({{"/10", "y"}, {"/2", "z"}}).unflatten() == json({{"10", "y"}, {"2", "z"}})); + CHECK(json({{"/0", 1}, {"/1", 2}}).unflatten() == json({1, 2})); + CHECK(json({{"/1", 2}, {"/0", 1}}).unflatten() == json({1, 2})); + CHECK(json({{"/0", 1}, {"/2", 3}}).unflatten() == json({1, nullptr, 3})); + CHECK(json({{"/a/1", 2}, {"/a/0", 1}}).unflatten() == json({{"a", {1, 2}}})); + CHECK(json({{"/a/1", 2}, {"/a/x", 1}}).unflatten() == json({{"a", {{"1", 2}, {"x", 1}}}})); + // roundtrip for primitive values json j_null; CHECK(j_null.flatten().unflatten() == j_null); diff --git a/tests/src/unit-large_json.cpp b/tests/src/unit-large_json.cpp index 98d16e336..f10c8be41 100644 --- a/tests/src/unit-large_json.cpp +++ b/tests/src/unit-large_json.cpp @@ -12,6 +12,7 @@ using nlohmann::json; #include +#include TEST_CASE("tests on very large JSONs") { @@ -27,3 +28,153 @@ TEST_CASE("tests on very large JSONs") } } +namespace +{ + +// Descend a chain of single-element containers and return the value at its end, +// reporting the number of levels traversed in @a depth. +// +// The values in the test case below are nested far deeper than the call stack +// can follow, so they must not be inspected with operator== or dump(): both are +// still recursive and would overflow the stack themselves. +const json* innermost_value(const json& j, std::size_t& depth) +{ + const json* current = &j; + depth = 0; + + while ((current->is_array() || current->is_object()) && !current->empty()) + { + current = current->is_array() + ? ¤t->front() + : ¤t->begin().value(); + ++depth; + } + + return current; +} + +} // namespace + +TEST_CASE("tests on deeply nested JSONs") +{ + // deep enough to exhaust the call stack, but small enough to stay cheap: + // parsing is iterative, so building the values below costs little + const std::size_t depth = 100000; + + SECTION("issue #5387 - stack overflow in the copy constructor") + { + SECTION("array") + { + const json j = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']')); + + const json copy(j); // NOLINT(performance-unnecessary-copy-initialization): the copy is what is tested + + std::size_t copy_depth = 0; + CHECK(*innermost_value(copy, copy_depth) == 0); + CHECK(copy_depth == depth); + } + + SECTION("object") + { + std::string s; + s.reserve((6 * depth) + 1); + for (std::size_t i = 0; i < depth; ++i) + { + s += "{\"a\":"; + } + s += '1'; + s.append(depth, '}'); + + const json j = json::parse(s); + + const json copy(j); // NOLINT(performance-unnecessary-copy-initialization): the copy is what is tested + + std::size_t copy_depth = 0; + CHECK(*innermost_value(copy, copy_depth) == 1); + CHECK(copy_depth == depth); + } + + SECTION("copy assignment") + { + // operator=(basic_json) takes its argument by value, so the deep + // copy happens in the copy constructor + const json j = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']')); + + json target; + target = j; + + std::size_t target_depth = 0; + CHECK(*innermost_value(target, target_depth) == 0); + CHECK(target_depth == depth); + } + + SECTION("depths around the bound of the recursive descent") + { + // The copy constructor descends into a bounded number of levels and + // completes whatever is below that without the call stack. Cover + // every depth around that bound, so that the two ways of copying + // are known to meet cleanly - wherever the bound is set. + for (std::size_t d = 1; d <= 300; ++d) + { + CAPTURE(d); + + const json array = json::parse(std::string(d, '[') + '0' + std::string(d, ']')); + const json array_copy(array); // NOLINT(performance-unnecessary-copy-initialization): the copy is what is tested + std::size_t array_depth = 0; + CHECK(*innermost_value(array_copy, array_depth) == 0); + CHECK(array_depth == d); + + std::string object_text; + for (std::size_t i = 0; i < d; ++i) + { + object_text += "{\"a\":"; + } + object_text += '1'; + object_text.append(d, '}'); + + const json object = json::parse(object_text); + const json object_copy(object); // NOLINT(performance-unnecessary-copy-initialization): the copy is what is tested + std::size_t object_depth = 0; + CHECK(*innermost_value(object_copy, object_depth) == 1); + CHECK(object_depth == d); + } + } + + SECTION("a value that is deep in one place only") + { + json j = json::object(); + j["shallow"] = 1; + j["deep"] = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']')); + j["also_shallow"] = json::array({1, 2, 3}); + + const json copy(j); + + CHECK(copy["shallow"] == 1); + CHECK(copy["also_shallow"] == json::array({1, 2, 3})); + + std::size_t deep_depth = 0; + CHECK(*innermost_value(copy["deep"], deep_depth) == 0); + CHECK(deep_depth == depth); + } + + SECTION("the copy is independent of the original") + { + const json j = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']')); + + json copy(j); + + // reach the innermost value without recursing and replace it + json* current = © + while (current->is_array() && !current->empty()) + { + current = ¤t->front(); + } + *current = 42; + + std::size_t unused = 0; + CHECK(*innermost_value(copy, unused) == 42); + CHECK(*innermost_value(j, unused) == 0); + } + } +} + diff --git a/tests/src/unit-merge_patch.cpp b/tests/src/unit-merge_patch.cpp index f02a1e991..c9741e85b 100644 --- a/tests/src/unit-merge_patch.cpp +++ b/tests/src/unit-merge_patch.cpp @@ -14,6 +14,60 @@ using nlohmann::json; using namespace nlohmann::literals; // NOLINT(google-build-using-namespace) #endif +#include + +namespace +{ +// RFC 7396's MergePatch, written recursively as in the RFC; only usable on +// values nested a few hundred levels deep +void reference_merge_patch(json& target, const json& patch) +{ + if (!patch.is_object()) + { + target = patch; + return; + } + if (!target.is_object()) + { + target = json::object(); + } + for (auto it = patch.begin(); it != patch.end(); ++it) + { + if (it.value().is_null()) + { + target.erase(it.key()); + } + else + { + reference_merge_patch(target[it.key()], it.value()); + } + } +} + +// objects nested `depth` levels deep under the key "a", with members that +// differ by `variant` on the way down +std::string nested_objects(const std::size_t depth, const int variant) +{ + std::string text; + for (std::size_t i = 0; i < depth; ++i) + { + text += "{"; + if ((i + static_cast(variant)) % 3 == 0) + { + text += "\"s" + std::to_string(variant) + "\":" + std::to_string(i) + ","; + } + if (variant == 2 && i % 5 == 0) + { + text += "\"s0\":null,"; + } + text += "\"a\":"; + } + text += variant == 1 ? R"({"x":1,"y":null})" : "{\"y\":2}"; + text.append(depth, '}'); + return text; +} +} // namespace + TEST_CASE("JSON Merge Patch") { SECTION("examples from RFC 7396") @@ -242,3 +296,52 @@ TEST_CASE("JSON Merge Patch") } } } + +TEST_CASE("JSON Merge Patch on deeply nested values") +{ + SECTION("patching past the descent bound gives the same result") + { + // every depth on either side of where the iterative version takes + // over (detail::recursion_depth_limit(), 128) + for (std::size_t depth = 0; depth <= 300; ++depth) + { + CAPTURE(depth); + for (int variant = 0; variant < 3; ++variant) + { + CAPTURE(variant); + const json patch = json::parse(nested_objects(depth, variant)); + + json result = json::parse(nested_objects(depth, (variant + 1) % 3)); + json expected = result; + result.merge_patch(patch); + reference_merge_patch(expected, patch); + CHECK(result == expected); + + // a target that is not an object, and an empty one + json from_null; + from_null.merge_patch(patch); + json expected_from_null; + reference_merge_patch(expected_from_null, patch); + CHECK(from_null == expected_from_null); + } + } + } + + SECTION("patches nested too deeply for the call stack (#5393)") + { + // applying a patch used to recurse once per nesting level. The result + // is only walked, never copied or compared, since those recurse too. + const std::size_t depth = 100000; + json target = json::parse(nested_objects(depth, 0)); + target.merge_patch(json::parse(nested_objects(depth, 1))); + + const json* p = ⌖ + for (std::size_t i = 0; i < depth; ++i) + { + p = &p->at("a"); + } + // {"y":2} patched with {"x":1,"y":null} + CHECK(p->size() == 1); + CHECK(p->at("x") == 1); + } +} diff --git a/tests/src/unit-modifiers.cpp b/tests/src/unit-modifiers.cpp index 18185ec00..c878ec15c 100644 --- a/tests/src/unit-modifiers.cpp +++ b/tests/src/unit-modifiers.cpp @@ -11,6 +11,53 @@ #include using nlohmann::json; +#include + +namespace +{ +// update(source, true) as documented, written recursively; only usable on +// values nested a few hundred levels deep +void reference_update(json& target, const json& source) +{ + for (auto it = source.begin(); it != source.end(); ++it) + { + const auto existing = target.find(it.key()); + if (it.value().is_object() && existing != target.end() && existing->is_object()) + { + reference_update(*existing, it.value()); + } + else + { + target[it.key()] = it.value(); + } + } +} + +// objects nested `depth` levels deep under the key "a", with members that +// differ by `variant` on the way down +std::string nested_objects(const std::size_t depth, const int variant) +{ + std::string text; + for (std::size_t i = 0; i < depth; ++i) + { + text += "{"; + if ((i + static_cast(variant)) % 3 == 0) + { + text += "\"s" + std::to_string(variant) + "\":" + std::to_string(i) + ","; + } + if (variant == 2 && i % 5 == 0) + { + // an object replacing a primitive, which is not merged + text += R"("s0":{"o":1},)"; + } + text += "\"a\":"; + } + text += variant == 1 ? "{\"x\":1}" : "{\"y\":2}"; + text.append(depth, '}'); + return text; +} +} // namespace + TEST_CASE("modifiers") { SECTION("clear()") @@ -641,6 +688,20 @@ TEST_CASE("modifiers") CHECK_THROWS_WITH_AS(j_array.insert(j_array.end(), j_other_array.begin(), j_other_array2.end()), "[json.exception.invalid_iterator.210] iterators do not fit", json::invalid_iterator&); } + + SECTION("iterators not pointing into an array") + { + json j_object2 = {{"k", 1}, {"l", 2}}; + json j_primitive = 5; + json j_null; + + CHECK_THROWS_WITH_AS(j_array.insert(j_array.begin(), j_object2.begin(), j_object2.end()), "[json.exception.invalid_iterator.202] iterators first and last must point to arrays", + json::invalid_iterator&); + CHECK_THROWS_WITH_AS(j_array.insert(j_array.begin(), j_primitive.begin(), j_primitive.end()), "[json.exception.invalid_iterator.202] iterators first and last must point to arrays", + json::invalid_iterator&); + CHECK_THROWS_WITH_AS(j_array.insert(j_array.begin(), j_null.begin(), j_null.end()), "[json.exception.invalid_iterator.202] iterators first and last must point to arrays", + json::invalid_iterator&); + } } SECTION("range for object") @@ -801,6 +862,30 @@ TEST_CASE("modifiers") j1.update(j2, true); CHECK(j1 == json({{"string", "t"}, {"numbers", 1}})); } + + SECTION("overwrite primitive with object") + { + json j1 = {{"k", 1}}; + json const j2 = {{"k", {{"x", 2}}}}; + j1.update(j2, true); + CHECK(j1 == json({{"k", {{"x", 2}}}})); + } + + SECTION("overwrite array with object") + { + json j1 = {{"k", {1, 2}}}; + json const j2 = {{"k", {{"x", 2}}}}; + j1.update(j2, true); + CHECK(j1 == json({{"k", {{"x", 2}}}})); + } + + SECTION("overwrite nested primitive with object") + { + json j1 = {{"k", {{"inner", 1}}}}; + json const j2 = {{"k", {{"inner", {{"x", 2}}}}}}; + j1.update(j2, true); + CHECK(j1 == json({{"k", {{"inner", {{"x", 2}}}}}})); + } } } } @@ -950,3 +1035,44 @@ TEST_CASE("modifiers") } } } + +TEST_CASE("update() on deeply nested values") +{ + SECTION("merging past the descent bound gives the same result") + { + // every depth on either side of where the iterative version takes + // over (detail::recursion_depth_limit(), 128) + for (std::size_t depth = 0; depth <= 300; ++depth) + { + CAPTURE(depth); + for (int variant = 0; variant < 3; ++variant) + { + CAPTURE(variant); + const json source = json::parse(nested_objects(depth, variant)); + json result = json::parse(nested_objects(depth, (variant + 1) % 3)); + json expected = result; + result.update(source, true); + reference_update(expected, source); + CHECK(result == expected); + } + } + } + + SECTION("objects nested too deeply for the call stack (#5545)") + { + // merging used to recurse once per nesting level. The result is only + // walked, never copied or compared, since those recurse too. + const std::size_t depth = 100000; + json target = json::parse(nested_objects(depth, 0)); + target.update(json::parse(nested_objects(depth, 1)), true); + + const json* p = ⌖ + for (std::size_t i = 0; i < depth; ++i) + { + p = &p->at("a"); + } + CHECK(p->size() == 2); + CHECK(p->at("x") == 1); + CHECK(p->at("y") == 2); + } +} diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index 4358cd464..a8892081d 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -1597,7 +1597,168 @@ TEST_CASE("MessagePack") } } +TEST_CASE("issue #5405 - array reserve for definite-length MessagePack arrays") +{ +#if !defined(JSON_NOEXCEPTION) + // this SECTION relies on catching a thrown exception to distinguish + // which of two acceptable, bounded rejections a hostile header took; + // under JSON_NOEXCEPTION, JSON_THROW never produces a catchable C++ + // exception (it aborts instead), so this cannot be tested that way here + SECTION("a huge claimed length with no element data must not over-allocate") + { + // 0xdd: array 32 (four-byte length); claims 0xFFFFFFFF (4294967295) + // elements but provides none. max_size() for a std::vector is far + // larger than this count, so it does not reject the header outright; + // the (capped) reservation must not attempt to allocate space for + // billions of elements before the missing data is detected. + json _; + const std::vector input = {0xdd, 0xFF, 0xFF, 0xFF, 0xFF}; + // On a platform where std::size_t is narrower than 64 bits (e.g. + // 32-bit), the claimed count 0xFFFFFFFF coincides with that + // platform's SIZE_MAX, which some size-narrowing checks treat the + // same as detail::unknown_size(); it may then be rejected before + // the SAX consumer's own max_size() check (out_of_range.408) rather + // than being accepted and only found short of data once the + // (capped) reservation looks for element bytes that were never + // provided (parse_error.110). Either is an acceptable, bounded + // rejection of the hostile header -- the property under test is + // that no path attempts to allocate space for billions of elements. + bool threw = false; + try + { + _ = json::from_msgpack(input); + } + catch (const json::parse_error& e) + { + threw = true; + CHECK(e.id == 110); + CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 6: syntax error while parsing MessagePack value: unexpected end of input"); + } + catch (const json::out_of_range& e) + { + threw = true; + CHECK(e.id == 408); + CHECK(std::string(e.what()).find("excessive") != std::string::npos); + } + CHECK(threw); + CHECK(json::from_msgpack(input, true, false).is_discarded()); + } +#endif + + SECTION("arrays of various sizes decode to the same value as before the reserve optimization") + { + for (const auto size : + { + std::size_t{0}, std::size_t{1}, std::size_t{5}, // small + std::size_t{16384}, // exactly at the reserve cap + std::size_t{20000} // above the reserve cap + }) + { + CAPTURE(size) + json j = json::array(); + for (std::size_t i = 0; i < size; ++i) + { + j.push_back(static_cast(i % 1000)); + } + + const auto packed = json::to_msgpack(j); + CHECK(json::from_msgpack(packed) == j); + } + } + + SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization") + { + // the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser; + // a custom SAX consumer that does not touch a DOM array sees identical events + json j = json::array(); + for (int i = 0; i < 100; ++i) + { + j.push_back(i); + } + const auto packed = json::to_msgpack(j); + + SaxCountdown scp(1000000); // large enough to never trigger an abort + CHECK(json::sax_parse(packed, &scp, json::input_format_t::msgpack)); + } +} + +TEST_CASE("regression test - MessagePack ext type rejects a subtype that doesn't fit a single byte") +{ + // subtype 0-255 must still round-trip correctly (regression guard, pre-existing behavior) + CHECK(json::from_msgpack(json::to_msgpack(json::binary({1, 2}, 0))).get_binary().subtype() == 0); + CHECK(json::from_msgpack(json::to_msgpack(json::binary({1, 2}, 200))).get_binary().subtype() == 200); + CHECK(json::from_msgpack(json::to_msgpack(json::binary({1, 2}, 255))).get_binary().subtype() == 255); + + // a subtype > 255 must throw instead of silently truncating + CHECK_THROWS_AS(json::to_msgpack(json::binary({1, 2}, 256)), json::out_of_range); + CHECK_THROWS_WITH_AS(json::to_msgpack(json::binary({1, 2}, 70000)), "[json.exception.out_of_range.415] subtype 70000 is too large for the MessagePack ext type (max 255)", json::out_of_range); + + // a binary value with no subtype at all must be unaffected + CHECK(json::from_msgpack(json::to_msgpack(json::binary({1, 2}))).get_binary().has_subtype() == false); +} + // use this testcase outside [hide] to run it with Valgrind +TEST_CASE("MessagePack nesting does not consume the call stack") +{ + // Reading a container used to call back into the value reader once per + // element, so the native call stack grew with the nesting depth of the + // input: one frame per byte for repeated 0x91 (a one-element array), which + // crashes the process long before the input is exhausted (#5104). The + // containers are kept on a heap stack now. + // + // Note that deeply nested values must not be compared, copied or dumped + // here: those operations are still recursive, and would reintroduce the + // very crash this checks for. Depth is measured by descending instead. + + SECTION("an unterminated chain is reported, not crashed on") + { + json _; + const std::vector input(300000, 0x91); + CHECK_THROWS_WITH_AS(_ = json::from_msgpack(input), "[json.exception.parse_error.110] parse error at byte 300001: syntax error while parsing MessagePack value: unexpected end of input", json::parse_error&); + CHECK(json::from_msgpack(input, true, false).is_discarded()); + } + + SECTION("a well-formed deep value is read through the SAX interface") + { + std::vector input(300000, 0x91); + input.push_back(0x01); // innermost value + + SaxCountdown accept_all(600001); + CHECK(json::sax_parse(input, &accept_all, json::input_format_t::msgpack)); + } + + SECTION("a well-formed deep value is read into a value") + { + const std::size_t depth = 10000; + std::vector input(depth, 0x91); + input.push_back(0x01); + + json j = json::from_msgpack(input); + + std::size_t measured = 0; + const json* p = &j; + while (p->is_array() && !p->empty()) + { + p = &p->front(); + ++measured; + } + CHECK(measured == depth); + CHECK(p->is_number()); + } + + SECTION("containers are still read the same way") + { + CHECK(json::from_msgpack(std::vector({0x90})) == json::array()); + CHECK(json::from_msgpack(std::vector({0x80})) == json::object()); + CHECK(json::from_msgpack(std::vector({0x92, 0x90, 0x80})) == json({json::array(), json::object()})); + CHECK(json::from_msgpack(std::vector({0x91, 0x91, 0x91, 0x90})) == json({{{json::array()}}})); + CHECK(json::from_msgpack(std::vector({0x81, 0xA1, 'a', 0x81, 0xA1, 'b', 0x92, 0x01, 0x02})) == json({{"a", {{"b", {1, 2}}}}})); + // array 16 and map 32, i.e. the counted forms + CHECK(json::from_msgpack(std::vector({0xDC, 0x00, 0x02, 0x01, 0x02})) == json({1, 2})); + CHECK(json::from_msgpack(std::vector({0xDF, 0x00, 0x00, 0x00, 0x01, 0xA1, 'k', 0xC3})) == json({{"k", true}})); + } +} + TEST_CASE("single MessagePack roundtrip") { SECTION("sample.json") diff --git a/tests/src/unit-no-macro-leak.cpp b/tests/src/unit-no-macro-leak.cpp new file mode 100644 index 000000000..c5184c52f --- /dev/null +++ b/tests/src/unit-no-macro-leak.cpp @@ -0,0 +1,34 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// This file makes sure that none of the internal JSON_HEDLEY_* macros (vendored +// from https://nemequ.github.io/hedley/, see +// include/nlohmann/thirdparty/hedley/hedley.hpp) leak into the including +// translation unit. include/nlohmann/detail/macro_unscope.hpp is supposed to +// #undef every JSON_HEDLEY_* macro (via hedley_undef.hpp) once json.hpp has +// been fully processed. See https://github.com/nlohmann/json/issues/5408, +// where JSON_HEDLEY_PRAGMA, JSON_HEDLEY_PREDICT_TRUE, JSON_HEDLEY_PREDICT_FALSE, +// and JSON_HEDLEY_CLANG_HAS_DECLSPEC_ATTRIBUTE escaped this cleanup because +// hedley_undef.hpp had no matching #undef for them. +// +// hedley_undef_checks.inc (included below) is generated at CMake configure/ +// build time by cmake/scripts/gen_hedley_undef_check.cmake, which derives the +// full list of JSON_HEDLEY_* macro names directly from hedley.hpp. That way +// this test covers every macro Hedley actually defines -- not a hardcoded +// snapshot that would silently go stale the next time `make update_hedley` +// runs -- and can never drift from the vendored header. + +#include "doctest_compatibility.h" + +#include + +TEST_CASE("JSON_HEDLEY macros do not leak after including json.hpp") +{ +#include "hedley_undef_checks.inc" + CHECK(true); // keep an assertion when nothing leaked +} diff --git a/tests/src/unit-no-mem-leak-on-adl-serialize.cpp b/tests/src/unit-no-mem-leak-on-adl-serialize.cpp index 469fc2c75..cfbdff008 100644 --- a/tests/src/unit-no-mem-leak-on-adl-serialize.cpp +++ b/tests/src/unit-no-mem-leak-on-adl-serialize.cpp @@ -70,7 +70,7 @@ TEST_CASE("check_for_mem_leak_on_adl_to_json-2") } } -TEST_CASE("check_for_mem_leak_on_adl_to_json-2") +TEST_CASE("check_for_mem_leak_on_adl_to_json-3") { try { diff --git a/tests/src/unit-no_io_and_user_exceptions.cpp b/tests/src/unit-no_io_and_user_exceptions.cpp new file mode 100644 index 000000000..667d114e7 --- /dev/null +++ b/tests/src/unit-no_io_and_user_exceptions.cpp @@ -0,0 +1,91 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// This translation unit is a dedicated, small compile-and-run check for two +// configuration macros that (per #5423) were never exercised anywhere in the +// test matrix: +// - JSON_NO_IO, which removes the library's / support +// (operator<<, operator>>, and the stream-based overloads of dump()/parse()) +// - the JSON_THROW_USER / JSON_TRY_USER / JSON_CATCH_USER trio, which lets a +// user replace the library's internal exception handling +// +// Both macros are about excluding/replacing a facility the library would +// otherwise pull in on its own, and defining one has no bearing on the other, +// so -- to keep the test matrix small -- they are exercised together in a +// single dedicated file instead of two. +// +// JSON_NO_IO requires this file itself to never rely on /; +// only string-based parsing/dumping is used below. +#define JSON_NO_IO 1 + +// The user-supplied exception macros below are a *conforming* replacement: +// they simply forward to the real throw/try/catch keywords (via a counter so +// the test can assert each macro was actually invoked, not just defined), so +// every exception-related behavior the library relies on internally -- +// including rethrowing std::out_of_range as json::out_of_range in at() -- +// keeps working exactly as it would with the library's own default macros. +static int json_throw_user_call_count = 0; // NOLINT(cppcoreguidelines-avoid-non-const-global-variables) + +#define JSON_THROW_USER(exception) do { ++json_throw_user_call_count; throw (exception); } while (false) // NOLINT(cppcoreguidelines-macro-usage) +#define JSON_TRY_USER try // NOLINT(cppcoreguidelines-macro-usage) +#define JSON_CATCH_USER(exception) catch (exception) // NOLINT(cppcoreguidelines-macro-usage) + +#include "doctest_compatibility.h" + +#include +using json = nlohmann::json; + +TEST_CASE("JSON_NO_IO") +{ + // everything that does not touch / must keep working: + // parsing from and dumping to std::string + const json j = json::parse(R"({"a":[1,2,3],"b":true})"); + CHECK(j.dump() == R"({"a":[1,2,3],"b":true})"); + CHECK(j.at("a").size() == 3); + CHECK(j.at("b").get() == true); +} + +// this test relies on CHECK_THROWS_AS() actually invoking the guarded +// expression so json_throw_user_call_count gets bumped and can be observed +// afterwards; doctest's "--no-throw" test filter (which ci_test_noexceptions +// passes, together with a global -DJSON_NOEXCEPTION added to CMAKE_CXX_FLAGS +// for every translation unit in that build, this file included) compiles +// CHECK_THROWS_AS() out to a no-op that never even invokes the given +// expression -- so json::parse()/at() below would never be called at all and +// the call-count assertions would fail even though our JSON_THROW_USER +// override (which always really throws, regardless of JSON_NOEXCEPTION) would +// have worked fine on its own +#if !defined(JSON_NOEXCEPTION) +TEST_CASE("JSON_THROW_USER, JSON_TRY_USER, JSON_CATCH_USER") +{ + json_throw_user_call_count = 0; + + // json::parse() is [[nodiscard]] (JSON_HEDLEY_WARN_UNUSED_RESULT); under + // GCC in C++11 mode that expands to __attribute__((warn_unused_result)), + // which -- unlike a [[nodiscard]] attribute proper -- GCC does not + // consider satisfied by doctest's CHECK_THROWS_AS() wrapping the + // expression in a (void) cast, so the discarded return value would still + // be flagged under -Werror=unused-result; assign it to discard it instead, + // matching the established `json _ = json::parse(...)` pattern used + // elsewhere in the test suite (see unit-class_parser.cpp) + json _; // NOLINT(readability-identifier-naming) + + // a parse error goes through JSON_THROW directly, i.e., through our + // JSON_THROW_USER override + CHECK_THROWS_AS(_ = json::parse("this is not JSON"), json::parse_error&); + CHECK(json_throw_user_call_count > 0); + + // at() on an out-of-range array index internally catches std::out_of_range + // (JSON_TRY_USER/JSON_CATCH_USER) and rethrows it as json::out_of_range + // (JSON_THROW_USER again), so this exercises all three macros together + const int count_before = json_throw_user_call_count; + const json arr = json::array({1, 2, 3}); + CHECK_THROWS_AS(arr.at(10), json::out_of_range&); + CHECK(json_throw_user_call_count > count_before); +} +#endif diff --git a/tests/src/unit-ordered_json.cpp b/tests/src/unit-ordered_json.cpp index a38a1a2b8..45fbf5493 100644 --- a/tests/src/unit-ordered_json.cpp +++ b/tests/src/unit-ordered_json.cpp @@ -81,3 +81,118 @@ TEST_CASE("regression test for issue #3732 - iteration_proxy_value(fn); } + +TEST_CASE("copying an ordered_json with nested values") +{ + // ordered_map is backed by a vector, so copying an object that has + // structured values takes a different route than copying a std::map-backed + // one; see https://github.com/nlohmann/json/issues/5387 + ordered_json oj; + oj["z"] = 1; + oj["a"]["y"] = 2; + oj["a"]["b"]["x"] = 3; + oj["m"] = {1, 2, {{"w", 4}}}; + + const ordered_json copy(oj); + + SECTION("the copy is equal to the original") + { + CHECK(copy == oj); + CHECK(copy.dump() == oj.dump()); + } + + SECTION("the key order is preserved at every level") + { + CHECK(copy.dump() == R"({"z":1,"a":{"y":2,"b":{"x":3}},"m":[1,2,{"w":4}]})"); + } + + SECTION("the copy is independent of the original") + { + ordered_json mutated(oj); + mutated["a"]["b"]["x"] = 99; + + CHECK(oj["a"]["b"]["x"] == 3); + CHECK(mutated["a"]["b"]["x"] == 99); + } +} + +TEST_CASE("regression test - diff() must account for ordered_json member order") +{ + SECTION("pure reorder, no value changes") + { + ordered_json a = {{"a", 1}, {"b", 2}}; + ordered_json b = {{"b", 2}, {"a", 1}}; + CHECK(a != b); // order-sensitive equality + CHECK(a.patch(ordered_json::diff(a, b)) == b); + } + + SECTION("new key must land at the front") + { + ordered_json c = {{"b", 2}}; + ordered_json e = {{"a", 1}, {"b", 2}}; + CHECK(c.patch(ordered_json::diff(c, e)) == e); + } + + SECTION("reorder plus a value change on one of the reordered keys") + { + ordered_json a = {{"a", 1}, {"b", 2}}; + ordered_json b = {{"b", 20}, {"a", 1}}; + CHECK(a != b); + CHECK(a.patch(ordered_json::diff(a, b)) == b); + } + + SECTION("reorder plus a deleted key") + { + ordered_json a = {{"a", 1}, {"b", 2}, {"c", 3}}; + ordered_json b = {{"b", 2}, {"a", 1}}; + CHECK(a != b); + CHECK(a.patch(ordered_json::diff(a, b)) == b); + } + + SECTION("reorder plus a nested value that itself needs a recursive diff") + { + ordered_json a = {{"a", {{"x", 1}, {"y", 2}}}, {"b", 2}}; + ordered_json b = {{"b", 2}, {"a", {{"x", 1}, {"y", 99}}}}; + CHECK(a != b); + CHECK(a.patch(ordered_json::diff(a, b)) == b); + } + + SECTION("three or more keys shuffled into a different order") + { + ordered_json a = {{"a", 1}, {"b", 2}, {"c", 3}, {"d", 4}}; + ordered_json b = {{"d", 4}, {"b", 2}, {"a", 1}, {"c", 3}}; + CHECK(a != b); + CHECK(a.patch(ordered_json::diff(a, b)) == b); + } + + SECTION("matching order still produces a minimal patch (fast path unaffected)") + { + ordered_json a = {{"a", 1}, {"b", 2}, {"c", 3}}; + ordered_json b = {{"a", 1}, {"b", 20}, {"c", 3}}; + auto p = ordered_json::diff(a, b); + // only the changed value should be touched, not a wholesale remove+add + CHECK(p.size() == 1); + CHECK(p[0]["op"] == "replace"); + CHECK(p[0]["path"] == "/b"); + CHECK(a.patch(p) == b); + } + + SECTION("plain json (std::map-backed) is unaffected by same-key-different-insertion-order") + { + json a; + a["b"] = 2; + a["a"] = 1; + + json b; + b["a"] = 1; + b["b"] = 2; + + // std::map iteration is always sorted by key, so a == b regardless of + // insertion order, and diff() must still produce the same minimal + // (empty) result as before this fix + CHECK(a == b); + auto p = json::diff(a, b); + CHECK(p.empty()); + CHECK(a.patch(p) == b); + } +} diff --git a/tests/src/unit-ordered_json2.cpp b/tests/src/unit-ordered_json2.cpp new file mode 100644 index 000000000..83eb7668d --- /dev/null +++ b/tests/src/unit-ordered_json2.cpp @@ -0,0 +1,489 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-FileCopyrightText: 2018 Vitaliy Manushkin +// SPDX-License-Identifier: MIT + +// This file closes a test-coverage gap described in GitHub issue #5421: +// nlohmann::ordered_json (and other non-default basic_json specializations, +// such as the alt_string-based one from unit-alt-string.cpp) were never +// exercised through the binary formats (CBOR/MessagePack/UBJSON/BSON/BJData) +// or through flatten()/unflatten()/diff()/patch()/merge_patch(). + +#include "doctest_compatibility.h" + +#include + +#include +#include +#include +#include + +using nlohmann::json; +using nlohmann::ordered_json; + +///////////////////////////////////////////////////////////////////////////// +// alt_json: a second, independent copy of the custom-string_t basic_json +// specialization defined in unit-alt-string.cpp. +// +// It is duplicated here (rather than shared via a header) because every +// unit-*.cpp file in this test suite is compiled into its own standalone +// executable (see tests/CMakeLists.txt), so there is no ODR concern in +// having the same class name defined in multiple translation units. +// +// Two members had to be added relative to the original alt_string +// (a constructor from std::string, and a find(char, pos) overload) because +// the original type was never used with the binary writers/readers before +// this file: BSON's array/document writer converts std::to_string() results +// and checks for embedded NUL characters via find(char), and the UBJSON/BSON +// high-precision-number path constructs the SAX string_t argument from a +// std::string. Neither path is exercised anywhere else in the test suite for +// this type, which is presumably why the gap was never noticed. +///////////////////////////////////////////////////////////////////////////// + +class alt_string; +bool operator<(const char* op1, const alt_string& op2) noexcept; // NOLINT(misc-use-internal-linkage) +void int_to_string(alt_string& target, std::size_t value); // NOLINT(misc-use-internal-linkage) + +class alt_string +{ + public: + using value_type = std::string::value_type; + + static constexpr auto npos = (std::numeric_limits::max)(); + + alt_string(const char* str): str_impl(str) {} + alt_string(const char* str, std::size_t count): str_impl(str, count) {} + alt_string(std::string str): str_impl(std::move(str)) {} + alt_string(size_t count, char chr): str_impl(count, chr) {} + alt_string() = default; + + alt_string& append(char ch) + { + str_impl.push_back(ch); + return *this; + } + + alt_string& append(const alt_string& str) + { + str_impl.append(str.str_impl); + return *this; + } + + alt_string& append(const char* s, std::size_t length) + { + str_impl.append(s, length); + return *this; + } + + void push_back(char c) + { + str_impl.push_back(c); + } + + template + bool operator==(const op_type& op) const + { + return str_impl == op; + } + + bool operator==(const alt_string& op) const + { + return str_impl == op.str_impl; + } + + template + bool operator!=(const op_type& op) const + { + return str_impl != op; + } + + bool operator!=(const alt_string& op) const + { + return str_impl != op.str_impl; + } + + std::size_t size() const noexcept + { + return str_impl.size(); + } + + void resize(std::size_t n) + { + str_impl.resize(n); + } + + void resize(std::size_t n, char c) + { + str_impl.resize(n, c); + } + + template + bool operator<(const op_type& op) const noexcept + { + return str_impl < op; + } + + bool operator<(const alt_string& op) const noexcept + { + return str_impl < op.str_impl; + } + + const char* c_str() const + { + return str_impl.c_str(); + } + + char& operator[](std::size_t index) + { + return str_impl[index]; + } + + const char& operator[](std::size_t index) const + { + return str_impl[index]; + } + + char& back() + { + return str_impl.back(); + } + + const char& back() const + { + return str_impl.back(); + } + + void clear() + { + str_impl.clear(); + } + + const value_type* data() const + { + return str_impl.data(); + } + + bool empty() const + { + return str_impl.empty(); + } + + std::size_t find(const alt_string& str, std::size_t pos = 0) const + { + return str_impl.find(str.str_impl, pos); + } + + // needed by binary_writer's BSON support, which probes string keys for + // embedded NUL characters via find(char) + std::size_t find(char c, std::size_t pos = 0) const + { + return str_impl.find(c, pos); + } + + std::size_t find_first_of(char c, std::size_t pos = 0) const + { + return str_impl.find_first_of(c, pos); + } + + alt_string substr(std::size_t pos = 0, std::size_t count = npos) const + { + const std::string s = str_impl.substr(pos, count); + return {s.data(), s.size()}; + } + + alt_string& replace(std::size_t pos, std::size_t count, const alt_string& str) + { + str_impl.replace(pos, count, str.str_impl); + return *this; + } + + void reserve(std::size_t new_cap = 0) + { + str_impl.reserve(new_cap); + } + + private: + std::string str_impl {}; // NOLINT(readability-redundant-member-init) + + friend bool operator<(const char* /*op1*/, const alt_string& /*op2*/) noexcept; +}; + +void int_to_string(alt_string& target, std::size_t value) +{ + target = std::to_string(value).c_str(); +} + +using alt_json = nlohmann::basic_json < + std::map, + std::vector, + alt_string, + bool, + std::int64_t, + std::uint64_t, + double, + std::allocator, + nlohmann::adl_serializer >; + +bool operator<(const char* op1, const alt_string& op2) noexcept +{ + return op1 < op2.str_impl; +} + +namespace +{ + +// collects the object keys of j, in iteration order +std::vector collect_keys(const ordered_json& j) +{ + std::vector result; + for (auto it = j.cbegin(); it != j.cend(); ++it) + { + result.push_back(it.key()); + } + return result; +} + +// a nested object/array value with keys inserted in non-alphabetical order, +// used to check both round-trip equality and (for ordered_json) that +// insertion order survives a trip through a binary format +ordered_json make_rich_ordered_json() +{ + ordered_json j; + j["zebra"] = 1; + j["apple"] = ordered_json::array({1, 2, 3}); + j["mango"]["z_nested"] = true; + j["mango"]["a_nested"] = nullptr; + j["banana"] = "some text"; + j["cherry"] = 3.14; + return j; +} + +alt_json make_rich_alt_json() +{ + alt_json j; + j["zebra"] = 1; + j["apple"] = alt_json::array({1, 2, 3}); + j["mango"]["z_nested"] = true; + j["mango"]["a_nested"] = nullptr; + j["banana"] = "some text"; + j["cherry"] = 3.14; + return j; +} + +} // namespace + +TEST_CASE("ordered_json across binary formats") +{ + const ordered_json original = make_rich_ordered_json(); + const std::vector original_keys = collect_keys(original); + const std::vector original_mango_keys = collect_keys(original["mango"]); + + SECTION("CBOR") + { + const auto bytes = ordered_json::to_cbor(original); + const auto restored = ordered_json::from_cbor(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("MessagePack") + { + const auto bytes = ordered_json::to_msgpack(original); + const auto restored = ordered_json::from_msgpack(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("UBJSON") + { + const auto bytes = ordered_json::to_ubjson(original); + const auto restored = ordered_json::from_ubjson(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("BSON") + { + const auto bytes = ordered_json::to_bson(original); + const auto restored = ordered_json::from_bson(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("BJData") + { + const auto bytes = ordered_json::to_bjdata(original); + const auto restored = ordered_json::from_bjdata(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } +} + +TEST_CASE("alt_json (custom string_t) across binary formats") +{ + const alt_json original = make_rich_alt_json(); + + SECTION("CBOR") + { + const auto bytes = alt_json::to_cbor(original); + const auto restored = alt_json::from_cbor(bytes); + CHECK(restored == original); + } + + SECTION("MessagePack") + { + const auto bytes = alt_json::to_msgpack(original); + const auto restored = alt_json::from_msgpack(bytes); + CHECK(restored == original); + } + + SECTION("UBJSON") + { + const auto bytes = alt_json::to_ubjson(original); + const auto restored = alt_json::from_ubjson(bytes); + CHECK(restored == original); + } + + SECTION("BSON") + { + const auto bytes = alt_json::to_bson(original); + const auto restored = alt_json::from_bson(bytes); + CHECK(restored == original); + } + + SECTION("BJData") + { + const auto bytes = alt_json::to_bjdata(original); + const auto restored = alt_json::from_bjdata(bytes); + CHECK(restored == original); + } +} + +TEST_CASE("ordered_json operator== is sensitive to key order") +{ + // Unlike nlohmann::json (whose object_t is a std::map, so equality never + // depends on insertion order), ordered_json's object_t (ordered_map) is a + // std::vector> under the hood, and does not define its + // own operator==: it inherits std::vector's element-wise comparison. As a + // result, two ordered_json objects holding the very same key/value pairs + // in different insertion order compare *unequal*. This is the property + // that makes the round-trip `CHECK(restored == original)` checks above a + // meaningful order-preservation check by themselves (the explicit + // collect_keys() comparisons make that check explicit/readable, and + // guard against this operator== behavior ever changing). + ordered_json a; + a["x"] = 1; + a["y"] = 2; + + ordered_json b; + b["y"] = 2; + b["x"] = 1; + + CHECK(a.size() == b.size()); + CHECK(a["x"] == b["x"]); + CHECK(a["y"] == b["y"]); + CHECK_FALSE(a == b); +} + +TEST_CASE("duplicate keys in a binary-encoded object") +{ + // CBOR encoding of a map with two entries under the same key "a": {"a": 1, "a": 2} + const std::vector cbor_bytes + { + 0xA2, 0x61, 'a', 0x01, 0x61, 'a', 0x02 + }; + + // Both json (std::map, via operator[]) and ordered_json (ordered_map, via + // operator[]) build binary-decoded objects by looking up/creating the + // entry for each incoming key and then assigning the value into it. This + // means a repeated key does *not* produce two entries in either case; + // instead, the *first* occurrence's position is kept (relevant only for + // ordered_json) while the *last* occurrence's value wins (for both) -- + // this matches operator[]'s "assign the referenced slot" semantics, and + // is worth noting because it differs from the initializer-list + // construction path (`ordered_json{{"a",1},{"a",2}}`), which builds + // through insert()/emplace() and therefore keeps the *first* value, not + // the last (see the "There are no dup keys..." case in + // unit-ordered_json.cpp). + const auto j = json::from_cbor(cbor_bytes); + const auto oj = ordered_json::from_cbor(cbor_bytes); + + CHECK(j.size() == 1); + CHECK(oj.size() == 1); + CHECK(j["a"] == 2); + CHECK(oj["a"] == 2); + CHECK(j == json(oj)); +} + +TEST_CASE("ordered_json through flatten/unflatten") +{ + const ordered_json original = make_rich_ordered_json(); + const std::vector original_keys = collect_keys(original); + const std::vector original_mango_keys = collect_keys(original["mango"]); + + const ordered_json flat = original.flatten(); + const ordered_json unflattened = flat.unflatten(); + + CHECK(unflattened == original); + // flatten() walks the value depth-first in iteration order and + // unflatten() re-inserts each flattened key via operator[] in the flat + // object's iteration order, so for ordered_json the original key order + // (both top-level and nested) is preserved end-to-end. + CHECK(collect_keys(unflattened) == original_keys); + CHECK(collect_keys(unflattened["mango"]) == original_mango_keys); +} + +TEST_CASE("ordered_json through diff/patch/patch_inplace") +{ + ordered_json original; + original["one"] = 1; + original["two"] = 2; + original["three"] = 3; + + ordered_json target = original; + target["one"] = 100; // replace + target.erase("two"); // remove + target["four"] = 4; // add + + const ordered_json patch = ordered_json::diff(original, target); + + SECTION("patch") + { + const ordered_json patched = original.patch(patch); + CHECK(patched == target); + } + + SECTION("patch_inplace") + { + ordered_json copy = original; + copy.patch_inplace(patch); + CHECK(copy == target); + } +} + +TEST_CASE("ordered_json through merge_patch") +{ + ordered_json original; + original["a"] = 1; + original["b"] = 2; + + const ordered_json patch = {{"b", nullptr}, {"c", 3}}; + + original.merge_patch(patch); + + ordered_json expected; + expected["a"] = 1; + expected["c"] = 3; + + CHECK(original == expected); + CHECK(collect_keys(original) == collect_keys(expected)); +} diff --git a/tests/src/unit-regression1.cpp b/tests/src/unit-regression1.cpp index 475ef511f..0529f83dd 100644 --- a/tests/src/unit-regression1.cpp +++ b/tests/src/unit-regression1.cpp @@ -29,10 +29,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" - -#ifdef JSON_HAS_CPP_17 - #include -#endif +#include "test_utils.hpp" #include "fifo_map.hpp" @@ -1373,7 +1370,8 @@ TEST_CASE("regression tests 1") std::array key1 = {{ 103, 92, 117, 48, 48, 48, 55, 92, 114, 215, 126, 214, 95, 92, 34, 174, 40, 71, 38, 174, 40, 71, 38, 223, 134, 247, 127, 0 }}; std::string const key1_str(reinterpret_cast(key1.data())); json const j = key1_str; - CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 10: 0x7E", json::type_error&); + // dump() is nodiscard; the exception is thrown by dump() itself before it would return + CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 10: 0x7E", json::type_error&); } #if JSON_USE_IMPLICIT_CONVERSIONS diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index 29b52d701..b128b7a73 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -31,6 +31,8 @@ using ordered_json = nlohmann::ordered_json; #include #include +#include "test_utils.hpp" + #ifdef JSON_HAS_CPP_17 #include #include @@ -239,209 +241,6 @@ class my_allocator : public std::allocator }; }; -///////////////////////////////////////////////////////////////////// -// for #3077 -///////////////////////////////////////////////////////////////////// - -class FooAlloc -{}; - -class Foo -{ - public: - explicit Foo(const FooAlloc& /* unused */ = FooAlloc()) {} - - bool value = false; -}; - -class FooBar -{ - public: - Foo foo{}; // NOLINT(readability-redundant-member-init) -}; - -inline void from_json(const nlohmann::json& j, FooBar& fb) // NOLINT(misc-use-internal-linkage) -{ - j.at("value").get_to(fb.foo.value); -} - -///////////////////////////////////////////////////////////////////// -// for #3171 -///////////////////////////////////////////////////////////////////// - -struct for_3171_base // NOLINT(cppcoreguidelines-special-member-functions) -{ - for_3171_base(const std::string& /*unused*/ = {}) {} - virtual ~for_3171_base(); - - for_3171_base(const for_3171_base& other) // NOLINT(hicpp-use-equals-default,modernize-use-equals-default) - : str(other.str) - {} - - for_3171_base& operator=(const for_3171_base& other) - { - if (this != &other) - { - str = other.str; - } - return *this; - } - - for_3171_base(for_3171_base&& other) noexcept - : str(std::move(other.str)) - {} - - for_3171_base& operator=(for_3171_base&& other) noexcept - { - if (this != &other) - { - str = std::move(other.str); - } - return *this; - } - - virtual void _from_json(const json& j) - { - j.at("str").get_to(str); - } - - std::string str{}; // NOLINT(readability-redundant-member-init) -}; - -for_3171_base::~for_3171_base() = default; - -struct for_3171_derived : public for_3171_base -{ - for_3171_derived() = default; - ~for_3171_derived() override; - explicit for_3171_derived(const std::string& /*unused*/) { } - - for_3171_derived(const for_3171_derived& other) // NOLINT(hicpp-use-equals-default,modernize-use-equals-default) - : for_3171_base(other) - {} - - for_3171_derived& operator=(const for_3171_derived& other) - { - if (this != &other) - { - for_3171_base::operator=(other); // Call base class assignment operator - } - return *this; - } - - for_3171_derived(for_3171_derived&& other) noexcept - : for_3171_base(std::move(other)) - {} - - for_3171_derived& operator=(for_3171_derived&& other) noexcept - { - if (this != &other) - { - for_3171_base::operator=(std::move(other)); // Call base class move assignment operator - } - return *this; - } -}; - -for_3171_derived::~for_3171_derived() = default; - -inline void from_json(const json& j, for_3171_base& tb) // NOLINT(misc-use-internal-linkage) -{ - tb._from_json(j); -} - -///////////////////////////////////////////////////////////////////// -// for #3312 -///////////////////////////////////////////////////////////////////// - -#ifdef JSON_HAS_CPP_20 -struct for_3312 -{ - std::string name; -}; - -inline void from_json(const json& j, for_3312& obj) // NOLINT(misc-use-internal-linkage) -{ - j.at("name").get_to(obj.name); -} -#endif - -///////////////////////////////////////////////////////////////////// -// for #3204 -///////////////////////////////////////////////////////////////////// - -struct for_3204_foo -{ - for_3204_foo() = default; - explicit for_3204_foo(std::string /*unused*/) {} // NOLINT(performance-unnecessary-value-param) -}; - -struct for_3204_bar -{ - enum constructed_from_t // NOLINT(cppcoreguidelines-use-enum-class) - { - constructed_from_none = 0, - constructed_from_foo = 1, - constructed_from_json = 2 - }; - - explicit for_3204_bar(std::function /*unused*/) noexcept // NOLINT(performance-unnecessary-value-param) - : constructed_from(constructed_from_foo) {} - explicit for_3204_bar(std::function /*unused*/) noexcept // NOLINT(performance-unnecessary-value-param) - : constructed_from(constructed_from_json) {} - - constructed_from_t constructed_from = constructed_from_none; -}; - -///////////////////////////////////////////////////////////////////// -// for #3333 -///////////////////////////////////////////////////////////////////// - -struct for_3333 final -{ - for_3333(int x_ = 0, int y_ = 0) : x(x_), y(y_) {} - - template - for_3333(const T& /*unused*/) - { - CHECK(false); - } - - int x = 0; - int y = 0; -}; - -template <> -inline for_3333::for_3333(const json& j) - : for_3333(j.value("x", 0), j.value("y", 0)) -{} - -///////////////////////////////////////////////////////////////////// -// for #3810 -///////////////////////////////////////////////////////////////////// - -struct Example_3810 -{ - int bla{}; - - Example_3810() = default; -}; - -NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE(Example_3810, bla) // NOLINT(misc-use-internal-linkage) - -///////////////////////////////////////////////////////////////////// -// for #4740 -///////////////////////////////////////////////////////////////////// - -#ifdef JSON_HAS_CPP_17 -struct Example_4740 -{ - std::optional host = std::nullopt; - std::optional port = std::nullopt; - NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(Example_4740, host, port) -}; -#endif - TEST_CASE("regression tests 2") { SECTION("issue #1001 - Fix memory leak during parser callback") @@ -639,7 +438,8 @@ TEST_CASE("regression tests 2") s += static_cast(i); } dump_test["1"] = s; - dump_test.dump(-1, ' ', true, nlohmann::json::error_handler_t::replace); + // dump() is nodiscard; this only checks that dumping does not throw/crash + utils::ignore_return_value(dump_test.dump(-1, ' ', true, nlohmann::json::error_handler_t::replace)); } } @@ -731,12 +531,14 @@ TEST_CASE("regression tests 2") { const std::array data = {{0x81, 0xA4, 0x64, 0x61, 0x74, 0x61, 0xC4, 0x0F, 0x33, 0x30, 0x30, 0x32, 0x33, 0x34, 0x30, 0x31, 0x30, 0x37, 0x30, 0x35, 0x30, 0x31, 0x30}}; const json j = json::from_msgpack(data.data(), data.size()); + // dump() is nodiscard; this only checks that dumping does not throw CHECK_NOTHROW( - j.dump(4, // Indent - ' ', // Indent char - false, // Ensure ascii - json::error_handler_t::strict // Error - )); + utils::ignore_return_value( + j.dump(4, // Indent + ' ', // Indent char + false, // Ensure ascii + json::error_handler_t::strict // Error + ))); } SECTION("PR #2181 - regression bug with lvalue") @@ -804,7 +606,11 @@ TEST_CASE("regression tests 2") SECTION("issue #2546 - parsing containers of std::byte") { const char DATA[] = R"("Hello, world!")"; // NOLINT(misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - const auto s = std::as_bytes(std::span(DATA)); + // exclude the trailing '\0' that string-literal initialization adds to + // DATA: std::span(DATA) would span the full array extent (including + // that NUL), which is only silently accepted as end-of-input by default + // and would fail under JSON_STRICT_NUL_HANDLING + const auto s = std::as_bytes(std::span(DATA, sizeof(DATA) - 1)); const json j = json::parse(s); CHECK(j.dump() == "\"Hello, world!\""); } @@ -959,600 +765,108 @@ TEST_CASE("regression tests 2") CHECK(j == k); } -#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM - // JSON_HAS_CPP_17 (do not remove; see note at top of file) - SECTION("issue #3070 - Version 3.10.3 breaks backward-compatibility with 3.10.2 ") +} + +TEST_CASE("regression test - parser callback must not lose a duplicate key's prior value") +{ + // a callback that rejects only the scalar value 2 + const json::parser_callback_t drop_value_2 = [](int /*depth*/, json::parse_event_t ev, json & v) noexcept { - nlohmann::detail::std_fs::path text_path("/tmp/text.txt"); - const json j(text_path); + return !(ev == json::parse_event_t::value && v == 2); + }; - const auto j_path = j.get(); - CHECK(j_path == text_path); - -#if DOCTEST_CLANG || DOCTEST_GCC >= DOCTEST_COMPILER(8, 4, 0) - // only known to work on Clang and GCC >=8.4 - CHECK_THROWS_WITH_AS(nlohmann::detail::std_fs::path(json(1)), "[json.exception.type_error.302] type must be string, but is number", json::type_error); -#endif - } -#endif - - SECTION("issue #3077 - explicit constructor with default does not compile") + SECTION("duplicate key, second (scalar) value rejected - prior value is restored") { - json j; - j[0]["value"] = true; - std::vector foo; - j.get_to(foo); + const json j = json::parse(R"({"a":1,"a":2})", drop_value_2); + CHECK(j.dump() == "{\"a\":1}"); } - SECTION("issue #3108 - ordered_json doesn't support range based erase") + SECTION("duplicate key, second value is an object rejected at object_end - prior value is restored") { - ordered_json j = {1, 2, 2, 4}; - - auto last = std::unique(j.begin(), j.end()); - j.erase(last, j.end()); - - CHECK(j.dump() == "[1,2,4]"); - - j.erase(std::remove_if(j.begin(), j.end(), [](const ordered_json & val) + const json j = json::parse(R"({"a":1,"a":{"x":2}})", + [](int depth, json::parse_event_t ev, json& /*parsed*/) noexcept { - return val == 2; - }), j.end()); - - CHECK(j.dump() == "[1,4]"); + return !(ev == json::parse_event_t::object_end && depth == 1); + }); + CHECK(j.dump() == "{\"a\":1}"); } - SECTION("issue #3343 - json and ordered_json are not interchangeable") + SECTION("duplicate key, second value is an array rejected at array_end - prior value is restored") { - json::object_t jobj({ { "product", "one" } }); - ordered_json::object_t ojobj({{"product", "one"}}); - - auto jit = jobj.begin(); - auto ojit = ojobj.begin(); - - CHECK(jit->first == ojit->first); - CHECK(jit->second.get() == ojit->second.get()); - } - - SECTION("issue #3171 - if class is_constructible from std::string wrong from_json overload is being selected, compilation failed") - { - const json j{{ "str", "value"}}; - - // failed with: error: no match for ‘operator=’ (operand types are ‘for_3171_derived’ and ‘const nlohmann::basic_json<>::string_t’ - // {aka ‘const std::__cxx11::basic_string’}) - // s = *j.template get_ptr(); - auto td = j.get(); - - CHECK(td.str == "value"); - } - -#ifdef JSON_HAS_CPP_20 - SECTION("issue #3312 - Parse to custom class from unordered_json breaks on G++11.2.0 with C++20") - { - // see test for #3171 - const ordered_json j = {{"name", "class"}}; - for_3312 obj{}; - - j.get_to(obj); - - CHECK(obj.name == "class"); - } -#endif - -#if defined(JSON_HAS_CPP_17) && JSON_USE_IMPLICIT_CONVERSIONS - SECTION("issue #3428 - Error occurred when converting nlohmann::json to std::any") - { - const json j; - const std::any a1 = j; - std::any&& a2 = j; - - CHECK(a1.type() == typeid(j)); - CHECK(a2.type() == typeid(j)); - } -#endif - - SECTION("issue #3204 - ambiguous regression") - { - const for_3204_bar bar_from_foo([](for_3204_foo) noexcept {}); // NOLINT(performance-unnecessary-value-param) - const for_3204_bar bar_from_json([](json) noexcept {}); // NOLINT(performance-unnecessary-value-param) - - CHECK(bar_from_foo.constructed_from == for_3204_bar::constructed_from_foo); - CHECK(bar_from_json.constructed_from == for_3204_bar::constructed_from_json); - } - - SECTION("issue #3333 - Ambiguous conversion from nlohmann::basic_json<> to custom class") - { - const json j + const json j = json::parse(R"({"a":1,"a":[9,9]})", + [](int depth, json::parse_event_t ev, json& /*parsed*/) noexcept { - {"x", 1}, - {"y", 2} - }; - const for_3333 p = j; - - CHECK(p.x == 1); - CHECK(p.y == 2); + return !(ev == json::parse_event_t::array_end && depth == 1); + }); + CHECK(j.dump() == "{\"a\":1}"); } - SECTION("issue #3810 - ordered_json doesn't support construction from C array of custom type") + SECTION("duplicate key, second value accepted (scalar) - last value wins") { - Example_3810 states[45]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - - // fix "not used" warning - states[0].bla = 1; - - const auto* const expected = R"([{"bla":1},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0}])"; - - // This works: - nlohmann::json j; - j["test"] = states; - CHECK(j["test"].dump() == expected); - - // This doesn't compile: - nlohmann::ordered_json oj; - oj["test"] = states; - CHECK(oj["test"].dump() == expected); - } - -#ifdef JSON_HAS_CPP_17 - SECTION("issue #4740 - build issue with std::optional") - { - const auto t1 = Example_4740(); - const auto j1 = nlohmann::json(t1); - CHECK(j1.dump() == "{\"host\":null,\"port\":null}"); - const auto t2 = j1.get(); - CHECK(!t2.host.has_value()); - CHECK(!t2.port.has_value()); - - // improve coverage - auto t3 = Example_4740(); - t3.port = 80; - t3.host = "example.com"; - const auto j2 = nlohmann::json(t3); - CHECK(j2.dump() == "{\"host\":\"example.com\",\"port\":80}"); - const auto t4 = j2.get(); - CHECK(t4.host.has_value()); - CHECK(t4.port.has_value()); - } -#endif - -#if !defined(_MSVC_LANG) - // MSVC returns garbage on invalid enum values, so this test is excluded - // there. - SECTION("issue #4762 - json exception 302 with unhelpful explanation : type must be number, but is number") - { - // In #4762, the main issue was that a json object with an invalid type - // returned "number" as type_name(), because this was the default case. - // This test makes sure we now return "invalid" instead. - json j; - j.m_data.m_type = static_cast(100); // NOLINT(clang-analyzer-optin.core.EnumCastOutOfRange) - CHECK(j.type_name() == "invalid"); - } -#endif - -#ifdef JSON_HAS_CPP_17 - SECTION("issue #4804: from_cbor incompatible with std::vector as binary_t") - { - const std::vector data = {0x80}; - const auto decoded = json_4804::from_cbor(data); - CHECK((decoded == json_4804::array())); - } - - SECTION("discussion #4209 - custom BinaryType direct assignment and round-tripping") - { - // Test that assigning a custom BinaryType directly creates a binary value, not an array - const std::vector original{std::byte{1}, std::byte{2}, std::byte{3}}; - const json_4804 j = original; - CHECK(j.is_binary()); - CHECK(!j.is_array()); - - // Test round-tripping: extracting the binary value back as the custom container type - const auto extracted = j.get>(); - CHECK(extracted == original); - - // Test that the default json alias behavior is unchanged: std::vector -> array - const json default_json = std::vector {1, 2, 3}; - CHECK(default_json.is_array()); - CHECK(!default_json.is_binary()); - } - - SECTION("discussion #4209 - custom BinaryType extraction from parsed array") - { - // Test that extracting a custom BinaryType from a parsed JSON array still works - // (not just from a binary-typed node) - const auto j = json_4804::parse("[1,2,3]"); - CHECK(j.is_array()); - CHECK(!j.is_binary()); - - // Extracting as custom BinaryType should work from arrays - const auto extracted = j.get>(); - CHECK(extracted.size() == 3); - CHECK(extracted[0] == std::byte{1}); - CHECK(extracted[1] == std::byte{2}); - CHECK(extracted[2] == std::byte{3}); - } - - SECTION("issue #5046 - implicit conversion of return json to std::optional no longer implicit") - { - const json jval{}; - auto GetValue = [](const json & valRoot) -> std::optional - { - if (valRoot.contains("default")) - { - return valRoot.at("default"); - } - return std::nullopt; - }; - auto result = GetValue(jval); - CHECK(!result.has_value()); - } -#endif - -#if JSON_HAS_RANGES == 1 - SECTION("issue #4440 - assert when using std::views::filter and GCC 10") - { - auto noOpFilter = std::views::filter([](auto&&) noexcept + const json j = json::parse(R"({"a":1,"a":2})", [](int, json::parse_event_t, json&) noexcept { return true; }); - json j = {1, 2, 3}; - auto filtered = j | noOpFilter; - CHECK(*filtered.begin() == 1); + CHECK(j.dump() == "{\"a\":2}"); } -#endif -#if JSON_HAS_RANGES && !defined(__MINGW32__) - SECTION("issue #4916 - constructing array from C++20 ranges view does not work") + SECTION("duplicate key, second value accepted (object) - last value wins") { - std::vector nums{1, 2, 37, 42, 21}; - auto filteredNums = nums | std::views::filter([](int i) + const json j = json::parse(R"({"a":1,"a":{"x":2}})", [](int, json::parse_event_t, json&) noexcept { - return i > 10; + return true; }); - json const j(filteredNums); - CHECK(j.type() == json::value_t::array); - CHECK(j == json({37, 42, 21})); + CHECK(j.dump() == "{\"a\":{\"x\":2}}"); } -#endif - // owning_view is not available in libstdc++ < 12 -#if JSON_HAS_RANGES && !defined(__MINGW32__) && !(defined(__GLIBCXX__) && _GLIBCXX_RELEASE < 12) - SECTION("issue #4916 - constructing array from prvalue C++20 ranges view (owning_view)") + SECTION("brand new (non-duplicate) key, value rejected - member is fully absent") { - json const j(std::vector {1, 2, 37, 42, 21} | std::views::filter([](int i) - { - return i > 10; - })); - CHECK(j.type() == json::value_t::array); - CHECK(j == json({37, 42, 21})); + const json j = json::parse(R"({"a":1,"b":2})", drop_value_2); + CHECK(j.dump() == "{\"a\":1}"); } -#endif -#if JSON_HAS_RANGES && !defined(__MINGW32__) - SECTION("issue #4916 - constructing array from C++20 transform view (prvalue elements)") + SECTION("duplicate key nested two levels deep") { - std::vector nums{1, 2, 3}; - auto t = nums | std::views::transform([](int i) noexcept - { - return i * 2; - }); - json const j(t); - CHECK(j.type() == json::value_t::array); - CHECK(j == json({2, 4, 6})); + const json j = json::parse(R"({"outer":{"a":1,"a":2}})", drop_value_2); + CHECK(j.dump() == "{\"outer\":{\"a\":1}}"); } -#endif -} -TEST_CASE_TEMPLATE("issue #4798 - nlohmann::json::to_msgpack() encode float NaN as double", T, double, float) // NOLINT(readability-math-missing-parentheses, bugprone-throwing-static-initialization) -{ - // With issue #4798, we encode NaN, infinity, and -infinity as float instead - // of double to allow for smaller encodings. - const json jx = std::numeric_limits::quiet_NaN(); - const json jy = std::numeric_limits::infinity(); - const json jz = -std::numeric_limits::infinity(); - - ///////////////////////////////////////////////////////////////////////// - // MessagePack - ///////////////////////////////////////////////////////////////////////// - - // expected MessagePack values - const std::vector msgpack_x = {{0xCA, 0x7F, 0xC0, 0x00, 0x00}}; - const std::vector msgpack_y = {{0xCA, 0x7F, 0x80, 0x00, 0x00}}; - const std::vector msgpack_z = {{0xCA, 0xFF, 0x80, 0x00, 0x00}}; - - CHECK(json::to_msgpack(jx) == msgpack_x); - CHECK(json::to_msgpack(jy) == msgpack_y); - CHECK(json::to_msgpack(jz) == msgpack_z); - - CHECK(std::isnan(json::from_msgpack(msgpack_x).get())); - CHECK(json::from_msgpack(msgpack_y).get() == std::numeric_limits::infinity()); - CHECK(json::from_msgpack(msgpack_z).get() == -std::numeric_limits::infinity()); - - // Make sure the other MessagePakc encodings for NaN, infinity, and - // -infinity are still supported. - const std::vector msgpack_x_2 = {{0xCB, 0x7F, 0xF8, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}}; - const std::vector msgpack_y_2 = {{0xCB, 0x7F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}}; - const std::vector msgpack_z_2 = {{0xCB, 0xFF, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}}; - CHECK(std::isnan(json::from_msgpack(msgpack_x_2).get())); - CHECK(json::from_msgpack(msgpack_y_2).get() == std::numeric_limits::infinity()); - CHECK(json::from_msgpack(msgpack_z_2).get() == -std::numeric_limits::infinity()); - - ///////////////////////////////////////////////////////////////////////// - // CBOR - ///////////////////////////////////////////////////////////////////////// - - // expected CBOR values - const std::vector cbor_x = {{0xF9, 0x7E, 0x00}}; - const std::vector cbor_y = {{0xF9, 0x7C, 0x00}}; - const std::vector cbor_z = {{0xF9, 0xfC, 0x00}}; - - CHECK(json::to_cbor(jx) == cbor_x); - CHECK(json::to_cbor(jy) == cbor_y); - CHECK(json::to_cbor(jz) == cbor_z); - - CHECK(std::isnan(json::from_cbor(cbor_x).get())); - CHECK(json::from_cbor(cbor_y).get() == std::numeric_limits::infinity()); - CHECK(json::from_cbor(cbor_z).get() == -std::numeric_limits::infinity()); - - // Make sure the other CBOR encodings for NaN, infinity, and -infinity are - // still supported. - const std::vector cbor_x_2 = {{0xFA, 0x7F, 0xC0, 0x00, 0x00}}; - const std::vector cbor_y_2 = {{0xFA, 0x7F, 0x80, 0x00, 0x00}}; - const std::vector cbor_z_2 = {{0xFA, 0xFF, 0x80, 0x00, 0x00}}; - const std::vector cbor_x_3 = {{0xFB, 0x7F, 0xF8, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}}; - const std::vector cbor_y_3 = {{0xFB, 0x7F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}}; - const std::vector cbor_z_3 = {{0xFB, 0xFF, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}}; - CHECK(std::isnan(json::from_cbor(cbor_x_2).get())); - CHECK(json::from_cbor(cbor_y_2).get() == std::numeric_limits::infinity()); - CHECK(json::from_cbor(cbor_z_2).get() == -std::numeric_limits::infinity()); - CHECK(std::isnan(json::from_cbor(cbor_x_3).get())); - CHECK(json::from_cbor(cbor_y_3).get() == std::numeric_limits::infinity()); - CHECK(json::from_cbor(cbor_z_3).get() == -std::numeric_limits::infinity()); -} - -TEST_CASE("regression test #5074 - portable workaround for single-element brace init") -{ - json const j_obj = {{"key", "value"}}; - - json const j = json::array({j_obj}); - CHECK(j.is_array()); - CHECK(j.size() == 1); - CHECK(j[0] == j_obj); -} - -#if defined(JSON_BRACE_INIT_COPY_SEMANTICS) && (JSON_BRACE_INIT_COPY_SEMANTICS == 1) -TEST_CASE("regression test #5074 - single-element brace init with JSON_BRACE_INIT_COPY_SEMANTICS") -{ - // with JSON_BRACE_INIT_COPY_SEMANTICS: single-element brace init copies/moves - json const j_obj = {{"key", "value"}, {"num", 42}}; - json const j_arr = {1, 2, 3}; - - // object: brace init copies instead of wrapping - json const j1{j_obj}; - CHECK(j1.is_object()); - CHECK(j1 == j_obj); - - // array: brace init copies instead of wrapping - json const j2{j_arr}; - CHECK(j2.is_array()); - CHECK(j2.size() == 3); - CHECK(j2 == j_arr); - - // primitives still work as initializer lists - json const j3{true}; - CHECK(j3.is_boolean()); - - json const j4{42}; - CHECK(j4.is_number_integer()); -} -#endif - -struct Example_5122 -{ - float b = 2; - nlohmann::ordered_map c{}; // NOLINT(readability-redundant-member-init): needed for GCC -Weffc++ - int a = 1; - NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(Example_5122, b, c, a) -}; - -TEST_CASE("regression test #5122 - from_json into types holding nlohmann::ordered_map") -{ - Example_5122 src; - src.c.emplace("first", "1"); - src.c.emplace("second", "2"); - - ordered_json const j = src; - Example_5122 const dst = j.get(); - - CHECK(dst.b == src.b); - CHECK(dst.a == src.a); - REQUIRE(dst.c.size() == src.c.size()); - auto src_it = src.c.begin(); - auto dst_it = dst.c.begin(); - for (; src_it != src.c.end(); ++src_it, ++dst_it) + SECTION("three occurrences of the same key - middle rejected, last accepted") { - CHECK(dst_it->first == src_it->first); - CHECK(dst_it->second == src_it->second); + const json j = json::parse(R"({"k":1,"k":2,"k":3})", drop_value_2); + CHECK(j.dump() == "{\"k\":3}"); } } -// -Wself-assign-overloaded was introduced in Clang 7. Gate the pragma on -// __has_warning so older Clang versions do not error with "unknown warning -// group". The __has_warning check has to stay inside the __clang__ branch -// because GCC does not provide it and would tokenize-error on the argument. -#if defined(__clang__) && defined(__has_warning) - #if __has_warning("-Wself-assign-overloaded") - DOCTEST_CLANG_SUPPRESS_WARNING_PUSH - DOCTEST_CLANG_SUPPRESS_WARNING("-Wself-assign-overloaded") - #endif -#endif - -TEST_CASE("regression test #5122 - nlohmann::ordered_map copy-assignment is self-assignment safe") +TEST_CASE("regression test - excessive binary container size honors allow_exceptions=false") { - nlohmann::ordered_map m; - m.emplace("first", "1"); - m.emplace("second", "2"); + // CBOR array with declared length 2^63 + const std::vector cbor = {0x9b, 0x80, 0, 0, 0, 0, 0, 0, 0}; + // CBOR map with declared length 2^63 + const std::vector cbor_m = {0xbb, 0x80, 0, 0, 0, 0, 0, 0, 0}; + // UBJSON array with declared length 2^63-1 + const std::vector ubj = {'[', '#', 'L', 0x7f, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff}; + // BJData array with declared length 2^63-1 (little endian) + const std::vector bjd = {'[', '#', 'L', 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x7f}; - // Insertion order is preserved by ordered_map, so we can check it directly. - m = m; + // allow_exceptions=false must report failure instead of throwing/aborting + CHECK(json::from_cbor(cbor, true, false).is_discarded()); + CHECK(json::from_cbor(cbor_m, true, false).is_discarded()); + CHECK(json::from_ubjson(ubj, true, false).is_discarded()); + CHECK(json::from_bjdata(bjd, true, false).is_discarded()); - REQUIRE(m.size() == 2); - auto it = m.begin(); - CHECK(it->first == "first"); - CHECK(it->second == "1"); - ++it; - CHECK(it->first == "second"); - CHECK(it->second == "2"); -} + // allow_exceptions=true (the default) must still throw exactly as before. + // The exact message text is not checked here: on platforms where + // std::size_t is 32-bit, the CBOR reader's own length-narrowing check + // (get_cbor_container_size(), unrelated to this fix) intercepts a + // declared length of 2^63 before it ever reaches the check this test + // targets, with different (but equally valid, and already correct) + // wording -- see unit-cbor.cpp for coverage of that message. + json _; + CHECK_THROWS_AS(_ = json::from_cbor(cbor), json::out_of_range); -#if defined(__clang__) && defined(__has_warning) - #if __has_warning("-Wself-assign-overloaded") - DOCTEST_CLANG_SUPPRESS_WARNING_POP - #endif -#endif - -TEST_CASE("regression test #5122 - nlohmann::ordered_map move-assignment transfers contents") -{ - nlohmann::ordered_map src; - src.emplace("first", "1"); - src.emplace("second", "2"); - - nlohmann::ordered_map dst; - dst.emplace("stale", "x"); - dst = std::move(src); - - REQUIRE(dst.size() == 2); - auto it = dst.begin(); - CHECK(it->first == "first"); - CHECK(it->second == "1"); - ++it; - CHECK(it->first == "second"); - CHECK(it->second == "2"); - - // Re-assigning into the moved-from object must leave it in a usable state. - src = nlohmann::ordered_map {}; - src.emplace("after-move", "3"); - REQUIRE(src.size() == 1); - CHECK(src.begin()->first == "after-move"); -} - -// Stand-in for a third-party library (e.g., Eigen as of 3.4, which added -// STL-compatible begin()/end() to its vector types), living in its own -// namespace with its own to_json overload for its vector type. -namespace issue_4320_eigen -{ -// "array-compatible" from the library's point of view (it has begin()/end()), -// but for which this (fake) third-party namespace provides its own to_json. -struct vector3 -{ - double v[3]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays,cppcoreguidelines-use-default-member-init,modernize-use-default-member-init) - vector3(double x, double y, double z) : v{x, y, z} {} // NOLINT(hicpp-member-init,cppcoreguidelines-pro-type-member-init) - double x() const - { - return v[0]; - } - double y() const - { - return v[1]; - } - double z() const - { - return v[2]; - } - double* begin() - { - return v; - } - double* end() - { - return v + 3; - } - const double* begin() const - { - return v; - } - const double* end() const - { - return v + 3; - } -}; - -inline void to_json(json& j, const vector3& v) // NOLINT(misc-use-internal-linkage) -{ - j = {{"x", v.x()}, {"y", v.y()}, {"z", v.z()}}; -} -} // namespace issue_4320_eigen - -// The user's own namespace, using the (fake) Eigen type as an implementation -// detail behind a payload type that has nothing to do with vectors/arrays. -namespace issue_4320 -{ -// Publicly derives from issue_4320_eigen::vector3 but does *not* define its -// own to_json - it is only ever used as a temporary to reach the base -// class's to_json via ADL. -struct vector3_wrapper : issue_4320_eigen::vector3 -{ - using issue_4320_eigen::vector3::vector3; -}; - -struct payload -{ - double x, y, z; -}; - -inline vector3_wrapper to_eigen(const payload& p) // NOLINT(misc-use-internal-linkage) -{ - return {p.x, p.y, p.z}; -} - -inline void to_json(json& j, const payload& p) // NOLINT(misc-use-internal-linkage) -{ - // Unqualified call, passing a *derived* vector3_wrapper: relies on ADL - // finding issue_4320_eigen::to_json(json&, const vector3&) through the - // vector3 base class, via a derived-to-base conversion. Must NOT resolve - // to the library's own generic array-compatible to_json (an exact-match - // template for vector3_wrapper, since it also has begin()/end()), which - // would serialize this as [x, y, z] instead of {"x":x, "y":y, "z":z}. - to_json(j, to_eigen(p)); -} -} // namespace issue_4320 - -TEST_CASE("issue #4320 - custom base class must not leak nlohmann::detail into ADL") -{ - // Before the fix, basic_json unconditionally derived from a type living in - // nlohmann::detail (json_default_base), which made nlohmann::detail an - // associated namespace of every basic_json for ADL purposes. That leaked - // the library's internal generic-array to_json overload into unqualified - // to_json() calls made from user code, silently bypassing user-defined - // to_json overloads reached via a derived-to-base conversion. - const issue_4320::payload p{1.0, 2.0, 3.0}; - - json j; - to_json(j, p); - CHECK(j == json({{"x", 1.0}, {"y", 2.0}, {"z", 3.0}})); -} - -TEST_CASE("issue #5338 - truncated CBOR tagged binary subtype is rejected") -{ - const std::vector> truncated_tags = - { - {0xD8}, - {0xD9, 0x00}, - {0xDA, 0x00, 0x00, 0x00}, - {0xDB, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00} - }; - - for (const auto& data : truncated_tags) - { - CAPTURE(data); - for (const auto tag_handler : - { - json::cbor_tag_handler_t::ignore, json::cbor_tag_handler_t::store - }) - { - CAPTURE(tag_handler); - const auto result = json::from_cbor(data, true, false, tag_handler); - CHECK(result.is_discarded()); - } - } + // regression guard: a genuinely truncated CBOR input must remain discarded + CHECK(json::from_cbor(std::vector {0x9b, 0, 0, 0, 0, 0, 0, 0, 0x02}, true, false).is_discarded()); } DOCTEST_CLANG_SUPPRESS_WARNING_POP diff --git a/tests/src/unit-regression3.cpp b/tests/src/unit-regression3.cpp new file mode 100644 index 000000000..882b3866b --- /dev/null +++ b/tests/src/unit-regression3.cpp @@ -0,0 +1,928 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// cmake/test.cmake selects the C++ standard versions with which to build a +// unit test based on the presence of JSON_HAS_CPP_ macros. +// When using macros that are only defined for particular versions of the standard +// (e.g., JSON_HAS_FILESYSTEM for C++17 and up), please mention the corresponding +// version macro in a comment close by, like this: +// JSON_HAS_CPP_ (do not remove; see note at top of file) + +#include "doctest_compatibility.h" + +// for some reason including this after the json header leads to linker errors with VS 2017... +#include + +// skip tests if JSON_DisableEnumSerialization=ON (#4384): std::byte is a +// scoped enum, so get() (needed below to get>() +// from a plain JSON array, not just from an already-binary value) relies on +// enum serialization being enabled +#if defined(JSON_DISABLE_ENUM_SERIALIZATION) && (JSON_DISABLE_ENUM_SERIALIZATION == 1) + #define SKIP_TESTS_FOR_ENUM_SERIALIZATION +#endif + +#define JSON_TESTS_PRIVATE +#include +using json = nlohmann::json; +using ordered_json = nlohmann::ordered_json; +#ifdef JSON_TEST_NO_GLOBAL_UDLS + using namespace nlohmann::literals; // NOLINT(google-build-using-namespace) +#endif + +#include +#include +#include +#include +#include + +#ifdef JSON_HAS_CPP_17 + #include + #include +#endif + +#ifdef JSON_HAS_CPP_17 + #if __has_include() + #include + #elif __has_include() + #endif + + ///////////////////////////////////////////////////////////////////// + // for #4804 + ///////////////////////////////////////////////////////////////////// + using json_4804 = nlohmann::basic_json, // BinaryType + void // CustomBaseClass + >; +#endif + +#ifdef JSON_HAS_CPP_20 + #if __has_include() + #include + #endif +#endif + +///////////////////////////////////////////////////////////////////// +// for #4825 - explicitly instantiating basic_json must compile; this +// forces instantiation of binary_writer::write_bjdata_ndarray, whose +// static_cast was ambiguous under explicit instantiation on +// C++17. Merely compiling this translation unit is the regression test. +///////////////////////////////////////////////////////////////////// +template class nlohmann::basic_json<>; + +///////////////////////////////////////////////////////////////////// +// for #4440 +///////////////////////////////////////////////////////////////////// +#if JSON_HAS_RANGES == 1 + #include +#endif + +// NLOHMANN_JSON_SERIALIZE_ENUM uses a static std::pair +DOCTEST_CLANG_SUPPRESS_WARNING_PUSH +DOCTEST_CLANG_SUPPRESS_WARNING("-Wexit-time-destructors") +///////////////////////////////////////////////////////////////////// +// for #3077 +///////////////////////////////////////////////////////////////////// + +class FooAlloc +{}; + +class Foo +{ + public: + explicit Foo(const FooAlloc& /* unused */ = FooAlloc()) {} + + bool value = false; +}; + +class FooBar +{ + public: + Foo foo{}; // NOLINT(readability-redundant-member-init) +}; + +inline void from_json(const nlohmann::json& j, FooBar& fb) // NOLINT(misc-use-internal-linkage) +{ + j.at("value").get_to(fb.foo.value); +} + +///////////////////////////////////////////////////////////////////// +// for #3171 +///////////////////////////////////////////////////////////////////// + +struct for_3171_base // NOLINT(cppcoreguidelines-special-member-functions) +{ + for_3171_base(const std::string& /*unused*/ = {}) {} + virtual ~for_3171_base(); + + for_3171_base(const for_3171_base& other) // NOLINT(hicpp-use-equals-default,modernize-use-equals-default) + : str(other.str) + {} + + for_3171_base& operator=(const for_3171_base& other) + { + if (this != &other) + { + str = other.str; + } + return *this; + } + + for_3171_base(for_3171_base&& other) noexcept + : str(std::move(other.str)) + {} + + for_3171_base& operator=(for_3171_base&& other) noexcept + { + if (this != &other) + { + str = std::move(other.str); + } + return *this; + } + + virtual void _from_json(const json& j) + { + j.at("str").get_to(str); + } + + std::string str{}; // NOLINT(readability-redundant-member-init) +}; + +for_3171_base::~for_3171_base() = default; + +struct for_3171_derived : public for_3171_base +{ + for_3171_derived() = default; + ~for_3171_derived() override; + explicit for_3171_derived(const std::string& /*unused*/) { } + + for_3171_derived(const for_3171_derived& other) // NOLINT(hicpp-use-equals-default,modernize-use-equals-default) + : for_3171_base(other) + {} + + for_3171_derived& operator=(const for_3171_derived& other) + { + if (this != &other) + { + for_3171_base::operator=(other); // Call base class assignment operator + } + return *this; + } + + for_3171_derived(for_3171_derived&& other) noexcept + : for_3171_base(std::move(other)) + {} + + for_3171_derived& operator=(for_3171_derived&& other) noexcept + { + if (this != &other) + { + for_3171_base::operator=(std::move(other)); // Call base class move assignment operator + } + return *this; + } +}; + +for_3171_derived::~for_3171_derived() = default; + +inline void from_json(const json& j, for_3171_base& tb) // NOLINT(misc-use-internal-linkage) +{ + tb._from_json(j); +} + +///////////////////////////////////////////////////////////////////// +// for #3312 +///////////////////////////////////////////////////////////////////// + +#ifdef JSON_HAS_CPP_20 +struct for_3312 +{ + std::string name; +}; + +inline void from_json(const json& j, for_3312& obj) // NOLINT(misc-use-internal-linkage) +{ + j.at("name").get_to(obj.name); +} +#endif + +///////////////////////////////////////////////////////////////////// +// for #3204 +///////////////////////////////////////////////////////////////////// + +struct for_3204_foo +{ + for_3204_foo() = default; + explicit for_3204_foo(std::string /*unused*/) {} // NOLINT(performance-unnecessary-value-param) +}; + +struct for_3204_bar +{ + enum constructed_from_t // NOLINT(cppcoreguidelines-use-enum-class) + { + constructed_from_none = 0, + constructed_from_foo = 1, + constructed_from_json = 2 + }; + + explicit for_3204_bar(std::function /*unused*/) noexcept // NOLINT(performance-unnecessary-value-param) + : constructed_from(constructed_from_foo) {} + explicit for_3204_bar(std::function /*unused*/) noexcept // NOLINT(performance-unnecessary-value-param) + : constructed_from(constructed_from_json) {} + + constructed_from_t constructed_from = constructed_from_none; +}; + +///////////////////////////////////////////////////////////////////// +// for #3333 +///////////////////////////////////////////////////////////////////// + +struct for_3333 final +{ + for_3333(int x_ = 0, int y_ = 0) : x(x_), y(y_) {} + + template + for_3333(const T& /*unused*/) + { + CHECK(false); + } + + int x = 0; + int y = 0; +}; + +template <> +inline for_3333::for_3333(const json& j) + : for_3333(j.value("x", 0), j.value("y", 0)) +{} + +///////////////////////////////////////////////////////////////////// +// for #3810 +///////////////////////////////////////////////////////////////////// + +struct Example_3810 +{ + int bla{}; + + Example_3810() = default; +}; + +NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE(Example_3810, bla) // NOLINT(misc-use-internal-linkage) + +///////////////////////////////////////////////////////////////////// +// for #4740 +///////////////////////////////////////////////////////////////////// + +#ifdef JSON_HAS_CPP_17 +struct Example_4740 +{ + std::optional host = std::nullopt; + std::optional port = std::nullopt; + NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(Example_4740, host, port) +}; +#endif + +TEST_CASE("regression tests 3") +{ +#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM + // JSON_HAS_CPP_17 (do not remove; see note at top of file) + SECTION("issue #3070 - Version 3.10.3 breaks backward-compatibility with 3.10.2 ") + { + nlohmann::detail::std_fs::path text_path("/tmp/text.txt"); + const json j(text_path); + + const auto j_path = j.get(); + CHECK(j_path == text_path); + +#if DOCTEST_CLANG || DOCTEST_GCC >= DOCTEST_COMPILER(8, 4, 0) + // only known to work on Clang and GCC >=8.4 + CHECK_THROWS_WITH_AS(nlohmann::detail::std_fs::path(json(1)), "[json.exception.type_error.302] type must be string, but is number", json::type_error); +#endif + } +#endif + + SECTION("issue #3077 - explicit constructor with default does not compile") + { + json j; + j[0]["value"] = true; + std::vector foo; + j.get_to(foo); + } + + SECTION("issue #3108 - ordered_json doesn't support range based erase") + { + ordered_json j = {1, 2, 2, 4}; + + auto last = std::unique(j.begin(), j.end()); + j.erase(last, j.end()); + + CHECK(j.dump() == "[1,2,4]"); + + j.erase(std::remove_if(j.begin(), j.end(), [](const ordered_json & val) + { + return val == 2; + }), j.end()); + + CHECK(j.dump() == "[1,4]"); + } + + SECTION("issue #3343 - json and ordered_json are not interchangeable") + { + json::object_t jobj({ { "product", "one" } }); + ordered_json::object_t ojobj({{"product", "one"}}); + + auto jit = jobj.begin(); + auto ojit = ojobj.begin(); + + CHECK(jit->first == ojit->first); + CHECK(jit->second.get() == ojit->second.get()); + } + + SECTION("issue #3171 - if class is_constructible from std::string wrong from_json overload is being selected, compilation failed") + { + const json j{{ "str", "value"}}; + + // failed with: error: no match for ‘operator=’ (operand types are ‘for_3171_derived’ and ‘const nlohmann::basic_json<>::string_t’ + // {aka ‘const std::__cxx11::basic_string’}) + // s = *j.template get_ptr(); + auto td = j.get(); + + CHECK(td.str == "value"); + } + +#ifdef JSON_HAS_CPP_20 + SECTION("issue #3312 - Parse to custom class from unordered_json breaks on G++11.2.0 with C++20") + { + // see test for #3171 + const ordered_json j = {{"name", "class"}}; + for_3312 obj{}; + + j.get_to(obj); + + CHECK(obj.name == "class"); + } +#endif + +#if defined(JSON_HAS_CPP_17) && JSON_USE_IMPLICIT_CONVERSIONS + SECTION("issue #3428 - Error occurred when converting nlohmann::json to std::any") + { + const json j; + const std::any a1 = j; + std::any&& a2 = j; + + CHECK(a1.type() == typeid(j)); + CHECK(a2.type() == typeid(j)); + } +#endif + + SECTION("issue #3204 - ambiguous regression") + { + const for_3204_bar bar_from_foo([](for_3204_foo) noexcept {}); // NOLINT(performance-unnecessary-value-param) + const for_3204_bar bar_from_json([](json) noexcept {}); // NOLINT(performance-unnecessary-value-param) + + CHECK(bar_from_foo.constructed_from == for_3204_bar::constructed_from_foo); + CHECK(bar_from_json.constructed_from == for_3204_bar::constructed_from_json); + } + + SECTION("issue #3333 - Ambiguous conversion from nlohmann::basic_json<> to custom class") + { + const json j + { + {"x", 1}, + {"y", 2} + }; + const for_3333 p = j; + + CHECK(p.x == 1); + CHECK(p.y == 2); + } + + SECTION("issue #3810 - ordered_json doesn't support construction from C array of custom type") + { + Example_3810 states[45]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + + // fix "not used" warning + states[0].bla = 1; + + const auto* const expected = R"([{"bla":1},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0},{"bla":0}])"; + + // This works: + nlohmann::json j; + j["test"] = states; + CHECK(j["test"].dump() == expected); + + // This doesn't compile: + nlohmann::ordered_json oj; + oj["test"] = states; + CHECK(oj["test"].dump() == expected); + } + +#ifdef JSON_HAS_CPP_17 + SECTION("issue #4740 - build issue with std::optional") + { + const auto t1 = Example_4740(); + const auto j1 = nlohmann::json(t1); + CHECK(j1.dump() == "{\"host\":null,\"port\":null}"); + const auto t2 = j1.get(); + CHECK(!t2.host.has_value()); + CHECK(!t2.port.has_value()); + + // improve coverage + auto t3 = Example_4740(); + t3.port = 80; + t3.host = "example.com"; + const auto j2 = nlohmann::json(t3); + CHECK(j2.dump() == "{\"host\":\"example.com\",\"port\":80}"); + const auto t4 = j2.get(); + CHECK(t4.host.has_value()); + CHECK(t4.port.has_value()); + } +#endif + +#if !defined(_MSVC_LANG) + // MSVC returns garbage on invalid enum values, so this test is excluded + // there. + SECTION("issue #4762 - json exception 302 with unhelpful explanation : type must be number, but is number") + { + // In #4762, the main issue was that a json object with an invalid type + // returned "number" as type_name(), because this was the default case. + // This test makes sure we now return "invalid" instead. + json j; + j.m_data.m_type = static_cast(100); // NOLINT(clang-analyzer-optin.core.EnumCastOutOfRange) + CHECK(j.type_name() == "invalid"); + } +#endif + +#ifdef JSON_HAS_CPP_17 + SECTION("issue #4804: from_cbor incompatible with std::vector as binary_t") + { + const std::vector data = {0x80}; + const auto decoded = json_4804::from_cbor(data); + CHECK((decoded == json_4804::array())); + } + +#ifndef SKIP_TESTS_FOR_ENUM_SERIALIZATION + SECTION("discussion #4209 - custom BinaryType direct assignment and round-tripping") + { + // Test that assigning a custom BinaryType directly creates a binary value, not an array + const std::vector original{std::byte{1}, std::byte{2}, std::byte{3}}; + const json_4804 j = original; + CHECK(j.is_binary()); + CHECK(!j.is_array()); + + // Test round-tripping: extracting the binary value back as the custom container type + const auto extracted = j.get>(); + CHECK(extracted == original); + + // Test that the default json alias behavior is unchanged: std::vector -> array + const json default_json = std::vector {1, 2, 3}; + CHECK(default_json.is_array()); + CHECK(!default_json.is_binary()); + } + + SECTION("discussion #4209 - custom BinaryType extraction from parsed array") + { + // Test that extracting a custom BinaryType from a parsed JSON array still works + // (not just from a binary-typed node) + const auto j = json_4804::parse("[1,2,3]"); + CHECK(j.is_array()); + CHECK(!j.is_binary()); + + // Extracting as custom BinaryType should work from arrays + const auto extracted = j.get>(); + CHECK(extracted.size() == 3); + CHECK(extracted[0] == std::byte{1}); + CHECK(extracted[1] == std::byte{2}); + CHECK(extracted[2] == std::byte{3}); + } +#endif + + SECTION("issue #5046 - implicit conversion of return json to std::optional no longer implicit") + { + const json jval{}; + auto GetValue = [](const json & valRoot) -> std::optional + { + if (valRoot.contains("default")) + { + return valRoot.at("default"); + } + return std::nullopt; + }; + auto result = GetValue(jval); + CHECK(!result.has_value()); + } +#endif + +#if JSON_HAS_RANGES == 1 + SECTION("issue #4440 - assert when using std::views::filter and GCC 10") + { + auto noOpFilter = std::views::filter([](auto&&) noexcept + { + return true; + }); + json j = {1, 2, 3}; + auto filtered = j | noOpFilter; + CHECK(*filtered.begin() == 1); + } +#endif + +#if JSON_HAS_RANGES && !defined(__MINGW32__) + SECTION("issue #4916 - constructing array from C++20 ranges view does not work") + { + std::vector nums{1, 2, 37, 42, 21}; + auto filteredNums = nums | std::views::filter([](int i) + { + return i > 10; + }); + json const j(filteredNums); + CHECK(j.type() == json::value_t::array); + CHECK(j == json({37, 42, 21})); + } +#endif + + // owning_view is not available in libstdc++ < 12 +#if JSON_HAS_RANGES && !defined(__MINGW32__) && !(defined(__GLIBCXX__) && _GLIBCXX_RELEASE < 12) + SECTION("issue #4916 - constructing array from prvalue C++20 ranges view (owning_view)") + { + json const j(std::vector {1, 2, 37, 42, 21} | std::views::filter([](int i) + { + return i > 10; + })); + CHECK(j.type() == json::value_t::array); + CHECK(j == json({37, 42, 21})); + } +#endif + +#if JSON_HAS_RANGES && !defined(__MINGW32__) + SECTION("issue #4916 - constructing array from C++20 transform view (prvalue elements)") + { + std::vector nums{1, 2, 3}; + auto t = nums | std::views::transform([](int i) noexcept + { + return i * 2; + }); + json const j(t); + CHECK(j.type() == json::value_t::array); + CHECK(j == json({2, 4, 6})); + } +#endif +} + +TEST_CASE_TEMPLATE("issue #4798 - nlohmann::json::to_msgpack() encode float NaN as double", T, double, float) // NOLINT(readability-math-missing-parentheses, bugprone-throwing-static-initialization) +{ + // With issue #4798, we encode NaN, infinity, and -infinity as float instead + // of double to allow for smaller encodings. + const json jx = std::numeric_limits::quiet_NaN(); + const json jy = std::numeric_limits::infinity(); + const json jz = -std::numeric_limits::infinity(); + + ///////////////////////////////////////////////////////////////////////// + // MessagePack + ///////////////////////////////////////////////////////////////////////// + + // expected MessagePack values + const std::vector msgpack_x = {{0xCA, 0x7F, 0xC0, 0x00, 0x00}}; + const std::vector msgpack_y = {{0xCA, 0x7F, 0x80, 0x00, 0x00}}; + const std::vector msgpack_z = {{0xCA, 0xFF, 0x80, 0x00, 0x00}}; + + CHECK(json::to_msgpack(jx) == msgpack_x); + CHECK(json::to_msgpack(jy) == msgpack_y); + CHECK(json::to_msgpack(jz) == msgpack_z); + + CHECK(std::isnan(json::from_msgpack(msgpack_x).get())); + CHECK(json::from_msgpack(msgpack_y).get() == std::numeric_limits::infinity()); + CHECK(json::from_msgpack(msgpack_z).get() == -std::numeric_limits::infinity()); + + // Make sure the other MessagePakc encodings for NaN, infinity, and + // -infinity are still supported. + const std::vector msgpack_x_2 = {{0xCB, 0x7F, 0xF8, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}}; + const std::vector msgpack_y_2 = {{0xCB, 0x7F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}}; + const std::vector msgpack_z_2 = {{0xCB, 0xFF, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}}; + CHECK(std::isnan(json::from_msgpack(msgpack_x_2).get())); + CHECK(json::from_msgpack(msgpack_y_2).get() == std::numeric_limits::infinity()); + CHECK(json::from_msgpack(msgpack_z_2).get() == -std::numeric_limits::infinity()); + + ///////////////////////////////////////////////////////////////////////// + // CBOR + ///////////////////////////////////////////////////////////////////////// + + // expected CBOR values + const std::vector cbor_x = {{0xF9, 0x7E, 0x00}}; + const std::vector cbor_y = {{0xF9, 0x7C, 0x00}}; + const std::vector cbor_z = {{0xF9, 0xfC, 0x00}}; + + CHECK(json::to_cbor(jx) == cbor_x); + CHECK(json::to_cbor(jy) == cbor_y); + CHECK(json::to_cbor(jz) == cbor_z); + + CHECK(std::isnan(json::from_cbor(cbor_x).get())); + CHECK(json::from_cbor(cbor_y).get() == std::numeric_limits::infinity()); + CHECK(json::from_cbor(cbor_z).get() == -std::numeric_limits::infinity()); + + // Make sure the other CBOR encodings for NaN, infinity, and -infinity are + // still supported. + const std::vector cbor_x_2 = {{0xFA, 0x7F, 0xC0, 0x00, 0x00}}; + const std::vector cbor_y_2 = {{0xFA, 0x7F, 0x80, 0x00, 0x00}}; + const std::vector cbor_z_2 = {{0xFA, 0xFF, 0x80, 0x00, 0x00}}; + const std::vector cbor_x_3 = {{0xFB, 0x7F, 0xF8, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}}; + const std::vector cbor_y_3 = {{0xFB, 0x7F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}}; + const std::vector cbor_z_3 = {{0xFB, 0xFF, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}}; + CHECK(std::isnan(json::from_cbor(cbor_x_2).get())); + CHECK(json::from_cbor(cbor_y_2).get() == std::numeric_limits::infinity()); + CHECK(json::from_cbor(cbor_z_2).get() == -std::numeric_limits::infinity()); + CHECK(std::isnan(json::from_cbor(cbor_x_3).get())); + CHECK(json::from_cbor(cbor_y_3).get() == std::numeric_limits::infinity()); + CHECK(json::from_cbor(cbor_z_3).get() == -std::numeric_limits::infinity()); +} + +TEST_CASE("regression test #5074 - portable workaround for single-element brace init") +{ + json const j_obj = {{"key", "value"}}; + + json const j = json::array({j_obj}); + CHECK(j.is_array()); + CHECK(j.size() == 1); + CHECK(j[0] == j_obj); +} + +struct Example_5122 +{ + float b = 2; + nlohmann::ordered_map c{}; // NOLINT(readability-redundant-member-init): needed for GCC -Weffc++ + int a = 1; + NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(Example_5122, b, c, a) +}; + +TEST_CASE("regression test #5122 - from_json into types holding nlohmann::ordered_map") +{ + Example_5122 src; + src.c.emplace("first", "1"); + src.c.emplace("second", "2"); + + ordered_json const j = src; + Example_5122 const dst = j.get(); + + CHECK(dst.b == src.b); + CHECK(dst.a == src.a); + REQUIRE(dst.c.size() == src.c.size()); + auto src_it = src.c.begin(); + auto dst_it = dst.c.begin(); + for (; src_it != src.c.end(); ++src_it, ++dst_it) + { + CHECK(dst_it->first == src_it->first); + CHECK(dst_it->second == src_it->second); + } +} + +// -Wself-assign-overloaded was introduced in Clang 7. Gate the pragma on +// __has_warning so older Clang versions do not error with "unknown warning +// group". The __has_warning check has to stay inside the __clang__ branch +// because GCC does not provide it and would tokenize-error on the argument. +#if defined(__clang__) && defined(__has_warning) + #if __has_warning("-Wself-assign-overloaded") + DOCTEST_CLANG_SUPPRESS_WARNING_PUSH + DOCTEST_CLANG_SUPPRESS_WARNING("-Wself-assign-overloaded") + #endif +#endif + +TEST_CASE("regression test #5122 - nlohmann::ordered_map copy-assignment is self-assignment safe") +{ + nlohmann::ordered_map m; + m.emplace("first", "1"); + m.emplace("second", "2"); + + // Insertion order is preserved by ordered_map, so we can check it directly. + m = m; + + REQUIRE(m.size() == 2); + auto it = m.begin(); + CHECK(it->first == "first"); + CHECK(it->second == "1"); + ++it; + CHECK(it->first == "second"); + CHECK(it->second == "2"); +} + +#if defined(__clang__) && defined(__has_warning) + #if __has_warning("-Wself-assign-overloaded") + DOCTEST_CLANG_SUPPRESS_WARNING_POP + #endif +#endif + +TEST_CASE("regression test #5122 - nlohmann::ordered_map move-assignment transfers contents") +{ + nlohmann::ordered_map src; + src.emplace("first", "1"); + src.emplace("second", "2"); + + nlohmann::ordered_map dst; + dst.emplace("stale", "x"); + dst = std::move(src); + + REQUIRE(dst.size() == 2); + auto it = dst.begin(); + CHECK(it->first == "first"); + CHECK(it->second == "1"); + ++it; + CHECK(it->first == "second"); + CHECK(it->second == "2"); + + // Re-assigning into the moved-from object must leave it in a usable state. + src = nlohmann::ordered_map {}; + src.emplace("after-move", "3"); + REQUIRE(src.size() == 1); + CHECK(src.begin()->first == "after-move"); +} + +// Stand-in for a third-party library (e.g., Eigen as of 3.4, which added +// STL-compatible begin()/end() to its vector types), living in its own +// namespace with its own to_json overload for its vector type. +namespace issue_4320_eigen +{ +// "array-compatible" from the library's point of view (it has begin()/end()), +// but for which this (fake) third-party namespace provides its own to_json. +struct vector3 +{ + double v[3]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays,cppcoreguidelines-use-default-member-init,modernize-use-default-member-init) + vector3(double x, double y, double z) : v{x, y, z} {} // NOLINT(hicpp-member-init,cppcoreguidelines-pro-type-member-init) + double x() const + { + return v[0]; + } + double y() const + { + return v[1]; + } + double z() const + { + return v[2]; + } + double* begin() + { + return v; + } + double* end() + { + return v + 3; + } + const double* begin() const + { + return v; + } + const double* end() const + { + return v + 3; + } +}; + +inline void to_json(json& j, const vector3& v) // NOLINT(misc-use-internal-linkage) +{ + j = {{"x", v.x()}, {"y", v.y()}, {"z", v.z()}}; +} +} // namespace issue_4320_eigen + +// The user's own namespace, using the (fake) Eigen type as an implementation +// detail behind a payload type that has nothing to do with vectors/arrays. +namespace issue_4320 +{ +// Publicly derives from issue_4320_eigen::vector3 but does *not* define its +// own to_json - it is only ever used as a temporary to reach the base +// class's to_json via ADL. +struct vector3_wrapper : issue_4320_eigen::vector3 +{ + using issue_4320_eigen::vector3::vector3; +}; + +struct payload +{ + double x, y, z; +}; + +inline vector3_wrapper to_eigen(const payload& p) // NOLINT(misc-use-internal-linkage) +{ + return {p.x, p.y, p.z}; +} + +inline void to_json(json& j, const payload& p) // NOLINT(misc-use-internal-linkage) +{ + // Unqualified call, passing a *derived* vector3_wrapper: relies on ADL + // finding issue_4320_eigen::to_json(json&, const vector3&) through the + // vector3 base class, via a derived-to-base conversion. Must NOT resolve + // to the library's own generic array-compatible to_json (an exact-match + // template for vector3_wrapper, since it also has begin()/end()), which + // would serialize this as [x, y, z] instead of {"x":x, "y":y, "z":z}. + to_json(j, to_eigen(p)); +} +} // namespace issue_4320 + +TEST_CASE("issue #4320 - custom base class must not leak nlohmann::detail into ADL") +{ + // Before the fix, basic_json unconditionally derived from a type living in + // nlohmann::detail (json_default_base), which made nlohmann::detail an + // associated namespace of every basic_json for ADL purposes. That leaked + // the library's internal generic-array to_json overload into unqualified + // to_json() calls made from user code, silently bypassing user-defined + // to_json overloads reached via a derived-to-base conversion. + const issue_4320::payload p{1.0, 2.0, 3.0}; + + json j; + to_json(j, p); + CHECK(j == json({{"x", 1.0}, {"y", 2.0}, {"z", 3.0}})); +} + +TEST_CASE("issue #5338 - truncated CBOR tagged binary subtype is rejected") +{ + const std::vector> truncated_tags = + { + {0xD8}, + {0xD9, 0x00}, + {0xDA, 0x00, 0x00, 0x00}, + {0xDB, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00} + }; + + for (const auto& data : truncated_tags) + { + CAPTURE(data); + for (const auto tag_handler : + { + json::cbor_tag_handler_t::ignore, json::cbor_tag_handler_t::store + }) + { + CAPTURE(tag_handler); + const auto result = json::from_cbor(data, true, false, tag_handler); + CHECK(result.is_discarded()); + } + } +} + +TEST_CASE("issue #5402 - update(merge_objects=true) overwrites a primitive with an object") +{ + json t = {{"k", 1}}; + t.update(json{{"k", {{"x", 2}}}}, true); + CHECK(t == json({{"k", {{"x", 2}}}})); + + json mixed = {{"keep", {{"a", 1}}}, {"replace", 1}}; + mixed.update(json{{"keep", {{"b", 2}}}, {"replace", {{"x", 2}}}}, true); + CHECK(mixed == json({{"keep", {{"a", 1}, {"b", 2}}}, {"replace", {{"x", 2}}}})); +} + + +TEST_CASE("regression test #5476 - array type without reserve()") +{ + // the capacity reserved for definite-length arrays must not require the + // array type to have a reserve() member function + using deque_json = nlohmann::basic_json; + + SECTION("std::deque") + { + const auto j = deque_json::parse(R"({"a":[1,[2,3]],"b":[]})"); + CHECK(j.dump() == R"({"a":[1,[2,3]],"b":[]})"); + + // the binary formats pass a definite length to start_array() + CHECK(deque_json::from_cbor(deque_json::to_cbor(j)) == j); + CHECK(deque_json::from_msgpack(deque_json::to_msgpack(j)) == j); + + // parse() instantiates the callback parser as well, which reserves too + const auto with_callback = deque_json::parse(R"([1,2,3])", [](int /*depth*/, deque_json::parse_event_t /*event*/, deque_json& /*parsed*/) noexcept + { + return true; + }); + CHECK(with_callback == deque_json({1, 2, 3})); + } + + SECTION("std::vector still reserves") + { + json array = json::array(); + for (int i = 0; i < 100; ++i) + { + array.push_back(i); + } + + const auto j = json::from_cbor(json::to_cbor(array)); + CHECK(j == array); + CHECK(j.get_ref().capacity() >= 100); + } + + SECTION("the reservation stays capped") + { + // CBOR array announcing 2^32-1 elements, but truncated right after the + // header: the input must be rejected without reserving that capacity + const std::vector truncated = {0x9A, 0xFF, 0xFF, 0xFF, 0xFF}; + CHECK(json::from_cbor(truncated, true, false).is_discarded()); + } +} + +DOCTEST_CLANG_SUPPRESS_WARNING_POP diff --git a/tests/src/unit-serialization.cpp b/tests/src/unit-serialization.cpp index f55ed8470..00b305a75 100644 --- a/tests/src/unit-serialization.cpp +++ b/tests/src/unit-serialization.cpp @@ -15,6 +15,8 @@ using nlohmann::json; #include #include +#include "test_utils.hpp" + TEST_CASE("serialization") { SECTION("operator<<") @@ -84,8 +86,9 @@ TEST_CASE("serialization") { const json j = "ä\xA9ü"; - CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&); - CHECK_THROWS_WITH_AS(j.dump(1, ' ', false, json::error_handler_t::strict), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&); + // dump() is nodiscard; the exception is thrown by dump() itself before it would return + CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&); + CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump(1, ' ', false, json::error_handler_t::strict)), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&); CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"äü\""); CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"ä\xEF\xBF\xBDü\""); CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"\\u00e4\\ufffd\\u00fc\""); @@ -95,8 +98,9 @@ TEST_CASE("serialization") { const json j = "123\xC2"; - CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] incomplete UTF-8 string; last byte: 0xC2", json::type_error&); - CHECK_THROWS_AS(j.dump(1, ' ', false, json::error_handler_t::strict), json::type_error&); + // dump() is nodiscard; the exception is thrown by dump() itself before it would return + CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] incomplete UTF-8 string; last byte: 0xC2", json::type_error&); + CHECK_THROWS_AS(utils::ignore_return_value(j.dump(1, ' ', false, json::error_handler_t::strict)), json::type_error&); CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123\""); CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\""); CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd\""); @@ -106,8 +110,9 @@ TEST_CASE("serialization") { const json j = "123\xF1\xB0\x34\x35\x36"; - CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0x34", json::type_error&); - CHECK_THROWS_AS(j.dump(1, ' ', false, json::error_handler_t::strict), json::type_error&); + // dump() is nodiscard; the exception is thrown by dump() itself before it would return + CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0x34", json::type_error&); + CHECK_THROWS_AS(utils::ignore_return_value(j.dump(1, ' ', false, json::error_handler_t::strict)), json::type_error&); CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123456\""); CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\x34\x35\x36\""); CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd456\""); @@ -382,3 +387,245 @@ TEST_CASE("dump for basic_json with long double number_float_t") check_same(100.0L, 100.0); } } + +TEST_CASE("serialization of strings (bulk fast path)") +{ + // These cases exercise the SWAR bulk-copy fast path in dump_escaped and the + // internal write buffer: long runs, escapes interrupting runs, 0x7F/DEL, + // multibyte UTF-8 under both ensure_ascii settings, and payloads larger than + // the write buffer. + + SECTION("long unescaped ASCII exceeds the write buffer") + { + const std::string big(3000, 'a'); + const json j = big; + CHECK(j.dump() == '"' + big + '"'); + CHECK(j.dump(-1, ' ', true) == '"' + big + '"'); + // round-trips + CHECK(json::parse(j.dump()) == j); + } + + SECTION("runs interrupted by escapes") + { + const json j = std::string(500, 'x') + "\n\"\\" + std::string(500, 'y'); + const std::string out = j.dump(); + CHECK(out == '"' + std::string(500, 'x') + "\\n\\\"\\\\" + std::string(500, 'y') + '"'); + CHECK(json::parse(out) == j); + } + + SECTION("DEL (0x7F) depends on ensure_ascii") + { + const json j = std::string("a\x7f" "b"); + CHECK(j.dump(-1, ' ', false) == "\"a\x7f" "b\""); // copied verbatim + CHECK(j.dump(-1, ' ', true) == "\"a\\u007fb\""); // escaped + } + + SECTION("multibyte UTF-8 under both ensure_ascii settings") + { + const json j = std::string("A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z"); // A é 你 😀 Z + // not escaping non-ASCII: bytes are copied through the bulk validator + CHECK(j.dump(-1, ' ', false) == "\"A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z\""); + // ensure_ascii: escaped (with a surrogate pair for the emoji) + CHECK(j.dump(-1, ' ', true) == "\"A\\u00e9\\u4f60\\ud83d\\ude00Z\""); + CHECK(json::parse(j.dump(-1, ' ', true)) == j); + } + + SECTION("many small structural writes exceed the write buffer") + { + json arr = json::array(); + for (int i = 0; i < 2000; ++i) + { + arr.push_back(i); + } + const std::string out = arr.dump(); + CHECK(out.front() == '['); + CHECK(out.back() == ']'); + CHECK(json::parse(out) == arr); + + json obj = json::object(); + for (int i = 0; i < 500; ++i) + { + obj["key" + std::to_string(i)] = i; + } + CHECK(json::parse(obj.dump()) == obj); + CHECK(json::parse(obj.dump(2)) == obj); + + // an array of many empty strings emits a long run of single-character + // writes ('"', '"', ',') at shallow nesting depth, so the write buffer + // fills and flushes mid-run without the deep recursion that would + // overflow the stack on some debug builds + json many_empty = json::array(); + for (int i = 0; i < 500; ++i) + { + many_empty.push_back(""); + } + const std::string out2 = many_empty.dump(); + CHECK(out2.size() > 1024); // spans multiple write-buffer flushes + CHECK(out2.front() == '['); + CHECK(out2.back() == ']'); + CHECK(json::parse(out2) == many_empty); + } + + SECTION("invalid UTF-8 handling is unaffected by the fast path") + { + const json j = std::string("valid\xff" "more"); + CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0xFF", json::type_error&); + CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"valid\xef\xbf\xbd" "more\""); + CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"valid\\ufffdmore\""); + CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"validmore\""); + } +} + +TEST_CASE("indentation is written straight into the write buffer") +{ + // put_indent() memsets the indentation into the write buffer instead of + // copying it out of a pre-grown indentation string. These cases cover an + // indentation wider than the buffer, a non-space indentation character, and + // nesting deep enough that the accumulated indentation spans several + // buffer-fulls - the situations the old grow-a-string approach got wrong. + + SECTION("indent_step wider than the write buffer") + { + const json j = {{"a", 1}}; + // 2000 > the 1024-byte write buffer, and > the 512 the indentation + // string used to start at + CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"a\": 1\n}"); + // several whole buffer-fulls, so the buffer is refilled once and then + // flushed repeatedly + CHECK(j.dump(5000) == "{\n" + std::string(5000, ' ') + "\"a\": 1\n}"); + CHECK(j.dump(5000, '\t') == "{\n" + std::string(5000, '\t') + "\"a\": 1\n}"); + // an exact multiple of the buffer size + CHECK(j.dump(4096) == "{\n" + std::string(4096, ' ') + "\"a\": 1\n}"); + } + + SECTION("a non-space indentation character is used throughout") + { + const json j = {{"a", 1}}; + // 600 is past the point where the indentation used to be grown, which + // is where a hard-coded space would have shown up + CHECK(j.dump(600, '\t') == "{\n" + std::string(600, '\t') + "\"a\": 1\n}"); + CHECK(j.dump(3, '.') == "{\n...\"a\": 1\n}"); + } + + SECTION("accumulated indentation spans several buffer-fulls") + { + // five levels deep at 400 per level: the innermost value is indented by + // 2000 characters, reached in steps that each straddle the buffer end + json j = json::array({1}); + for (int i = 0; i < 4; ++i) + { + j = json::array({j}); + } + + const std::string out = j.dump(400); + CHECK(out.find(std::string("\n") + std::string(2000, ' ') + "1\n") != std::string::npos); + CHECK(json::parse(out) == j); + } + + SECTION("binary values are indented the same way") + { + // a binary value is serialized as an object with "bytes" and + // "subtype" keys; the byte array itself is always written compactly + // (see dump_byte()), so only the surrounding object's indentation + // goes through put_indent() + const json j = json::binary({1, 2, 3}, 128); + CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"bytes\": [1, 2, 3],\n" + + std::string(2000, ' ') + "\"subtype\": 128\n}"); + CHECK(j.dump(2000, '\t') == "{\n" + std::string(2000, '\t') + "\"bytes\": [1, 2, 3],\n" + + std::string(2000, '\t') + "\"subtype\": 128\n}"); + } + + SECTION("indentation is unchanged for ordinary widths") + { + const json j = {{"a", {1, 2}}, {"b", nullptr}}; + CHECK(j.dump(2) == "{\n \"a\": [\n 1,\n 2\n ],\n \"b\": null\n}"); + CHECK(j.dump(0) == "{\n\"a\": [\n1,\n2\n],\n\"b\": null\n}"); + } +} + +TEST_CASE("serialization of deeply nested values") +{ + // dump() descends into a bounded number of levels and writes out whatever + // is nested deeper than that without the call stack; see + // https://github.com/nlohmann/json/issues/5387 + + SECTION("nested deeper than the call stack could follow") + { + // parsing is iterative, so building these costs little + const std::size_t depth = 100000; + + const std::string array_text = std::string(depth, '[') + '0' + std::string(depth, ']'); + CHECK(json::parse(array_text).dump() == array_text); + + std::string object_text; + object_text.reserve((6 * depth) + 1); + for (std::size_t i = 0; i < depth; ++i) + { + object_text += "{\"a\":"; + } + object_text += '1'; + object_text.append(depth, '}'); + CHECK(json::parse(object_text).dump() == object_text); + } + + SECTION("depths around the bound of the descent") + { + // Cover every depth around the bound, so that the two ways of writing a + // value are known to meet cleanly - wherever the bound is set. + for (std::size_t d = 1; d <= 300; ++d) + { + CAPTURE(d); + + const std::string array_text = std::string(d, '[') + '7' + std::string(d, ']'); + CHECK(json::parse(array_text).dump() == array_text); + + std::string object_text; + for (std::size_t i = 0; i < d; ++i) + { + object_text += "{\"k\":"; + } + object_text += '7'; + object_text.append(d, '}'); + CHECK(json::parse(object_text).dump() == object_text); + } + } + + SECTION("pretty-printing across the bound") + { + for (std::size_t d = 120; d <= 140; ++d) + { + CAPTURE(d); + + const json j = json::parse(std::string(d, '[') + '7' + std::string(d, ']')); + + std::string expected; + for (std::size_t i = 0; i < d; ++i) + { + expected += std::string(2 * i, ' ') + "[\n"; + } + expected += std::string(2 * d, ' ') + '7'; + for (std::size_t i = d; i > 0; --i) + { + expected += '\n' + std::string(2 * (i - 1), ' ') + ']'; + } + + CHECK(j.dump(2) == expected); + } + } + + SECTION("an empty container below the bound") + { + // an empty container is written out in full and never descended into, + // so it must not gain a newline when it is reached iteratively + for (std::size_t d = 125; d <= 135; ++d) + { + CAPTURE(d); + + const std::string compact = std::string(d, '[') + "[]" + std::string(d, ']'); + CHECK(json::parse(compact).dump() == compact); + + const std::string with_object = std::string(d, '[') + "{}" + std::string(d, ']'); + CHECK(json::parse(with_object).dump() == with_object); + } + } +} diff --git a/tests/src/unit-std-format.cpp b/tests/src/unit-std-format.cpp index f7a364884..58cbbf5cc 100644 --- a/tests/src/unit-std-format.cpp +++ b/tests/src/unit-std-format.cpp @@ -17,6 +17,7 @@ #include using json = nlohmann::json; +using ordered_json = nlohmann::ordered_json; // JSON_HAS_CPP_20 (do not remove; see note at top of file) #if JSON_HAS_STD_FORMAT @@ -52,6 +53,23 @@ TEST_CASE("std::formatter") CHECK(std::format("{:2}", j) == j.dump(2)); CHECK(std::format("{:#2}", j) == j.dump(2)); CHECK(std::format("{:8}", j) == j.dump(8)); + // multi-digit widths must accumulate every digit, not just the first + CHECK(std::format("{:12}", j) == j.dump(12)); + CHECK(std::format("{:#12}", j) == j.dump(12)); + CHECK(std::format("{:10}", j) == j.dump(10)); + } + + SECTION("bare alignment with no fill character defaults to a space indent character") + { + const json j = {{"foo", 1}, {"bar", {1, 2, 3}}}; + // without a preceding fill character, the alignment character itself must not + // be mistaken for the indent character -- the default space is kept + CHECK(std::format("{:<}", j) == j.dump()); + CHECK(std::format("{:>}", j) == j.dump()); + CHECK(std::format("{:^}", j) == j.dump()); + CHECK(std::format("{:<3}", j) == j.dump(3, ' ')); + CHECK(std::format("{:>3}", j) == j.dump(3, ' ')); + CHECK(std::format("{:^3}", j) == j.dump(3, ' ')); } SECTION("fill-and-align sets the indent character, like dump(indent, indent_char)") @@ -93,4 +111,16 @@ TEST_CASE("std::formatter") } } +TEST_CASE("std::formatter") +{ + // spot-check a non-default basic_json instantiation, since the formatter + // is written against the generic NLOHMANN_BASIC_JSON_TPL_DECLARATION + // template and must actually instantiate (and behave correctly) for + // template arguments other than nlohmann::json + const ordered_json j = {{"foo", 1}, {"bar", {1, 2, 3}}}; + CHECK(std::format("{}", j) == j.dump()); + CHECK(std::format("{:#}", j) == j.dump(4)); + CHECK(std::format("{:2}", j) == j.dump(2)); +} + #endif diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index e1e327563..c8458c44d 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -15,6 +15,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "round_trip_corpus.hpp" #include "test_utils.hpp" namespace @@ -2149,6 +2150,280 @@ TEST_CASE("UBJSON") } } +TEST_CASE("UBJSON nesting does not consume the call stack") +{ + // Containers used to be read by calling back into the value reader once + // per element, so the native call stack grew with the nesting depth of the + // input. '[' alone opens a container, so a payload of repeated '[' crashed + // the process (#5104), as did the optimized forms, which reach the same + // path through a type or size annotation. The containers are kept on a + // heap stack now. + // + // Deeply nested values must not be compared, copied or dumped here: those + // operations are still recursive and would reintroduce the crash. + json _; + + SECTION("containers that end at a marker") + { + const std::vector input(500000, '['); + CHECK_THROWS_WITH_AS(_ = json::from_ubjson(input), "[json.exception.parse_error.110] parse error at byte 500001: syntax error while parsing UBJSON value: unexpected end of input", json::parse_error&); + CHECK(json::from_ubjson(input, true, false).is_discarded()); + } + + SECTION("containers with a size") + { + std::vector input; + for (std::size_t i = 0; i < 100000; ++i) + { + input.push_back('['); + input.push_back('#'); + input.push_back('i'); + input.push_back(1); + } + CHECK_THROWS_AS(_ = json::from_ubjson(input), json::parse_error&); + CHECK(json::from_ubjson(input, true, false).is_discarded()); + } + + SECTION("containers with a type and a size") + { + // '[' is a permitted optimized type in UBJSON, so each element of such + // a container is itself a container, read without a marker of its own + std::vector input; + for (std::size_t i = 0; i < 100000; ++i) + { + const std::vector level = {'[', '$', '[', '#', 'i', 1}; + input.insert(input.end(), level.begin(), level.end()); + } + CHECK_THROWS_AS(_ = json::from_ubjson(input), json::parse_error&); + CHECK(json::from_ubjson(input, true, false).is_discarded()); + } + + SECTION("a well-formed deep value is read through the SAX interface") + { + std::vector input(100000, '['); + input.insert(input.end(), 100000, ']'); + + SaxCountdown accept_all(1000000); + CHECK(json::sax_parse(input, &accept_all, json::input_format_t::ubjson)); + } + + SECTION("a well-formed deep value is read into a value") + { + const std::size_t depth = 10000; + std::vector input(depth, '['); + input.insert(input.end(), depth, ']'); + + json j = json::from_ubjson(input); + + std::size_t measured = 0; + const json* p = &j; + while (p->is_array() && !p->empty()) + { + p = &p->front(); + ++measured; + } + // the innermost array is empty, so the descent stops one level short + CHECK(measured == depth - 1); + } + + SECTION("containers are still read the same way") + { + CHECK(json::from_ubjson(std::vector({'[', ']'})) == json::array()); + CHECK(json::from_ubjson(std::vector({'{', '}'})) == json::object()); + CHECK(json::from_ubjson(std::vector({'[', '#', 'i', 0})) == json::array()); + CHECK(json::from_ubjson(std::vector({'{', '#', 'i', 0})) == json::object()); + CHECK(json::from_ubjson(std::vector({'[', '$', 'i', '#', 'i', 2, 1, 2})) == json({1, 2})); + CHECK(json::from_ubjson(std::vector({'[', '#', 'i', 2, 'i', 1, 'i', 2})) == json({1, 2})); + CHECK(json::from_ubjson(std::vector({'{', '$', 'i', '#', 'i', 1, 'i', 1, 'a', 1})) == json({{"a", 1}})); + // a no-op is not a value, so a container of them holds none + CHECK(json::from_ubjson(std::vector({'[', '$', 'N', '#', 'i', 2})) == json::array()); + // sized and unsized forms nested inside one another + CHECK(json::from_ubjson(std::vector({'[', '[', '#', 'i', 2, 'i', 1, 'i', 2, ']'})) == json({{1, 2}})); + CHECK(json::from_ubjson(std::vector({'[', '#', 'i', 1, '[', 'i', 1, ']'})) == json({{1}})); + // an optimized container of containers + CHECK(json::from_ubjson(std::vector({'[', '$', '[', '#', 'i', 2, 'i', 1, ']', 'i', 2, ']'})) == json({{1}, {2}})); + } + + SECTION("BJData containers are still read the same way") + { + // the ND-array wrapper and the binary shortcut are complete values, + // not containers the reader descends into + CHECK(json::from_bjdata(std::vector({'[', '$', 'U', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6})) == + json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}})); + CHECK(json::from_bjdata(std::vector({'[', '$', 'i', '#', 'i', 2, 1, 2})) == json({1, 2})); + CHECK(json::from_bjdata(std::vector({'[', '[', 'i', 1, ']', ']'})) == json({{1}})); + } +} + +TEST_CASE("UBJSON optimized arrays of a valueless type are bounded") +{ + // An element of type 'Z', 'T' or 'F' is encoded by its marker alone, so an + // optimized array of one of those has no payload and the declared count is + // the only thing deciding how much is allocated. Ten bytes used to produce + // billions of values (#2793); every other type costs at least one byte per + // element and is bounded by the end of the input. + json _; + + SECTION("an excessive count is rejected") + { + // 'l' is a big-endian int32: 0x7FFFFFFF elements, about 34 GB of value; + // OSS-Fuzz reported this shape as a parse_ubjson_fuzzer timeout + // (testcase 6347769435193344, no issue filed) + for (const auto marker : + {'Z', 'T', 'F' + }) + { + const std::vector input = {'[', '$', static_cast(marker), '#', 'l', 0x7F, 0xFF, 0xFF, 0xFF}; + CHECK_THROWS_WITH_AS(_ = json::from_ubjson(input), "[json.exception.out_of_range.408] syntax error while parsing UBJSON size: excessive array size", json::out_of_range&); + CHECK(json::from_ubjson(input, true, false).is_discarded()); + } + } + + SECTION("ordinary counts are unaffected") + { + CHECK(json::from_ubjson(std::vector({'[', '$', 'Z', '#', 'i', 3})) == json({nullptr, nullptr, nullptr})); + CHECK(json::from_ubjson(std::vector({'[', '$', 'T', '#', 'i', 2})) == json({true, true})); + CHECK(json::from_ubjson(std::vector({'[', '$', 'F', '#', 'i', 2})) == json({false, false})); + // 'N' is a no-op rather than a value, and still yields an empty array + CHECK(json::from_ubjson(std::vector({'[', '$', 'N', '#', 'i', 2})) == json::array()); + } + + SECTION("a type with a payload is unaffected") + { + // A count past the limit is not rejected for 'U', which costs a byte + // per element and is bounded by the end of the input instead. The + // count is kept just past the limit rather than made huge, because a + // count that also exceeds the array's max_size() is reported as + // out_of_range before the input runs out, and max_size() depends on + // the width of std::size_t. + const std::vector input = {'[', '$', 'U', '#', 'l', 0x00, 0x10, 0x00, 0x01}; + CHECK_THROWS_WITH_AS(_ = json::from_ubjson(input), "[json.exception.parse_error.110] parse error at byte 10: syntax error while parsing UBJSON number: unexpected end of input", json::parse_error&); + CHECK(json::from_ubjson(input, true, false).is_discarded()); + } + + SECTION("the writer stays within what the reader accepts") + { + // below the limit the optimized form is used and is tiny; above it the + // writer falls back so that the result can still be read back + json const at_limit(1048576, nullptr); + const auto v_at_limit = json::to_ubjson(at_limit, true, true); + CHECK(v_at_limit.size() == 9); + CHECK(v_at_limit.at(1) == '$'); + CHECK(json::from_ubjson(v_at_limit) == at_limit); + + json const above_limit(1048577, nullptr); + const auto v_above_limit = json::to_ubjson(above_limit, true, true); + CHECK(v_above_limit.at(1) != '$'); + CHECK(json::from_ubjson(v_above_limit) == above_limit); + } +} + +TEST_CASE("issue #5405 - array reserve for definite-length UBJSON arrays") +{ +#if !defined(JSON_NOEXCEPTION) + // this SECTION relies on catching a thrown exception to distinguish + // which of two acceptable, bounded rejections a hostile header took; + // under JSON_NOEXCEPTION, JSON_THROW never produces a catchable C++ + // exception (it aborts instead), so this cannot be tested that way here + SECTION("a huge claimed length with no element data must not over-allocate") + { + // optimized form [$type#count: type 'i' (int8), count as a four-byte + // 'l' (int32) of 0x7FFFFFFF (2147483647), but no element data at all. + // max_size() for a std::vector is far larger than this count, so it + // does not reject the header outright; the (capped) reservation must + // not attempt to allocate space for billions of elements before the + // missing data is detected. + json _; + const std::vector input = {'[', '$', 'i', '#', 'l', 0x7F, 0xFF, 0xFF, 0xFF}; + // On a platform where std::vector::max_size() is smaller than + // the claimed count (e.g. 32-bit, where max_size() is bounded by a + // 32-bit SIZE_MAX divided by sizeof(json)), the SAX consumer's own + // check rejects the header outright (out_of_range.408, with the + // claimed count in the message) instead of accepting it and only + // finding it short of data once the (capped) reservation looks for + // element bytes that were never provided (parse_error.110). Either + // is an acceptable, bounded rejection of the hostile header -- the + // property under test is that no path attempts to allocate space + // for billions of elements. + bool threw = false; + try + { + _ = json::from_ubjson(input); + } + catch (const json::parse_error& e) + { + threw = true; + CHECK(e.id == 110); + CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 10: syntax error while parsing UBJSON number: unexpected end of input"); + } + catch (const json::out_of_range& e) + { + threw = true; + CHECK(e.id == 408); + CHECK(std::string(e.what()).find("excessive array size") != std::string::npos); + } + CHECK(threw); + + // json_sax_dom_parser::start_array()'s max_size() check (unlike the + // scanner's own parse_error path) throws unconditionally via + // JSON_THROW rather than going through sax->parse_error(), so it is + // not gated by allow_exceptions=false on a platform where this + // header hits that check (e.g. 32-bit, see above) -- allow either + // a discarded result or the same out_of_range it throws with + // exceptions enabled. + try + { + CHECK(json::from_ubjson(input, true, false).is_discarded()); + } + catch (const json::out_of_range& e) + { + CHECK(e.id == 408); + } + } +#endif + + SECTION("arrays of various sizes decode to the same value as before the reserve optimization") + { + for (const auto size : + { + std::size_t{0}, std::size_t{1}, std::size_t{5}, // small + std::size_t{16384}, // exactly at the reserve cap + std::size_t{20000} // above the reserve cap + }) + { + CAPTURE(size) + json j = json::array(); + for (std::size_t i = 0; i < size; ++i) + { + j.push_back(static_cast(i % 1000)); + } + + // exercise both the plain and the optimized [$type#count encoding + const auto packed_plain = json::to_ubjson(j); + CHECK(json::from_ubjson(packed_plain) == j); + + const auto packed_optimized = json::to_ubjson(j, true, true); + CHECK(json::from_ubjson(packed_optimized) == j); + } + } + + SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization") + { + // the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser; + // a custom SAX consumer that does not touch a DOM array sees identical events + json j = json::array(); + for (int i = 0; i < 100; ++i) + { + j.push_back(i); + } + const auto packed = json::to_ubjson(j, true, true); + + SaxCountdown scp(1000000); // large enough to never trigger an abort + CHECK(json::sax_parse(packed, &scp, json::input_format_t::ubjson)); + } +} + + TEST_CASE("Universal Binary JSON Specification Examples 1") { SECTION("Null Value") @@ -2503,6 +2778,93 @@ TEST_CASE("all UBJSON first bytes") } #endif +TEST_CASE("UBJSON use_type requires use_size") +{ + SECTION("non-empty array throws other_error.502") + { + const json j = {1, 2, 3}; + CHECK_THROWS_WITH_AS(json::to_ubjson(j, false, true), + "[json.exception.other_error.502] use_type requires use_size = true", + json::other_error&); + } + + SECTION("non-empty object throws other_error.502") + { + const json j = {{"a", 1}, {"b", 2}}; + CHECK_THROWS_WITH_AS(json::to_ubjson(j, false, true), + "[json.exception.other_error.502] use_type requires use_size = true", + json::other_error&); + } + + SECTION("scalars do not throw with use_type=true, use_count=false") + { + CHECK_NOTHROW(json::to_ubjson(42, false, true)); + CHECK_NOTHROW(json::to_ubjson(3.14, false, true)); + CHECK_NOTHROW(json::to_ubjson("hello", false, true)); + CHECK_NOTHROW(json::to_ubjson(true, false, true)); + CHECK_NOTHROW(json::to_ubjson(nullptr, false, true)); + } + + SECTION("empty containers do not throw with use_type=true, use_count=false") + { + CHECK_NOTHROW(json::to_ubjson(json::array(), false, true)); + CHECK_NOTHROW(json::to_ubjson(json::object(), false, true)); + } + + SECTION("valid combinations on non-empty containers") + { + const json j = {1, 2, 3}; + CHECK_NOTHROW(json::to_ubjson(j, false, false)); + CHECK_NOTHROW(json::to_ubjson(j, true, false)); + CHECK_NOTHROW(json::to_ubjson(j, true, true)); + } +} + +TEST_CASE("UBJSON round-trip invariants") +{ + // This checks what the parse_ubjson_fuzzer driver checks (see + // tests/src/fuzzer-parse_ubjson.cpp), so that a regression shows up in CI + // rather than as an OSS-Fuzz report: every value from_ubjson() returns + // (j1) can be serialized with any combination of options, the result can + // be parsed back (j2), and serializing j2 again with the same options + // reproduces the exact bytes. Beyond the driver, this also checks that j2 + // equals j1. Values are compared with dump() rather than operator==, + // because a NaN never compares equal to itself. + struct options + { + bool use_size; + bool use_type; + }; + const std::vector all_options = + { + {false, false}, + {true, false}, + {true, true}, + }; + + for (const auto& j0 : utils::round_trip_corpus::values()) + { + // turn the corpus value into a value as from_ubjson() returns it; this + // has no binary values, as UBJSON writes them as arrays of integers + for (const auto& initial : all_options) + { + const json j1 = json::from_ubjson(json::to_ubjson(j0, initial.use_size, initial.use_type)); + + for (const auto& o : all_options) + { + INFO("j1 = " << j1.dump() << ", use_size = " << o.use_size << ", use_type = " << o.use_type); + + const std::vector vec = json::to_ubjson(j1, o.use_size, o.use_type); + json j2; + // anything the library writes must be parsable by the library + REQUIRE_NOTHROW(j2 = json::from_ubjson(vec)); + CHECK(j2.dump() == j1.dump()); + CHECK(json::to_ubjson(j2, o.use_size, o.use_type) == vec); + } + } + } +} + TEST_CASE("UBJSON roundtrips" * doctest::skip()) { SECTION("input from self-generated UBJSON files") diff --git a/tests/src/unit-unicode1.cpp b/tests/src/unit-unicode1.cpp index 174ce1395..2d744003a 100644 --- a/tests/src/unit-unicode1.cpp +++ b/tests/src/unit-unicode1.cpp @@ -17,6 +17,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "test_utils.hpp" TEST_CASE("Unicode (1/5)" * doctest::skip()) { @@ -240,7 +241,8 @@ void roundtrip(bool success_expected, const std::string& s) if (success_expected) { // serialization succeeds - CHECK_NOTHROW(j.dump()); + // dump() is nodiscard; this only checks that dumping does not throw + CHECK_NOTHROW(utils::ignore_return_value(j.dump())); // exclude parse test for U+0000 if (s[0] != '\0') @@ -259,7 +261,8 @@ void roundtrip(bool success_expected, const std::string& s) else { // serialization fails - CHECK_THROWS_AS(j.dump(), json::type_error&); + // dump() is nodiscard; the exception is thrown by dump() itself before it would return + CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&); // parsing JSON text fails CHECK_THROWS_AS(_ = json::parse(ps), json::parse_error&); diff --git a/tests/src/unit-unicode2.cpp b/tests/src/unit-unicode2.cpp index fb68815ba..a9649b4de 100644 --- a/tests/src/unit-unicode2.cpp +++ b/tests/src/unit-unicode2.cpp @@ -19,6 +19,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "test_utils.hpp" // this test suite uses static variables with non-trivial destructors DOCTEST_CLANG_SUPPRESS_WARNING_PUSH @@ -97,7 +98,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 else { // strict mode must throw if success is not expected - CHECK_THROWS_AS(j.dump(), json::type_error&); + // dump() is nodiscard; the exception is thrown by dump() itself before it would return + CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&); // ignore and replace must create different dumps CHECK(s_ignored != s_replaced); diff --git a/tests/src/unit-unicode3.cpp b/tests/src/unit-unicode3.cpp index 739a3dad3..12c12eea4 100644 --- a/tests/src/unit-unicode3.cpp +++ b/tests/src/unit-unicode3.cpp @@ -19,6 +19,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "test_utils.hpp" // this test suite uses static variables with non-trivial destructors DOCTEST_CLANG_SUPPRESS_WARNING_PUSH @@ -97,7 +98,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 else { // strict mode must throw if success is not expected - CHECK_THROWS_AS(j.dump(), json::type_error&); + // dump() is nodiscard; the exception is thrown by dump() itself before it would return + CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&); // ignore and replace must create different dumps CHECK(s_ignored != s_replaced); @@ -304,8 +306,8 @@ TEST_CASE("Unicode (3/5)" * doctest::skip()) { for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { - // skip fourth second byte - if (0x80 <= byte3 && byte3 <= 0xBF) + // skip correct fourth byte + if (0x80 <= byte4 && byte4 <= 0xBF) { continue; } diff --git a/tests/src/unit-unicode4.cpp b/tests/src/unit-unicode4.cpp index f7047201c..43cf7095e 100644 --- a/tests/src/unit-unicode4.cpp +++ b/tests/src/unit-unicode4.cpp @@ -19,6 +19,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "test_utils.hpp" // this test suite uses static variables with non-trivial destructors DOCTEST_CLANG_SUPPRESS_WARNING_PUSH @@ -97,7 +98,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 else { // strict mode must throw if success is not expected - CHECK_THROWS_AS(j.dump(), json::type_error&); + // dump() is nodiscard; the exception is thrown by dump() itself before it would return + CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&); // ignore and replace must create different dumps CHECK(s_ignored != s_replaced); @@ -305,7 +307,7 @@ TEST_CASE("Unicode (4/5)" * doctest::skip()) for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { // skip correct fourth byte - if (0x80 <= byte3 && byte3 <= 0xBF) + if (0x80 <= byte4 && byte4 <= 0xBF) { continue; } diff --git a/tests/src/unit-unicode5.cpp b/tests/src/unit-unicode5.cpp index e4dcc2131..bc0312820 100644 --- a/tests/src/unit-unicode5.cpp +++ b/tests/src/unit-unicode5.cpp @@ -19,6 +19,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "test_utils.hpp" // this test suite uses static variables with non-trivial destructors DOCTEST_CLANG_SUPPRESS_WARNING_PUSH @@ -97,7 +98,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 else { // strict mode must throw if success is not expected - CHECK_THROWS_AS(j.dump(), json::type_error&); + // dump() is nodiscard; the exception is thrown by dump() itself before it would return + CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&); // ignore and replace must create different dumps CHECK(s_ignored != s_replaced); @@ -305,7 +307,7 @@ TEST_CASE("Unicode (5/5)" * doctest::skip()) for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { // skip correct fourth byte - if (0x80 <= byte3 && byte3 <= 0xBF) + if (0x80 <= byte4 && byte4 <= 0xBF) { continue; } diff --git a/tests/src/unit-user_defined_input.cpp b/tests/src/unit-user_defined_input.cpp index 823e82862..f07a8a608 100644 --- a/tests/src/unit-user_defined_input.cpp +++ b/tests/src/unit-user_defined_input.cpp @@ -18,7 +18,12 @@ #include using nlohmann::json; +#include // array +#include // size_t +#include // uint8_t #include +#include // string +#include // vector #if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) #include @@ -212,6 +217,66 @@ TEST_CASE("Parse with heterogeneous iterator and sentinel types") CHECK(j2.at(0) == 1); } +// A type whose data() hands out raw bytes but whose size() counts something +// else - here fixed-size records. Reading [data(), data() + size()) as bytes +// would silently truncate the input, so data() and size() alone must not be +// taken as evidence of contiguous byte storage. +struct record_buffer +{ + using value_type = std::array; + + std::string bytes; + + const char* data() const noexcept + { + return bytes.data(); + } + std::size_t size() const noexcept + { + return bytes.size() / sizeof(value_type); + } + const char* begin() const noexcept + { + return bytes.data(); + } + const char* end() const noexcept + { + return bytes.data() + bytes.size(); + } +}; + +TEST_CASE("Contiguous byte containers take the pointer adapter") +{ + // Containers with contiguous single-byte storage are routed through the + // pointer-based adapter so the bulk fast paths apply in every standard, not + // only in C++20 where the library iterators model std::contiguous_iterator. + CHECK(nlohmann::detail::is_contiguous_byte_container::value); + CHECK(nlohmann::detail::is_contiguous_byte_container>::value); + CHECK(nlohmann::detail::is_contiguous_byte_container>::value); + CHECK(nlohmann::detail::is_contiguous_byte_container>::value); + + // input_adapter() takes its container by forwarding reference, so the trait + // is also asked about reference types + CHECK(nlohmann::detail::is_contiguous_byte_container::value); + CHECK(nlohmann::detail::is_contiguous_byte_container::value); + + // everything else keeps the iterator-based adapter + CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container>::value); + CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container>::value); + CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container::value); + + // including a type that has data() and size() but whose size() does not + // count the units data() points at: its value_type says so + CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container::value); + + // and such a container still parses through its iterators, in full - taking + // it for a byte container would stop after data() + size() bytes + const record_buffer buffer{"[1,2,3,4,5]"}; + CHECK(buffer.data() == buffer.bytes.data()); + CHECK(buffer.size() * sizeof(record_buffer::value_type) < buffer.bytes.size()); + CHECK(json::parse(buffer) == json({1, 2, 3, 4, 5})); +} + #if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) // JSON_HAS_CPP_20 (do not remove; see note at top of file) TEST_CASE("Parse with std::counted_iterator and std::default_sentinel_t") @@ -228,6 +293,180 @@ TEST_CASE("Parse with std::counted_iterator and std::default_sentinel_t") const std::counted_iterator first2(json_str.begin(), len); CHECK(json::accept(first2, std::default_sentinel)); } + +TEST_CASE("std::counted_iterator reaches the contiguous fast paths") +{ + // A sized sentinel makes the remaining element count computable in O(1), so + // std::counted_iterator over a contiguous iterator must reach the same bulk + // string/number scanners as a plain pointer - not just the byte-at-a-time + // fallback (see #5268 for the equivalent memcpy fast path). +#if JSON_HAS_RANGES + // JSON_HAS_RANGES is 0 on standard libraries with an incomplete + // (libstdc++ < 11, libc++ < 16), where the adapter deliberately falls back + // to the byte-at-a-time scanner; everything below still has to work there. + using adapter_type = nlohmann::detail::iterator_input_adapter, std::default_sentinel_t>; + CHECK(adapter_type::supports_bulk_scan); + CHECK(adapter_type::supports_seek); +#endif + + // exercise every fast path: long ASCII run, multibyte UTF-8, escapes, and + // integer/floating-point numbers + const std::string json_str = + R"({"ascii":"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",)" + "\"utf8\":\"\xe4\xb8\xad\xe6\x96\x87\xf0\x9f\x98\x80\xc3\xa9\"," + R"("escaped":"aéb\n\\","ints":[0,-1,18446744073709551615,-9223372036854775808],)" + R"("floats":[1.5,-2.25e3,0.30000000000000004]})"; + const auto len = static_cast>(json_str.size()); + + const std::counted_iterator first(json_str.data(), len); + const json j = json::parse(first, std::default_sentinel); + + // parsing through the pointer adapter must give exactly the same result + CHECK(j == json::parse(json_str)); + +#if !defined(JSON_NOEXCEPTION) + // Diagnostics that quote the offending token are reconstructed from the + // already-consumed input (supports_seek), a path a sized sentinel only + // reaches now; check a few that include the "last read" text. Parsing + // invalid input aborts when exceptions are off, hence the guard. + // Raw strings and explicit bytes: an escaped literal and two literals + // written next to each other both read as mistakes to static analysis. + const auto byte = [](int value) + { + return std::string(1, static_cast(value)); + }; + const std::vector diagnostic_docs = + { + "1\nx", + "truX", + "[tru]", + R"("abc)", + R"(["\ud834"])", + R"(["a)" + byte(0x01) + R"(b"])", + R"([")" + byte(0xC3) + byte(0x28) + R"("])", + "[1e]", + R"(["aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaX)" + }; + + for (const auto& text : diagnostic_docs) + { + CAPTURE(text); + const std::counted_iterator it(text.data(), static_cast>(text.size())); + std::string counted_message; + std::string string_message; + try + { + const json counted_result = json::parse(it, std::default_sentinel); + static_cast(counted_result); + } + catch (const json::parse_error& e) + { + counted_message = e.what(); + } + try + { + const json string_result = json::parse(text); + static_cast(string_result); + } + catch (const json::parse_error& e) + { + string_message = e.what(); + } + CHECK_FALSE(counted_message.empty()); + CHECK(counted_message == string_message); + } + + // and errors must still be reported identically + const std::string bad = "[01\n]"; + const std::counted_iterator bad_first(bad.data(), static_cast>(bad.size())); + std::string counted_what; + std::string string_what; + try + { + const json counted_result = json::parse(bad_first, std::default_sentinel); + static_cast(counted_result); + } + catch (const json::parse_error& e) + { + counted_what = e.what(); + } + try + { + const json string_result = json::parse(bad); + static_cast(string_result); + } + catch (const json::parse_error& e) + { + string_what = e.what(); + } + CHECK_FALSE(counted_what.empty()); + CHECK(counted_what == string_what); +#endif +} + +#if !defined(JSON_NOEXCEPTION) +// several cases below are truncated on purpose, and parsing invalid input +// aborts when exceptions are off +TEST_CASE("std::counted_iterator bulk scanning stops at the counted end") +{ + // The count, not the size of the underlying buffer, is the end of the + // input: the bulk scanners must never look at the bytes behind it, even + // though they are readable. Each case is compared against parsing the + // equivalent prefix as a std::string. + const auto via_counted = [](const std::string & buf, std::size_t n) -> std::string + { + const std::counted_iterator first(buf.data(), static_cast>(n)); + try + { + const json j = json::parse(first, std::default_sentinel); + return "OK|" + j.dump(); + } + catch (const json::parse_error& e) + { + return {e.what()}; + } + }; + const auto via_prefix = [](const std::string & buf, std::size_t n) -> std::string + { + try + { + const json j = json::parse(buf.substr(0, n)); + return "OK|" + j.dump(); + } + catch (const json::parse_error& e) + { + return {e.what()}; + } + }; + + struct testcase // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init) + { + const char* buffer; + std::size_t count; + }; + const std::vector cases = + { + {"[\"abc\"]____TRAILING____", 7}, // exact fit, tail hidden + {"[\"abcdefghijklmnop\"]____", 8}, // cut inside a string + {"[\"abc\"]____", 6}, // cut just before the closing quote + {"[12345]xxxxx", 4}, // cut inside a number + {"[123]999999", 5}, // number ends exactly at the count + {"[\"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\"]", 12}, // closing quote only behind the count + {"[\"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\"]", 19}, // cut inside an 8-byte SWAR stride + {"[\"\xe4\xb8\xad\xe6\x96\x87\"]", 5}, // cut inside a UTF-8 sequence + {"[\"\xe4\xb8\xad\xe6\x96\x87\"]____", 10}, // complete UTF-8, tail hidden + {"[1.25e3]TRAILINGDIGITS999", 7}, // number token reaches the count + }; + + for (const auto& tc : cases) + { + CAPTURE(tc.buffer); + CAPTURE(tc.count); + const std::string buffer = tc.buffer; + CHECK(via_counted(buffer, tc.count) == via_prefix(buffer, tc.count)); + } +} +#endif #endif } // namespace diff --git a/tools/generate_natvis/generate_natvis.py b/tools/generate_natvis/generate_natvis.py index 9266050c5..968690abe 100755 --- a/tools/generate_natvis/generate_natvis.py +++ b/tools/generate_natvis/generate_natvis.py @@ -20,7 +20,7 @@ if __name__ == '__main__': namespaces = ['nlohmann'] abi_prefix = 'json_abi' - abi_tags = ['_diag', '_ldvcmp'] + abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics'] version = '_v' + args.version.replace('.', '_') inline_namespaces = [] diff --git a/tools/serve_header/README.md b/tools/serve_header/README.md index 0d0ed69f6..bdf2c60ec 100644 --- a/tools/serve_header/README.md +++ b/tools/serve_header/README.md @@ -60,6 +60,10 @@ int main() { `serve_header.py` will try to read a configuration file `serve_header.yml` in the top level or project root directory, and will fall back on built-in defaults if the file cannot be read. An annotated example configuration can be found in `tools/serve_header/serve_header.yml.example`. +By default, the server listens on `localhost` only, and only web pages from Compiler Explorer (`https://godbolt.org` and `https://compiler-explorer.com`) may read the header. +Set `bind` to serve other machines as well; anyone who can reach the server can then trigger `make` runs in your working trees. +Set `cors_origins` to allow other web pages. + ## Serving `json.hpp` from multiple project directory instances or working trees `serve_header.py` was designed with the goal of supporting multiple project roots or working trees at the same time. diff --git a/tools/serve_header/serve_header.py b/tools/serve_header/serve_header.py index e2da2dad0..1f29cb589 100755 --- a/tools/serve_header/serve_header.py +++ b/tools/serve_header/serve_header.py @@ -26,6 +26,10 @@ HEADER = 'json.hpp' DATETIME_FORMAT = '%Y-%m-%d %H:%M:%S' +# origins whose pages may read the served header from a browser; Compiler +# Explorer downloads #include headers client-side +DEFAULT_CORS_ORIGINS = ['https://godbolt.org', 'https://compiler-explorer.com'] + JSON_VERSION_RE = re.compile(r'\s*#\s*define\s+NLOHMANN_JSON_VERSION_MAJOR\s+') class ExitHandler(logging.StreamHandler): @@ -247,6 +251,8 @@ class WorkTrees(FileSystemEventHandler): self.observer.join() class HeaderRequestHandler(SimpleHTTPRequestHandler): # lgtm[py/missing-call-to-init] + cors_origins = DEFAULT_CORS_ORIGINS + def __init__(self, request, client_address, server): """.""" self.worktrees = server.worktrees @@ -310,8 +316,11 @@ class HeaderRequestHandler(SimpleHTTPRequestHandler): # lgtm[py/missing-call-to- # set content length super().send_header('Content-Length', length) - # CORS header - self.send_header('Access-Control-Allow-Origin', '*') + # CORS header; only for the configured origins + origin = self.headers.get('Origin') + if origin in self.cors_origins: + self.send_header('Access-Control-Allow-Origin', origin) + self.send_header('Vary', 'Origin') # prevent caching self.send_header('Cache-Control', 'no-cache, no-store, must-revalidate') self.send_header('Pragma', 'no-cache') @@ -383,8 +392,15 @@ if __name__ == '__main__': # find and monitor working trees worktrees = WorkTrees(config.get('root', '.')) - # start web server - infos = socket.getaddrinfo(config.get('bind', None), config.get('port', 8443), + # origins allowed to read the header from a browser + cors_origins = config.get('cors_origins', DEFAULT_CORS_ORIGINS) + if isinstance(cors_origins, str): + cors_origins = [cors_origins] + HeaderRequestHandler.cors_origins = cors_origins + + # start web server; only reachable from this machine unless configured + # otherwise (bind: null listens on all interfaces) + infos = socket.getaddrinfo(config.get('bind', 'localhost'), config.get('port', 8443), type=socket.SOCK_STREAM, flags=socket.AI_PASSIVE) DualStackServer.address_family = infos[0][0] HeaderRequestHandler.protocol_version = 'HTTP/1.0' diff --git a/tools/serve_header/serve_header.yml.example b/tools/serve_header/serve_header.yml.example index 42310910e..ec75e2b49 100644 --- a/tools/serve_header/serve_header.yml.example +++ b/tools/serve_header/serve_header.yml.example @@ -10,6 +10,13 @@ # cert_file: localhost.pem # key_file: localhost-key.pem -# address and port for the server to listen on -# bind: null +# address and port for the server to listen on; by default, only this machine +# can connect. Binding to a network address, or to null for all interfaces, +# lets other machines connect, and every request runs make in a working tree. +# bind: localhost # port: 8443 + +# origins whose web pages may read the header (CORS) +# cors_origins: +# - https://godbolt.org +# - https://compiler-explorer.com