mirror of
https://github.com/nlohmann/json.git
synced 2026-09-29 14:24:56 +01:00
* Add tests for uncovered code paths Cover code the test suite did not reach, found from the Coveralls report of develop and a local coverage run of HEAD: - dump() of every kind of value below the bound of the recursive descent (pretty-printed objects, binary values, discarded values, scalars), and flushes of the escape and write buffers mid-string and mid-binary - the iterative comparison: objects with different keys, containers that are a prefix of each other, and elements that cannot be ordered, each both at the top level and below the nesting bound - SAX handlers that stop at any event, including the end of a nested container, in the BSON, CBOR, MessagePack, UBJSON and BJData readers - from_bson/cbor/msgpack/ubjson/bjdata returning a discarded value through the iterator and pointer overloads - JSON Patch, diff, merge_patch and update(..., true) on ordered_json - smaller gaps: get_allocator(), to_ubjson/to_bjdata into a string, value() with an unresolvable JSON pointer, integer/float comparison below the integer range and with negative fractions, conversion to a custom binary type, std::formatter::parse on a spec without '}', unescape() of a lone '~', and the callback parser's start_array() Signed-off-by: Niels Lohmann <mail@nlohmann.me> * Cover more paths that were thought unreachable - parse_float_fast() declining malformed or inexact input, called directly since the lexer only passes well-formed numbers to it - a UTF-16 high surrogate followed by a unit above the low surrogates - self-assignment of a const_iterator - a truncated CBOR string read through non-contiguous iterators - serializing a long double under the de_DE locale, which undoes the locale's decimal point and thousands separator - values read from a binary format carrying no diagnostic positions, with and without a parser callback Signed-off-by: Niels Lohmann <mail@nlohmann.me> * Fix the CI failures of the new coverage tests - declare the self-assignment reference const (misc-const-correctness) - expect the (/path) prefix that JSON_DIAGNOSTICS adds to the messages of the failing ordered_json patch operations Signed-off-by: Niels Lohmann <mail@nlohmann.me> * Expect the byte range JSON_DIAGNOSTIC_POSITIONS adds to the patch errors Signed-off-by: Niels Lohmann <mail@nlohmann.me> * Build the expected dump of the nested-object test with += clang-tidy (performance-inefficient-string-concatenation) reported the chain of operator+ calls that assembled the expected indented output. Signed-off-by: Niels Lohmann <mail@nlohmann.me> * Compare the BJData and UBJSON test outputs byte by byte Building a std::string from the byte vector converts each byte implicitly, which -fsanitize=integer reports for bytes of 0x80 and above (ci_test_clang_sanitizer). Signed-off-by: Niels Lohmann <mail@nlohmann.me> --------- Signed-off-by: Niels Lohmann <mail@nlohmann.me>
146 lines
7.4 KiB
C++
146 lines
7.4 KiB
C++
// __ _____ _____ _____
|
|
// __| | __| | | | JSON for Modern C++ (supporting code)
|
|
// | | |__ | | | | | | version 3.12.0
|
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
|
//
|
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
#include "doctest_compatibility.h"
|
|
|
|
#include <nlohmann/json.hpp>
|
|
using nlohmann::json;
|
|
|
|
// ICPC errors out on multibyte character sequences in source files
|
|
#ifndef __INTEL_COMPILER
|
|
namespace
|
|
{
|
|
bool wstring_is_utf16();
|
|
bool wstring_is_utf16()
|
|
{
|
|
return (std::wstring(L"💩") == std::wstring(L"\U0001F4A9"));
|
|
}
|
|
|
|
bool u16string_is_utf16();
|
|
bool u16string_is_utf16()
|
|
{
|
|
return (std::u16string(u"💩") == std::u16string(u"\U0001F4A9"));
|
|
}
|
|
|
|
bool u32string_is_utf32();
|
|
bool u32string_is_utf32()
|
|
{
|
|
return (std::u32string(U"💩") == std::u32string(U"\U0001F4A9"));
|
|
}
|
|
} // namespace
|
|
|
|
TEST_CASE("wide strings")
|
|
{
|
|
SECTION("std::wstring")
|
|
{
|
|
if (wstring_is_utf16())
|
|
{
|
|
std::wstring const w = L"[12.2,\"Ⴥaäö💤🧢\"]";
|
|
json const j = json::parse(w);
|
|
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
|
|
}
|
|
}
|
|
|
|
SECTION("invalid std::wstring")
|
|
{
|
|
if (wstring_is_utf16())
|
|
{
|
|
std::wstring const w = L"\"\xDBFF";
|
|
json _;
|
|
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
|
|
|
// the exact message depends on the width of wchar_t: a 16-bit
|
|
// wchar_t passes the lone surrogate to the UTF-8 decoder unchanged
|
|
// (rejected as a single ill-formed byte at column 2), while a
|
|
// 32-bit wchar_t first encodes it as an ill-formed three-byte
|
|
// sequence (rejected one byte later, at column 3)
|
|
const char* const error_low_surrogate = sizeof(wchar_t) == 2
|
|
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
|
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xB0'";
|
|
const char* const error_high_surrogate = sizeof(wchar_t) == 2
|
|
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
|
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xA0'";
|
|
|
|
// a lone low surrogate cannot start a pair
|
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'"'}), error_low_surrogate, json::parse_error&);
|
|
// a high surrogate followed by a non-low-surrogate unit is invalid
|
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&);
|
|
// ... also when the unit is above the low surrogates
|
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), static_cast<wchar_t>(0xE000), L'"'}), error_high_surrogate, json::parse_error&);
|
|
// a lone low surrogate must not swallow the following unit: pairing
|
|
// it with any second unit would produce valid UTF-8, so the error
|
|
// has to report an ill-formed byte at the surrogate's own position
|
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'a', L'"'}), error_low_surrogate, json::parse_error&);
|
|
}
|
|
}
|
|
|
|
SECTION("std::u16string")
|
|
{
|
|
if (u16string_is_utf16())
|
|
{
|
|
std::u16string const w = u"[12.2,\"Ⴥaäö💤🧢\"]";
|
|
json const j = json::parse(w);
|
|
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
|
|
}
|
|
}
|
|
|
|
SECTION("invalid std::u16string")
|
|
{
|
|
if (u16string_is_utf16())
|
|
{
|
|
std::u16string const w = u"\"\xDBFF";
|
|
json _;
|
|
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
|
|
|
// a lone low surrogate cannot start a pair
|
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
|
// a high surrogate followed by a non-low-surrogate unit is invalid
|
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
|
// ... also when the unit is above the low surrogates
|
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
|
// a lone low surrogate must not swallow the following unit: pairing
|
|
// it with any second unit would produce valid UTF-8, so the error
|
|
// has to report an ill-formed byte at the surrogate's own position
|
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
|
// a valid surrogate pair is still decoded (U+1F600)
|
|
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
|
|
}
|
|
}
|
|
|
|
SECTION("std::u32string")
|
|
{
|
|
if (u32string_is_utf32())
|
|
{
|
|
std::u32string const w = U"[12.2,\"Ⴥaäö💤🧢\"]";
|
|
json const j = json::parse(w);
|
|
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
|
|
}
|
|
}
|
|
|
|
SECTION("invalid std::u32string")
|
|
{
|
|
if (u32string_is_utf32())
|
|
{
|
|
std::u32string const w = U"\"\x110000";
|
|
json _;
|
|
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
|
|
|
// a code unit above U+10FFFF must not be narrowed onto the EOF
|
|
// sentinel: 0xFFFFFFFF would otherwise end the document silently and
|
|
// let everything following it pass the strict end-of-input check
|
|
std::u32string const trailing{U'[', U'1', U']', static_cast<char32_t>(0xFFFFFFFF), U'x'};
|
|
CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
|
|
CHECK(!json::accept(trailing));
|
|
|
|
// the same unit inside a string is reported as an ill-formed byte
|
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
|
|
}
|
|
}
|
|
}
|
|
#endif
|