Files
json/tests/src/unit-wstring.cpp
Niels Lohmann 85f8b21e1c Add tests for uncovered code paths (#5581)
* Add tests for uncovered code paths

Cover code the test suite did not reach, found from the Coveralls report
of develop and a local coverage run of HEAD:

- dump() of every kind of value below the bound of the recursive descent
  (pretty-printed objects, binary values, discarded values, scalars), and
  flushes of the escape and write buffers mid-string and mid-binary
- the iterative comparison: objects with different keys, containers that
  are a prefix of each other, and elements that cannot be ordered, each
  both at the top level and below the nesting bound
- SAX handlers that stop at any event, including the end of a nested
  container, in the BSON, CBOR, MessagePack, UBJSON and BJData readers
- from_bson/cbor/msgpack/ubjson/bjdata returning a discarded value
  through the iterator and pointer overloads
- JSON Patch, diff, merge_patch and update(..., true) on ordered_json
- smaller gaps: get_allocator(), to_ubjson/to_bjdata into a string,
  value() with an unresolvable JSON pointer, integer/float comparison
  below the integer range and with negative fractions, conversion to a
  custom binary type, std::formatter::parse on a spec without '}',
  unescape() of a lone '~', and the callback parser's start_array()

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

* Cover more paths that were thought unreachable

- parse_float_fast() declining malformed or inexact input, called
  directly since the lexer only passes well-formed numbers to it
- a UTF-16 high surrogate followed by a unit above the low surrogates
- self-assignment of a const_iterator
- a truncated CBOR string read through non-contiguous iterators
- serializing a long double under the de_DE locale, which undoes the
  locale's decimal point and thousands separator
- values read from a binary format carrying no diagnostic positions,
  with and without a parser callback

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

* Fix the CI failures of the new coverage tests

- declare the self-assignment reference const (misc-const-correctness)
- expect the (/path) prefix that JSON_DIAGNOSTICS adds to the messages
  of the failing ordered_json patch operations

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

* Expect the byte range JSON_DIAGNOSTIC_POSITIONS adds to the patch errors

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

* Build the expected dump of the nested-object test with +=

clang-tidy (performance-inefficient-string-concatenation) reported the
chain of operator+ calls that assembled the expected indented output.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

* Compare the BJData and UBJSON test outputs byte by byte

Building a std::string from the byte vector converts each byte
implicitly, which -fsanitize=integer reports for bytes of 0x80 and
above (ci_test_clang_sanitizer).

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

---------

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-27 14:19:22 +02:00

146 lines
7.4 KiB
C++

// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
using nlohmann::json;
// ICPC errors out on multibyte character sequences in source files
#ifndef __INTEL_COMPILER
namespace
{
bool wstring_is_utf16();
bool wstring_is_utf16()
{
return (std::wstring(L"💩") == std::wstring(L"\U0001F4A9"));
}
bool u16string_is_utf16();
bool u16string_is_utf16()
{
return (std::u16string(u"💩") == std::u16string(u"\U0001F4A9"));
}
bool u32string_is_utf32();
bool u32string_is_utf32()
{
return (std::u32string(U"💩") == std::u32string(U"\U0001F4A9"));
}
} // namespace
TEST_CASE("wide strings")
{
SECTION("std::wstring")
{
if (wstring_is_utf16())
{
std::wstring const w = L"[12.2,\"Ⴥaäö💤🧢\"]";
json const j = json::parse(w);
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
}
}
SECTION("invalid std::wstring")
{
if (wstring_is_utf16())
{
std::wstring const w = L"\"\xDBFF";
json _;
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
// the exact message depends on the width of wchar_t: a 16-bit
// wchar_t passes the lone surrogate to the UTF-8 decoder unchanged
// (rejected as a single ill-formed byte at column 2), while a
// 32-bit wchar_t first encodes it as an ill-formed three-byte
// sequence (rejected one byte later, at column 3)
const char* const error_low_surrogate = sizeof(wchar_t) == 2
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xB0'";
const char* const error_high_surrogate = sizeof(wchar_t) == 2
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xA0'";
// a lone low surrogate cannot start a pair
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'"'}), error_low_surrogate, json::parse_error&);
// a high surrogate followed by a non-low-surrogate unit is invalid
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&);
// ... also when the unit is above the low surrogates
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), static_cast<wchar_t>(0xE000), L'"'}), error_high_surrogate, json::parse_error&);
// a lone low surrogate must not swallow the following unit: pairing
// it with any second unit would produce valid UTF-8, so the error
// has to report an ill-formed byte at the surrogate's own position
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'a', L'"'}), error_low_surrogate, json::parse_error&);
}
}
SECTION("std::u16string")
{
if (u16string_is_utf16())
{
std::u16string const w = u"[12.2,\"Ⴥaäö💤🧢\"]";
json const j = json::parse(w);
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
}
}
SECTION("invalid std::u16string")
{
if (u16string_is_utf16())
{
std::u16string const w = u"\"\xDBFF";
json _;
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
// a lone low surrogate cannot start a pair
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a high surrogate followed by a non-low-surrogate unit is invalid
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// ... also when the unit is above the low surrogates
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a lone low surrogate must not swallow the following unit: pairing
// it with any second unit would produce valid UTF-8, so the error
// has to report an ill-formed byte at the surrogate's own position
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a valid surrogate pair is still decoded (U+1F600)
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
}
}
SECTION("std::u32string")
{
if (u32string_is_utf32())
{
std::u32string const w = U"[12.2,\"Ⴥaäö💤🧢\"]";
json const j = json::parse(w);
CHECK(j.dump() == "[12.2,\"Ⴥaäö💤🧢\"]");
}
}
SECTION("invalid std::u32string")
{
if (u32string_is_utf32())
{
std::u32string const w = U"\"\x110000";
json _;
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
// a code unit above U+10FFFF must not be narrowed onto the EOF
// sentinel: 0xFFFFFFFF would otherwise end the document silently and
// let everything following it pass the strict end-of-input check
std::u32string const trailing{U'[', U'1', U']', static_cast<char32_t>(0xFFFFFFFF), U'x'};
CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
CHECK(!json::accept(trailing));
// the same unit inside a string is reported as an ill-formed byte
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
}
}
}
#endif