From d304501fa7d451332b1a47e117d3709dc1e6952a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 22:00:35 +0200 Subject: [PATCH] Cover every byte class in the ill-formed UTF-8 sweeps Pinning the bytes around the invalid one to 0x80 lost coverage: with error_handler_t::ignore/replace, the serializer re-reads the invalid byte and decodes the following bytes, and its decoder distinguishes the continuation classes 0x80-0x8F, 0x90-0x9F, and 0xA0-0xBF, as do the lexer's range checks. Iterate those positions over the first and last byte of each class within the valid range instead (utils::utf8_continuation_bytes). The invalid byte still takes all 256 values. Defining JSON_TEST_UTF8_EXHAUSTIVE restores the full Cartesian product, with the same assertion counts as before #5418. Signed-off-by: Niels Lohmann --- tests/src/test_utils.hpp | 28 +++++++++ tests/src/unit-unicode2.cpp | 120 ++++++++++++++++++++---------------- tests/src/unit-unicode3.cpp | 87 +++++++++++++++----------- tests/src/unit-unicode4.cpp | 64 +++++++++++-------- tests/src/unit-unicode5.cpp | 85 +++++++++++++++---------- 5 files changed, 235 insertions(+), 149 deletions(-) diff --git a/tests/src/test_utils.hpp b/tests/src/test_utils.hpp index b2a381fb0..e73985aa6 100644 --- a/tests/src/test_utils.hpp +++ b/tests/src/test_utils.hpp @@ -8,6 +8,7 @@ #pragma once +#include // array #include // uint8_t #include // size_t #include // ifstream, istreambuf_iterator, ios @@ -42,6 +43,33 @@ T next_integer_sample(T i, T last, T stride) return n < last ? n : last; } +// UTF-8 continuation bytes in [lo, hi] that stand in for all of them in the +// ill-formed UTF-8 tests. Both the lexer's range checks and the serializer's +// decoder (detail::decode) only distinguish the classes 0x80..0x8F, 0x90..0x9F, +// and 0xA0..0xBF, so the first and last byte of each class within [lo, hi] +// exercise every behavior while a test sweeps another byte position through +// all 256 values (#5418). Define JSON_TEST_UTF8_EXHAUSTIVE to get every byte. +inline std::vector utf8_continuation_bytes(int lo, int hi) +{ + std::vector result; +#ifdef JSON_TEST_UTF8_EXHAUSTIVE + for (int byte = lo; byte <= hi; ++byte) + { + result.push_back(byte); + } +#else + static const std::array class_ends = {{0x80, 0x8F, 0x90, 0x9F, 0xA0, 0xBF}}; + for (const int byte : class_ends) + { + if (lo <= byte && byte <= hi) + { + result.push_back(byte); + } + } +#endif + return result; +} + inline std::vector read_binary_file(const std::string& filename) { std::ifstream file(filename, std::ios::binary); diff --git a/tests/src/unit-unicode2.cpp b/tests/src/unit-unicode2.cpp index 645341e79..375b79b1c 100644 --- a/tests/src/unit-unicode2.cpp +++ b/tests/src/unit-unicode2.cpp @@ -323,8 +323,7 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { - // Pin the 3rd byte to one valid continuation (#5418). - const int byte3 = 0x80; + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) for (int byte1 = 0xE0; byte1 <= 0xE0; ++byte1) { for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) @@ -335,28 +334,33 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) continue; } - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); + for (const int byte3 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); + } } } } SECTION("ill-formed: wrong third byte") { - // Pin the 2nd byte to one valid continuation (#5418). - const int byte2 = 0xA0; + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) for (int byte1 = 0xE0; byte1 <= 0xE0; ++byte1) { - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + for (const int byte2 : utils::utf8_continuation_bytes(0xA0, 0xBF)) { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) { - continue; - } + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) + { + continue; + } - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); + } } } } @@ -402,8 +406,7 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { - // Pin the 3rd byte to one valid continuation (#5418). - const int byte3 = 0x80; + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) for (int byte1 = 0xE1; byte1 <= 0xEC; ++byte1) { for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) @@ -414,28 +417,33 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) continue; } - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); + for (const int byte3 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); + } } } } SECTION("ill-formed: wrong third byte") { - // Pin the 2nd byte to one valid continuation (#5418). - const int byte2 = 0x80; + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) for (int byte1 = 0xE1; byte1 <= 0xEC; ++byte1) { - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + for (const int byte2 : utils::utf8_continuation_bytes(0x80, 0xBF)) { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) { - continue; - } + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) + { + continue; + } - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); + } } } } @@ -481,8 +489,7 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { - // Pin the 3rd byte to one valid continuation (#5418). - const int byte3 = 0x80; + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) for (int byte1 = 0xED; byte1 <= 0xED; ++byte1) { for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) @@ -493,28 +500,33 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) continue; } - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); + for (const int byte3 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); + } } } } SECTION("ill-formed: wrong third byte") { - // Pin the 2nd byte to one valid continuation (#5418). - const int byte2 = 0x80; + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) for (int byte1 = 0xED; byte1 <= 0xED; ++byte1) { - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + for (const int byte2 : utils::utf8_continuation_bytes(0x80, 0x9F)) { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) { - continue; - } + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) + { + continue; + } - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); + } } } } @@ -560,8 +572,7 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { - // Pin the 3rd byte to one valid continuation (#5418). - const int byte3 = 0x80; + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) for (int byte1 = 0xEE; byte1 <= 0xEF; ++byte1) { for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) @@ -572,28 +583,33 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) continue; } - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); + for (const int byte3 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); + } } } } SECTION("ill-formed: wrong third byte") { - // Pin the 2nd byte to one valid continuation (#5418). - const int byte2 = 0x80; + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) for (int byte1 = 0xEE; byte1 <= 0xEF; ++byte1) { - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + for (const int byte2 : utils::utf8_continuation_bytes(0x80, 0xBF)) { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) { - continue; - } + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) + { + continue; + } - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); + } } } } diff --git a/tests/src/unit-unicode3.cpp b/tests/src/unit-unicode3.cpp index dcf9a6097..dd92fb4ef 100644 --- a/tests/src/unit-unicode3.cpp +++ b/tests/src/unit-unicode3.cpp @@ -250,61 +250,76 @@ TEST_CASE("Unicode (3/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { - // Property: an out-of-range 2nd byte is rejected regardless of - // later continuation bytes. Pin those to one valid value so - // this section does not sweep 192*64*64 combinations (#5418). - const int byte1 = 0xF0; - const int byte3 = 0x80; - const int byte4 = 0x80; - for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) + for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1) { - // skip correct second byte - if (0x90 <= byte2 && byte2 <= 0xBF) + for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) { - continue; - } + // skip correct second byte + if (0x90 <= byte2 && byte2 <= 0xBF) + { + continue; + } - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); + for (const int byte3 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + for (const int byte4 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); + } + } + } } } SECTION("ill-formed: wrong third byte") { - // Property: an out-of-range 3rd byte is rejected regardless of - // the 2nd/4th bytes. Pin those to one valid value (#5418). - const int byte1 = 0xF0; - const int byte2 = 0x90; - const int byte4 = 0x80; - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) + for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1) { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) + for (const int byte2 : utils::utf8_continuation_bytes(0x90, 0xBF)) { - continue; - } + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + { + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) + { + continue; + } - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); + for (const int byte4 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); + } + } + } } } SECTION("ill-formed: wrong fourth byte") { - // Pin 2nd/3rd bytes to one valid continuation (#5418). - const int byte1 = 0xF0; - const int byte2 = 0x90; - const int byte3 = 0x80; - for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) + for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1) { - // skip correct fourth byte - if (0x80 <= byte4 && byte4 <= 0xBF) + for (const int byte2 : utils::utf8_continuation_bytes(0x90, 0xBF)) { - continue; - } + for (const int byte3 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) + { + // skip correct fourth byte + if (0x80 <= byte4 && byte4 <= 0xBF) + { + continue; + } - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); + } + } + } } } } diff --git a/tests/src/unit-unicode4.cpp b/tests/src/unit-unicode4.cpp index 9ec106af9..2c5406b8d 100644 --- a/tests/src/unit-unicode4.cpp +++ b/tests/src/unit-unicode4.cpp @@ -250,11 +250,7 @@ TEST_CASE("Unicode (4/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { - // Pin later continuation bytes; the 2nd-byte property does not - // depend on them. Lead bytes F1-F3 stay, as they define this - // sequence class (#5418). - const int byte3 = 0x80; - const int byte4 = 0x80; + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1) { for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) @@ -265,50 +261,64 @@ TEST_CASE("Unicode (4/5)" * doctest::skip()) continue; } - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); + for (const int byte3 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + for (const int byte4 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); + } + } } } } SECTION("ill-formed: wrong third byte") { - // Pin 2nd/4th bytes to one valid continuation (#5418). - const int byte2 = 0x80; - const int byte4 = 0x80; + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1) { - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + for (const int byte2 : utils::utf8_continuation_bytes(0x80, 0xBF)) { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) { - continue; - } + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) + { + continue; + } - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); + for (const int byte4 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); + } + } } } } SECTION("ill-formed: wrong fourth byte") { - // Pin 2nd/3rd bytes to one valid continuation (#5418). - const int byte2 = 0x80; - const int byte3 = 0x80; + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1) { - for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) + for (const int byte2 : utils::utf8_continuation_bytes(0x80, 0xBF)) { - // skip correct fourth byte - if (0x80 <= byte4 && byte4 <= 0xBF) + for (const int byte3 : utils::utf8_continuation_bytes(0x80, 0xBF)) { - continue; - } + for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) + { + // skip correct fourth byte + if (0x80 <= byte4 && byte4 <= 0xBF) + { + continue; + } - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); + } + } } } } diff --git a/tests/src/unit-unicode5.cpp b/tests/src/unit-unicode5.cpp index e1b8bacee..1832196ad 100644 --- a/tests/src/unit-unicode5.cpp +++ b/tests/src/unit-unicode5.cpp @@ -250,59 +250,76 @@ TEST_CASE("Unicode (5/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { - // Pin later continuation bytes (#5418). F4's valid 2nd byte - // range is 0x80-0x8F. - const int byte1 = 0xF4; - const int byte3 = 0x80; - const int byte4 = 0x80; - for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) + for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1) { - // skip correct second byte - if (0x80 <= byte2 && byte2 <= 0x8F) + for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) { - continue; - } + // skip correct second byte + if (0x80 <= byte2 && byte2 <= 0x8F) + { + continue; + } - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); + for (const int byte3 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + for (const int byte4 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); + } + } + } } } SECTION("ill-formed: wrong third byte") { - // Pin 2nd/4th bytes to one valid continuation (#5418). - const int byte1 = 0xF4; - const int byte2 = 0x80; - const int byte4 = 0x80; - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) + for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1) { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) + for (const int byte2 : utils::utf8_continuation_bytes(0x80, 0x8F)) { - continue; - } + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + { + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) + { + continue; + } - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); + for (const int byte4 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); + } + } + } } } SECTION("ill-formed: wrong fourth byte") { - // Pin 2nd/3rd bytes to one valid continuation (#5418). - const int byte1 = 0xF4; - const int byte2 = 0x80; - const int byte3 = 0x80; - for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) + // the other bytes take the ends of each byte class, see utils::utf8_continuation_bytes (#5418) + for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1) { - // skip correct fourth byte - if (0x80 <= byte4 && byte4 <= 0xBF) + for (const int byte2 : utils::utf8_continuation_bytes(0x80, 0x8F)) { - continue; - } + for (const int byte3 : utils::utf8_continuation_bytes(0x80, 0xBF)) + { + for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) + { + // skip correct fourth byte + if (0x80 <= byte4 && byte4 <= 0xBF) + { + continue; + } - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); + } + } + } } } }