From c6bb0844f366f56e6c9a914693948ad85a0ec530 Mon Sep 17 00:00:00 2001 From: elix3r <157088510+22elix3r@users.noreply.github.com> Date: Thu, 27 Aug 2026 07:39:29 +0530 Subject: [PATCH] Cut Unicode ill-formed byte sweeps to one representative prefix Wrong-2nd and wrong-3rd-byte sections iterated every valid trailing byte, which dominated Linux --no-skip runtime without testing extra properties. Pin those bytes to a single valid continuation. The fourth-byte typo is left to #5416 so this change stands alone. Signed-off-by: elix3r <157088510+22elix3r@users.noreply.github.com> Signed-off-by: Niels Lohmann --- tests/src/unit-unicode3.cpp | 57 ++++++++++++++++--------------------- tests/src/unit-unicode4.cpp | 38 +++++++++++-------------- tests/src/unit-unicode5.cpp | 55 +++++++++++++++-------------------- 3 files changed, 65 insertions(+), 85 deletions(-) diff --git a/tests/src/unit-unicode3.cpp b/tests/src/unit-unicode3.cpp index 12c12eea4..0c76a1217 100644 --- a/tests/src/unit-unicode3.cpp +++ b/tests/src/unit-unicode3.cpp @@ -250,49 +250,42 @@ TEST_CASE("Unicode (3/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { - for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1) + // Property: an out-of-range 2nd byte is rejected regardless of + // later continuation bytes. Pin those to one valid value so + // this section does not sweep 192*64*64 combinations (#5418). + const int byte1 = 0xF0; + const int byte3 = 0x80; + const int byte4 = 0x80; + for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) { - for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) + // skip correct second byte + if (0x90 <= byte2 && byte2 <= 0xBF) { - // skip correct second byte - if (0x90 <= byte2 && byte2 <= 0xBF) - { - continue; - } - - for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3) - { - for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4) - { - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); - } - } + continue; } + + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); } } SECTION("ill-formed: wrong third byte") { - for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1) + // Property: an out-of-range 3rd byte is rejected regardless of + // the 2nd/4th bytes. Pin those to one valid value (#5418). + const int byte1 = 0xF0; + const int byte2 = 0x90; + const int byte4 = 0x80; + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) { - for (int byte2 = 0x90; byte2 <= 0xBF; ++byte2) + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) { - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) - { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) - { - continue; - } - - for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4) - { - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); - } - } + continue; } + + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); } } diff --git a/tests/src/unit-unicode4.cpp b/tests/src/unit-unicode4.cpp index 43cf7095e..10271d6fc 100644 --- a/tests/src/unit-unicode4.cpp +++ b/tests/src/unit-unicode4.cpp @@ -250,6 +250,11 @@ TEST_CASE("Unicode (4/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { + // Pin later continuation bytes; the 2nd-byte property does not + // depend on them. Lead bytes F1-F3 stay, as they define this + // sequence class (#5418). + const int byte3 = 0x80; + const int byte4 = 0x80; for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1) { for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) @@ -260,38 +265,29 @@ TEST_CASE("Unicode (4/5)" * doctest::skip()) continue; } - for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3) - { - for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4) - { - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); - } - } + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); } } } SECTION("ill-formed: wrong third byte") { + // Pin 2nd/4th bytes to one valid continuation (#5418). + const int byte2 = 0x80; + const int byte4 = 0x80; for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1) { - for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2) + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) { - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) - { - continue; - } - - for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4) - { - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); - } + continue; } + + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); } } } diff --git a/tests/src/unit-unicode5.cpp b/tests/src/unit-unicode5.cpp index bc0312820..cf321d5ff 100644 --- a/tests/src/unit-unicode5.cpp +++ b/tests/src/unit-unicode5.cpp @@ -250,49 +250,40 @@ TEST_CASE("Unicode (5/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { - for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1) + // Pin later continuation bytes (#5418). F4's valid 2nd byte + // range is 0x80-0x8F. + const int byte1 = 0xF4; + const int byte3 = 0x80; + const int byte4 = 0x80; + for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) { - for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) + // skip correct second byte + if (0x80 <= byte2 && byte2 <= 0x8F) { - // skip correct second byte - if (0x80 <= byte2 && byte2 <= 0x8F) - { - continue; - } - - for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3) - { - for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4) - { - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); - } - } + continue; } + + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); } } SECTION("ill-formed: wrong third byte") { - for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1) + // Pin 2nd/4th bytes to one valid continuation (#5418). + const int byte1 = 0xF4; + const int byte2 = 0x80; + const int byte4 = 0x80; + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) { - for (int byte2 = 0x80; byte2 <= 0x8F; ++byte2) + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) { - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) - { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) - { - continue; - } - - for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4) - { - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); - } - } + continue; } + + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); } }