From e802c98da801f5de3c45bad4e3f328c97fac9721 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 21:46:08 +0200 Subject: [PATCH] Pin unrelated bytes in the remaining ill-formed UTF-8 sweeps The wrong-4th-byte sections of unit-unicode3/4/5 became live with #5499 and swept every valid 2nd/3rd byte again (2.4M iterations in unicode4 alone). Pin those bytes like the 2nd/3rd-byte sections, and do the same for the 3-byte sequences in unit-unicode2. Signed-off-by: Niels Lohmann --- tests/src/unit-unicode2.cpp | 112 +++++++++++++++++------------------- tests/src/unit-unicode3.cpp | 27 ++++----- tests/src/unit-unicode4.cpp | 23 ++++---- tests/src/unit-unicode5.cpp | 27 ++++----- 4 files changed, 84 insertions(+), 105 deletions(-) diff --git a/tests/src/unit-unicode2.cpp b/tests/src/unit-unicode2.cpp index a9649b4de..645341e79 100644 --- a/tests/src/unit-unicode2.cpp +++ b/tests/src/unit-unicode2.cpp @@ -323,6 +323,8 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { + // Pin the 3rd byte to one valid continuation (#5418). + const int byte3 = 0x80; for (int byte1 = 0xE0; byte1 <= 0xE0; ++byte1) { for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) @@ -333,32 +335,28 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) continue; } - for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3) - { - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); - } + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); } } } SECTION("ill-formed: wrong third byte") { + // Pin the 2nd byte to one valid continuation (#5418). + const int byte2 = 0xA0; for (int byte1 = 0xE0; byte1 <= 0xE0; ++byte1) { - for (int byte2 = 0xA0; byte2 <= 0xBF; ++byte2) + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) { - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) - { - continue; - } - - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); + continue; } + + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); } } } @@ -404,6 +402,8 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { + // Pin the 3rd byte to one valid continuation (#5418). + const int byte3 = 0x80; for (int byte1 = 0xE1; byte1 <= 0xEC; ++byte1) { for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) @@ -414,32 +414,28 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) continue; } - for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3) - { - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); - } + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); } } } SECTION("ill-formed: wrong third byte") { + // Pin the 2nd byte to one valid continuation (#5418). + const int byte2 = 0x80; for (int byte1 = 0xE1; byte1 <= 0xEC; ++byte1) { - for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2) + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) { - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) - { - continue; - } - - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); + continue; } + + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); } } } @@ -485,6 +481,8 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { + // Pin the 3rd byte to one valid continuation (#5418). + const int byte3 = 0x80; for (int byte1 = 0xED; byte1 <= 0xED; ++byte1) { for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) @@ -495,32 +493,28 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) continue; } - for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3) - { - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); - } + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); } } } SECTION("ill-formed: wrong third byte") { + // Pin the 2nd byte to one valid continuation (#5418). + const int byte2 = 0x80; for (int byte1 = 0xED; byte1 <= 0xED; ++byte1) { - for (int byte2 = 0x80; byte2 <= 0x9F; ++byte2) + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) { - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) - { - continue; - } - - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); + continue; } + + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); } } } @@ -566,6 +560,8 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) SECTION("ill-formed: wrong second byte") { + // Pin the 3rd byte to one valid continuation (#5418). + const int byte3 = 0x80; for (int byte1 = 0xEE; byte1 <= 0xEF; ++byte1) { for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2) @@ -576,32 +572,28 @@ TEST_CASE("Unicode (2/5)" * doctest::skip()) continue; } - for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3) - { - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); - } + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); } } } SECTION("ill-formed: wrong third byte") { + // Pin the 2nd byte to one valid continuation (#5418). + const int byte2 = 0x80; for (int byte1 = 0xEE; byte1 <= 0xEF; ++byte1) { - for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2) + for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) { - for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3) + // skip correct third byte + if (0x80 <= byte3 && byte3 <= 0xBF) { - // skip correct third byte - if (0x80 <= byte3 && byte3 <= 0xBF) - { - continue; - } - - check_utf8string(false, byte1, byte2, byte3); - check_utf8dump(false, byte1, byte2, byte3); + continue; } + + check_utf8string(false, byte1, byte2, byte3); + check_utf8dump(false, byte1, byte2, byte3); } } } diff --git a/tests/src/unit-unicode3.cpp b/tests/src/unit-unicode3.cpp index 0c76a1217..dcf9a6097 100644 --- a/tests/src/unit-unicode3.cpp +++ b/tests/src/unit-unicode3.cpp @@ -291,25 +291,20 @@ TEST_CASE("Unicode (3/5)" * doctest::skip()) SECTION("ill-formed: wrong fourth byte") { - for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1) + // Pin 2nd/3rd bytes to one valid continuation (#5418). + const int byte1 = 0xF0; + const int byte2 = 0x90; + const int byte3 = 0x80; + for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { - for (int byte2 = 0x90; byte2 <= 0xBF; ++byte2) + // skip correct fourth byte + if (0x80 <= byte4 && byte4 <= 0xBF) { - for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3) - { - for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) - { - // skip correct fourth byte - if (0x80 <= byte4 && byte4 <= 0xBF) - { - continue; - } - - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); - } - } + continue; } + + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); } } } diff --git a/tests/src/unit-unicode4.cpp b/tests/src/unit-unicode4.cpp index 10271d6fc..9ec106af9 100644 --- a/tests/src/unit-unicode4.cpp +++ b/tests/src/unit-unicode4.cpp @@ -294,24 +294,21 @@ TEST_CASE("Unicode (4/5)" * doctest::skip()) SECTION("ill-formed: wrong fourth byte") { + // Pin 2nd/3rd bytes to one valid continuation (#5418). + const int byte2 = 0x80; + const int byte3 = 0x80; for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1) { - for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2) + for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { - for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3) + // skip correct fourth byte + if (0x80 <= byte4 && byte4 <= 0xBF) { - for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) - { - // skip correct fourth byte - if (0x80 <= byte4 && byte4 <= 0xBF) - { - continue; - } - - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); - } + continue; } + + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); } } } diff --git a/tests/src/unit-unicode5.cpp b/tests/src/unit-unicode5.cpp index cf321d5ff..e1b8bacee 100644 --- a/tests/src/unit-unicode5.cpp +++ b/tests/src/unit-unicode5.cpp @@ -289,25 +289,20 @@ TEST_CASE("Unicode (5/5)" * doctest::skip()) SECTION("ill-formed: wrong fourth byte") { - for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1) + // Pin 2nd/3rd bytes to one valid continuation (#5418). + const int byte1 = 0xF4; + const int byte2 = 0x80; + const int byte3 = 0x80; + for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { - for (int byte2 = 0x80; byte2 <= 0x8F; ++byte2) + // skip correct fourth byte + if (0x80 <= byte4 && byte4 <= 0xBF) { - for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3) - { - for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) - { - // skip correct fourth byte - if (0x80 <= byte4 && byte4 <= 0xBF) - { - continue; - } - - check_utf8string(false, byte1, byte2, byte3, byte4); - check_utf8dump(false, byte1, byte2, byte3, byte4); - } - } + continue; } + + check_utf8string(false, byte1, byte2, byte3, byte4); + check_utf8dump(false, byte1, byte2, byte3, byte4); } } }