Pin unrelated bytes in the remaining ill-formed UTF-8 sweeps

The wrong-4th-byte sections of unit-unicode3/4/5 became live with #5499
and swept every valid 2nd/3rd byte again (2.4M iterations in unicode4
alone). Pin those bytes like the 2nd/3rd-byte sections, and do the
same for the 3-byte sequences in unit-unicode2.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-27 21:46:08 +02:00
parent c6bb0844f3
commit e802c98da8
4 changed files with 84 additions and 105 deletions
+52 -60
View File
@@ -323,6 +323,8 @@ TEST_CASE("Unicode (2/5)" * doctest::skip())
SECTION("ill-formed: wrong second byte")
{
// Pin the 3rd byte to one valid continuation (#5418).
const int byte3 = 0x80;
for (int byte1 = 0xE0; byte1 <= 0xE0; ++byte1)
{
for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2)
@@ -333,32 +335,28 @@ TEST_CASE("Unicode (2/5)" * doctest::skip())
continue;
}
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
SECTION("ill-formed: wrong third byte")
{
// Pin the 2nd byte to one valid continuation (#5418).
const int byte2 = 0xA0;
for (int byte1 = 0xE0; byte1 <= 0xE0; ++byte1)
{
for (int byte2 = 0xA0; byte2 <= 0xBF; ++byte2)
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
{
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
continue;
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
@@ -404,6 +402,8 @@ TEST_CASE("Unicode (2/5)" * doctest::skip())
SECTION("ill-formed: wrong second byte")
{
// Pin the 3rd byte to one valid continuation (#5418).
const int byte3 = 0x80;
for (int byte1 = 0xE1; byte1 <= 0xEC; ++byte1)
{
for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2)
@@ -414,32 +414,28 @@ TEST_CASE("Unicode (2/5)" * doctest::skip())
continue;
}
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
SECTION("ill-formed: wrong third byte")
{
// Pin the 2nd byte to one valid continuation (#5418).
const int byte2 = 0x80;
for (int byte1 = 0xE1; byte1 <= 0xEC; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
{
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
continue;
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
@@ -485,6 +481,8 @@ TEST_CASE("Unicode (2/5)" * doctest::skip())
SECTION("ill-formed: wrong second byte")
{
// Pin the 3rd byte to one valid continuation (#5418).
const int byte3 = 0x80;
for (int byte1 = 0xED; byte1 <= 0xED; ++byte1)
{
for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2)
@@ -495,32 +493,28 @@ TEST_CASE("Unicode (2/5)" * doctest::skip())
continue;
}
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
SECTION("ill-formed: wrong third byte")
{
// Pin the 2nd byte to one valid continuation (#5418).
const int byte2 = 0x80;
for (int byte1 = 0xED; byte1 <= 0xED; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0x9F; ++byte2)
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
{
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
continue;
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
@@ -566,6 +560,8 @@ TEST_CASE("Unicode (2/5)" * doctest::skip())
SECTION("ill-formed: wrong second byte")
{
// Pin the 3rd byte to one valid continuation (#5418).
const int byte3 = 0x80;
for (int byte1 = 0xEE; byte1 <= 0xEF; ++byte1)
{
for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2)
@@ -576,32 +572,28 @@ TEST_CASE("Unicode (2/5)" * doctest::skip())
continue;
}
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
SECTION("ill-formed: wrong third byte")
{
// Pin the 2nd byte to one valid continuation (#5418).
const int byte2 = 0x80;
for (int byte1 = 0xEE; byte1 <= 0xEF; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
{
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
continue;
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
+11 -16
View File
@@ -291,25 +291,20 @@ TEST_CASE("Unicode (3/5)" * doctest::skip())
SECTION("ill-formed: wrong fourth byte")
{
for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1)
// Pin 2nd/3rd bytes to one valid continuation (#5418).
const int byte1 = 0xF0;
const int byte2 = 0x90;
const int byte3 = 0x80;
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
for (int byte2 = 0x90; byte2 <= 0xBF; ++byte2)
// skip correct fourth byte
if (0x80 <= byte4 && byte4 <= 0xBF)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip correct fourth byte
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
continue;
}
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
}
+10 -13
View File
@@ -294,24 +294,21 @@ TEST_CASE("Unicode (4/5)" * doctest::skip())
SECTION("ill-formed: wrong fourth byte")
{
// Pin 2nd/3rd bytes to one valid continuation (#5418).
const int byte2 = 0x80;
const int byte3 = 0x80;
for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
// skip correct fourth byte
if (0x80 <= byte4 && byte4 <= 0xBF)
{
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip correct fourth byte
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
continue;
}
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
}
+11 -16
View File
@@ -289,25 +289,20 @@ TEST_CASE("Unicode (5/5)" * doctest::skip())
SECTION("ill-formed: wrong fourth byte")
{
for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1)
// Pin 2nd/3rd bytes to one valid continuation (#5418).
const int byte1 = 0xF4;
const int byte2 = 0x80;
const int byte3 = 0x80;
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
for (int byte2 = 0x80; byte2 <= 0x8F; ++byte2)
// skip correct fourth byte
if (0x80 <= byte4 && byte4 <= 0xBF)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip correct fourth byte
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
continue;
}
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
}