mirror of
https://github.com/nlohmann/json.git
synced 2026-10-04 06:11:14 +01:00
Read BON8 strings in bulk from contiguous input
- copy the valid UTF-8 of a string in one step when the input is contiguous (twitter.json is read in 1.68 instead of 2.52 ms, jeopardy.json in 196 instead of 297 ms, close to CBOR and MessagePack) - share the new valid_utf8_prefix() with the writer's UTF-8 check, which now skips ASCII 8 bytes at a time - let the fuzzer check that contiguous and stream input give the same value or error, and test both paths in the unit tests - clarify that a second 0xFF after a string is an empty string Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -28,6 +28,7 @@
|
||||
#include <nlohmann/detail/input/input_adapters.hpp>
|
||||
#include <nlohmann/detail/input/json_sax.hpp>
|
||||
#include <nlohmann/detail/input/lexer.hpp>
|
||||
#include <nlohmann/detail/input/string_scan.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/meta/is_sax.hpp>
|
||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||
@@ -97,6 +98,11 @@ class binary_reader
|
||||
using char_type = typename InputAdapterType::char_type;
|
||||
using char_int_type = typename char_traits<char_type>::int_type;
|
||||
|
||||
/// whether the input is a contiguous block of bytes that BON8 strings can
|
||||
/// be copied from in bulk; see @ref get_bon8_string_bulk
|
||||
static constexpr bool bon8_bulk_scan =
|
||||
input_adapter_supports_bulk_scan<InputAdapterType>(is_detected<detect_supports_bulk_scan, InputAdapterType> {});
|
||||
|
||||
public:
|
||||
/*!
|
||||
@brief create a binary reader
|
||||
@@ -3589,6 +3595,42 @@ class binary_reader
|
||||
return bon8_error("expected a string; last byte", "key");
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief append the run of valid UTF-8 at the read position to a string
|
||||
|
||||
For contiguous input, the ASCII characters and complete well-formed UTF-8
|
||||
sequences at the read position are appended to @a result in one step. The
|
||||
byte that stops the run (an end-of-string marker, the first byte of the
|
||||
next value, or an ill-formed byte) is left for @ref get_bon8_string, so
|
||||
that strings end and errors are reported exactly as without this step.
|
||||
|
||||
@param[in,out] result the string to append to
|
||||
*/
|
||||
void get_bon8_string_bulk(string_t& result, std::true_type /*bulk*/)
|
||||
{
|
||||
// bytes handed back must be read through get_bon8() first
|
||||
if (bon8_pushback_size != 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
const std::size_t remaining = ia.bulk_remaining();
|
||||
if (remaining == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
const auto* const data = reinterpret_cast<const unsigned char*>(ia.bulk_data());
|
||||
const std::size_t length = valid_utf8_prefix(data, remaining);
|
||||
if (length != 0)
|
||||
{
|
||||
result.append(reinterpret_cast<const typename string_t::value_type*>(data), length);
|
||||
ia.bulk_skip(length);
|
||||
chars_read += length;
|
||||
}
|
||||
}
|
||||
|
||||
/// input that is not contiguous: strings are read byte by byte
|
||||
void get_bon8_string_bulk(string_t& /*result*/, std::false_type /*bulk*/) const noexcept {}
|
||||
|
||||
/*!
|
||||
@brief read a string
|
||||
|
||||
@@ -3605,6 +3647,8 @@ class binary_reader
|
||||
{
|
||||
while (true)
|
||||
{
|
||||
get_bon8_string_bulk(result, std::integral_constant<bool, bon8_bulk_scan> {});
|
||||
|
||||
const auto byte = get_bon8();
|
||||
|
||||
if (byte == char_traits<char_type>::eof())
|
||||
|
||||
Reference in New Issue
Block a user