Diff deeply nested values without recursing per nesting level

diff() descended into both values once per nesting level, and compared
them with operator== on every level on the way, which recurses as well.
Values nested deeply enough - 25,000 levels on an 8 MiB stack - exhausted
the call stack and terminated the process, although parse() accepts
them without complaint. On such a chain the per-level comparisons and
path strings also made diff() quadratic in time and memory.

Both the recursion and operator== only descend as far as the source is
nested. So diff() first checks, recursing at most diff_depth_limit()
(128) levels, whether the source is nested more deeply than that. If not
- all but a vanishing minority of values - the recursive algorithm
diffs it exactly as before, now as diff_recursively(). Otherwise
diff_iteratively() walks the two values on an explicit stack, emitting
the same operations in the same order. It does not compare arrays and
objects with operator== up front (equal ones yield no operations
anyway), keeps the path in one buffer instead of a new string per
level, and hands every subtree that is not nested too deeply back to
diff_recursively(), so equal parts are still skipped quickly.

The check costs one pass over the source. On a 3,000-object document
that is about 30% of diffing two equal values (which is just an
operator== call), about 10% of diffing values that differ in a few
places, and noise when arrays change length. Once operator== no longer
recurses (#5390), the check can go.

Tests check that the patch reproduces the target at every depth up to
300, for json and ordered_json, including reordered members. They also
check the exact operation for a difference deep inside, and diff values
nested 100,000 levels deep.

Fixes #5393 for diff().

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-24 09:38:22 +02:00
parent 3379ed5e33
commit a3f42f3f95
3 changed files with 928 additions and 4 deletions
+394 -2
View File
@@ -5381,6 +5381,73 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
JSON_HEDLEY_WARN_UNUSED_RESULT
static basic_json diff(const basic_json& source, const basic_json& target,
const string_t& path = "")
{
// Diffing descends into both values once per nesting level and
// compares them with operator== on the way, which recurses as well,
// so values nested deeply enough used to exhaust the call stack.
// Both only descend as far as the source is nested, so a source
// nested no more than diff_depth_limit() levels deep - all but a
// vanishing minority - is diffed recursively as before; deeper ones
// are diffed without the call stack.
if (JSON_HEDLEY_LIKELY(!nesting_exceeds(source, diff_depth_limit())))
{
return diff_recursively(source, target, path);
}
return diff_iteratively(source, target, path);
}
JSON_PRIVATE_UNLESS_TESTED:
/// the nesting depth up to which @ref diff recurses
static constexpr std::size_t diff_depth_limit() noexcept
{
return 128;
}
private:
/*!
@brief whether @a j is nested more than @a limit levels deep
A primitive value is not nested at all, an array or object one level more
than its most deeply nested element. Recurses at most @a limit levels.
*/
static bool nesting_exceeds(const basic_json& j, const std::size_t limit)
{
switch (j.m_data.m_type)
{
case value_t::array:
{
return limit == 0 || std::any_of(j.m_data.m_value.array->cbegin(), j.m_data.m_value.array->cend(),
[limit](const basic_json & element)
{
return element.is_structured() && nesting_exceeds(element, limit - 1);
});
}
case value_t::object:
{
return limit == 0 || std::any_of(j.m_data.m_value.object->cbegin(), j.m_data.m_value.object->cend(),
[limit](const typename object_t::value_type & element)
{
return element.second.is_structured() && nesting_exceeds(element.second, limit - 1);
});
}
case value_t::null:
case value_t::string:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::binary:
case value_t::discarded:
default:
return false;
}
}
/// @ref diff for a @a source nested no more than @ref diff_depth_limit levels deep
static basic_json diff_recursively(const basic_json& source, const basic_json& target,
const string_t& path)
{
// the patch
basic_json result(value_t::array);
@@ -5410,7 +5477,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
while (i < source.size() && i < target.size())
{
// recursive call to compare array values at index i
auto temp_diff = diff(source[i], target[i], detail::concat<string_t>(path, '/', detail::to_string<string_t>(i)));
auto temp_diff = diff_recursively(source[i], target[i], detail::concat<string_t>(path, '/', detail::to_string<string_t>(i)));
result.insert(result.end(), temp_diff.begin(), temp_diff.end());
++i;
}
@@ -5529,7 +5596,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
if (common_it != common_keys_source_order.cend() && it.key() == *common_it)
{
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
auto temp_diff = diff(it.value(), target[it.key()], path_key);
auto temp_diff = diff_recursively(it.value(), target[it.key()], path_key);
result.insert(result.end(), temp_diff.begin(), temp_diff.end());
++common_it;
}
@@ -5613,6 +5680,331 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
return result;
}
/*!
@brief @ref diff without the call stack
Produces the same patch as @ref diff_recursively. Only used for a source
nested more deeply than @ref diff_depth_limit; any arrays and objects
below it that are not nested that deeply are diffed recursively.
*/
static basic_json diff_iteratively(const basic_json& source, const basic_json& target,
const string_t& path)
{
// the patch
basic_json result(value_t::array);
// The arrays and objects being diffed are kept on an explicit stack,
// and every pair of elements is still diffed completely before the
// next one, so the operations come out in the same order as in
// diff_recursively. The path of the values being diffed is kept in
// one buffer that grows and shrinks with the stack, rather than in a
// new string per level.
struct diff_frame
{
diff_frame(const basic_json* source_, const basic_json* target_, const std::size_t path_length_)
: source(source_), target(target_), path_length(path_length_)
{}
/// the values being diffed, both arrays or both objects
const basic_json* source;
const basic_json* target;
/// the length of their path in `current_path`
std::size_t path_length;
/// arrays: the next index to diff
std::size_t index = 0;
/// objects: the next member of source to look at
const_iterator member{};
/// objects: the keys common to both, in source's order
std::vector<typename object_t::key_type> common_keys{};
/// objects: the next entry of common_keys
std::size_t next_common = 0;
/// objects: the "add" operations for keys only target has
basic_json added_ops{};
};
std::vector<diff_frame> stack;
string_t current_path = path;
// diff `s` against `t`, whose path is current_path: primitives,
// values of different types, and objects whose members were reordered
// are handled right away; arrays and other objects get a frame
const auto enter = [&result, &stack, &current_path](const basic_json & s, const basic_json & t)
{
// if the values are the same, there is nothing to do. Arrays and
// objects are not compared up front: comparing them recurses into
// everything below them - equal ones yield no operations anyway.
if ((!s.is_structured() || !t.is_structured()) && s == t)
{
return;
}
if (s.type() != t.type())
{
// different types: replace value
result.push_back(
{
{"op", "replace"}, {"path", current_path}, {"value", t}
});
return;
}
// arrays and objects that are not nested too deeply for the call
// stack are diffed recursively, which can skip equal parts
if (!nesting_exceeds(s, diff_depth_limit()))
{
const basic_json partial = diff_recursively(s, t, current_path);
result.insert(result.end(), partial.begin(), partial.end());
return;
}
switch (s.type())
{
case value_t::array:
{
stack.emplace_back(&s, &t, current_path.size());
return;
}
case value_t::object:
{
// first pass: record, for every source key, whether it is
// common to both objects (in source's iteration order) or
// was deleted (i.e., in source but not in target) -- this is
// a by-product of the t.find() call already needed to
// tell the two cases apart, so it adds no extra lookups. The
// "remove" ops themselves are emitted later, interleaved
// with the per-key diffs in the fast path below, to match
// source's original iteration order (as the original,
// pre-reordering-aware implementation did) instead of
// grouping all removes before all per-key diffs.
std::vector<typename object_t::key_type> common_keys_source_order;
for (auto it = s.cbegin(); it != s.cend(); ++it)
{
if (t.find(it.key()) != t.end())
{
common_keys_source_order.push_back(it.key());
}
}
// second pass: find keys that were added (i.e., in target but
// not in source), and record the keys common to both, in
// target's iteration order -- again a by-product of the
// s.find() call already needed to detect added keys. At
// the same time, determine whether every added key comes
// after every common key in target's order (a precondition
// for the fast path below, which only ever appends new keys
// at the very end): for an object_t whose iteration order is
// a pure function of the key set (e.g. the default std::map,
// which always iterates in sorted key order), the order
// check further below is always true and this whole
// mechanism is effectively a no-op; it only matters for a
// reorderable object_t such as the one backing `ordered_json`.
// patch ops for keys that were added (i.e., in target but not
// in source); built here so the fast path below can reuse
// them without a second s.find() per target key. Only
// used by the fast path -- the slow (reordering) path
// rebuilds "add" ops for every key itself.
std::vector<typename object_t::key_type> common_keys_target_order;
basic_json added_ops(value_t::array);
bool new_keys_form_suffix = true;
bool seen_new_key = false;
for (auto it = t.cbegin(); it != t.cend(); ++it)
{
if (s.find(it.key()) == s.end())
{
seen_new_key = true;
const auto path_key = detail::concat<string_t>(current_path, '/', detail::escape(it.key()));
added_ops.push_back(
{
{"op", "add"}, {"path", path_key},
{"value", it.value()}
});
}
else
{
common_keys_target_order.push_back(it.key());
if (seen_new_key)
{
new_keys_form_suffix = false;
}
}
}
if (common_keys_source_order == common_keys_target_order && new_keys_form_suffix)
{
// fast path: order of common keys already matches (or the
// object_t's iteration order does not depend on
// insertion history), so a plain per-key diff is correct
// and minimal, as before. The frame walks source in
// lockstep with common_keys_source_order, which is, by
// construction, the subsequence of source's keys that
// are common to both objects, in source's iteration
// order -- so a cheap key comparison replaces another
// lookup. Deleted keys are interleaved there too, in
// source's original order, and the "add" ops collected
// above are appended once all members are done.
stack.emplace_back(&s, &t, current_path.size());
stack.back().member = s.cbegin();
stack.back().common_keys = std::move(common_keys_source_order);
stack.back().added_ops = std::move(added_ops);
return;
}
// slow path: the common keys are in a different relative
// order in source and target (only possible for a
// reorderable object_t like ordered_map). Building a
// minimal reordering patch is a nontrivial (LCS-like)
// problem; instead, remove every source key -- both
// deleted keys (which must be removed regardless) and
// common keys (removed so they can be re-added in
// target's order) -- and re-add every key that should
// remain, with its final target value, in target's
// order. basic_json::patch()'s "add" operation on an
// object uses operator[], which appends at the end for a
// vector-backed insertion-ordered map when the key does
// not already exist -- so removing a key and then adding
// it moves it to the end, fixing its position.
for (auto it = s.cbegin(); it != s.cend(); ++it)
{
const auto path_key = detail::concat<string_t>(current_path, '/', detail::escape(it.key()));
result.push_back(object(
{
{"op", "remove"}, {"path", path_key}
}));
}
// add every key that is either common (just removed
// above) or brand new, in target's iteration order, so
// that the final order after applying the patch matches
// target exactly
for (auto it = t.cbegin(); it != t.cend(); ++it)
{
const auto path_key = detail::concat<string_t>(current_path, '/', detail::escape(it.key()));
result.push_back(
{
{"op", "add"}, {"path", path_key},
{"value", it.value()}
});
}
return;
}
case value_t::null:
case value_t::string:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::binary:
case value_t::discarded:
default:
{
// both primitive types: replace value
result.push_back(
{
{"op", "replace"}, {"path", current_path}, {"value", t}
});
return;
}
}
};
enter(source, target);
while (!stack.empty())
{
diff_frame& frame = stack.back();
const std::size_t path_length = frame.path_length;
const std::size_t depth = stack.size();
if (frame.source->is_array())
{
const auto& source_array = *frame.source->m_data.m_value.array;
const auto& target_array = *frame.target->m_data.m_value.array;
// first pass: traverse common elements
if (frame.index < source_array.size() && frame.index < target_array.size())
{
const std::size_t i = frame.index++;
detail::concat_into(current_path, '/', detail::to_string<string_t>(i));
enter(source_array[i], target_array[i]); // may push, which invalidates `frame`
if (stack.size() == depth)
{
current_path.resize(path_length);
}
continue;
}
// We now reached the end of at least one array
// in a second pass, traverse the remaining elements
// remove my remaining elements, highest index first; appending
// in that order avoids the quadratic reinsertion done before
for (std::size_t j = source_array.size(); j > frame.index; --j)
{
result.push_back(object(
{
{"op", "remove"},
{"path", detail::concat<string_t>(current_path, '/', detail::to_string<string_t>(j - 1))}
}));
}
// add other remaining elements
for (std::size_t i = source_array.size(); i < target_array.size(); ++i)
{
result.push_back(
{
{"op", "add"},
{"path", detail::concat<string_t>(current_path, "/-")},
{"value", target_array[i]}
});
}
}
else
{
if (frame.member != frame.source->cend())
{
const const_iterator it = frame.member;
++frame.member;
if (frame.next_common < frame.common_keys.size() && it.key() == frame.common_keys[frame.next_common])
{
++frame.next_common;
const basic_json& target_value = (*frame.target)[it.key()];
detail::concat_into(current_path, '/', detail::escape(it.key()));
enter(it.value(), target_value); // may push, which invalidates `frame`
if (stack.size() == depth)
{
current_path.resize(path_length);
}
}
else
{
// found a key that is not in target -> remove it
const auto path_key = detail::concat<string_t>(current_path, '/', detail::escape(it.key()));
result.push_back(object(
{
{"op", "remove"}, {"path", path_key}
}));
}
continue;
}
// append the "add" ops for brand-new keys collected when the
// object was entered
result.insert(result.end(), frame.added_ops.begin(), frame.added_ops.end());
}
// this array or object is done: continue with the one it is in
stack.pop_back();
if (!stack.empty())
{
current_path.resize(stack.back().path_length);
}
}
return result;
}
public:
/// @}
////////////////////////////////
+394 -2
View File
@@ -29740,6 +29740,73 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
JSON_HEDLEY_WARN_UNUSED_RESULT
static basic_json diff(const basic_json& source, const basic_json& target,
const string_t& path = "")
{
// Diffing descends into both values once per nesting level and
// compares them with operator== on the way, which recurses as well,
// so values nested deeply enough used to exhaust the call stack.
// Both only descend as far as the source is nested, so a source
// nested no more than diff_depth_limit() levels deep - all but a
// vanishing minority - is diffed recursively as before; deeper ones
// are diffed without the call stack.
if (JSON_HEDLEY_LIKELY(!nesting_exceeds(source, diff_depth_limit())))
{
return diff_recursively(source, target, path);
}
return diff_iteratively(source, target, path);
}
JSON_PRIVATE_UNLESS_TESTED:
/// the nesting depth up to which @ref diff recurses
static constexpr std::size_t diff_depth_limit() noexcept
{
return 128;
}
private:
/*!
@brief whether @a j is nested more than @a limit levels deep
A primitive value is not nested at all, an array or object one level more
than its most deeply nested element. Recurses at most @a limit levels.
*/
static bool nesting_exceeds(const basic_json& j, const std::size_t limit)
{
switch (j.m_data.m_type)
{
case value_t::array:
{
return limit == 0 || std::any_of(j.m_data.m_value.array->cbegin(), j.m_data.m_value.array->cend(),
[limit](const basic_json & element)
{
return element.is_structured() && nesting_exceeds(element, limit - 1);
});
}
case value_t::object:
{
return limit == 0 || std::any_of(j.m_data.m_value.object->cbegin(), j.m_data.m_value.object->cend(),
[limit](const typename object_t::value_type & element)
{
return element.second.is_structured() && nesting_exceeds(element.second, limit - 1);
});
}
case value_t::null:
case value_t::string:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::binary:
case value_t::discarded:
default:
return false;
}
}
/// @ref diff for a @a source nested no more than @ref diff_depth_limit levels deep
static basic_json diff_recursively(const basic_json& source, const basic_json& target,
const string_t& path)
{
// the patch
basic_json result(value_t::array);
@@ -29769,7 +29836,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
while (i < source.size() && i < target.size())
{
// recursive call to compare array values at index i
auto temp_diff = diff(source[i], target[i], detail::concat<string_t>(path, '/', detail::to_string<string_t>(i)));
auto temp_diff = diff_recursively(source[i], target[i], detail::concat<string_t>(path, '/', detail::to_string<string_t>(i)));
result.insert(result.end(), temp_diff.begin(), temp_diff.end());
++i;
}
@@ -29888,7 +29955,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
if (common_it != common_keys_source_order.cend() && it.key() == *common_it)
{
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
auto temp_diff = diff(it.value(), target[it.key()], path_key);
auto temp_diff = diff_recursively(it.value(), target[it.key()], path_key);
result.insert(result.end(), temp_diff.begin(), temp_diff.end());
++common_it;
}
@@ -29972,6 +30039,331 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
return result;
}
/*!
@brief @ref diff without the call stack
Produces the same patch as @ref diff_recursively. Only used for a source
nested more deeply than @ref diff_depth_limit; any arrays and objects
below it that are not nested that deeply are diffed recursively.
*/
static basic_json diff_iteratively(const basic_json& source, const basic_json& target,
const string_t& path)
{
// the patch
basic_json result(value_t::array);
// The arrays and objects being diffed are kept on an explicit stack,
// and every pair of elements is still diffed completely before the
// next one, so the operations come out in the same order as in
// diff_recursively. The path of the values being diffed is kept in
// one buffer that grows and shrinks with the stack, rather than in a
// new string per level.
struct diff_frame
{
diff_frame(const basic_json* source_, const basic_json* target_, const std::size_t path_length_)
: source(source_), target(target_), path_length(path_length_)
{}
/// the values being diffed, both arrays or both objects
const basic_json* source;
const basic_json* target;
/// the length of their path in `current_path`
std::size_t path_length;
/// arrays: the next index to diff
std::size_t index = 0;
/// objects: the next member of source to look at
const_iterator member{};
/// objects: the keys common to both, in source's order
std::vector<typename object_t::key_type> common_keys{};
/// objects: the next entry of common_keys
std::size_t next_common = 0;
/// objects: the "add" operations for keys only target has
basic_json added_ops{};
};
std::vector<diff_frame> stack;
string_t current_path = path;
// diff `s` against `t`, whose path is current_path: primitives,
// values of different types, and objects whose members were reordered
// are handled right away; arrays and other objects get a frame
const auto enter = [&result, &stack, &current_path](const basic_json & s, const basic_json & t)
{
// if the values are the same, there is nothing to do. Arrays and
// objects are not compared up front: comparing them recurses into
// everything below them - equal ones yield no operations anyway.
if ((!s.is_structured() || !t.is_structured()) && s == t)
{
return;
}
if (s.type() != t.type())
{
// different types: replace value
result.push_back(
{
{"op", "replace"}, {"path", current_path}, {"value", t}
});
return;
}
// arrays and objects that are not nested too deeply for the call
// stack are diffed recursively, which can skip equal parts
if (!nesting_exceeds(s, diff_depth_limit()))
{
const basic_json partial = diff_recursively(s, t, current_path);
result.insert(result.end(), partial.begin(), partial.end());
return;
}
switch (s.type())
{
case value_t::array:
{
stack.emplace_back(&s, &t, current_path.size());
return;
}
case value_t::object:
{
// first pass: record, for every source key, whether it is
// common to both objects (in source's iteration order) or
// was deleted (i.e., in source but not in target) -- this is
// a by-product of the t.find() call already needed to
// tell the two cases apart, so it adds no extra lookups. The
// "remove" ops themselves are emitted later, interleaved
// with the per-key diffs in the fast path below, to match
// source's original iteration order (as the original,
// pre-reordering-aware implementation did) instead of
// grouping all removes before all per-key diffs.
std::vector<typename object_t::key_type> common_keys_source_order;
for (auto it = s.cbegin(); it != s.cend(); ++it)
{
if (t.find(it.key()) != t.end())
{
common_keys_source_order.push_back(it.key());
}
}
// second pass: find keys that were added (i.e., in target but
// not in source), and record the keys common to both, in
// target's iteration order -- again a by-product of the
// s.find() call already needed to detect added keys. At
// the same time, determine whether every added key comes
// after every common key in target's order (a precondition
// for the fast path below, which only ever appends new keys
// at the very end): for an object_t whose iteration order is
// a pure function of the key set (e.g. the default std::map,
// which always iterates in sorted key order), the order
// check further below is always true and this whole
// mechanism is effectively a no-op; it only matters for a
// reorderable object_t such as the one backing `ordered_json`.
// patch ops for keys that were added (i.e., in target but not
// in source); built here so the fast path below can reuse
// them without a second s.find() per target key. Only
// used by the fast path -- the slow (reordering) path
// rebuilds "add" ops for every key itself.
std::vector<typename object_t::key_type> common_keys_target_order;
basic_json added_ops(value_t::array);
bool new_keys_form_suffix = true;
bool seen_new_key = false;
for (auto it = t.cbegin(); it != t.cend(); ++it)
{
if (s.find(it.key()) == s.end())
{
seen_new_key = true;
const auto path_key = detail::concat<string_t>(current_path, '/', detail::escape(it.key()));
added_ops.push_back(
{
{"op", "add"}, {"path", path_key},
{"value", it.value()}
});
}
else
{
common_keys_target_order.push_back(it.key());
if (seen_new_key)
{
new_keys_form_suffix = false;
}
}
}
if (common_keys_source_order == common_keys_target_order && new_keys_form_suffix)
{
// fast path: order of common keys already matches (or the
// object_t's iteration order does not depend on
// insertion history), so a plain per-key diff is correct
// and minimal, as before. The frame walks source in
// lockstep with common_keys_source_order, which is, by
// construction, the subsequence of source's keys that
// are common to both objects, in source's iteration
// order -- so a cheap key comparison replaces another
// lookup. Deleted keys are interleaved there too, in
// source's original order, and the "add" ops collected
// above are appended once all members are done.
stack.emplace_back(&s, &t, current_path.size());
stack.back().member = s.cbegin();
stack.back().common_keys = std::move(common_keys_source_order);
stack.back().added_ops = std::move(added_ops);
return;
}
// slow path: the common keys are in a different relative
// order in source and target (only possible for a
// reorderable object_t like ordered_map). Building a
// minimal reordering patch is a nontrivial (LCS-like)
// problem; instead, remove every source key -- both
// deleted keys (which must be removed regardless) and
// common keys (removed so they can be re-added in
// target's order) -- and re-add every key that should
// remain, with its final target value, in target's
// order. basic_json::patch()'s "add" operation on an
// object uses operator[], which appends at the end for a
// vector-backed insertion-ordered map when the key does
// not already exist -- so removing a key and then adding
// it moves it to the end, fixing its position.
for (auto it = s.cbegin(); it != s.cend(); ++it)
{
const auto path_key = detail::concat<string_t>(current_path, '/', detail::escape(it.key()));
result.push_back(object(
{
{"op", "remove"}, {"path", path_key}
}));
}
// add every key that is either common (just removed
// above) or brand new, in target's iteration order, so
// that the final order after applying the patch matches
// target exactly
for (auto it = t.cbegin(); it != t.cend(); ++it)
{
const auto path_key = detail::concat<string_t>(current_path, '/', detail::escape(it.key()));
result.push_back(
{
{"op", "add"}, {"path", path_key},
{"value", it.value()}
});
}
return;
}
case value_t::null:
case value_t::string:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::binary:
case value_t::discarded:
default:
{
// both primitive types: replace value
result.push_back(
{
{"op", "replace"}, {"path", current_path}, {"value", t}
});
return;
}
}
};
enter(source, target);
while (!stack.empty())
{
diff_frame& frame = stack.back();
const std::size_t path_length = frame.path_length;
const std::size_t depth = stack.size();
if (frame.source->is_array())
{
const auto& source_array = *frame.source->m_data.m_value.array;
const auto& target_array = *frame.target->m_data.m_value.array;
// first pass: traverse common elements
if (frame.index < source_array.size() && frame.index < target_array.size())
{
const std::size_t i = frame.index++;
detail::concat_into(current_path, '/', detail::to_string<string_t>(i));
enter(source_array[i], target_array[i]); // may push, which invalidates `frame`
if (stack.size() == depth)
{
current_path.resize(path_length);
}
continue;
}
// We now reached the end of at least one array
// in a second pass, traverse the remaining elements
// remove my remaining elements, highest index first; appending
// in that order avoids the quadratic reinsertion done before
for (std::size_t j = source_array.size(); j > frame.index; --j)
{
result.push_back(object(
{
{"op", "remove"},
{"path", detail::concat<string_t>(current_path, '/', detail::to_string<string_t>(j - 1))}
}));
}
// add other remaining elements
for (std::size_t i = source_array.size(); i < target_array.size(); ++i)
{
result.push_back(
{
{"op", "add"},
{"path", detail::concat<string_t>(current_path, "/-")},
{"value", target_array[i]}
});
}
}
else
{
if (frame.member != frame.source->cend())
{
const const_iterator it = frame.member;
++frame.member;
if (frame.next_common < frame.common_keys.size() && it.key() == frame.common_keys[frame.next_common])
{
++frame.next_common;
const basic_json& target_value = (*frame.target)[it.key()];
detail::concat_into(current_path, '/', detail::escape(it.key()));
enter(it.value(), target_value); // may push, which invalidates `frame`
if (stack.size() == depth)
{
current_path.resize(path_length);
}
}
else
{
// found a key that is not in target -> remove it
const auto path_key = detail::concat<string_t>(current_path, '/', detail::escape(it.key()));
result.push_back(object(
{
{"op", "remove"}, {"path", path_key}
}));
}
continue;
}
// append the "add" ops for brand-new keys collected when the
// object was entered
result.insert(result.end(), frame.added_ops.begin(), frame.added_ops.end());
}
// this array or object is done: continue with the one it is in
stack.pop_back();
if (!stack.empty())
{
current_path.resize(stack.back().path_length);
}
}
return result;
}
public:
/// @}
////////////////////////////////
+140
View File
@@ -15,8 +15,64 @@ using nlohmann::json;
#endif
#include <fstream>
#include <string>
#include "make_test_data_available.hpp"
namespace
{
// alternating objects and arrays nested `depth` levels deep, with members that
// depend on `variant` at some levels, so diffing two variants yields
// operations on many levels: replacing the innermost value, adding, removing,
// and (for ordered_json) reordering members, and changing array lengths
template<typename BasicJsonType>
BasicJsonType nested(const std::size_t depth, const int variant)
{
BasicJsonType value = variant;
for (std::size_t i = 0; i < depth; ++i)
{
if (i % 2 == 0)
{
BasicJsonType object = BasicJsonType::object();
if ((i + static_cast<std::size_t>(variant)) % 7 == 0)
{
object["x"] = i;
}
if (variant == 2 && i % 11 == 0)
{
object["z"] = "z";
}
object["a"] = std::move(value);
if (variant == 1 && i % 5 == 0)
{
object["y"] = 1;
}
value = std::move(object);
}
else
{
BasicJsonType array = BasicJsonType::array({std::move(value)});
if ((i + static_cast<std::size_t>(variant)) % 3 == 0)
{
array.push_back(i);
}
value = std::move(array);
}
}
return value;
}
// a path of `depth` reference tokens, as nested() nests its values
std::string nested_path(const std::size_t depth)
{
std::string path;
for (std::size_t i = depth; i > 0; --i)
{
path += (i - 1) % 2 == 0 ? "/a" : "/0";
}
return path;
}
} // namespace
TEST_CASE("JSON patch")
{
SECTION("examples from RFC 6902")
@@ -1751,3 +1807,87 @@ TEST_CASE("JSON patch - diff emits array removals in descending index order")
CHECK(source.patch(patch) == target);
}
}
TEST_CASE("JSON patch: diff of deeply nested values")
{
SECTION("the diff reproduces the target at every depth")
{
// every depth on either side of the nesting depth up to which diff()
// recurses (basic_json::diff_depth_limit(), 128)
for (std::size_t depth = 0; depth <= 300; ++depth)
{
CAPTURE(depth);
for (int from = 0; from < 3; ++from)
{
for (int to = 0; to < 3; ++to)
{
CAPTURE(from);
CAPTURE(to);
const auto source = nested<json>(depth, from);
const auto target = nested<json>(depth, to);
const auto patch = json::diff(source, target);
CHECK(source.patch(patch) == target);
CHECK(patch.empty() == (from == to));
const auto ordered_source = nested<nlohmann::ordered_json>(depth, from);
const auto ordered_target = nested<nlohmann::ordered_json>(depth, to);
CHECK(ordered_source.patch(nlohmann::ordered_json::diff(ordered_source, ordered_target)) == ordered_target);
}
}
}
}
SECTION("a difference only in the innermost value is one replace operation")
{
for (std::size_t depth = 0; depth <= 300; ++depth)
{
CAPTURE(depth);
json source = 1;
json target = 2;
for (std::size_t i = 0; i < depth; ++i)
{
source = i % 2 == 0 ? json::object({{"a", std::move(source)}}) : json::array({std::move(source)});
target = i % 2 == 0 ? json::object({{"a", std::move(target)}}) : json::array({std::move(target)});
}
CHECK(json::diff(source, target, "/root") == json::array({{{"op", "replace"}, {"path", "/root" + nested_path(depth)}, {"value", 2}}}));
}
}
SECTION("values nested too deeply for the call stack (#5393)")
{
// diff() used to recurse once per nesting level, and compared the
// values with operator== on every level. The values are only
// parsed and diffed, never copied or compared, since those recurse
// too.
const std::size_t depth = 100000;
for (const bool objects :
{
false, true
})
{
CAPTURE(objects);
std::string source_text;
std::string target_text;
std::string equal_text;
std::string path;
for (std::size_t i = 0; i < depth; ++i)
{
source_text += objects ? "{\"a\":" : "[";
path += objects ? "/a" : "/0";
}
target_text = source_text + "2";
equal_text = source_text + "1";
source_text += "1";
const std::string closing(depth, objects ? '}' : ']');
const auto source = json::parse(source_text + closing);
const auto patch = json::diff(source, json::parse(target_text + closing));
REQUIRE(patch.size() == 1);
CHECK(patch[0]["op"] == "replace");
CHECK(patch[0]["path"] == path);
CHECK(patch[0]["value"] == 2);
CHECK(json::diff(source, json::parse(equal_text + closing)).empty());
}
}
}