From 9be501874881aac92e0dceb8bfb07e9e2f1f3bdb Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 23 Sep 2026 21:53:09 +0200 Subject: [PATCH] Hash deeply nested values without recursing per nesting level std::hash hashed an array or object by hashing each element, which called detail::hash again once per nesting level. A value nested deeply enough - 50,000 levels of objects on an 8 MiB stack - exhausted the call stack and terminated the process. parse() accepts such values without complaint, since the parser is iterative, and a parsed value is hashed wherever it is used as a key in an unordered container. Bound the descent the same way dump() does: detail::hash takes the nesting level, and once hash_depth_limit() (128) levels have been entered, hash_iteratively() hashes what is left on an explicit stack. It combines the seeds in exactly the same order, so hash values are unchanged. A value nested less deeply than the bound is hashed by the same code as before, without allocating, and is as fast as before. Tests check that every depth up to twice the bound hashes exactly like the recursive definition of the hash, and that values nested 100,000 levels deep hash without crashing. Fixes #5545 for std::hash. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/hash.hpp | 107 ++++++++++++++++++++++++++++- single_include/nlohmann/json.hpp | 107 ++++++++++++++++++++++++++++- tests/src/unit-hash.cpp | 113 +++++++++++++++++++++++++++++++ 3 files changed, 321 insertions(+), 6 deletions(-) diff --git a/include/nlohmann/detail/hash.hpp b/include/nlohmann/detail/hash.hpp index be8063f89..0b3d1e2a5 100644 --- a/include/nlohmann/detail/hash.hpp +++ b/include/nlohmann/detail/hash.hpp @@ -11,6 +11,7 @@ #include // uint8_t #include // size_t #include // hash +#include // vector #include #include @@ -26,6 +27,16 @@ inline std::size_t combine(std::size_t seed, std::size_t h) noexcept return seed; } +/// the number of levels @ref hash descends into before handing over to +/// @ref hash_iteratively +constexpr std::size_t hash_depth_limit() noexcept +{ + return 128; +} + +template +std::size_t hash_iteratively(const BasicJsonType& j); + /*! @brief hash a JSON value @@ -33,12 +44,21 @@ The hash function tries to rely on std::hash where possible. Furthermore, the type of the JSON value is taken into account to have different hash values for null, 0, 0U, and false, etc. +Hashing an array or an object hashes its elements, which used to call this +function again once per nesting level, so a value nested deeply enough +exhausted the call stack and terminated the process. The descent is bounded +here: once @ref hash_depth_limit levels have been entered, @ref +hash_iteratively hashes what is left without the call stack. A value nested +less deeply than that - all but a vanishing minority - is hashed exactly as +before, without allocating. + @tparam BasicJsonType basic_json specialization @param j JSON value to hash +@param depth nesting level of @a j, counted from the value passed by the caller @return hash value of j */ template -std::size_t hash(const BasicJsonType& j) +std::size_t hash(const BasicJsonType& j, const std::size_t depth = 0) { using string_t = typename BasicJsonType::string_t; using number_integer_t = typename BasicJsonType::number_integer_t; @@ -56,22 +76,32 @@ std::size_t hash(const BasicJsonType& j) case BasicJsonType::value_t::object: { + if (JSON_HEDLEY_UNLIKELY(depth >= hash_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j.items()) { const auto h = std::hash {}(element.key()); seed = combine(seed, h); - seed = combine(seed, hash(element.value())); + seed = combine(seed, hash(element.value(), depth + 1)); } return seed; } case BasicJsonType::value_t::array: { + if (JSON_HEDLEY_UNLIKELY(depth >= hash_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j) { - seed = combine(seed, hash(element)); + seed = combine(seed, hash(element, depth + 1)); } return seed; } @@ -127,5 +157,76 @@ std::size_t hash(const BasicJsonType& j) } } +/// an array or object whose elements @ref hash_iteratively is hashing +template +struct hash_frame +{ + hash_frame(const BasicJsonType* value_, std::size_t seed_) + : value(value_), position(value_->cbegin()), seed(seed_) + {} + + const BasicJsonType* value; + typename BasicJsonType::const_iterator position; + std::size_t seed; +}; + +/*! +@brief hash the array or object @a j without the call stack + +Computes the same value as @ref hash, keeping the arrays and objects it has +entered on an explicit stack instead of descending into them. Only reached for +values nested deeper than @ref hash_depth_limit. + +@tparam BasicJsonType basic_json specialization +@param j array or object to hash +@return hash value of j +*/ +template +std::size_t hash_iteratively(const BasicJsonType& j) +{ + using string_t = typename BasicJsonType::string_t; + + std::vector> stack; + stack.emplace_back(&j, combine(static_cast(j.type()), j.size())); + + while (true) + { + hash_frame& frame = stack.back(); + + if (frame.position == frame.value->cend()) + { + // all elements are hashed: fold this value's hash into its parent's + // seed, exactly where the recursive version returns it + const std::size_t h = frame.seed; + stack.pop_back(); + if (stack.empty()) + { + return h; + } + stack.back().seed = combine(stack.back().seed, h); + continue; + } + + if (frame.value->is_object()) + { + frame.seed = combine(frame.seed, std::hash {}(frame.position.key())); + } + + // read the element and advance before entering it: entering can + // reallocate the stack and so invalidate `frame` + const BasicJsonType& element = *frame.position; + ++frame.position; + + if (element.is_structured()) + { + stack.emplace_back(&element, combine(static_cast(element.type()), element.size())); + } + else + { + frame.seed = combine(frame.seed, hash(element)); + } + } +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index c11fa27e5..8769d6e30 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7028,6 +7028,7 @@ NLOHMANN_JSON_NAMESPACE_END #include // uint8_t #include // size_t #include // hash +#include // vector // #include @@ -7045,6 +7046,16 @@ inline std::size_t combine(std::size_t seed, std::size_t h) noexcept return seed; } +/// the number of levels @ref hash descends into before handing over to +/// @ref hash_iteratively +constexpr std::size_t hash_depth_limit() noexcept +{ + return 128; +} + +template +std::size_t hash_iteratively(const BasicJsonType& j); + /*! @brief hash a JSON value @@ -7052,12 +7063,21 @@ The hash function tries to rely on std::hash where possible. Furthermore, the type of the JSON value is taken into account to have different hash values for null, 0, 0U, and false, etc. +Hashing an array or an object hashes its elements, which used to call this +function again once per nesting level, so a value nested deeply enough +exhausted the call stack and terminated the process. The descent is bounded +here: once @ref hash_depth_limit levels have been entered, @ref +hash_iteratively hashes what is left without the call stack. A value nested +less deeply than that - all but a vanishing minority - is hashed exactly as +before, without allocating. + @tparam BasicJsonType basic_json specialization @param j JSON value to hash +@param depth nesting level of @a j, counted from the value passed by the caller @return hash value of j */ template -std::size_t hash(const BasicJsonType& j) +std::size_t hash(const BasicJsonType& j, const std::size_t depth = 0) { using string_t = typename BasicJsonType::string_t; using number_integer_t = typename BasicJsonType::number_integer_t; @@ -7075,22 +7095,32 @@ std::size_t hash(const BasicJsonType& j) case BasicJsonType::value_t::object: { + if (JSON_HEDLEY_UNLIKELY(depth >= hash_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j.items()) { const auto h = std::hash {}(element.key()); seed = combine(seed, h); - seed = combine(seed, hash(element.value())); + seed = combine(seed, hash(element.value(), depth + 1)); } return seed; } case BasicJsonType::value_t::array: { + if (JSON_HEDLEY_UNLIKELY(depth >= hash_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j) { - seed = combine(seed, hash(element)); + seed = combine(seed, hash(element, depth + 1)); } return seed; } @@ -7146,6 +7176,77 @@ std::size_t hash(const BasicJsonType& j) } } +/// an array or object whose elements @ref hash_iteratively is hashing +template +struct hash_frame +{ + hash_frame(const BasicJsonType* value_, std::size_t seed_) + : value(value_), position(value_->cbegin()), seed(seed_) + {} + + const BasicJsonType* value; + typename BasicJsonType::const_iterator position; + std::size_t seed; +}; + +/*! +@brief hash the array or object @a j without the call stack + +Computes the same value as @ref hash, keeping the arrays and objects it has +entered on an explicit stack instead of descending into them. Only reached for +values nested deeper than @ref hash_depth_limit. + +@tparam BasicJsonType basic_json specialization +@param j array or object to hash +@return hash value of j +*/ +template +std::size_t hash_iteratively(const BasicJsonType& j) +{ + using string_t = typename BasicJsonType::string_t; + + std::vector> stack; + stack.emplace_back(&j, combine(static_cast(j.type()), j.size())); + + while (true) + { + hash_frame& frame = stack.back(); + + if (frame.position == frame.value->cend()) + { + // all elements are hashed: fold this value's hash into its parent's + // seed, exactly where the recursive version returns it + const std::size_t h = frame.seed; + stack.pop_back(); + if (stack.empty()) + { + return h; + } + stack.back().seed = combine(stack.back().seed, h); + continue; + } + + if (frame.value->is_object()) + { + frame.seed = combine(frame.seed, std::hash {}(frame.position.key())); + } + + // read the element and advance before entering it: entering can + // reallocate the stack and so invalidate `frame` + const BasicJsonType& element = *frame.position; + ++frame.position; + + if (element.is_structured()) + { + stack.emplace_back(&element, combine(static_cast(element.type()), element.size())); + } + else + { + frame.seed = combine(frame.seed, hash(element)); + } + } +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/tests/src/unit-hash.cpp b/tests/src/unit-hash.cpp index c161efa6e..3f2900443 100644 --- a/tests/src/unit-hash.cpp +++ b/tests/src/unit-hash.cpp @@ -13,6 +13,78 @@ using json = nlohmann::json; using ordered_json = nlohmann::ordered_json; #include +#include + +namespace +{ +// how detail::hash defines the hash of an array or object: the seeds of the +// elements, combined in order. Recursive, so only usable on values nested a +// few hundred levels deep - which is exactly what is needed to check that the +// iterative path taken below detail::hash_depth_limit() computes the same. +template +std::size_t reference_hash(const BasicJsonType& j) +{ + using nlohmann::detail::combine; + using string_t = typename BasicJsonType::string_t; + + if (!j.is_structured()) + { + return std::hash {}(j); + } + + auto seed = combine(static_cast(j.type()), j.size()); + for (const auto& element : j.items()) + { + if (j.is_object()) + { + seed = combine(seed, std::hash {}(element.key())); + } + seed = combine(seed, reference_hash(element.value())); + } + return seed; +} + +// a value nested `depth` levels deep, with siblings on every level +template +BasicJsonType nested(const std::size_t depth, const bool objects) +{ + BasicJsonType value = "leaf"; + for (std::size_t i = 0; i < depth; ++i) + { + if (objects) + { + value = BasicJsonType{{"before", i}, {"nested", std::move(value)}, {"after", {i, "x"}}}; + } + else + { + value = BasicJsonType::array({i, std::move(value), BasicJsonType::object({{"k", i}})}); + } + } + return value; +} + +std::string nested_text(const std::size_t depth, const bool objects) +{ + std::string text; + if (objects) + { + text.reserve(6 * depth + 1); + for (std::size_t i = 0; i < depth; ++i) + { + text += "{\"a\":"; + } + text += "1"; + text.append(depth, '}'); + } + else + { + text.assign(depth, '['); + text += "1"; + text.append(depth, ']'); + } + return text; +} +} // namespace TEST_CASE("hash") { @@ -111,3 +183,44 @@ TEST_CASE("hash") CHECK(hashes.size() == 21); } + +TEST_CASE("hash of deeply nested values") +{ + SECTION("hashing past the descent bound computes the same values") + { + // every depth on either side of where the iterative path takes over + for (std::size_t depth = 0; depth <= 2 * nlohmann::detail::hash_depth_limit() + 10; ++depth) + { + CAPTURE(depth); + const auto arrays = nested(depth, false); + const auto objects = nested(depth, true); + const auto ordered = nested(depth, true); + CHECK(std::hash {}(arrays) == reference_hash(arrays)); + CHECK(std::hash {}(objects) == reference_hash(objects)); + CHECK(std::hash {}(ordered) == reference_hash(ordered)); + } + } + + SECTION("values nested too deeply for the call stack (#5545)") + { + // recursing once per level used to exhaust the call stack here; the + // values are only parsed and hashed, never copied or compared, since + // those recurse as well + const std::size_t depth = 100000; + for (const bool objects : + { + false, true + }) + { + CAPTURE(objects); + const auto text = nested_text(depth, objects); + const auto a = json::parse(text); + const auto b = json::parse(text); + CHECK(std::hash {}(a) == std::hash {}(b)); + + const auto c = ordered_json::parse(text); + const auto d = ordered_json::parse(text); + CHECK(std::hash {}(c) == std::hash {}(d)); + } + } +}