Compare commits

..
Author SHA1 Message Date
Niels Lohmann 3f6a037b9c Flatten deeply nested values without recursing per nesting level
json_pointer::flatten() called itself once per nesting level, so
flatten() on a value nested deeply enough exhausted the call stack.
#5547 and #5548 fixed merge_patch() and diff() from #5393, but flatten()
was left out.

flatten() now walks the value with an explicit stack and keeps the path
in one buffer that grows and shrinks with it. It has a single code path
and no depth limit: the old version built a new path string per child,
so the iterative one is no slower on shallow values and much faster on
deep ones. The output, including the order of an ordered_json result,
is unchanged.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-10-09 09:16:55 +02:00
5 changed files with 304 additions and 290 deletions
+111 -48
View File
@@ -878,64 +878,127 @@ class json_pointer
@param[in,out] result the result object to insert values to
@note Empty objects or arrays are flattened to `null`.
The value is walked with an explicit stack rather than the call stack, so
arbitrarily deeply nested values can be flattened.
@sa https://github.com/nlohmann/json/issues/5393
*/
template<typename BasicJsonType>
static void flatten(const string_t& reference_string,
const BasicJsonType& value,
BasicJsonType& result)
{
switch (value.type())
using object_const_iterator = typename BasicJsonType::object_t::const_iterator;
// an array or object being walked: the container, the array index or
// object iterator of the next child, and the length of the path of the
// container itself
struct frame
{
case detail::value_t::array:
{
if (value.m_data.m_value.array->empty())
{
// flatten empty array as null
result[reference_string] = nullptr;
}
else
{
// iterate array and use index as a reference string
for (std::size_t i = 0; i < value.m_data.m_value.array->size(); ++i)
{
flatten(detail::concat<string_t>(reference_string, '/', std::to_string(i)),
value.m_data.m_value.array->operator[](i), result);
}
}
break;
}
const BasicJsonType* container;
std::size_t index;
object_const_iterator member;
std::size_t path_length;
};
case detail::value_t::object:
{
if (value.m_data.m_value.object->empty())
{
// flatten empty object as null
result[reference_string] = nullptr;
}
else
{
// iterate object and use keys as reference string
for (const auto& element : *value.m_data.m_value.object)
{
flatten(detail::concat<string_t>(reference_string, '/', detail::escape(element.first)), element.second, result);
}
}
break;
}
// The containers being flattened are kept on an explicit stack, and
// every child is flattened completely before the next one, so the
// entries come out in the same order as with a recursive walk. The
// path of the value being flattened is kept in one buffer that grows
// and shrinks with the stack, rather than in a new string per level.
std::vector<frame> stack;
string_t path = reference_string;
case detail::value_t::null:
case detail::value_t::string:
case detail::value_t::boolean:
case detail::value_t::number_integer:
case detail::value_t::number_unsigned:
case detail::value_t::number_float:
case detail::value_t::binary:
case detail::value_t::discarded:
default:
// flatten `v`, whose path is `path`: primitives and empty containers
// are added to the result right away; other containers get a frame
const auto enter = [&stack, &path, &result](const BasicJsonType & v)
{
switch (v.type())
{
// add a primitive value with its reference string
result[reference_string] = value;
break;
case detail::value_t::array:
{
if (v.m_data.m_value.array->empty())
{
// flatten empty array as null
result[path] = nullptr;
}
else
{
stack.push_back({&v, 0, object_const_iterator(), path.size()});
}
return;
}
case detail::value_t::object:
{
if (v.m_data.m_value.object->empty())
{
// flatten empty object as null
result[path] = nullptr;
}
else
{
stack.push_back({&v, 0, v.m_data.m_value.object->begin(), path.size()});
}
return;
}
case detail::value_t::null:
case detail::value_t::string:
case detail::value_t::boolean:
case detail::value_t::number_integer:
case detail::value_t::number_unsigned:
case detail::value_t::number_float:
case detail::value_t::binary:
case detail::value_t::discarded:
default:
{
// add a primitive value with its reference string
result[path] = v;
return;
}
}
};
enter(value);
while (!stack.empty())
{
// the frame is changed through stack.back(): enter() may push a
// frame, which would invalidate a reference to it
const BasicJsonType* const container = stack.back().container;
// drop the path of the previous child
path.resize(stack.back().path_length);
if (container->is_array())
{
const auto& array = *container->m_data.m_value.array;
const std::size_t i = stack.back().index;
if (i == array.size())
{
stack.pop_back();
continue;
}
// iterate array and use index as a reference string
++stack.back().index;
detail::concat_into(path, '/', detail::to_string<string_t>(i));
enter(array[i]);
}
else
{
const object_const_iterator it = stack.back().member;
if (it == container->m_data.m_value.object->end())
{
stack.pop_back();
continue;
}
// iterate object and use keys as reference string
++stack.back().member;
detail::concat_into(path, '/', detail::escape(it->first));
enter(it->second);
}
}
}
@@ -85,12 +85,6 @@ template<typename BasicJsonType, typename CharType, typename OutputSinkType = ou
class binary_writer
{
using string_t = typename BasicJsonType::string_t;
/// an object key as string_t: a reference when object_t::key_type already is
/// string_t, otherwise a converted copy that outlives sanitize_utf8_for_write's result
using object_key_string_t = typename std::conditional <
std::is_same<typename BasicJsonType::object_t::key_type, string_t>::value,
const string_t&, string_t >::type;
using binary_t = typename BasicJsonType::binary_t;
using number_float_t = typename BasicJsonType::number_float_t;
@@ -825,10 +819,8 @@ class binary_writer
for (const auto& el : *j.m_data.m_value.object)
{
// a converted key must outlive the reference returned by sanitize_utf8_for_write
const object_key_string_t key_string = el.first;
string_t storage;
const string_t& key = sanitize_utf8_for_write(key_string, j, storage);
const string_t& key = sanitize_utf8_for_write(el.first, j, storage);
write_number_with_ubjson_prefix(key.size(), true, use_bjdata);
oa.write_characters(
reinterpret_cast<const CharType*>(key.data()),
@@ -1374,10 +1366,8 @@ class binary_writer
continue;
}
// a converted key must outlive the reference returned by sanitize_utf8_for_write
const object_key_string_t key_string = current.object_it->first;
string_t storage;
const string_t& key = sanitize_utf8_for_write(key_string, j, storage);
const string_t& key = sanitize_utf8_for_write(current.object_it->first, j, storage);
write_number_with_ubjson_prefix(key.size(), true, use_bjdata);
oa.write_characters(
reinterpret_cast<const CharType*>(key.data()),
@@ -2740,11 +2730,6 @@ class binary_writer
itself in every case but a sanitized `replace`/`ignore` one, so @a
storage must outlive the returned reference only then.
@a s must be an lvalue that outlives the returned reference. An object key
whose `key_type` is not @ref string_t must therefore first be converted
into a named string_t (see @ref object_key_string_t); the deleted overload
below enforces this at compile time.
@param[in] s the string (value or object key) to write
@param[in] context the value @a s belongs to (for diagnostics)
@param[out] storage backing storage for a sanitized copy
@@ -2774,10 +2759,6 @@ class binary_writer
}
}
/// deleted: anything but a string_t would bind a temporary that dies before the returned reference is used
template < typename T, enable_if_t < !std::is_same<T, string_t>::value, int > = 0 >
const string_t& sanitize_utf8_for_write(const T& /*s*/, const BasicJsonType& /*context*/, string_t& /*storage*/) const = delete;
/*!
@brief write an integer in the shortest encoding
+113 -69
View File
@@ -20796,64 +20796,127 @@ class json_pointer
@param[in,out] result the result object to insert values to
@note Empty objects or arrays are flattened to `null`.
The value is walked with an explicit stack rather than the call stack, so
arbitrarily deeply nested values can be flattened.
@sa https://github.com/nlohmann/json/issues/5393
*/
template<typename BasicJsonType>
static void flatten(const string_t& reference_string,
const BasicJsonType& value,
BasicJsonType& result)
{
switch (value.type())
using object_const_iterator = typename BasicJsonType::object_t::const_iterator;
// an array or object being walked: the container, the array index or
// object iterator of the next child, and the length of the path of the
// container itself
struct frame
{
case detail::value_t::array:
{
if (value.m_data.m_value.array->empty())
{
// flatten empty array as null
result[reference_string] = nullptr;
}
else
{
// iterate array and use index as a reference string
for (std::size_t i = 0; i < value.m_data.m_value.array->size(); ++i)
{
flatten(detail::concat<string_t>(reference_string, '/', std::to_string(i)),
value.m_data.m_value.array->operator[](i), result);
}
}
break;
}
const BasicJsonType* container;
std::size_t index;
object_const_iterator member;
std::size_t path_length;
};
case detail::value_t::object:
{
if (value.m_data.m_value.object->empty())
{
// flatten empty object as null
result[reference_string] = nullptr;
}
else
{
// iterate object and use keys as reference string
for (const auto& element : *value.m_data.m_value.object)
{
flatten(detail::concat<string_t>(reference_string, '/', detail::escape(element.first)), element.second, result);
}
}
break;
}
// The containers being flattened are kept on an explicit stack, and
// every child is flattened completely before the next one, so the
// entries come out in the same order as with a recursive walk. The
// path of the value being flattened is kept in one buffer that grows
// and shrinks with the stack, rather than in a new string per level.
std::vector<frame> stack;
string_t path = reference_string;
case detail::value_t::null:
case detail::value_t::string:
case detail::value_t::boolean:
case detail::value_t::number_integer:
case detail::value_t::number_unsigned:
case detail::value_t::number_float:
case detail::value_t::binary:
case detail::value_t::discarded:
default:
// flatten `v`, whose path is `path`: primitives and empty containers
// are added to the result right away; other containers get a frame
const auto enter = [&stack, &path, &result](const BasicJsonType & v)
{
switch (v.type())
{
// add a primitive value with its reference string
result[reference_string] = value;
break;
case detail::value_t::array:
{
if (v.m_data.m_value.array->empty())
{
// flatten empty array as null
result[path] = nullptr;
}
else
{
stack.push_back({&v, 0, object_const_iterator(), path.size()});
}
return;
}
case detail::value_t::object:
{
if (v.m_data.m_value.object->empty())
{
// flatten empty object as null
result[path] = nullptr;
}
else
{
stack.push_back({&v, 0, v.m_data.m_value.object->begin(), path.size()});
}
return;
}
case detail::value_t::null:
case detail::value_t::string:
case detail::value_t::boolean:
case detail::value_t::number_integer:
case detail::value_t::number_unsigned:
case detail::value_t::number_float:
case detail::value_t::binary:
case detail::value_t::discarded:
default:
{
// add a primitive value with its reference string
result[path] = v;
return;
}
}
};
enter(value);
while (!stack.empty())
{
// the frame is changed through stack.back(): enter() may push a
// frame, which would invalidate a reference to it
const BasicJsonType* const container = stack.back().container;
// drop the path of the previous child
path.resize(stack.back().path_length);
if (container->is_array())
{
const auto& array = *container->m_data.m_value.array;
const std::size_t i = stack.back().index;
if (i == array.size())
{
stack.pop_back();
continue;
}
// iterate array and use index as a reference string
++stack.back().index;
detail::concat_into(path, '/', detail::to_string<string_t>(i));
enter(array[i]);
}
else
{
const object_const_iterator it = stack.back().member;
if (it == container->m_data.m_value.object->end())
{
stack.pop_back();
continue;
}
// iterate object and use keys as reference string
++stack.back().member;
detail::concat_into(path, '/', detail::escape(it->first));
enter(it->second);
}
}
}
@@ -21590,12 +21653,6 @@ template<typename BasicJsonType, typename CharType, typename OutputSinkType = ou
class binary_writer
{
using string_t = typename BasicJsonType::string_t;
/// an object key as string_t: a reference when object_t::key_type already is
/// string_t, otherwise a converted copy that outlives sanitize_utf8_for_write's result
using object_key_string_t = typename std::conditional <
std::is_same<typename BasicJsonType::object_t::key_type, string_t>::value,
const string_t&, string_t >::type;
using binary_t = typename BasicJsonType::binary_t;
using number_float_t = typename BasicJsonType::number_float_t;
@@ -22330,10 +22387,8 @@ class binary_writer
for (const auto& el : *j.m_data.m_value.object)
{
// a converted key must outlive the reference returned by sanitize_utf8_for_write
const object_key_string_t key_string = el.first;
string_t storage;
const string_t& key = sanitize_utf8_for_write(key_string, j, storage);
const string_t& key = sanitize_utf8_for_write(el.first, j, storage);
write_number_with_ubjson_prefix(key.size(), true, use_bjdata);
oa.write_characters(
reinterpret_cast<const CharType*>(key.data()),
@@ -22879,10 +22934,8 @@ class binary_writer
continue;
}
// a converted key must outlive the reference returned by sanitize_utf8_for_write
const object_key_string_t key_string = current.object_it->first;
string_t storage;
const string_t& key = sanitize_utf8_for_write(key_string, j, storage);
const string_t& key = sanitize_utf8_for_write(current.object_it->first, j, storage);
write_number_with_ubjson_prefix(key.size(), true, use_bjdata);
oa.write_characters(
reinterpret_cast<const CharType*>(key.data()),
@@ -24245,11 +24298,6 @@ class binary_writer
itself in every case but a sanitized `replace`/`ignore` one, so @a
storage must outlive the returned reference only then.
@a s must be an lvalue that outlives the returned reference. An object key
whose `key_type` is not @ref string_t must therefore first be converted
into a named string_t (see @ref object_key_string_t); the deleted overload
below enforces this at compile time.
@param[in] s the string (value or object key) to write
@param[in] context the value @a s belongs to (for diagnostics)
@param[out] storage backing storage for a sanitized copy
@@ -24279,10 +24327,6 @@ class binary_writer
}
}
/// deleted: anything but a string_t would bind a temporary that dies before the returned reference is used
template < typename T, enable_if_t < !std::is_same<T, string_t>::value, int > = 0 >
const string_t& sanitize_utf8_for_write(const T& /*s*/, const BasicJsonType& /*context*/, string_t& /*storage*/) const = delete;
/*!
@brief write an integer in the shortest encoding
@@ -12,10 +12,7 @@
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <map>
#include <memory>
#include <string>
#include <utility>
#include <vector>
namespace
@@ -58,49 +55,6 @@ std::string dump_and_parse(const std::string& raw, eh error_handler)
return json::parse(json(raw).dump(-1, ' ', false, error_handler)).get<std::string>();
}
// an object key type that is not string_t, but converts implicitly to it
class converting_key
{
public:
converting_key(const char* s) : m_value(s) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions)
converting_key(std::string s) : m_value(std::move(s)) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions)
// the conversion yields a temporary string_t
operator std::string() const // NOLINT(google-explicit-constructor,hicpp-explicit-conversions)
{
return m_value;
}
// read by the exception messages when JSON_DIAGNOSTICS is enabled
const char* data() const noexcept
{
return m_value.data();
}
friend bool operator<(const converting_key& lhs, const converting_key& rhs)
{
return lhs.m_value < rhs.m_value;
}
private:
std::string m_value;
};
// ObjectType using converting_key; the Key template argument is ignored
template<typename Key, typename Value, typename Compare, typename Allocator>
class converting_key_object : public std::map<converting_key, Value, std::less<converting_key>, // NOLINT(modernize-use-transparent-functors)
typename std::allocator_traits<Allocator>::template rebind_alloc<std::pair<const converting_key, Value>>>
{
using base_type = std::map<converting_key, Value, std::less<converting_key>, // NOLINT(modernize-use-transparent-functors)
typename std::allocator_traits<Allocator>::template rebind_alloc<std::pair<const converting_key, Value>>>;
public:
using base_type::base_type;
using base_type::operator=;
};
using converting_key_json = nlohmann::basic_json<converting_key_object>;
} // namespace
TEST_CASE("UTF-8 error_handler for the binary readers and writers")
@@ -416,109 +370,3 @@ TEST_CASE("UTF-8 error_handler for the binary readers and writers")
CHECK(json::from_bson(bson_bytes)["k"].get<std::string>() == ill_formed_cases()[0].bytes);
}
}
// The UBJSON and BJData writers bind the (possibly sanitized) key to a const
// string_t&. If key_type is not string_t but converts to it, the converted
// temporary must outlive that reference; this was a use-after-scope found by
// AddressSanitizer. Keys exceed the small string optimization on purpose.
TEST_CASE("UBJSON and BJData writers with an object_t whose key_type is not string_t")
{
const std::string long_prefix(70, 'k');
SECTION("well-formed keys, every error_handler")
{
const std::string key1 = long_prefix + "-first";
const std::string key2 = long_prefix + "-second";
converting_key_json::object_t o;
o.emplace(converting_key(key1), 1);
o.emplace(converting_key(key2), "value");
const converting_key_json v(std::move(o));
json expected;
expected[key1] = 1;
expected[key2] = "value";
const bool combos[3][2] = {{false, false}, {true, false}, {true, true}};
for (const auto h : all_handlers())
{
CAPTURE(static_cast<int>(h))
for (const auto& combo : combos)
{
const bool use_count = combo[0];
const bool use_type = combo[1];
CAPTURE(use_count)
CAPTURE(use_type)
CHECK(json::from_ubjson(converting_key_json::to_ubjson(v, use_count, use_type, h)) == expected);
CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, use_count, use_type, json::bjdata_version_t::draft2, h)) == expected);
CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, use_count, use_type, json::bjdata_version_t::draft3, h)) == expected);
}
}
}
SECTION("ill-formed keys")
{
for (const auto& c : ill_formed_cases())
{
CAPTURE(c.name)
const std::string key = long_prefix + c.bytes;
converting_key_json::object_t o;
o.emplace(converting_key(key), 1);
const converting_key_json v(std::move(o));
CHECK_THROWS_AS(converting_key_json::to_ubjson(v, false, false, eh::strict), converting_key_json::type_error&);
CHECK_THROWS_AS(converting_key_json::to_bjdata(v, false, false, json::bjdata_version_t::draft2, eh::strict), converting_key_json::type_error&);
for (const auto h :
{
eh::replace, eh::ignore
})
{
CAPTURE(static_cast<int>(h))
const std::string expected = dump_and_parse(key, h);
CHECK(json::from_ubjson(converting_key_json::to_ubjson(v, false, false, h)).begin().key() == expected);
CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, false, false, json::bjdata_version_t::draft2, h)).begin().key() == expected);
}
CHECK(json::from_ubjson(converting_key_json::to_ubjson(v, false, false, eh::keep)).begin().key() == key);
CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, false, false, json::bjdata_version_t::draft2, eh::keep)).begin().key() == key);
}
}
SECTION("nested deeper than the recursion limit")
{
// wrap the previous value, innermost first
converting_key_json v = 42;
json expected = 42;
for (int i = 199; i >= 0; --i)
{
const std::string key = "level-" + std::to_string(i) + "-" + std::string(64, 'x');
converting_key_json::object_t o;
o.emplace(converting_key(key), std::move(v));
v = converting_key_json(std::move(o));
json e;
e[key] = std::move(expected);
expected = std::move(e);
}
for (const auto h : all_handlers())
{
CAPTURE(static_cast<int>(h))
for (const bool use_count :
{
false, true
})
{
CAPTURE(use_count)
CHECK(json::from_ubjson(converting_key_json::to_ubjson(v, use_count, false, h)) == expected);
CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, use_count, false, json::bjdata_version_t::draft2, h)) == expected);
}
}
}
}
+78
View File
@@ -939,3 +939,81 @@ TEST_CASE("unescaping keeps a '~' that does not start an escape sequence")
nlohmann::detail::unescape(s);
CHECK(s == "~/~");
}
TEST_CASE("flatten of structured values")
{
SECTION("values nested too deeply for the call stack (#5393)")
{
// flatten() used to recurse once per nesting level
const std::size_t depth = 100000;
for (const bool objects :
{
false, true
})
{
CAPTURE(objects)
std::string text;
std::string path;
for (std::size_t i = 0; i < depth; ++i)
{
text += objects ? "{\"a\":" : "[";
path += objects ? "/a" : "/0";
}
text += "0";
text += std::string(depth, objects ? '}' : ']');
const auto value = json::parse(text);
const auto flat = value.flatten();
REQUIRE(flat.size() == 1);
REQUIRE(flat.begin().key().size() == path.size());
CHECK(flat.begin().key() == path);
CHECK(flat.begin().value() == 0);
// unflatten() is not iterative: it takes time and memory
// quadratic in the depth, so it is only roundtripped for a
// moderate depth
std::string small_text;
for (std::size_t i = 0; i < 500; ++i)
{
small_text += objects ? "{\"a\":" : "[";
}
small_text += "0";
small_text += std::string(500, objects ? '}' : ']');
const auto small_value = json::parse(small_text);
CHECK(small_value.flatten().unflatten() == small_value);
}
}
SECTION("objects and arrays interleaved")
{
const json value =
{
{"a", {1, {{"b", json::array()}, {"c", json::object()}}, json::array({{{"x~/", {true, nullptr}}}})}},
{"a/b", {{"~", 1}}},
{"z", "s"}
};
const json expected =
{
{"/a/0", 1},
{"/a/1/b", nullptr},
{"/a/1/c", nullptr},
{"/a/2/0/x~0~1/0", true},
{"/a/2/0/x~0~1/1", nullptr},
{"/a~1b/~0", 1},
{"/z", "s"}
};
CHECK(value.flatten() == expected);
}
SECTION("order of the entries of an ordered_json")
{
const auto value = nlohmann::ordered_json::parse(
R"({"z":"s","a/b":{"~":1,"k":[]},"a":[1,{"c":{},"b":[]},[{"x~/":[true,null],"w":2}]]})");
const auto flat = value.flatten();
CHECK(flat.dump() ==
R"({"/z":"s","/a~1b/~0":1,"/a~1b/k":null,"/a/0":1,"/a/1/c":null,"/a/1/b":null,"/a/2/0/x~0~1/0":true,"/a/2/0/x~0~1/1":null,"/a/2/0/w":2})");
}
}