Size the dump buffer of a small value by its own extent

The source extent of a value was read from the next node, falling back to the rest of the document when that node held a decoded string. dump() of a small value could thus allocate a buffer as large as the document. Skip a few such nodes, cap the fallback estimate, and shrink a buffer that is much larger than its output.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann committed 2026-10-09 16:11:14 +02:00
1 parent 8860bf6f3a
commit 2ba331e4d0
3 files changed
+62 -8

No files matched your search

+7 -1
View File
@@ -44,7 +44,13 @@ class output_buffer
void finish()
{
m_out.resize(static_cast<std::size_t>(m_pos - m_out.data()));
const auto size = static_cast<std::size_t>(m_pos - m_out.data());
m_out.resize(size);
// do not keep a buffer that was sized for a much larger output
if (m_out.capacity() > 1024 && m_out.capacity() / 2 > size)
{
m_out.shrink_to_fit();
}
}
NLOHMANN_VIEW_ALWAYS_INLINE void reserve(std::size_t n)
+21 -7
View File
@@ -24,6 +24,7 @@
#ifndef INCLUDE_NLOHMANN_JSON_VIEW_HPP_
#define INCLUDE_NLOHMANN_JSON_VIEW_HPP_
#include <algorithm> // min
#include <cstddef> // size_t
#include <cstring> // memcpy, strlen
#include <iterator> // distance, input_iterator_tag, iterator_traits
@@ -682,20 +683,33 @@ class basic_json_view
}
/// the number of source bytes of this value (estimated for values with
/// decoded strings)
/// decoded strings); the estimate sizes the output buffer of dump()
std::size_t source_extent() const noexcept
{
const node* const next = document_data::after(m_node);
const bool in_source = (m_node->flags & detail::view::node_flags::storage) == 0;
if (!in_source)
const node* const end = m_doc->tape + m_doc->tape_size;
if ((m_node->flags & detail::view::node_flags::storage) != 0)
{
return m_node->len;
}
if (next != m_doc->tape + m_doc->tape_size && (next->flags & detail::view::node_flags::storage) == 0 && next->off >= m_node->off)
// the value ends where the next node in the source begins; nodes
// with decoded strings (their offset is in the arena) are skipped,
// but only a few of them, to keep the walk short
const node* next = document_data::after(m_node);
for (int skipped = 0; next != end && skipped < 16; ++skipped, ++next)
{
return next->off - m_node->off;
if ((next->flags & detail::view::node_flags::storage) == 0)
{
return next->off >= m_node->off ? next->off - m_node->off : 0;
}
}
return m_doc->size - m_node->off;
if (next == end)
{
return m_doc->size - m_node->off;
}
// the end is unknown: assume a few bytes per node, the output buffer
// grows should the value be larger
const auto nodes = static_cast<std::size_t>(document_data::after(m_node) - m_node);
return (std::min)(m_doc->size - m_node->off, static_cast<std::size_t>(1024) + nodes * 16);
}
/// the value of the first member with this key, or a discarded view
+34
View File
@@ -1183,6 +1183,40 @@ TEST_CASE("json_view dump")
CHECK(json_document::parse(deep).root().dump() == deep);
}
SECTION("the output buffer of a small value is small")
{
// an escaped key after the value: its node lies in the arena, so the
// source extent of the value cannot be read from the next node
const std::string big(100000, 'a');
const std::string text = R"({"small":1,"list":[1,2,3],"k\n":")" + big + R"("})";
const json_document d = json_document::parse(text);
const auto small = d.root()["small"].dump();
CHECK(small == "1");
CHECK(small.capacity() < 4096);
const auto list = d.root()["list"].dump();
CHECK(list == "[1,2,3]");
CHECK(list.capacity() < 4096);
CHECK(d.root()["list"].dump(2).capacity() < 4096);
// the whole document and the large value are unaffected
CHECK(d.root().dump() == ordered_json::parse(text).dump());
CHECK(d.root()["k\n"].dump() == "\"" + big + "\"");
}
SECTION("output that outgrows the estimate")
{
// ensure_ascii writes six bytes for each two-byte character
std::string chars;
for (int i = 0; i < 5000; ++i)
{
chars += "\xC3\xA9";
}
const json_document d = json_document::parse("{\"a\":\"" + chars + R"(","k\n":1})");
const json expected = json::parse("\"" + chars + "\"");
CHECK(d.root()["a"].dump(-1, ' ', true) == expected.dump(-1, ' ', true));
CHECK(d.root()["a"].dump(-1, ' ', true).size() == 2 + 5000 * 6);
}
SECTION("streams and discarded views")
{
const json_document d = json_document::parse(R"({"a": [1, 2]})");