mirror of
https://github.com/nlohmann/json.git
synced 2026-10-10 08:27:13 +00:00
Size the dump buffer of a small value by its own extent
The source extent of a value was read from the next node, falling back to the rest of the document when that node held a decoded string. dump() of a small value could thus allocate a buffer as large as the document. Skip a few such nodes, cap the fallback estimate, and shrink a buffer that is much larger than its output. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
3 files changed
+62
-8
No files matched your search
@@ -44,7 +44,13 @@ class output_buffer
|
||||
|
||||
void finish()
|
||||
{
|
||||
m_out.resize(static_cast<std::size_t>(m_pos - m_out.data()));
|
||||
const auto size = static_cast<std::size_t>(m_pos - m_out.data());
|
||||
m_out.resize(size);
|
||||
// do not keep a buffer that was sized for a much larger output
|
||||
if (m_out.capacity() > 1024 && m_out.capacity() / 2 > size)
|
||||
{
|
||||
m_out.shrink_to_fit();
|
||||
}
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE void reserve(std::size_t n)
|
||||
|
||||
@@ -24,6 +24,7 @@
|
||||
#ifndef INCLUDE_NLOHMANN_JSON_VIEW_HPP_
|
||||
#define INCLUDE_NLOHMANN_JSON_VIEW_HPP_
|
||||
|
||||
#include <algorithm> // min
|
||||
#include <cstddef> // size_t
|
||||
#include <cstring> // memcpy, strlen
|
||||
#include <iterator> // distance, input_iterator_tag, iterator_traits
|
||||
@@ -682,20 +683,33 @@ class basic_json_view
|
||||
}
|
||||
|
||||
/// the number of source bytes of this value (estimated for values with
|
||||
/// decoded strings)
|
||||
/// decoded strings); the estimate sizes the output buffer of dump()
|
||||
std::size_t source_extent() const noexcept
|
||||
{
|
||||
const node* const next = document_data::after(m_node);
|
||||
const bool in_source = (m_node->flags & detail::view::node_flags::storage) == 0;
|
||||
if (!in_source)
|
||||
const node* const end = m_doc->tape + m_doc->tape_size;
|
||||
if ((m_node->flags & detail::view::node_flags::storage) != 0)
|
||||
{
|
||||
return m_node->len;
|
||||
}
|
||||
if (next != m_doc->tape + m_doc->tape_size && (next->flags & detail::view::node_flags::storage) == 0 && next->off >= m_node->off)
|
||||
// the value ends where the next node in the source begins; nodes
|
||||
// with decoded strings (their offset is in the arena) are skipped,
|
||||
// but only a few of them, to keep the walk short
|
||||
const node* next = document_data::after(m_node);
|
||||
for (int skipped = 0; next != end && skipped < 16; ++skipped, ++next)
|
||||
{
|
||||
return next->off - m_node->off;
|
||||
if ((next->flags & detail::view::node_flags::storage) == 0)
|
||||
{
|
||||
return next->off >= m_node->off ? next->off - m_node->off : 0;
|
||||
}
|
||||
}
|
||||
return m_doc->size - m_node->off;
|
||||
if (next == end)
|
||||
{
|
||||
return m_doc->size - m_node->off;
|
||||
}
|
||||
// the end is unknown: assume a few bytes per node, the output buffer
|
||||
// grows should the value be larger
|
||||
const auto nodes = static_cast<std::size_t>(document_data::after(m_node) - m_node);
|
||||
return (std::min)(m_doc->size - m_node->off, static_cast<std::size_t>(1024) + nodes * 16);
|
||||
}
|
||||
|
||||
/// the value of the first member with this key, or a discarded view
|
||||
|
||||
@@ -1183,6 +1183,40 @@ TEST_CASE("json_view dump")
|
||||
CHECK(json_document::parse(deep).root().dump() == deep);
|
||||
}
|
||||
|
||||
SECTION("the output buffer of a small value is small")
|
||||
{
|
||||
// an escaped key after the value: its node lies in the arena, so the
|
||||
// source extent of the value cannot be read from the next node
|
||||
const std::string big(100000, 'a');
|
||||
const std::string text = R"({"small":1,"list":[1,2,3],"k\n":")" + big + R"("})";
|
||||
const json_document d = json_document::parse(text);
|
||||
const auto small = d.root()["small"].dump();
|
||||
CHECK(small == "1");
|
||||
CHECK(small.capacity() < 4096);
|
||||
const auto list = d.root()["list"].dump();
|
||||
CHECK(list == "[1,2,3]");
|
||||
CHECK(list.capacity() < 4096);
|
||||
CHECK(d.root()["list"].dump(2).capacity() < 4096);
|
||||
|
||||
// the whole document and the large value are unaffected
|
||||
CHECK(d.root().dump() == ordered_json::parse(text).dump());
|
||||
CHECK(d.root()["k\n"].dump() == "\"" + big + "\"");
|
||||
}
|
||||
|
||||
SECTION("output that outgrows the estimate")
|
||||
{
|
||||
// ensure_ascii writes six bytes for each two-byte character
|
||||
std::string chars;
|
||||
for (int i = 0; i < 5000; ++i)
|
||||
{
|
||||
chars += "\xC3\xA9";
|
||||
}
|
||||
const json_document d = json_document::parse("{\"a\":\"" + chars + R"(","k\n":1})");
|
||||
const json expected = json::parse("\"" + chars + "\"");
|
||||
CHECK(d.root()["a"].dump(-1, ' ', true) == expected.dump(-1, ' ', true));
|
||||
CHECK(d.root()["a"].dump(-1, ' ', true).size() == 2 + 5000 * 6);
|
||||
}
|
||||
|
||||
SECTION("streams and discarded views")
|
||||
{
|
||||
const json_document d = json_document::parse(R"({"a": [1, 2]})");
|
||||
|
||||
Reference in new issue
Block a user