diff --git a/include/nlohmann/detail/view/serializer.hpp b/include/nlohmann/detail/view/serializer.hpp index 26bfb3225..7870601bc 100644 --- a/include/nlohmann/detail/view/serializer.hpp +++ b/include/nlohmann/detail/view/serializer.hpp @@ -44,7 +44,13 @@ class output_buffer void finish() { - m_out.resize(static_cast(m_pos - m_out.data())); + const auto size = static_cast(m_pos - m_out.data()); + m_out.resize(size); + // do not keep a buffer that was sized for a much larger output + if (m_out.capacity() > 1024 && m_out.capacity() / 2 > size) + { + m_out.shrink_to_fit(); + } } NLOHMANN_VIEW_ALWAYS_INLINE void reserve(std::size_t n) diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 882101336..f984bdd33 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -24,6 +24,7 @@ #ifndef INCLUDE_NLOHMANN_JSON_VIEW_HPP_ #define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ +#include // min #include // size_t #include // memcpy, strlen #include // distance, input_iterator_tag, iterator_traits @@ -682,20 +683,33 @@ class basic_json_view } /// the number of source bytes of this value (estimated for values with - /// decoded strings) + /// decoded strings); the estimate sizes the output buffer of dump() std::size_t source_extent() const noexcept { - const node* const next = document_data::after(m_node); - const bool in_source = (m_node->flags & detail::view::node_flags::storage) == 0; - if (!in_source) + const node* const end = m_doc->tape + m_doc->tape_size; + if ((m_node->flags & detail::view::node_flags::storage) != 0) { return m_node->len; } - if (next != m_doc->tape + m_doc->tape_size && (next->flags & detail::view::node_flags::storage) == 0 && next->off >= m_node->off) + // the value ends where the next node in the source begins; nodes + // with decoded strings (their offset is in the arena) are skipped, + // but only a few of them, to keep the walk short + const node* next = document_data::after(m_node); + for (int skipped = 0; next != end && skipped < 16; ++skipped, ++next) { - return next->off - m_node->off; + if ((next->flags & detail::view::node_flags::storage) == 0) + { + return next->off >= m_node->off ? next->off - m_node->off : 0; + } } - return m_doc->size - m_node->off; + if (next == end) + { + return m_doc->size - m_node->off; + } + // the end is unknown: assume a few bytes per node, the output buffer + // grows should the value be larger + const auto nodes = static_cast(document_data::after(m_node) - m_node); + return (std::min)(m_doc->size - m_node->off, static_cast(1024) + nodes * 16); } /// the value of the first member with this key, or a discarded view diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 3f35c6440..07ad71dcd 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -1183,6 +1183,40 @@ TEST_CASE("json_view dump") CHECK(json_document::parse(deep).root().dump() == deep); } + SECTION("the output buffer of a small value is small") + { + // an escaped key after the value: its node lies in the arena, so the + // source extent of the value cannot be read from the next node + const std::string big(100000, 'a'); + const std::string text = R"({"small":1,"list":[1,2,3],"k\n":")" + big + R"("})"; + const json_document d = json_document::parse(text); + const auto small = d.root()["small"].dump(); + CHECK(small == "1"); + CHECK(small.capacity() < 4096); + const auto list = d.root()["list"].dump(); + CHECK(list == "[1,2,3]"); + CHECK(list.capacity() < 4096); + CHECK(d.root()["list"].dump(2).capacity() < 4096); + + // the whole document and the large value are unaffected + CHECK(d.root().dump() == ordered_json::parse(text).dump()); + CHECK(d.root()["k\n"].dump() == "\"" + big + "\""); + } + + SECTION("output that outgrows the estimate") + { + // ensure_ascii writes six bytes for each two-byte character + std::string chars; + for (int i = 0; i < 5000; ++i) + { + chars += "\xC3\xA9"; + } + const json_document d = json_document::parse("{\"a\":\"" + chars + R"(","k\n":1})"); + const json expected = json::parse("\"" + chars + "\""); + CHECK(d.root()["a"].dump(-1, ' ', true) == expected.dump(-1, ' ', true)); + CHECK(d.root()["a"].dump(-1, ' ', true).size() == 2 + 5000 * 6); + } + SECTION("streams and discarded views") { const json_document d = json_document::parse(R"({"a": [1, 2]})");