mirror of
https://github.com/nlohmann/json.git
synced 2026-10-06 22:47:13 +00:00
Grow the json_view dump's output in steps and compare raw-number dumps
- The output string reserves the size estimate (the source extent) and grows in steps of 64 KiB within it, instead of being resized to the estimate at once: resize() zero-fills, and for a pretty-printed source the estimate is far larger than the compact output. citm_catalog dump: 342 -> 283 us on x86-64, 144 -> 134 us on Apple M1. - bench_corpus compares the dump with source numbers with yyjson writing numbers read as raw text (YYJSON_READ_NUMBER_AS_RAW); both write the same bytes. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
4 files changed
+51
-8
No files matched your search
@@ -31,7 +31,11 @@ namespace view
|
||||
{
|
||||
|
||||
/// append-only output buffer: writes through a raw pointer into a string that
|
||||
/// is resized ahead, and trimmed by finish()
|
||||
/// is resized ahead, and trimmed by finish(). The estimate is reserved, and the
|
||||
/// string grows in steps of 64 KiB within it: resize() fills the new bytes
|
||||
/// with zeros (before C++23, a string cannot grow without), and a small step
|
||||
/// is filled while the writer is about to use it, in the cache, instead of
|
||||
/// filling the whole estimate in memory first.
|
||||
template<typename StringType>
|
||||
class output_buffer
|
||||
{
|
||||
@@ -93,16 +97,31 @@ class output_buffer
|
||||
}
|
||||
|
||||
private:
|
||||
/// the size of a growth step (a function: std::min() takes a reference,
|
||||
/// which a static constexpr member does not have before C++17)
|
||||
static constexpr std::size_t step() noexcept
|
||||
{
|
||||
return 65536;
|
||||
}
|
||||
|
||||
static StringType& sized(StringType& out, std::size_t estimate)
|
||||
{
|
||||
out.resize((std::max)(estimate, static_cast<std::size_t>(64)));
|
||||
out.reserve(estimate);
|
||||
out.resize((std::min)((std::max)(estimate, static_cast<std::size_t>(64)), step()));
|
||||
return out;
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_NOINLINE void grow(std::size_t n)
|
||||
{
|
||||
const auto used = static_cast<std::size_t>(m_pos - m_out.data());
|
||||
m_out.resize((std::max)(m_out.size() * 2, used + n + 256));
|
||||
// (a step does not go beyond the reserved estimate, so that a good
|
||||
// estimate is never copied to a larger allocation)
|
||||
const std::size_t size = (std::max)((std::min)(m_out.size() + step(), m_out.capacity()), used + n + 256);
|
||||
if (size > m_out.capacity())
|
||||
{
|
||||
m_out.reserve((std::max)(m_out.capacity() * 2, size));
|
||||
}
|
||||
m_out.resize(size);
|
||||
m_pos = &m_out[0] + used;
|
||||
m_end = &m_out[0] + m_out.size();
|
||||
}
|
||||
|
||||
@@ -5409,7 +5409,11 @@ namespace view
|
||||
{
|
||||
|
||||
/// append-only output buffer: writes through a raw pointer into a string that
|
||||
/// is resized ahead, and trimmed by finish()
|
||||
/// is resized ahead, and trimmed by finish(). The estimate is reserved, and the
|
||||
/// string grows in steps of 64 KiB within it: resize() fills the new bytes
|
||||
/// with zeros (before C++23, a string cannot grow without), and a small step
|
||||
/// is filled while the writer is about to use it, in the cache, instead of
|
||||
/// filling the whole estimate in memory first.
|
||||
template<typename StringType>
|
||||
class output_buffer
|
||||
{
|
||||
@@ -5471,16 +5475,31 @@ class output_buffer
|
||||
}
|
||||
|
||||
private:
|
||||
/// the size of a growth step (a function: std::min() takes a reference,
|
||||
/// which a static constexpr member does not have before C++17)
|
||||
static constexpr std::size_t step() noexcept
|
||||
{
|
||||
return 65536;
|
||||
}
|
||||
|
||||
static StringType& sized(StringType& out, std::size_t estimate)
|
||||
{
|
||||
out.resize((std::max)(estimate, static_cast<std::size_t>(64)));
|
||||
out.reserve(estimate);
|
||||
out.resize((std::min)((std::max)(estimate, static_cast<std::size_t>(64)), step()));
|
||||
return out;
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_NOINLINE void grow(std::size_t n)
|
||||
{
|
||||
const auto used = static_cast<std::size_t>(m_pos - m_out.data());
|
||||
m_out.resize((std::max)(m_out.size() * 2, used + n + 256));
|
||||
// (a step does not go beyond the reserved estimate, so that a good
|
||||
// estimate is never copied to a larger allocation)
|
||||
const std::size_t size = (std::max)((std::min)(m_out.size() + step(), m_out.capacity()), used + n + 256);
|
||||
if (size > m_out.capacity())
|
||||
{
|
||||
m_out.reserve((std::max)(m_out.capacity() * 2, size));
|
||||
}
|
||||
m_out.resize(size);
|
||||
m_pos = &m_out[0] + used;
|
||||
m_end = &m_out[0] + m_out.size();
|
||||
}
|
||||
|
||||
@@ -50,7 +50,8 @@ JSON-RPC request (`rpc`):
|
||||
| dump | serialize a parsed document (compact) |
|
||||
|
||||
`bench_corpus.cpp` runs parse, traverse, and dump on any list of files, so that no library is tuned to a handful of
|
||||
documents.
|
||||
documents. Its dump also writes the numbers as they are in the input: `json_view` with `number_format::source`, and
|
||||
yyjson with numbers read as raw text (`YYJSON_READ_NUMBER_AS_RAW`), without converting them.
|
||||
|
||||
`bench_edit.cpp` measures read-modify-write: parse, apply the same logical edits with each library's own API, and
|
||||
serialize (compact). Workloads: `patch` (a handful of edits at fixed places) and `update` (edits in every record).
|
||||
|
||||
@@ -15,7 +15,8 @@
|
||||
// count, string bytes, sum of numbers) before anything is timed. Workloads:
|
||||
// parse (build and free a document), traverse (visit every value, convert
|
||||
// every number), dump (compact), and for json_view also dump with the source
|
||||
// number text. Results go to bench_corpus.csv.
|
||||
// number text, compared with yyjson writing numbers read as raw text
|
||||
// (YYJSON_READ_NUMBER_AS_RAW). Results go to bench_corpus.csv.
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
@@ -272,6 +273,7 @@ int main(int argc, char** argv)
|
||||
};
|
||||
json_document vd = json_document::parse(s);
|
||||
yyjson_doc* yd = yyjson_read(s.data(), s.size(), 0);
|
||||
yyjson_doc* yd_raw = yyjson_read(s.data(), s.size(), YYJSON_READ_NUMBER_AS_RAW);
|
||||
simdjson::dom::parser sjd;
|
||||
const simdjson::dom::element se = sjd.parse(ps).value_unsafe();
|
||||
const std::vector<std::pair<std::string, std::vector<engine>>> workloads =
|
||||
@@ -304,6 +306,7 @@ int main(int argc, char** argv)
|
||||
{"yyjson", [&] { std::size_t n = 0; char* o = yyjson_write(yd, 0, &n); g_sink = static_cast<double>(n); std::free(o); }},
|
||||
{"simdjson DOM", [&] { std::string o = simdjson::to_string(se); g_sink = static_cast<double>(o.size()); }},
|
||||
{"json_view (source numbers)", [&] { std::string o = vd.root().dump(-1, ' ', false, json_view::number_format::source); g_sink = static_cast<double>(o.size()); }},
|
||||
{"yyjson (raw numbers)", [&] { std::size_t n = 0; char* o = yyjson_write(yd_raw, 0, &n); g_sink = static_cast<double>(n); std::free(o); }},
|
||||
}
|
||||
},
|
||||
};
|
||||
@@ -333,6 +336,7 @@ int main(int argc, char** argv)
|
||||
std::fflush(stdout);
|
||||
}
|
||||
yyjson_doc_free(yd);
|
||||
yyjson_doc_free(yd_raw);
|
||||
}
|
||||
std::fclose(csv);
|
||||
}
|
||||
Reference in new issue
Block a user