Make the json_view comparison fair to fresh documents and robust

- compare.py: make --data, --corpus, and --build-dir absolute, since the
  benchmarks run in the build directory; download into a .part file and
  remove an archive whose SHA-256 does not match, so that an interrupted
  download is not kept
- bench_view/bench_corpus/bench_edit: report files that cannot be opened instead of
  aborting; run each engine once untimed before its timed call, so that
  no engine pays for the allocator cleaning up after the previous one
  (with glibc, json_view after json::parse looked 1.7x slower on
  citm_catalog traverse); add "simdjson DOM (fresh)" and time
  "json_view (reused)" for traverse and select too
- README: explain fresh vs. reused documents and page faults on Linux

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-10-05 20:25:36 +02:00
parent 606c1e710b
commit 42f8043413
5 changed files with 57 additions and 6 deletions
+12 -2
View File
@@ -60,7 +60,10 @@ outputs are checked to describe the same value.
Before anything is timed, all engines must accept each document and agree on the traversal: the number of values, the
bytes of all strings and keys, and the sum of all numbers. All engines run interleaved in every round, and the best
round is reported, as time and as a factor of the `json_view` time (below 1 means faster than `json_view`).
round is reported, as time and as a factor of the `json_view` time (below 1 means faster than `json_view`). Each timed
call follows an untimed call of the same engine: otherwise the engine after `json::parse` pays for the allocator
cleaning up the tens of thousands of nodes `json::parse` just freed (with glibc, this made `json_view` look 1.7 times
slower on citm_catalog traverse).
The engines do not all offer the same features, which the numbers should be read with:
@@ -68,11 +71,18 @@ The engines do not all offer the same features, which the numbers should be read
|---|---|---|---|---|
| `json_view` | immutable index into the text | yes | no | a fresh document per parse; "reused" parses into the same document |
| yyjson | immutable (`yyjson_read`) | yes | via a mutable copy | |
| simdjson DOM | immutable, parser reused | yes | no | |
| simdjson DOM | immutable, parser reused | yes | no | "fresh" uses a new parser per parse |
| simdjson On-Demand | none: forward-only, lazy | no | no | only traverse and select |
| Boost.JSON | owning, mutable DOM | yes | yes | monotonic resource |
| `json::parse` | owning, mutable DOM | yes | yes | |
Reusing memory matters as much as the parser. simdjson DOM reuses its parser, so it writes into memory it already
touched; a fresh `json_view` document or yyjson document gets new memory for every parse. On Linux, glibc returns large
blocks to the system when they are freed, so every fresh parse of a large document pays a page fault per 4 KiB page:
on x86-64 Linux, a fresh `json_view` parse of jeopardy took about twice as long as a reused one. On macOS on Apple
silicon, with 16 KiB pages, the difference is much smaller. Compare "json_view (reused)" with "simdjson DOM", and the
fresh `json_view` with "simdjson DOM (fresh)" and yyjson.
## Published results
Results are only published with the file `compare.py` wrote, which names the machine and the versions; see
@@ -29,6 +29,7 @@
#include <chrono>
#include <cmath>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <fstream>
#include <functional>
@@ -195,6 +196,11 @@ static void walk(const boost::json::value& v, stats& st)
static std::string slurp(const std::string& p)
{
std::ifstream f(p, std::ios::binary);
if (!f)
{
std::fprintf(stderr, "cannot open %s\n", p.c_str());
std::exit(1);
}
std::stringstream ss;
ss << f.rdbuf();
return ss.str();
@@ -222,6 +228,7 @@ int main(int argc, char** argv)
}
std::FILE* csv = std::fopen("bench_corpus.csv", "w");
std::fprintf(csv, "file,bytes,workload,engine,ns\n");
json_document reused;
simdjson::dom::parser sj;
for (const auto& path : files)
{
@@ -272,8 +279,10 @@ int main(int argc, char** argv)
{
"parse", {
{"json_view", [&] { auto x = json_document::parse(s); g_sink = static_cast<double>(x.node_count()); }},
{"json_view (reused)", [&] { reused.read(s); g_sink = static_cast<double>(reused.node_count()); }},
{"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast<double>(yyjson_doc_get_val_count(x)); yyjson_doc_free(x); }},
{"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }},
{"simdjson DOM (fresh)", [&] { simdjson::dom::parser p; auto e = p.parse(ps).value_unsafe(); g_sink = e.is_object(); }},
#if JSON_VIEW_BENCH_BOOST
{"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }},
#endif
@@ -306,6 +315,9 @@ int main(int argc, char** argv)
{
for (std::size_t k = 0; k < wl.second.size(); ++k)
{
// an untimed call first: whatever the previous engine left to the allocator
// (e.g. thousands of freed json nodes) is cleaned up here, not in the timing
wl.second[k].fn();
const auto t0 = std::chrono::steady_clock::now();
wl.second[k].fn();
best[k] = std::min(best[k], std::chrono::duration<double, std::nano>(std::chrono::steady_clock::now() - t0).count());
@@ -479,6 +479,11 @@ static std::string edit_boost(const std::string& name, const std::string& s, boo
static std::string slurp(const std::string& p)
{
std::ifstream f(p, std::ios::binary);
if (!f)
{
std::fprintf(stderr, "cannot open %s\n", p.c_str());
std::exit(1);
}
std::stringstream ss;
ss << f.rdbuf();
return ss.str();
@@ -550,6 +555,9 @@ int main(int argc, char** argv)
{
for (std::size_t k = 0; k < engines.size(); ++k)
{
// an untimed call first: whatever the previous engine left to the allocator
// (e.g. thousands of freed json nodes) is cleaned up here, not in the timing
g_sink = engines[k].second(dc.name, dc.text, update).size();
const auto t0 = std::chrono::steady_clock::now();
for (int b = 0; b < dc.batch; ++b)
{
+16 -2
View File
@@ -10,7 +10,8 @@
//
// json_view nlohmann/json_view.hpp (fresh document per parse / reused)
// yyjson yyjson_read(): immutable document, random access
// simdjson DOM dom::parser (reused, as recommended): immutable, random access
// simdjson DOM dom::parser (reused, as recommended; "fresh": a new parser
// per parse): immutable, random access
// references (different feature sets):
// simdjson OD On-Demand: forward-only, lazy
// Boost.JSON owning, mutable DOM (monotonic resource)
@@ -19,7 +20,8 @@
// Workloads: parse (build + free), traverse (visit everything, convert every
// number, touch every string and key), select (a few fields per document),
// dump (compact serialization of the parsed document).
// All engines run interleaved in every round; the best round is reported.
// All engines run interleaved in every round, each timed call after an untimed
// one of the same engine; the best round is reported.
#include <nlohmann/json_view.hpp>
#if JSON_VIEW_BENCH_BOOST
@@ -33,6 +35,7 @@
#include <chrono>
#include <cmath>
#include <cstdio>
#include <cstdlib>
#include <fstream>
#include <functional>
#include <map>
@@ -581,6 +584,11 @@ static double pick_od(const std::string& name, simdjson::ondemand::document& d)
static std::string slurp(const std::string& p)
{
std::ifstream f(p, std::ios::binary);
if (!f)
{
std::fprintf(stderr, "cannot open %s\n", p.c_str());
std::exit(1);
}
std::stringstream ss;
ss << f.rdbuf();
return ss.str();
@@ -657,6 +665,7 @@ int main(int argc, char** argv)
{"json_view (reused)", [&] { reused.read(s); g_sink = static_cast<double>(reused.node_count()); }},
{"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast<double>(yyjson_doc_get_val_count(d)); yyjson_doc_free(d); }},
{"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }},
{"simdjson DOM (fresh)", [&] { simdjson::dom::parser p; auto e = p.parse(ps).value_unsafe(); g_sink = e.is_object(); }},
#if JSON_VIEW_BENCH_BOOST
{"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }},
#endif
@@ -664,6 +673,7 @@ int main(int argc, char** argv)
}});
workloads.push_back({"traverse", {
{"json_view", [&] { auto d = json_document::parse(s); stats st; walk(d.root(), st); g_sink = st.num; }},
{"json_view (reused)", [&] { reused.read(s); stats st; walk(reused.root(), st); g_sink = st.num; }},
{"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(d), st); g_sink = st.num; yyjson_doc_free(d); }},
{"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }},
{"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); stats st; walk_od(d.get_value().value_unsafe(), st); g_sink = st.num; }},
@@ -674,6 +684,7 @@ int main(int argc, char** argv)
}});
workloads.push_back({"select", {
{"json_view", [&] { auto d = json_document::parse(s); g_sink = pick(name, d.root()); }},
{"json_view (reused)", [&] { reused.read(s); g_sink = pick(name, reused.root()); }},
{"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = pick(name, yyjson_doc_get_root(d)); yyjson_doc_free(d); }},
{"simdjson DOM", [&] { g_sink = pick(name, sj.parse(ps).value_unsafe()); }},
{"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); g_sink = pick_od(name, d); }},
@@ -710,6 +721,9 @@ int main(int argc, char** argv)
{
for (std::size_t k = 0; k < wl.second.size(); ++k)
{
// an untimed call first: whatever the previous engine left to the allocator
// (e.g. thousands of freed json nodes) is cleaned up here, not in the timing
wl.second[k].fn();
const auto t0 = std::chrono::steady_clock::now();
for (int b = 0; b < dc.batch; ++b)
{
+9 -2
View File
@@ -154,11 +154,14 @@ def download_library(name, work):
if not os.path.isfile(archive):
print(f'downloading {pin["url"]}', flush=True)
# the URLs are the https constants in PINNED, and the SHA-256 is checked below
urllib.request.urlretrieve(pin['url'], archive) # nosec B310
# (into a .part file first, so that an interrupted download is not kept)
urllib.request.urlretrieve(pin['url'], archive + '.part') # nosec B310
os.replace(archive + '.part', archive)
with open(archive, 'rb') as f:
digest = hashlib.sha256(f.read()).hexdigest()
if digest != pin['sha256']:
sys.exit(f'error: SHA-256 of {archive} is {digest}, expected {pin["sha256"]}')
os.remove(archive) # downloaded again by the next run
sys.exit(f'error: SHA-256 of {archive} is {digest}, expected {pin["sha256"]} (removed)')
src = os.path.join(work, 'download', pin['dir'])
if not os.path.isdir(src):
with tarfile.open(archive) as t:
@@ -214,6 +217,10 @@ def main():
ap.add_argument('--corpus', nargs='*', default=[], help='more files for bench_corpus')
ap.add_argument('--build-dir', default=os.path.join(HERE, 'build'), help='where to build (default: build/ next to this script)')
args = ap.parse_args()
# the benchmarks run in the build directory: make the paths absolute
args.data = os.path.abspath(args.data)
args.corpus = [os.path.abspath(f) for f in args.corpus]
args.build_dir = os.path.abspath(args.build_dir)
cxx = os.environ.get('CXX', 'c++')
cc = os.environ.get('CC', 'cc')