Make the json_view comparison fair to fresh documents and robust

- compare.py: make --data, --corpus, and --build-dir absolute, since the
  benchmarks run in the build directory; download into a .part file and
  remove an archive whose SHA-256 does not match, so that an interrupted
  download is not kept
- bench_view/bench_corpus/bench_edit: report files that cannot be opened instead of
  aborting; run each engine once untimed before its timed call, so that
  no engine pays for the allocator cleaning up after the previous one
  (with glibc, json_view after json::parse looked 1.7x slower on
  citm_catalog traverse); add "simdjson DOM (fresh)" and time
  "json_view (reused)" for traverse and select too
- README: explain fresh vs. reused documents and page faults on Linux

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-10-05 20:25:36 +02:00
parent 606c1e710b
commit 42f8043413
5 changed files with 57 additions and 6 deletions
@@ -479,6 +479,11 @@ static std::string edit_boost(const std::string& name, const std::string& s, boo
static std::string slurp(const std::string& p)
{
std::ifstream f(p, std::ios::binary);
if (!f)
{
std::fprintf(stderr, "cannot open %s\n", p.c_str());
std::exit(1);
}
std::stringstream ss;
ss << f.rdbuf();
return ss.str();
@@ -550,6 +555,9 @@ int main(int argc, char** argv)
{
for (std::size_t k = 0; k < engines.size(); ++k)
{
// an untimed call first: whatever the previous engine left to the allocator
// (e.g. thousands of freed json nodes) is cleaned up here, not in the timing
g_sink = engines[k].second(dc.name, dc.text, update).size();
const auto t0 = std::chrono::steady_clock::now();
for (int b = 0; b < dc.batch; ++b)
{