diff --git a/.github/labeler.yml b/.github/labeler.yml index 5e76f915c..bc1621a06 100644 --- a/.github/labeler.yml +++ b/.github/labeler.yml @@ -52,6 +52,7 @@ labels: - "single_include/nlohmann/json_view\\.hpp" - "tests/src/unit-json_view.*" - "tests/src/fuzzer-(parse_json_view|json_view_image)\\.cpp" + - "tests/benchmarks/json_view/.*" - "tools/amalgamate/config_json_view\\.json" - "docs/mkdocs/docs/features/json_view\\.md" - "docs/mkdocs/docs/api/basic_json_(document|view)/.*" diff --git a/.github/workflows/json_view_benchmarks.yml b/.github/workflows/json_view_benchmarks.yml new file mode 100644 index 000000000..ee41eac9d --- /dev/null +++ b/.github/workflows/json_view_benchmarks.yml @@ -0,0 +1,78 @@ +name: "json_view benchmarks" + +# On demand only: runs the comparison of json_view with yyjson, simdjson, and +# Boost.JSON (tests/benchmarks/json_view/compare.py) on GitHub-hosted runners, +# for numbers from x86-64 and AArch64 Linux. It runs when started by hand, or +# when a pull request gets the label "benchmark" (on both architectures, with +# GCC and the default settings). Shared runners are noisy: the results show +# where json_view stands, but published numbers need a quiet machine (see +# tests/benchmarks/json_view/README.md). + +on: + pull_request: + types: [labeled] + workflow_dispatch: + inputs: + runner: + description: "Runner image" + type: choice + options: + - ubuntu-24.04 + - ubuntu-24.04-arm + default: ubuntu-24.04 + compiler: + description: "Compiler" + type: choice + options: + - g++ + - clang++ + default: g++ + native: + description: "Compile for the runner's CPU (-march=native)" + type: boolean + default: false + rounds: + description: "Rounds of bench_view" + type: number + default: 30 + +permissions: + contents: read + +jobs: + compare: + if: github.event_name == 'workflow_dispatch' || github.event.label.name == 'benchmark' + strategy: + matrix: + runner: ${{ fromJSON(github.event_name == 'workflow_dispatch' && format('["{0}"]', inputs.runner) || '["ubuntu-24.04", "ubuntu-24.04-arm"]') }} + runs-on: ${{ matrix.runner }} + steps: + - name: Harden Runner + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 + with: + egress-policy: audit + + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - name: Download test data + run: | + cmake -S . -B build -DJSON_BuildTests=On + cmake --build build --target download_test_data + + - name: Run the comparison + env: + CXX: ${{ inputs.compiler || 'g++' }} + CC: ${{ inputs.compiler == 'clang++' && 'clang' || 'gcc' }} + ROUNDS: ${{ inputs.rounds || 30 }} + NATIVE: ${{ inputs.native && '--native' || '' }} + run: python3 tests/benchmarks/json_view/compare.py --data build/test_files --download --rounds "$ROUNDS" $NATIVE + + - name: Summary + run: cat tests/benchmarks/json_view/results/*.md >> "$GITHUB_STEP_SUMMARY" + + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: json_view-benchmarks-${{ matrix.runner }}-${{ inputs.compiler || 'g++' }} + path: tests/benchmarks/json_view/results/ diff --git a/tests/benchmarks/json_view/.gitignore b/tests/benchmarks/json_view/.gitignore new file mode 100644 index 000000000..567609b12 --- /dev/null +++ b/tests/benchmarks/json_view/.gitignore @@ -0,0 +1 @@ +build/ diff --git a/tests/benchmarks/json_view/README.md b/tests/benchmarks/json_view/README.md new file mode 100644 index 000000000..c4a05d35a --- /dev/null +++ b/tests/benchmarks/json_view/README.md @@ -0,0 +1,90 @@ +# json_view compared with other libraries + +The in-tree benchmarks in [`tests/benchmarks`](../README.md) measure `json_document` against `json::parse` only. The +programs here compare it with [yyjson](https://github.com/ibireme/yyjson), +[simdjson](https://github.com/simdjson/simdjson), and [Boost.JSON](https://github.com/boostorg/json): the question +users ask when they pick a library. They are not built by CMake or run by CI. + +## Reproducing the numbers + +`compare.py` builds the programs against `include/` of this checkout, runs them, and writes the results together with +everything needed to reproduce them to `results/-.md` (and `.csv`): the date, the commit, the CPU, the +OS, the compiler, the flags, and the versions of all libraries. + +```sh +python3 tests/benchmarks/json_view/compare.py --data [--native] [--rounds 30] +``` + +- `--data` is the downloaded [test data](https://github.com/nlohmann/json_test_data), e.g. the `test_files` directory + of a CMake build directory. It needs `nativejson-benchmark/{twitter,citm_catalog,canada}.json` and + `jeopardy/jeopardy.json`. +- The other libraries come from the system: pkg-config, or Homebrew (`brew install yyjson simdjson boost`). With + `--download`, pinned releases are downloaded instead and checked against their SHA-256. Without Boost headers (or + with `--no-boost`), the Boost.JSON columns are skipped, and the results say so. +- `--corpus file...` adds files to the corpus benchmark, e.g. those of + [simdjson-data](https://github.com/simdjson/simdjson-data) or the + [yyjson benchmark](https://github.com/ibireme/yyjson_benchmark). +- Only the Python 3 standard library is used; a C++17 compiler is needed (`CXX` and `CC` are honored). + +For numbers worth publishing, use a quiet machine (see [Getting stable numbers](../README.md#getting-stable-numbers)), +the default 30 rounds or more, and `--native` only if the other libraries were built for the same CPU. + +### On GitHub-hosted runners + +The workflow [json_view benchmarks](../../../.github/workflows/json_view_benchmarks.yml) runs `compare.py --download` +on demand: by hand (Actions → "json_view benchmarks" → "Run workflow"), on an x86-64 or AArch64 Ubuntu runner with GCC +or Clang, or when a pull request gets the label `benchmark`, on both architectures with GCC. The results appear as the +job summary and as an artifact. Shared runners are noisy, so these numbers show +where `json_view` stands on another architecture; they are not meant for publication. + +## What is measured + +`bench_view.cpp` runs four workloads on twitter, citm_catalog, canada, jeopardy, a single tweet (`status`), and a +JSON-RPC request (`rpc`): + +| workload | what it does | +|---|---| +| parse | build and free a document | +| traverse | parse, then visit every value, convert every number, touch every string and key | +| select | parse, then read a few fields per record (e.g. id, user name, and retweet count of each tweet) | +| dump | serialize a parsed document (compact) | + +`bench_corpus.cpp` runs parse, traverse, and dump on any list of files, so that no library is tuned to a handful of +documents. Its dump also writes the numbers as they are in the input: `json_view` with `number_format::source`, and +yyjson with numbers read as raw text (`YYJSON_READ_NUMBER_AS_RAW`), without converting them. + +`bench_edit.cpp` measures read-modify-write: parse, apply the same logical edits with each library's own API, and +serialize (compact). Workloads: `patch` (a handful of edits at fixed places) and `update` (edits in every record). +An editable `json_document` edits in place; yyjson copies its immutable document into a mutable one first +(`yyjson_doc_mut_copy`); Boost.JSON and `json::parse` build mutable DOMs; simdjson cannot edit a document. All +outputs are checked to describe the same value. + +Before anything is timed, all engines must accept each document and agree on the traversal: the number of values, the +bytes of all strings and keys, and the sum of all numbers. All engines run interleaved in every round, and the best +round is reported, as time and as a factor of the `json_view` time (below 1 means faster than `json_view`). Each timed +call follows an untimed call of the same engine: otherwise the engine after `json::parse` pays for the allocator +cleaning up the tens of thousands of nodes `json::parse` just freed (with glibc, this made `json_view` look 1.7 times +slower on citm_catalog traverse). + +The engines do not all offer the same features, which the numbers should be read with: + +| engine | document | random access | editable | notes | +|---|---|---|---|---| +| `json_view` | immutable index into the text | yes | no | a fresh document per parse; "reused" parses into the same document | +| yyjson | immutable (`yyjson_read`) | yes | via a mutable copy | | +| simdjson DOM | immutable, parser reused | yes | no | "fresh" uses a new parser per parse | +| simdjson On-Demand | none: forward-only, lazy | no | no | only traverse and select | +| Boost.JSON | owning, mutable DOM | yes | yes | monotonic resource | +| `json::parse` | owning, mutable DOM | yes | yes | | + +Reusing memory matters as much as the parser. simdjson DOM reuses its parser, so it writes into memory it already +touched; a fresh `json_view` document or yyjson document gets new memory for every parse. On Linux, glibc returns large +blocks to the system when they are freed, so every fresh parse of a large document pays a page fault per 4 KiB page: +on x86-64 Linux, a fresh `json_view` parse of jeopardy took about twice as long as a reused one. On macOS on Apple +silicon, with 16 KiB pages, the difference is much smaller. Compare "json_view (reused)" with "simdjson DOM", and the +fresh `json_view` with "simdjson DOM (fresh)" and yyjson. + +## Published results + +Results are only published with the file `compare.py` wrote, which names the machine and the versions; see +`results/`. Numbers from one machine and compiler do not carry over to another: rerun the script. diff --git a/tests/benchmarks/json_view/bench_corpus.cpp b/tests/benchmarks/json_view/bench_corpus.cpp new file mode 100644 index 000000000..d3806f131 --- /dev/null +++ b/tests/benchmarks/json_view/bench_corpus.cpp @@ -0,0 +1,342 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// Corpus benchmark: the read-only workloads of bench_view.cpp on any list of +// JSON files (for example the benchmark sets of simdjson and yyjson). +// +// ./bench_corpus [--rounds N] file... +// +// For every file, all engines must accept it and agree on a traversal (value +// count, string bytes, sum of numbers) before anything is timed. Workloads: +// parse (build and free a document), traverse (visit every value, convert +// every number), dump (compact), and for json_view also dump with the source +// number text, compared with yyjson writing numbers read as raw text +// (YYJSON_READ_NUMBER_AS_RAW). Results go to bench_corpus.csv. +#include + +#if JSON_VIEW_BENCH_BOOST + #include + #include +#endif +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using nlohmann::json; +using nlohmann::json_document; +using nlohmann::json_view; + +static volatile double g_sink; + +struct stats +{ + double num = 0; + std::size_t str = 0, nodes = 0; +}; + +static void walk(json_view v, stats& st) +{ + ++st.nodes; + switch (v.type()) + { + case json::value_t::object: + for (auto it = v.begin(); it != v.end(); ++it) + { + st.str += it.key().size(); + walk(*it, st); + } + break; + case json::value_t::array: + for (const json_view e : v) + { + walk(e, st); + } + break; + case json::value_t::string: + st.str += v.get_string().size(); + break; + case json::value_t::number_integer: + st.num += static_cast(v.get()); + break; + case json::value_t::number_unsigned: + st.num += static_cast(v.get()); + break; + case json::value_t::number_float: + st.num += v.get(); + break; + default: + break; + } +} + +static void walk(yyjson_val* v, stats& st) +{ + ++st.nodes; + switch (yyjson_get_type(v)) + { + case YYJSON_TYPE_OBJ: + { + std::size_t idx, max; + yyjson_val* k, * val; + yyjson_obj_foreach(v, idx, max, k, val) + { + st.str += yyjson_get_len(k); + walk(val, st); + } + break; + } + case YYJSON_TYPE_ARR: + { + std::size_t idx, max; + yyjson_val* val; + yyjson_arr_foreach(v, idx, max, val) + { + walk(val, st); + } + break; + } + case YYJSON_TYPE_STR: + st.str += yyjson_get_len(v); + break; + case YYJSON_TYPE_NUM: + st.num += yyjson_is_sint(v) ? static_cast(yyjson_get_sint(v)) : yyjson_is_uint(v) ? static_cast(yyjson_get_uint(v)) : yyjson_get_real(v); + break; + default: + break; + } +} + +static void walk(simdjson::dom::element e, stats& st) +{ + ++st.nodes; + switch (e.type()) + { + case simdjson::dom::element_type::OBJECT: + for (auto f : simdjson::dom::object(e)) + { + st.str += f.key.size(); + walk(f.value, st); + } + break; + case simdjson::dom::element_type::ARRAY: + for (auto c : simdjson::dom::array(e)) + { + walk(c, st); + } + break; + case simdjson::dom::element_type::STRING: + st.str += std::string_view(e).size(); + break; + case simdjson::dom::element_type::INT64: + st.num += static_cast(int64_t(e)); + break; + case simdjson::dom::element_type::UINT64: + st.num += static_cast(uint64_t(e)); + break; + case simdjson::dom::element_type::DOUBLE: + st.num += double(e); + break; + default: + break; + } +} + +#if JSON_VIEW_BENCH_BOOST +static void walk(const boost::json::value& v, stats& st) +{ + ++st.nodes; + switch (v.kind()) + { + case boost::json::kind::object: + for (const auto& kv : v.get_object()) + { + st.str += kv.key().size(); + walk(kv.value(), st); + } + break; + case boost::json::kind::array: + for (const auto& c : v.get_array()) + { + walk(c, st); + } + break; + case boost::json::kind::string: + st.str += v.get_string().size(); + break; + case boost::json::kind::int64: + st.num += static_cast(v.get_int64()); + break; + case boost::json::kind::uint64: + st.num += static_cast(v.get_uint64()); + break; + case boost::json::kind::double_: + st.num += v.get_double(); + break; + default: + break; + } +} +#endif + +static std::string slurp(const std::string& p) +{ + std::ifstream f(p, std::ios::binary); + if (!f) + { + std::fprintf(stderr, "cannot open %s\n", p.c_str()); + std::exit(1); + } + std::stringstream ss; + ss << f.rdbuf(); + return ss.str(); +} + +static bool same(const stats& a, const stats& b) +{ + return a.nodes == b.nodes && a.str == b.str && (a.num == b.num || std::fabs(a.num - b.num) <= 1e-9 * std::fabs(a.num)); +} + +int main(int argc, char** argv) +{ + int rounds = 0; // 0: by size + std::vector files; + for (int i = 1; i < argc; ++i) + { + if (std::strcmp(argv[i], "--rounds") == 0 && i + 1 < argc) + { + rounds = std::atoi(argv[++i]); + } + else + { + files.push_back(argv[i]); + } + } + std::FILE* csv = std::fopen("bench_corpus.csv", "w"); + std::fprintf(csv, "file,bytes,workload,engine,ns\n"); + json_document reused; + simdjson::dom::parser sj; + for (const auto& path : files) + { + const std::string s = slurp(path); + const std::string name = path.substr(path.rfind('/') + 1); + const simdjson::padded_string ps(s); + + // all engines must agree before timing + stats a, b, c, d; + const json_document doc = json_document::parse(s); + walk(doc.root(), a); + yyjson_doc* y = yyjson_read(s.data(), s.size(), 0); + auto sjr = sj.parse(ps); +#if JSON_VIEW_BENCH_BOOST + boost::json::parse_options opt; + opt.numbers = boost::json::number_precision::precise; + boost::json::monotonic_resource mr0; + const boost::json::value bv = boost::json::parse(s, &mr0, opt); +#endif + if (y == nullptr || sjr.error()) + { + std::printf("%-34s skipped (an engine rejects it)\n", name.c_str()); + yyjson_doc_free(y); + continue; + } + walk(yyjson_doc_get_root(y), b); + walk(sjr.value_unsafe(), c); +#if JSON_VIEW_BENCH_BOOST + walk(bv, d); +#else + d = a; +#endif + yyjson_doc_free(y); + const bool ok = same(a, b) && same(a, c) && same(a, d); + + const int r = rounds > 0 ? rounds : static_cast(std::max(3, std::min(60, 400000000 / (s.size() + 1)))); + struct engine + { + std::string name; + std::function fn; + }; + json_document vd = json_document::parse(s); + yyjson_doc* yd = yyjson_read(s.data(), s.size(), 0); + yyjson_doc* yd_raw = yyjson_read(s.data(), s.size(), YYJSON_READ_NUMBER_AS_RAW); + simdjson::dom::parser sjd; + const simdjson::dom::element se = sjd.parse(ps).value_unsafe(); + const std::vector>> workloads = + { + { + "parse", { + {"json_view", [&] { auto x = json_document::parse(s); g_sink = static_cast(x.node_count()); }}, + {"json_view (reused)", [&] { reused.read(s); g_sink = static_cast(reused.node_count()); }}, + {"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast(yyjson_doc_get_val_count(x)); yyjson_doc_free(x); }}, + {"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }}, + {"simdjson DOM (fresh)", [&] { simdjson::dom::parser p; auto e = p.parse(ps).value_unsafe(); g_sink = e.is_object(); }}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }}, +#endif + } + }, + { + "traverse", { + {"json_view", [&] { auto x = json_document::parse(s); stats st; walk(x.root(), st); g_sink = st.num; }}, + {"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(x), st); g_sink = st.num; yyjson_doc_free(x); }}, + {"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); stats st; walk(v, st); g_sink = st.num; }}, +#endif + } + }, + { + "dump", { + {"json_view", [&] { std::string o = vd.root().dump(); g_sink = static_cast(o.size()); }}, + {"yyjson", [&] { std::size_t n = 0; char* o = yyjson_write(yd, 0, &n); g_sink = static_cast(n); std::free(o); }}, + {"simdjson DOM", [&] { std::string o = simdjson::to_string(se); g_sink = static_cast(o.size()); }}, + {"json_view (source numbers)", [&] { std::string o = vd.root().dump(-1, ' ', false, json_view::number_format::source); g_sink = static_cast(o.size()); }}, + {"yyjson (raw numbers)", [&] { std::size_t n = 0; char* o = yyjson_write(yd_raw, 0, &n); g_sink = static_cast(n); std::free(o); }}, + } + }, + }; + std::printf("%-34s %9zu B%s\n", name.c_str(), s.size(), ok ? "" : " [ENGINES DISAGREE]"); + for (const auto& wl : workloads) + { + std::vector best(wl.second.size(), 1e300); + for (int i = 0; i < r; ++i) + { + for (std::size_t k = 0; k < wl.second.size(); ++k) + { + // an untimed call first: whatever the previous engine left to the allocator + // (e.g. thousands of freed json nodes) is cleaned up here, not in the timing + wl.second[k].fn(); + const auto t0 = std::chrono::steady_clock::now(); + wl.second[k].fn(); + best[k] = std::min(best[k], std::chrono::duration(std::chrono::steady_clock::now() - t0).count()); + } + } + std::printf(" %-9s", wl.first.c_str()); + for (std::size_t k = 0; k < wl.second.size(); ++k) + { + std::printf(" %s %.2f GB/s (%.2fx)", wl.second[k].name.c_str(), static_cast(s.size()) / best[k], best[k] / best[0]); + std::fprintf(csv, "%s,%zu,%s,%s,%.1f\n", name.c_str(), s.size(), wl.first.c_str(), wl.second[k].name.c_str(), best[k]); + } + std::printf("\n"); + std::fflush(stdout); + } + yyjson_doc_free(yd); + yyjson_doc_free(yd_raw); + } + std::fclose(csv); +} diff --git a/tests/benchmarks/json_view/bench_edit.cpp b/tests/benchmarks/json_view/bench_edit.cpp new file mode 100644 index 000000000..d5b35f912 --- /dev/null +++ b/tests/benchmarks/json_view/bench_edit.cpp @@ -0,0 +1,583 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// Read-modify-write benchmark: parse a document, apply the same logical edits +// with each library's own API, and serialize it (compact). +// +// json_view json_editable_document: edits in place, unchanged values stay in the index +// yyjson yyjson_read + yyjson_doc_mut_copy (the way to edit a parsed document) +// Boost.JSON parse into a mutable DOM (monotonic resource, precise numbers), serialize +// json::parse nlohmann::json today +// simdjson has no mutable document and is not part of this comparison. +// +// Workloads: +// patch a handful of edits at fixed places (scalars, a new member, a new array element) +// update edits in every record (twitter: 100 statuses, citm: 243 performances, +// canada: 480 rings, jeopardy: 216,930 questions): set scalars, erase a +// member, add a member (canada: replace the first point of every ring) +// +// Build: see README.md (same flags as bench_view.cpp). +#include + +#if JSON_VIEW_BENCH_BOOST + #include + #include +#endif +#include + +#include +#include +#include +#include +#include +#include +#include + +using nlohmann::json; +using nlohmann::json_editable_document; +using nlohmann::json_editable_view; +#if JSON_VIEW_BENCH_BOOST + namespace bj = boost::json; +#endif + +static volatile std::size_t g_sink; + +// ---------------- json_view ---------------- + +static std::string edit_view(const std::string& name, const std::string& s, bool update) +{ + json_editable_document d = json_editable_document::parse(s); + const json_editable_view r = d.root(); + if (name == "twitter") + { + if (update) + { + std::int64_t i = 0; + for (const json_editable_view st : r["statuses"]) + { + d.set(st, "retweet_count", i++); + d.set(st, "favorited", true); + d.set(st, "text", "redacted"); + d.erase(st, "entities"); + d.set(st, "edited", true); + } + } + else + { + d.set(r["search_metadata"], "count", 200); + d.set(r["statuses"][0], "text", "patched"); + d.set(r["statuses"][0]["user"], "followers_count", 1); + d.set(r["statuses"][99], "favorited", true); + d.set(r, "patched", true); + } + } + else if (name == "citm_catalog") + { + if (update) + { + for (const json_editable_view p : r["performances"]) + { + d.set(p, "name", "performance"); + d.set(p, "start", p["start"].get() + 1); + d.erase(p, "seatMapImage"); + d.set(p, "edited", true); + } + } + else + { + d.set(r["events"]["138586341"], "name", "patched"); + d.set(r["performances"][0], "start", 0); + d.set(r["venueNames"], "PLEYEL_PLEYEL", "Salle"); + d.set(r, "patched", true); + } + } + else if (name == "canada") + { + const json_editable_view coords = r["features"][0]["geometry"]["coordinates"]; + if (update) + { + for (const json_editable_view ring : coords) + { + d.set(ring, 0, json::array({0.5, 0.5})); + } + } + else + { + d.set(r["features"][0]["properties"], "name", "patched"); + d.set(r, "type", "FeatureCollection2"); + d.set(coords[0], 0, json::array({0.0, 0.0})); + } + } + else if (name == "jeopardy") + { + if (update) + { + for (const json_editable_view q : r) + { + d.set(q, "value", "$1"); + d.erase(q, "air_date"); + } + } + else + { + d.set(r[0], "value", "$0"); + d.set(r[100000], "answer", "patched"); + d.set(r[216929], "round", "x"); + d.push_back(r, json::object({{"category", "NEW"}, {"value", "$5"}})); + } + } + else if (name == "status") + { + d.set(r, "retweet_count", 1); + d.set(r["user"], "name", "x"); + } + else if (name == "rpc") + { + d.set(r, "id", 4); + d.set(r["params"], "subtrahend", 24); + } + return r.dump(); +} + +// ---------------- nlohmann::json ---------------- + +static std::string edit_json(const std::string& name, const std::string& s, bool update) +{ + json r = json::parse(s); + if (name == "twitter") + { + if (update) + { + std::int64_t i = 0; + for (auto& st : r["statuses"]) + { + st["retweet_count"] = i++; + st["favorited"] = true; + st["text"] = "redacted"; + st.erase("entities"); + st["edited"] = true; + } + } + else + { + r["search_metadata"]["count"] = 200; + r["statuses"][0]["text"] = "patched"; + r["statuses"][0]["user"]["followers_count"] = 1; + r["statuses"][99]["favorited"] = true; + r["patched"] = true; + } + } + else if (name == "citm_catalog") + { + if (update) + { + for (auto& p : r["performances"]) + { + p["name"] = "performance"; + p["start"] = p["start"].get() + 1; + p.erase("seatMapImage"); + p["edited"] = true; + } + } + else + { + r["events"]["138586341"]["name"] = "patched"; + r["performances"][0]["start"] = 0; + r["venueNames"]["PLEYEL_PLEYEL"] = "Salle"; + r["patched"] = true; + } + } + else if (name == "canada") + { + json& coords = r["features"][0]["geometry"]["coordinates"]; + if (update) + { + for (auto& ring : coords) + { + ring[0] = json::array({0.5, 0.5}); + } + } + else + { + r["features"][0]["properties"]["name"] = "patched"; + r["type"] = "FeatureCollection2"; + coords[0][0] = json::array({0.0, 0.0}); + } + } + else if (name == "jeopardy") + { + if (update) + { + for (auto& q : r) + { + q["value"] = "$1"; + q.erase("air_date"); + } + } + else + { + r[0]["value"] = "$0"; + r[100000]["answer"] = "patched"; + r[216929]["round"] = "x"; + r.push_back(json::object({{"category", "NEW"}, {"value", "$5"}})); + } + } + else if (name == "status") + { + r["retweet_count"] = 1; + r["user"]["name"] = "x"; + } + else if (name == "rpc") + { + r["id"] = 4; + r["params"]["subtrahend"] = 24; + } + return r.dump(); +} + +// ---------------- yyjson ---------------- + +static std::string edit_yyjson(const std::string& name, const std::string& s, bool update) +{ + yyjson_doc* idoc = yyjson_read(s.data(), s.size(), 0); + yyjson_mut_doc* d = yyjson_doc_mut_copy(idoc, nullptr); + yyjson_doc_free(idoc); + yyjson_mut_val* r = yyjson_mut_doc_get_root(d); + auto get = [](yyjson_mut_val * o, const char* k) + { + return yyjson_mut_obj_get(o, k); + }; + if (name == "twitter") + { + yyjson_mut_val* sts = get(r, "statuses"); + if (update) + { + std::size_t idx, max; + yyjson_mut_val* st; + std::int64_t i = 0; + yyjson_mut_arr_foreach(sts, idx, max, st) + { + yyjson_mut_set_sint(get(st, "retweet_count"), i++); + yyjson_mut_set_bool(get(st, "favorited"), true); + yyjson_mut_set_str(get(st, "text"), "redacted"); + yyjson_mut_obj_remove_key(st, "entities"); + yyjson_mut_obj_add_bool(d, st, "edited", true); + } + } + else + { + yyjson_mut_set_sint(get(get(r, "search_metadata"), "count"), 200); + yyjson_mut_val* s0 = yyjson_mut_arr_get(sts, 0); + yyjson_mut_set_str(get(s0, "text"), "patched"); + yyjson_mut_set_sint(get(get(s0, "user"), "followers_count"), 1); + yyjson_mut_set_bool(get(yyjson_mut_arr_get(sts, 99), "favorited"), true); + yyjson_mut_obj_add_bool(d, r, "patched", true); + } + } + else if (name == "citm_catalog") + { + if (update) + { + std::size_t idx, max; + yyjson_mut_val* p; + yyjson_mut_arr_foreach(get(r, "performances"), idx, max, p) + { + yyjson_mut_set_str(get(p, "name"), "performance"); + yyjson_mut_val* start = get(p, "start"); + yyjson_mut_set_sint(start, yyjson_mut_get_sint(start) + 1); + yyjson_mut_obj_remove_key(p, "seatMapImage"); + yyjson_mut_obj_add_bool(d, p, "edited", true); + } + } + else + { + yyjson_mut_set_str(get(get(get(r, "events"), "138586341"), "name"), "patched"); + yyjson_mut_set_sint(get(yyjson_mut_arr_get(get(r, "performances"), 0), "start"), 0); + yyjson_mut_set_str(get(get(r, "venueNames"), "PLEYEL_PLEYEL"), "Salle"); + yyjson_mut_obj_add_bool(d, r, "patched", true); + } + } + else if (name == "canada") + { + yyjson_mut_val* f0 = yyjson_mut_arr_get(get(r, "features"), 0); + yyjson_mut_val* coords = get(get(f0, "geometry"), "coordinates"); + if (update) + { + static const double half[2] = {0.5, 0.5}; + std::size_t idx, max; + yyjson_mut_val* ring; + yyjson_mut_arr_foreach(coords, idx, max, ring) + { + yyjson_mut_arr_replace(ring, 0, yyjson_mut_arr_with_real(d, half, 2)); + } + } + else + { + static const double zero[2] = {0.0, 0.0}; + yyjson_mut_set_str(get(get(f0, "properties"), "name"), "patched"); + yyjson_mut_set_str(get(r, "type"), "FeatureCollection2"); + yyjson_mut_arr_replace(yyjson_mut_arr_get(coords, 0), 0, yyjson_mut_arr_with_real(d, zero, 2)); + } + } + else if (name == "jeopardy") + { + if (update) + { + std::size_t idx, max; + yyjson_mut_val* q; + yyjson_mut_arr_foreach(r, idx, max, q) + { + yyjson_mut_set_str(get(q, "value"), "$1"); + yyjson_mut_obj_remove_key(q, "air_date"); + } + } + else + { + yyjson_mut_set_str(get(yyjson_mut_arr_get(r, 0), "value"), "$0"); + yyjson_mut_set_str(get(yyjson_mut_arr_get(r, 100000), "answer"), "patched"); + yyjson_mut_set_str(get(yyjson_mut_arr_get(r, 216929), "round"), "x"); + yyjson_mut_val* o = yyjson_mut_obj(d); + yyjson_mut_obj_add_str(d, o, "category", "NEW"); + yyjson_mut_obj_add_str(d, o, "value", "$5"); + yyjson_mut_arr_append(r, o); + } + } + else if (name == "status") + { + yyjson_mut_set_sint(get(r, "retweet_count"), 1); + yyjson_mut_set_str(get(get(r, "user"), "name"), "x"); + } + else if (name == "rpc") + { + yyjson_mut_set_sint(get(r, "id"), 4); + yyjson_mut_set_sint(get(get(r, "params"), "subtrahend"), 24); + } + std::size_t n = 0; + char* out = yyjson_mut_write(d, 0, &n); + std::string result(out, n); + std::free(out); + yyjson_mut_doc_free(d); + return result; +} + +#if JSON_VIEW_BENCH_BOOST +// ---------------- Boost.JSON ---------------- + +static std::string edit_boost(const std::string& name, const std::string& s, bool update) +{ + bj::monotonic_resource mr; + bj::parse_options opt; + opt.numbers = bj::number_precision::precise; // correctly rounded, like the others + bj::value v = bj::parse(s, &mr, opt); + bj::object* const obj = v.if_object(); // nullptr for jeopardy (an array) + if (name == "twitter") + { + bj::array& sts = (*obj)["statuses"].as_array(); + if (update) + { + std::int64_t i = 0; + for (auto& e : sts) + { + bj::object& st = e.as_object(); + st["retweet_count"] = i++; + st["favorited"] = true; + st["text"] = "redacted"; + st.erase("entities"); + st["edited"] = true; + } + } + else + { + (*obj)["search_metadata"].as_object()["count"] = 200; + bj::object& s0 = sts[0].as_object(); + s0["text"] = "patched"; + s0["user"].as_object()["followers_count"] = 1; + sts[99].as_object()["favorited"] = true; + (*obj)["patched"] = true; + } + } + else if (name == "citm_catalog") + { + if (update) + { + for (auto& e : (*obj)["performances"].as_array()) + { + bj::object& p = e.as_object(); + p["name"] = "performance"; + p["start"] = p["start"].as_int64() + 1; + p.erase("seatMapImage"); + p["edited"] = true; + } + } + else + { + (*obj)["events"].as_object()["138586341"].as_object()["name"] = "patched"; + (*obj)["performances"].as_array()[0].as_object()["start"] = 0; + (*obj)["venueNames"].as_object()["PLEYEL_PLEYEL"] = "Salle"; + (*obj)["patched"] = true; + } + } + else if (name == "canada") + { + bj::object& f0 = (*obj)["features"].as_array()[0].as_object(); + bj::array& coords = f0["geometry"].as_object()["coordinates"].as_array(); + if (update) + { + for (auto& ring : coords) + { + ring.as_array()[0] = bj::array({0.5, 0.5}); + } + } + else + { + f0["properties"].as_object()["name"] = "patched"; + (*obj)["type"] = "FeatureCollection2"; + coords[0].as_array()[0] = bj::array({0.0, 0.0}); + } + } + else if (name == "jeopardy") + { + bj::array& a = v.as_array(); + if (update) + { + for (auto& e : a) + { + bj::object& q = e.as_object(); + q["value"] = "$1"; + q.erase("air_date"); + } + } + else + { + a[0].as_object()["value"] = "$0"; + a[100000].as_object()["answer"] = "patched"; + a[216929].as_object()["round"] = "x"; + a.push_back(bj::object({{"category", "NEW"}, {"value", "$5"}})); + } + } + else if (name == "status") + { + (*obj)["retweet_count"] = 1; + (*obj)["user"].as_object()["name"] = "x"; + } + else if (name == "rpc") + { + (*obj)["id"] = 4; + (*obj)["params"].as_object()["subtrahend"] = 24; + } + return bj::serialize(v); +} +#endif + +// ---------------- harness ---------------- + +static std::string slurp(const std::string& p) +{ + std::ifstream f(p, std::ios::binary); + if (!f) + { + std::fprintf(stderr, "cannot open %s\n", p.c_str()); + std::exit(1); + } + std::stringstream ss; + ss << f.rdbuf(); + return ss.str(); +} + +int main(int argc, char** argv) +{ + if (argc < 2) + { + std::fprintf(stderr, "usage: %s [rounds] [document]\n", argv[0]); + return 1; + } + const std::string T = std::string(argv[1]) + "/"; + const int rounds = argc > 2 ? std::atoi(argv[2]) : 20; + const std::string only = argc > 3 ? argv[3] : ""; + struct doc + { + std::string name, text; + int batch; + }; + std::vector docs; + for (const char* f : + {"nativejson-benchmark/twitter.json", "nativejson-benchmark/citm_catalog.json", "nativejson-benchmark/canada.json", "jeopardy/jeopardy.json" + }) + { + std::string n = std::string(f).substr(std::string(f).find('/') + 1); + docs.push_back({n.substr(0, n.size() - 5), slurp(T + f), 1}); + } + docs.push_back({"status", json::parse(docs[0].text)["statuses"][0].dump(), 200}); + docs.push_back({"rpc", R"({"jsonrpc": "2.0", "method": "subtract", "params": {"minuend": 42, "subtrahend": 23}, "id": 3})", 5000}); + + using fn = std::string (*)(const std::string&, const std::string&, bool); + const std::vector> engines = + { + {"json_view", edit_view}, {"yyjson", edit_yyjson}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", edit_boost}, +#endif + {"json::parse", edit_json} + }; + + std::FILE* csv = std::fopen("bench_edit.csv", "w"); + std::fprintf(csv, "doc,bytes,workload,engine,ns\n"); + for (const auto& dc : docs) + { + if (!only.empty() && dc.name != only) + { + continue; + } + for (const bool update : + { + false, true + }) + { + if (update && (dc.name == "status" || dc.name == "rpc")) + { + continue; + } + // all engines must produce the same value + const json expected = json::parse(edit_json(dc.name, dc.text, update)); + bool ok = true; + for (const auto& e : engines) + { + ok = ok && json::parse(e.second(dc.name, dc.text, update)) == expected; + } + std::vector best(engines.size(), 1e300); + const int r = dc.text.size() > 10000000 ? std::max(3, rounds / 4) : rounds; + for (int i = 0; i < r; ++i) + { + for (std::size_t k = 0; k < engines.size(); ++k) + { + // an untimed call first: whatever the previous engine left to the allocator + // (e.g. thousands of freed json nodes) is cleaned up here, not in the timing + g_sink = engines[k].second(dc.name, dc.text, update).size(); + const auto t0 = std::chrono::steady_clock::now(); + for (int b = 0; b < dc.batch; ++b) + { + g_sink = engines[k].second(dc.name, dc.text, update).size(); + } + const double ns = std::chrono::duration(std::chrono::steady_clock::now() - t0).count() / dc.batch; + best[k] = std::min(best[k], ns); + } + } + const char* wl = update ? "update" : "patch"; + std::printf("%-13s %-7s %s", dc.name.c_str(), wl, ok ? "" : "[OUTPUT MISMATCH] "); + for (std::size_t k = 0; k < engines.size(); ++k) + { + const double us = best[k] / 1e3; + std::printf(" %s %.*fus (%.2fx)", engines[k].first.c_str(), us < 10 ? 3 : (us < 1000 ? 1 : 0), us, best[k] / best[0]); + std::fprintf(csv, "%s,%zu,%s,%s,%.1f\n", dc.name.c_str(), dc.text.size(), wl, engines[k].first.c_str(), best[k]); + } + std::printf("\n"); + std::fflush(stdout); + } + } + std::fclose(csv); +} diff --git a/tests/benchmarks/json_view/bench_view.cpp b/tests/benchmarks/json_view/bench_view.cpp new file mode 100644 index 000000000..4f4234db5 --- /dev/null +++ b/tests/benchmarks/json_view/bench_view.cpp @@ -0,0 +1,749 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// Same-feature-set benchmark: read-only JSON documents with random access. +// +// json_view nlohmann/json_view.hpp (fresh document per parse / reused) +// yyjson yyjson_read(): immutable document, random access +// simdjson DOM dom::parser (reused, as recommended; "fresh": a new parser +// per parse): immutable, random access +// references (different feature sets): +// simdjson OD On-Demand: forward-only, lazy +// Boost.JSON owning, mutable DOM (monotonic resource) +// json::parse owning, mutable DOM (nlohmann today) +// +// Workloads: parse (build + free), traverse (visit everything, convert every +// number, touch every string and key), select (a few fields per document), +// dump (compact serialization of the parsed document). +// All engines run interleaved in every round, each timed call after an untimed +// one of the same engine; the best round is reported. +#include + +#if JSON_VIEW_BENCH_BOOST + #include + #include +#endif +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using nlohmann::json; +using nlohmann::json_document; +using nlohmann::json_view; + +static volatile double g_sink; + +struct stats +{ + double num = 0; + std::size_t str = 0, nodes = 0; +}; + +// ---------------- traversal ---------------- + +static void walk(json_view v, stats& st) +{ + ++st.nodes; + switch (v.type()) + { + case json::value_t::object: + for (auto it = v.begin(); it != v.end(); ++it) + { + st.str += it.key().size(); + walk(*it, st); + } + break; + case json::value_t::array: + for (const json_view e : v) + { + walk(e, st); + } + break; + case json::value_t::string: + st.str += v.get_string().size(); + break; + case json::value_t::number_integer: + st.num += static_cast(v.get()); + break; + case json::value_t::number_unsigned: + st.num += static_cast(v.get()); + break; + case json::value_t::number_float: + st.num += v.get(); + break; + default: + break; + } +} + +static void walk(const json& j, stats& st) +{ + ++st.nodes; + switch (j.type()) + { + case json::value_t::object: + for (const auto& kv : j.get_ref()) + { + st.str += kv.first.size(); + walk(kv.second, st); + } + break; + case json::value_t::array: + for (const auto& e : j.get_ref()) + { + walk(e, st); + } + break; + case json::value_t::string: + st.str += j.get_ref().size(); + break; + case json::value_t::number_integer: + st.num += static_cast(*j.get_ptr()); + break; + case json::value_t::number_unsigned: + st.num += static_cast(*j.get_ptr()); + break; + case json::value_t::number_float: + st.num += *j.get_ptr(); + break; + default: + break; + } +} + +static void walk(yyjson_val* v, stats& st) +{ + ++st.nodes; + switch (yyjson_get_type(v)) + { + case YYJSON_TYPE_OBJ: + { + std::size_t idx, max; + yyjson_val* k, * val; + yyjson_obj_foreach(v, idx, max, k, val) + { + st.str += yyjson_get_len(k); + walk(val, st); + } + break; + } + case YYJSON_TYPE_ARR: + { + std::size_t idx, max; + yyjson_val* val; + yyjson_arr_foreach(v, idx, max, val) + { + walk(val, st); + } + break; + } + case YYJSON_TYPE_STR: + st.str += yyjson_get_len(v); + break; + case YYJSON_TYPE_NUM: + if (yyjson_is_sint(v)) + { + st.num += static_cast(yyjson_get_sint(v)); + } + else if (yyjson_is_uint(v)) + { + st.num += static_cast(yyjson_get_uint(v)); + } + else + { + st.num += yyjson_get_real(v); + } + break; + default: + break; + } +} + +static void walk(simdjson::dom::element e, stats& st) +{ + ++st.nodes; + switch (e.type()) + { + case simdjson::dom::element_type::OBJECT: + for (auto f : simdjson::dom::object(e)) + { + st.str += f.key.size(); + walk(f.value, st); + } + break; + case simdjson::dom::element_type::ARRAY: + for (auto c : simdjson::dom::array(e)) + { + walk(c, st); + } + break; + case simdjson::dom::element_type::STRING: + st.str += std::string_view(e).size(); + break; + case simdjson::dom::element_type::INT64: + st.num += static_cast(int64_t(e)); + break; + case simdjson::dom::element_type::UINT64: + st.num += static_cast(uint64_t(e)); + break; + case simdjson::dom::element_type::DOUBLE: + st.num += double(e); + break; + default: + break; + } +} + +static void walk_od(simdjson::ondemand::value v, stats& st) +{ + ++st.nodes; + switch (v.type()) + { + case simdjson::ondemand::json_type::object: + for (auto f : v.get_object()) + { + st.str += std::string_view(f.unescaped_key()).size(); + walk_od(f.value(), st); + } + break; + case simdjson::ondemand::json_type::array: + for (auto c : v.get_array()) + { + walk_od(c.value(), st); + } + break; + case simdjson::ondemand::json_type::string: + st.str += std::string_view(v.get_string()).size(); + break; + case simdjson::ondemand::json_type::number: + { + simdjson::ondemand::number n = v.get_number(); + switch (n.get_number_type()) + { + case simdjson::ondemand::number_type::signed_integer: + st.num += static_cast(n.get_int64()); + break; + case simdjson::ondemand::number_type::unsigned_integer: + st.num += static_cast(n.get_uint64()); + break; + default: + st.num += n.get_double(); + break; + } + break; + } + case simdjson::ondemand::json_type::boolean: + (void)bool(v.get_bool()); + break; + default: + (void)v.is_null(); + break; + } +} + +#if JSON_VIEW_BENCH_BOOST +static void walk(const boost::json::value& v, stats& st) +{ + ++st.nodes; + switch (v.kind()) + { + case boost::json::kind::object: + for (const auto& kv : v.get_object()) + { + st.str += kv.key().size(); + walk(kv.value(), st); + } + break; + case boost::json::kind::array: + for (const auto& c : v.get_array()) + { + walk(c, st); + } + break; + case boost::json::kind::string: + st.str += v.get_string().size(); + break; + case boost::json::kind::int64: + st.num += static_cast(v.get_int64()); + break; + case boost::json::kind::uint64: + st.num += static_cast(v.get_uint64()); + break; + case boost::json::kind::double_: + st.num += v.get_double(); + break; + default: + break; + } +} +#endif + +// ---------------- selective access ---------------- +// twitter: per status id, user.screen_name, retweet_count +// citm: per performance id, eventId, #seatCategories; #events +// canada: type, features[0].geometry.type, #coordinates +// jeopardy: per question: round == "Final Jeopardy!", len(category) +// status (one tweet): id, user.screen_name, retweet_count +// rpc: method, params.minuend, id + +static double pick(const std::string& name, json_view r) +{ + double acc = 0; + if (name == "twitter" || name == "status") + { + auto one = [&](json_view s) + { + acc += static_cast(s["id"].get()); + acc += static_cast(s["user"]["screen_name"].get_string().size()); + acc += static_cast(s["retweet_count"].get()); + }; + if (name == "twitter") + { + for (const json_view s : r["statuses"]) + { + one(s); + } + } + else + { + one(r); + } + } + else if (name == "citm_catalog") + { + for (const json_view p : r["performances"]) + { + acc += static_cast(p["id"].get() + p["eventId"].get() + p["seatCategories"].size()); + } + acc += static_cast(r["events"].size()); + } + else if (name == "canada") + { + const json_view g = r["features"][0]["geometry"]; + acc += static_cast(r["type"].get_string().size() + g["type"].get_string().size() + g["coordinates"].size()); + } + else if (name == "jeopardy") + { + for (const json_view q : r) + { + acc += q["round"].get_string() == "Final Jeopardy!" ? 1 : 0; + acc += static_cast(q["category"].get_string().size()); + } + } + else if (name == "rpc") + { + acc += static_cast(r["method"].get_string().size()); + acc += static_cast(r["params"]["minuend"].get() + r["id"].get()); + } + return acc; +} + +static double pick(const std::string& name, const json& r) +{ + double acc = 0; + if (name == "twitter" || name == "status") + { + auto one = [&](const json & s) + { + acc += static_cast(s["id"].get()); + acc += static_cast(s["user"]["screen_name"].get_ref().size()); + acc += static_cast(s["retweet_count"].get()); + }; + if (name == "twitter") + { + for (const auto& s : r["statuses"]) + { + one(s); + } + } + else + { + one(r); + } + } + else if (name == "citm_catalog") + { + for (const auto& p : r["performances"]) + { + acc += static_cast(p["id"].get() + p["eventId"].get() + p["seatCategories"].size()); + } + acc += static_cast(r["events"].size()); + } + else if (name == "canada") + { + const json& g = r["features"][0]["geometry"]; + acc += static_cast(r["type"].get_ref().size() + g["type"].get_ref().size() + g["coordinates"].size()); + } + else if (name == "jeopardy") + { + for (const auto& q : r) + { + acc += q["round"].get_ref() == "Final Jeopardy!" ? 1 : 0; + acc += static_cast(q["category"].get_ref().size()); + } + } + else if (name == "rpc") + { + acc += static_cast(r["method"].get_ref().size()); + acc += static_cast(r["params"]["minuend"].get() + r["id"].get()); + } + return acc; +} + +static double pick(const std::string& name, yyjson_val* r) +{ + double acc = 0; + auto get = [](yyjson_val * o, const char* k) + { + return yyjson_obj_get(o, k); + }; + if (name == "twitter" || name == "status") + { + auto one = [&](yyjson_val * s) + { + acc += static_cast(yyjson_get_uint(get(s, "id"))); + acc += static_cast(yyjson_get_len(get(get(s, "user"), "screen_name"))); + acc += static_cast(yyjson_get_sint(get(s, "retweet_count"))); + }; + if (name == "twitter") + { + std::size_t idx, max; + yyjson_val* s; + yyjson_arr_foreach(get(r, "statuses"), idx, max, s) + { + one(s); + } + } + else + { + one(r); + } + } + else if (name == "citm_catalog") + { + std::size_t idx, max; + yyjson_val* p; + yyjson_arr_foreach(get(r, "performances"), idx, max, p) + { + acc += static_cast(yyjson_get_uint(get(p, "id")) + yyjson_get_uint(get(p, "eventId")) + yyjson_arr_size(get(p, "seatCategories"))); + } + acc += static_cast(yyjson_obj_size(get(r, "events"))); + } + else if (name == "canada") + { + yyjson_val* g = get(yyjson_arr_get(get(r, "features"), 0), "geometry"); + acc += static_cast(yyjson_get_len(get(r, "type")) + yyjson_get_len(get(g, "type")) + yyjson_arr_size(get(g, "coordinates"))); + } + else if (name == "jeopardy") + { + std::size_t idx, max; + yyjson_val* q; + yyjson_arr_foreach(r, idx, max, q) + { + acc += yyjson_equals_str(get(q, "round"), "Final Jeopardy!") ? 1 : 0; + acc += static_cast(yyjson_get_len(get(q, "category"))); + } + } + else if (name == "rpc") + { + acc += static_cast(yyjson_get_len(get(r, "method"))); + acc += static_cast(yyjson_get_sint(get(get(r, "params"), "minuend")) + yyjson_get_sint(get(r, "id"))); + } + return acc; +} + +static double pick(const std::string& name, simdjson::dom::element r) +{ + double acc = 0; + if (name == "twitter" || name == "status") + { + auto one = [&](simdjson::dom::element s) + { + acc += static_cast(uint64_t(s["id"])); + acc += static_cast(std::string_view(s["user"]["screen_name"]).size()); + acc += static_cast(int64_t(s["retweet_count"])); + }; + if (name == "twitter") + { + for (auto s : simdjson::dom::array(r["statuses"])) + { + one(s); + } + } + else + { + one(r); + } + } + else if (name == "citm_catalog") + { + for (auto p : simdjson::dom::array(r["performances"])) + { + acc += static_cast(uint64_t(p["id"]) + uint64_t(p["eventId"]) + simdjson::dom::array(p["seatCategories"]).size()); + } + acc += static_cast(simdjson::dom::object(r["events"]).size()); + } + else if (name == "canada") + { + auto g = r["features"].at(0)["geometry"]; + acc += static_cast(std::string_view(r["type"]).size() + std::string_view(g["type"]).size() + simdjson::dom::array(g["coordinates"]).size()); + } + else if (name == "jeopardy") + { + for (auto q : simdjson::dom::array(r)) + { + acc += std::string_view(q["round"]) == "Final Jeopardy!" ? 1 : 0; + acc += static_cast(std::string_view(q["category"]).size()); + } + } + else if (name == "rpc") + { + acc += static_cast(std::string_view(r["method"]).size()); + acc += static_cast(int64_t(r["params"]["minuend"]) + int64_t(r["id"])); + } + return acc; +} + +static double pick_od(const std::string& name, simdjson::ondemand::document& d) +{ + double acc = 0; + if (name == "twitter" || name == "status") + { + auto one = [&](simdjson::ondemand::object s) + { + acc += static_cast(uint64_t(s["id"])); + acc += static_cast(std::string_view(s["user"]["screen_name"]).size()); + acc += static_cast(int64_t(s["retweet_count"])); + }; + if (name == "twitter") + { + for (auto s : d["statuses"]) + { + one(s.get_object()); + } + } + else + { + one(d.get_object()); + } + } + else if (name == "citm_catalog") + { + simdjson::ondemand::object ev = d["events"].get_object(); + acc += static_cast(ev.count_fields()); + for (auto p : d["performances"]) + { + simdjson::ondemand::object o = p.get_object(); + const auto a = uint64_t(o["eventId"]) + uint64_t(o["id"]); + simdjson::ondemand::array sc = o["seatCategories"].get_array(); + acc += static_cast(a + sc.count_elements()); + } + } + else if (name == "canada") + { + acc += static_cast(std::string_view(d["type"]).size()); + auto g = d["features"].at(0)["geometry"]; + acc += static_cast(std::string_view(g["type"]).size()); + simdjson::ondemand::array co = g["coordinates"].get_array(); + acc += static_cast(co.count_elements()); + } + else if (name == "jeopardy") + { + for (auto q : d) + { + simdjson::ondemand::object o = q.get_object(); + acc += static_cast(std::string_view(o["category"]).size()); + acc += std::string_view(o["round"]) == "Final Jeopardy!" ? 1 : 0; + } + } + else if (name == "rpc") + { + acc += static_cast(std::string_view(d["method"]).size()); + acc += static_cast(int64_t(d["params"]["minuend"])); + acc += static_cast(int64_t(d["id"])); + } + return acc; +} + +// ---------------- harness ---------------- + +static std::string slurp(const std::string& p) +{ + std::ifstream f(p, std::ios::binary); + if (!f) + { + std::fprintf(stderr, "cannot open %s\n", p.c_str()); + std::exit(1); + } + std::stringstream ss; + ss << f.rdbuf(); + return ss.str(); +} + +struct engine +{ + std::string name; + std::function fn; +}; + +int main(int argc, char** argv) +{ + if (argc < 2) + { + std::fprintf(stderr, "usage: %s [rounds] [document]\n", argv[0]); + return 1; + } + const std::string T = std::string(argv[1]) + "/"; + const int rounds = argc > 2 ? std::atoi(argv[2]) : 30; + const std::string only = argc > 3 ? argv[3] : ""; + struct doc + { + std::string name, text; + int batch; + }; + std::vector docs; + for (const char* f : + {"nativejson-benchmark/twitter.json", "nativejson-benchmark/citm_catalog.json", "nativejson-benchmark/canada.json", "jeopardy/jeopardy.json" + }) + { + std::string n = std::string(f).substr(std::string(f).find('/') + 1); + docs.push_back({n.substr(0, n.size() - 5), slurp(T + f), 1}); + } + docs.push_back({"status", json::parse(docs[0].text)["statuses"][0].dump(), 200}); + docs.push_back({"rpc", R"({"jsonrpc": "2.0", "method": "subtract", "params": {"minuend": 42, "subtrahend": 23}, "id": 3})", 5000}); + + // correctness cross-check of the workloads + for (const auto& dc : docs) + { + stats a, b, c, dd; + walk(json::parse(dc.text), a); + auto d = json_document::parse(dc.text); + walk(d.root(), b); + yyjson_doc* y = yyjson_read(dc.text.data(), dc.text.size(), 0); + walk(yyjson_doc_get_root(y), c); + simdjson::dom::parser p; + walk(p.parse(dc.text).value(), dd); + const bool ok = a.nodes == b.nodes && a.nodes == c.nodes && a.nodes == dd.nodes && a.str == b.str && a.str == c.str && a.str == dd.str + && std::fabs(a.num - b.num) <= 1e-9 * std::fabs(a.num) && std::fabs(a.num - c.num) <= 1e-9 * std::fabs(a.num); + const double pa = pick(dc.name, json::parse(dc.text)), pb = pick(dc.name, d.root()), pc = pick(dc.name, yyjson_doc_get_root(y)), pd = pick(dc.name, p.parse(dc.text).value()); + std::printf("check %-13s traverse %s select %s\n", dc.name.c_str(), ok ? "OK" : "MISMATCH", (pa == pb && pa == pc && pa == pd) ? "OK" : "MISMATCH"); + yyjson_doc_free(y); + } + + std::FILE* csv = std::fopen("bench_view.csv", "w"); + std::fprintf(csv, "doc,bytes,workload,engine,ns\n"); + json_document reused; + simdjson::dom::parser sj; + simdjson::ondemand::parser od; + for (const auto& dc : docs) + { + if (!only.empty() && dc.name != only) + { + continue; + } + const std::string& s = dc.text; + const simdjson::padded_string ps(s); + const std::string name = dc.name; + std::vector>> workloads; + + workloads.push_back({"parse", { + {"json_view", [&] { auto d = json_document::parse(s); g_sink = static_cast(d.node_count()); }}, + {"json_view (reused)", [&] { reused.read(s); g_sink = static_cast(reused.node_count()); }}, + {"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast(yyjson_doc_get_val_count(d)); yyjson_doc_free(d); }}, + {"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }}, + {"simdjson DOM (fresh)", [&] { simdjson::dom::parser p; auto e = p.parse(ps).value_unsafe(); g_sink = e.is_object(); }}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }}, +#endif + {"json::parse", [&] { json j = json::parse(s); g_sink = static_cast(j.size()); }}, + }}); + workloads.push_back({"traverse", { + {"json_view", [&] { auto d = json_document::parse(s); stats st; walk(d.root(), st); g_sink = st.num; }}, + {"json_view (reused)", [&] { reused.read(s); stats st; walk(reused.root(), st); g_sink = st.num; }}, + {"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(d), st); g_sink = st.num; yyjson_doc_free(d); }}, + {"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }}, + {"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); stats st; walk_od(d.get_value().value_unsafe(), st); g_sink = st.num; }}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); stats st; walk(v, st); g_sink = st.num; }}, +#endif + {"json::parse", [&] { json j = json::parse(s); stats st; walk(j, st); g_sink = st.num; }}, + }}); + workloads.push_back({"select", { + {"json_view", [&] { auto d = json_document::parse(s); g_sink = pick(name, d.root()); }}, + {"json_view (reused)", [&] { reused.read(s); g_sink = pick(name, reused.root()); }}, + {"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = pick(name, yyjson_doc_get_root(d)); yyjson_doc_free(d); }}, + {"simdjson DOM", [&] { g_sink = pick(name, sj.parse(ps).value_unsafe()); }}, + {"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); g_sink = pick_od(name, d); }}, + {"json::parse", [&] { json j = json::parse(s); g_sink = pick(name, j); }}, + }}); + { + // serialization of an already parsed document + static json_document vd; + vd.read(s); + static yyjson_doc* yd = nullptr; + if (yd) + { + yyjson_doc_free(yd); + } + yd = yyjson_read(s.data(), s.size(), 0); + static simdjson::dom::parser sjd; + static simdjson::dom::element se; + se = sjd.parse(ps).value_unsafe(); + static json jd; + jd = json::parse(s); + workloads.push_back({"dump", { + {"json_view", [&] { std::string o = vd.root().dump(); g_sink = static_cast(o.size()); }}, + {"yyjson", [&] { std::size_t n = 0; char* o = yyjson_write(yd, 0, &n); g_sink = static_cast(n); std::free(o); }}, + {"simdjson DOM", [&] { std::string o = simdjson::to_string(se); g_sink = static_cast(o.size()); }}, + {"json::parse", [&] { std::string o = jd.dump(); g_sink = static_cast(o.size()); }}, + }}); + } + + for (auto& wl : workloads) + { + std::vector best(wl.second.size(), 1e300); + const int r = s.size() > 10000000 ? std::max(3, rounds / 5) : rounds; + for (int i = 0; i < r; ++i) + { + for (std::size_t k = 0; k < wl.second.size(); ++k) + { + // an untimed call first: whatever the previous engine left to the allocator + // (e.g. thousands of freed json nodes) is cleaned up here, not in the timing + wl.second[k].fn(); + const auto t0 = std::chrono::steady_clock::now(); + for (int b = 0; b < dc.batch; ++b) + { + wl.second[k].fn(); + } + const double ns = std::chrono::duration(std::chrono::steady_clock::now() - t0).count() / dc.batch; + best[k] = std::min(best[k], ns); + } + } + const double ref = best[0]; + std::printf("%-13s %-9s", dc.name.c_str(), wl.first.c_str()); + for (std::size_t k = 0; k < wl.second.size(); ++k) + { + const double us = best[k] / 1e3; + std::printf(" %s %s%s (%.2fx)", wl.second[k].name.c_str(), us >= 100 ? "" : "", (us >= 1000 ? std::to_string(static_cast(us)) + "us" : (std::to_string(us).substr(0, 5) + "us")).c_str(), best[k] / ref); + std::fprintf(csv, "%s,%zu,%s,%s,%.1f\n", dc.name.c_str(), s.size(), wl.first.c_str(), wl.second[k].name.c_str(), best[k]); + } + std::printf("\n"); + std::fflush(stdout); + } + } + std::fclose(csv); +} diff --git a/tests/benchmarks/json_view/compare.py b/tests/benchmarks/json_view/compare.py new file mode 100755 index 000000000..32a28b115 --- /dev/null +++ b/tests/benchmarks/json_view/compare.py @@ -0,0 +1,313 @@ +#!/usr/bin/env python3 +# __ _____ _____ _____ +# __| | __| | | | JSON for Modern C++ (supporting code) +# | | |__ | | | | | | version 3.12.0 +# |_____|_____|_____|_|___| https://github.com/nlohmann/json +# +# SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +# SPDX-License-Identifier: MIT + +"""Compare json_view with yyjson, simdjson, Boost.JSON, and json::parse. + +Builds bench_view.cpp, bench_corpus.cpp, and bench_edit.cpp against the include/ directory of +this checkout, runs them, and writes the results with everything needed to +reproduce them (date, commit, CPU, OS, compiler, library versions, flags) to +results/-.md and .csv next to this script. + +The other libraries come from the system (--system, the default: pkg-config +or Homebrew) or are downloaded as pinned releases and checked against their +SHA-256 (--download). Boost.JSON is optional: without Boost headers, its +columns are skipped, and the results say so. + +Only the Python 3 standard library is used; a C++17 compiler is needed. +""" + +import argparse +import datetime +import hashlib +import os +import platform +import re +import shlex +import shutil +# runs only the compilers and benchmark binaries this script builds +import subprocess # nosec B404 +import sys +import tarfile +import urllib.request + +HERE = os.path.dirname(os.path.abspath(__file__)) +REPO = os.path.abspath(os.path.join(HERE, '..', '..', '..')) + +# pinned releases for --download; the hashes are those of the archives +PINNED = { + 'yyjson': { + 'version': '0.13.0', + 'url': 'https://github.com/ibireme/yyjson/archive/refs/tags/0.13.0.tar.gz', + 'sha256': '34e0f62a2bc11ab20d601e8ca1cc2b2079503aa45119a19133d89d19b94a0fae', + 'dir': 'yyjson-0.13.0', + }, + 'simdjson': { + 'version': '4.6.11', + 'url': 'https://github.com/simdjson/simdjson/archive/refs/tags/v4.6.11.tar.gz', + 'sha256': '61d948fc24f0d793829ad658058e7597d064988a89b4607ea02e401a82df98ff', + 'dir': 'simdjson-4.6.11', + }, + 'boost': { + 'version': '1.92.0', + 'url': 'https://archives.boost.io/release/1.92.0/source/boost_1_92_0.tar.gz', + 'sha256': 'c4a3b310ddd2472416e091067166b0713be97c63f38c212c484ada022fd296ce', + 'dir': 'boost_1_92_0', + }, +} + +# the documents of bench_view.cpp, relative to the json_test_data directory +DEFAULT_CORPUS = [ + 'nativejson-benchmark/twitter.json', + 'nativejson-benchmark/citm_catalog.json', + 'nativejson-benchmark/canada.json', + 'jeopardy/jeopardy.json', +] + + +def run(cmd, **kwargs): + print('+ ' + ' '.join(shlex.quote(c) for c in cmd), flush=True) + # cmd is an argument list built by this script, never a shell string + return subprocess.run(cmd, check=True, **kwargs) # nosec B603 + + +def output(cmd): + try: + # cmd is an argument list built by this script, never a shell string + return subprocess.run(cmd, check=True, capture_output=True, text=True).stdout.strip() # nosec B603 + except (OSError, subprocess.CalledProcessError): + return '' + + +# --------------------------------------------------------------------------- +# libraries +# --------------------------------------------------------------------------- + +class Library: + """include directories, sources to compile, and linker flags of a library""" + + def __init__(self, name, include=None, sources=None, link=None, version=''): + self.name = name + self.include = include or [] + self.sources = sources or [] + self.link = link or [] + self.version = version + + +def header_version(path, pattern): + try: + with open(path, encoding='utf-8', errors='replace') as f: + m = re.search(pattern, f.read()) + return m.group(1) if m else '' + except OSError: + return '' + + +def library_version(name, include_dirs): + patterns = { + 'yyjson': ('yyjson.h', r'#define\s+YYJSON_VERSION_STRING\s+"([^"]+)"'), + 'simdjson': ('simdjson.h', r'#define\s+SIMDJSON_VERSION\s+"?([0-9.]+)"?'), + 'boost': (os.path.join('boost', 'version.hpp'), r'#define\s+BOOST_LIB_VERSION\s+"([^"]+)"'), + } + header, pattern = patterns[name] + for d in include_dirs: + v = header_version(os.path.join(d, header), pattern) + if v: + return v.replace('_', '.') + return '' + + +def system_library(name): + """a library found with pkg-config or Homebrew, or None""" + flags = output(['pkg-config', '--cflags', '--libs', name]).split() + if flags: + include = [f[2:] for f in flags if f.startswith('-I')] + link = [f for f in flags if f.startswith('-L') or f.startswith('-l')] + libdirs = [f[2:] for f in link if f.startswith('-L')] + link += ['-Wl,-rpath,' + d for d in libdirs] + return Library(name, include, [], link, library_version(name, include)) + prefix = output(['brew', '--prefix', name]) if shutil.which('brew') else '' + if prefix and os.path.isdir(os.path.join(prefix, 'include')): + include = [os.path.join(prefix, 'include')] + link = [] + if name != 'boost': + lib = os.path.join(prefix, 'lib') + link = ['-L' + lib, '-l' + name, '-Wl,-rpath,' + lib] + return Library(name, include, [], link, library_version(name, include)) + if name == 'boost': + for d in ['/usr/include', '/usr/local/include']: + if os.path.isfile(os.path.join(d, 'boost', 'json.hpp')): + return Library(name, [d], [], [], library_version(name, [d])) + return None + + +def download_library(name, work): + """a pinned release, downloaded and checked, or an error""" + pin = PINNED[name] + archive = os.path.join(work, 'download', os.path.basename(pin['url'])) + os.makedirs(os.path.dirname(archive), exist_ok=True) + if not os.path.isfile(archive): + print(f'downloading {pin["url"]}', flush=True) + # the URLs are the https constants in PINNED, and the SHA-256 is checked below + # (into a .part file first, so that an interrupted download is not kept) + urllib.request.urlretrieve(pin['url'], archive + '.part') # nosec B310 + os.replace(archive + '.part', archive) + with open(archive, 'rb') as f: + digest = hashlib.sha256(f.read()).hexdigest() + if digest != pin['sha256']: + os.remove(archive) # downloaded again by the next run + sys.exit(f'error: SHA-256 of {archive} is {digest}, expected {pin["sha256"]} (removed)') + src = os.path.join(work, 'download', pin['dir']) + if not os.path.isdir(src): + with tarfile.open(archive) as t: + # (the 'data' filter rejects links and paths outside the target where Python has it) + kwargs = {'filter': 'data'} if hasattr(tarfile, 'data_filter') else {} + t.extractall(os.path.join(work, 'download'), **kwargs) # noqa: S202 (checked archive) # nosec B202 + if name == 'yyjson': + return Library(name, [os.path.join(src, 'src')], [os.path.join(src, 'src', 'yyjson.c')], [], pin['version']) + if name == 'simdjson': + single = os.path.join(src, 'singleheader') + return Library(name, [single], [os.path.join(single, 'simdjson.cpp')], [], pin['version']) + return Library(name, [src], [], [], pin['version']) + + +# --------------------------------------------------------------------------- +# machine description +# --------------------------------------------------------------------------- + +def cpu_model(): + if sys.platform == 'darwin': + return output(['sysctl', '-n', 'machdep.cpu.brand_string']) + try: + with open('/proc/cpuinfo', encoding='utf-8') as f: + for line in f: + if line.startswith('model name') or line.startswith('Model'): + return line.split(':', 1)[1].strip() + except OSError: + pass + # (AArch64 Linux: /proc/cpuinfo has no model name, lscpu knows it) + for line in output(['lscpu']).splitlines(): + if line.startswith('Model name:'): + return line.split(':', 1)[1].strip() + return platform.processor() or platform.machine() + + +def git_commit(): + commit = output(['git', '-C', REPO, 'rev-parse', '--short=12', 'HEAD']) + dirty = output(['git', '-C', REPO, 'status', '--porcelain', '--untracked-files=no']) + return commit + (' (with local changes)' if dirty else '') + + +# --------------------------------------------------------------------------- +# main +# --------------------------------------------------------------------------- + +def main(): + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument('--data', required=True, help='json_test_data directory (with nativejson-benchmark/ and jeopardy/)') + ap.add_argument('--download', action='store_true', help='use pinned downloads instead of system libraries') + ap.add_argument('--no-boost', action='store_true', help='skip Boost.JSON') + ap.add_argument('--native', action='store_true', help='compile for this CPU (-march=native / -mcpu=native)') + ap.add_argument('--rounds', type=int, default=30, help='rounds of bench_view (default: 30)') + ap.add_argument('--corpus', nargs='*', default=[], help='more files for bench_corpus') + ap.add_argument('--build-dir', default=os.path.join(HERE, 'build'), help='where to build (default: build/ next to this script)') + args = ap.parse_args() + # the benchmarks run in the build directory: make the paths absolute + args.data = os.path.abspath(args.data) + args.corpus = [os.path.abspath(f) for f in args.corpus] + args.build_dir = os.path.abspath(args.build_dir) + + cxx = os.environ.get('CXX', 'c++') + cc = os.environ.get('CC', 'cc') + os.makedirs(args.build_dir, exist_ok=True) + + libs = {} + for name in ['yyjson', 'simdjson', 'boost']: + if name == 'boost' and args.no_boost: + continue + lib = download_library(name, args.build_dir) if args.download else system_library(name) + if lib is None and name != 'boost': + sys.exit(f'error: {name} not found; install it, or use --download') + if lib is not None: + libs[name] = lib + with_boost = 'boost' in libs + if not with_boost: + print('Boost.JSON not found: its columns are skipped', flush=True) + + flags = ['-std=c++17', '-O3', '-DNDEBUG', f'-DJSON_VIEW_BENCH_BOOST={1 if with_boost else 0}'] + if args.native: + flags.append('-mcpu=native' if platform.machine().lower() in ('arm64', 'aarch64') else '-march=native') + include = ['-I' + os.path.join(REPO, 'include')] + ['-I' + d for lib in libs.values() for d in lib.include] + link = [f for lib in libs.values() for f in lib.link] + + # C sources of downloaded libraries are compiled once + objects = [] + for lib in libs.values(): + for src in lib.sources: + obj = os.path.join(args.build_dir, os.path.basename(src) + '.o') + compiler = cc if src.endswith('.c') else cxx + run([compiler] + (['-std=c++17'] if compiler == cxx else []) + ['-O3', '-DNDEBUG', '-c', src, '-o', obj] + + ['-I' + d for d in lib.include]) + objects.append(obj) + + binaries = {} + for bench in ['bench_view', 'bench_corpus', 'bench_edit']: + exe = os.path.join(args.build_dir, bench) + run([cxx] + flags + include + [os.path.join(HERE, bench + '.cpp')] + objects + link + ['-o', exe]) + binaries[bench] = exe + + # run: bench_view on its documents, bench_corpus on those and the given files + corpus = [os.path.join(args.data, f) for f in DEFAULT_CORPUS] + args.corpus + outputs = {} + outputs['bench_view'] = run([binaries['bench_view'], args.data, str(args.rounds)], cwd=args.build_dir, + capture_output=True, text=True).stdout + outputs['bench_corpus'] = run([binaries['bench_corpus']] + corpus, cwd=args.build_dir, + capture_output=True, text=True).stdout + outputs['bench_edit'] = run([binaries['bench_edit'], args.data, str(max(1, args.rounds // 2))], cwd=args.build_dir, + capture_output=True, text=True).stdout + for name, text in outputs.items(): + print(text) + + # results with their metadata + now = datetime.datetime.now() + host = re.sub(r'[^A-Za-z0-9-]+', '-', platform.node().split('.')[0]) or 'host' + stem = os.path.join(HERE, 'results', f'{now:%Y-%m-%d}-{host}') + os.makedirs(os.path.dirname(stem), exist_ok=True) + meta = [ + ('date', f'{now:%Y-%m-%d %H:%M}'), + ('commit', git_commit()), + ('CPU', cpu_model()), + ('OS', f'{platform.system()} {platform.release()} ({platform.machine()})'), + ('compiler', output([cxx, '--version']).splitlines()[0] if output([cxx, '--version']) else cxx), + ('flags', ' '.join(flags)), + ('yyjson', libs['yyjson'].version), + ('simdjson', libs['simdjson'].version), + ('Boost.JSON', libs['boost'].version if with_boost else 'skipped (not found)'), + ('libraries from', 'pinned downloads' if args.download else 'the system'), + ('rounds', str(args.rounds)), + ] + with open(stem + '.md', 'w', encoding='utf-8') as f: + f.write(f'# json_view comparison, {now:%Y-%m-%d}\n\n') + f.write('Generated by `tests/benchmarks/json_view/compare.py`; best of the interleaved rounds.\n\n') + f.write('| | |\n|---|---|\n') + for key, value in meta: + f.write(f'| {key} | {value} |\n') + for name, text in outputs.items(): + f.write(f'\n## {name}\n\n```\n{text.rstrip()}\n```\n') + with open(stem + '.csv', 'w', encoding='utf-8') as out: + out.write(''.join(f'# {key}: {value}\n' for key, value in meta)) + for name in ['bench_view', 'bench_corpus', 'bench_edit']: + path = os.path.join(args.build_dir, name + '.csv') + if os.path.isfile(path): + with open(path, encoding='utf-8') as f: + out.write(f'# {name}\n' + f.read()) + print(f'results: {stem}.md, {stem}.csv') + + +if __name__ == '__main__': + main()