diff --git a/.github/labeler.yml b/.github/labeler.yml index 7884e69fb..a135a10f1 100644 --- a/.github/labeler.yml +++ b/.github/labeler.yml @@ -52,6 +52,7 @@ labels: - "single_include/nlohmann/json_view\\.hpp" - "tests/src/unit-json_view.*" - "tests/src/fuzzer-parse_json_view\\.cpp" + - "tests/benchmarks/json_view/.*" - "tools/amalgamate/config_json_view\\.json" - "docs/mkdocs/docs/features/json_view\\.md" - "docs/mkdocs/docs/api/basic_json_(document|view)/.*" diff --git a/tests/benchmarks/json_view/.gitignore b/tests/benchmarks/json_view/.gitignore new file mode 100644 index 000000000..567609b12 --- /dev/null +++ b/tests/benchmarks/json_view/.gitignore @@ -0,0 +1 @@ +build/ diff --git a/tests/benchmarks/json_view/README.md b/tests/benchmarks/json_view/README.md new file mode 100644 index 000000000..d5e0678c9 --- /dev/null +++ b/tests/benchmarks/json_view/README.md @@ -0,0 +1,65 @@ +# json_view compared with other libraries + +The in-tree benchmarks in [`tests/benchmarks`](../README.md) measure `json_document` against `json::parse` only. The +programs here compare it with [yyjson](https://github.com/ibireme/yyjson), +[simdjson](https://github.com/simdjson/simdjson), and [Boost.JSON](https://github.com/boostorg/json): the question +users ask when they pick a library. They are not built by CMake or run by CI. + +## Reproducing the numbers + +`compare.py` builds both programs against `include/` of this checkout, runs them, and writes the results together with +everything needed to reproduce them to `results/-.md` (and `.csv`): the date, the commit, the CPU, the +OS, the compiler, the flags, and the versions of all libraries. + +```sh +python3 tests/benchmarks/json_view/compare.py --data [--native] [--rounds 30] +``` + +- `--data` is the downloaded [test data](https://github.com/nlohmann/json_test_data), e.g. the `test_files` directory + of a CMake build directory. It needs `nativejson-benchmark/{twitter,citm_catalog,canada}.json` and + `jeopardy/jeopardy.json`. +- The other libraries come from the system: pkg-config, or Homebrew (`brew install yyjson simdjson boost`). With + `--download`, pinned releases are downloaded instead and checked against their SHA-256. Without Boost headers (or + with `--no-boost`), the Boost.JSON columns are skipped, and the results say so. +- `--corpus file...` adds files to the corpus benchmark, e.g. those of + [simdjson-data](https://github.com/simdjson/simdjson-data) or the + [yyjson benchmark](https://github.com/ibireme/yyjson_benchmark). +- Only the Python 3 standard library is used; a C++17 compiler is needed (`CXX` and `CC` are honored). + +For numbers worth publishing, use a quiet machine (see [Getting stable numbers](../README.md#getting-stable-numbers)), +the default 30 rounds or more, and `--native` only if the other libraries were built for the same CPU. + +## What is measured + +`bench_view.cpp` runs four workloads on twitter, citm_catalog, canada, jeopardy, a single tweet (`status`), and a +JSON-RPC request (`rpc`): + +| workload | what it does | +|---|---| +| parse | build and free a document | +| traverse | parse, then visit every value, convert every number, touch every string and key | +| select | parse, then read a few fields per record (e.g. id, user name, and retweet count of each tweet) | +| dump | serialize a parsed document (compact) | + +`bench_corpus.cpp` runs parse, traverse, and dump on any list of files, so that no library is tuned to a handful of +documents. + +Before anything is timed, all engines must accept each document and agree on the traversal: the number of values, the +bytes of all strings and keys, and the sum of all numbers. All engines run interleaved in every round, and the best +round is reported, as time and as a factor of the `json_view` time (below 1 means faster than `json_view`). + +The engines do not all offer the same features, which the numbers should be read with: + +| engine | document | random access | editable | notes | +|---|---|---|---|---| +| `json_view` | immutable index into the text | yes | no | a fresh document per parse; "reused" parses into the same document | +| yyjson | immutable (`yyjson_read`) | yes | via a mutable copy | | +| simdjson DOM | immutable, parser reused | yes | no | | +| simdjson On-Demand | none: forward-only, lazy | no | no | only traverse and select | +| Boost.JSON | owning, mutable DOM | yes | yes | monotonic resource | +| `json::parse` | owning, mutable DOM | yes | yes | | + +## Published results + +Results are only published with the file `compare.py` wrote, which names the machine and the versions; see +`results/`. Numbers from one machine and compiler do not carry over to another: rerun the script. diff --git a/tests/benchmarks/json_view/bench_corpus.cpp b/tests/benchmarks/json_view/bench_corpus.cpp new file mode 100644 index 000000000..7d4f5b2f9 --- /dev/null +++ b/tests/benchmarks/json_view/bench_corpus.cpp @@ -0,0 +1,326 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// Corpus benchmark: the read-only workloads of bench_view.cpp on any list of +// JSON files (for example the benchmark sets of simdjson and yyjson). +// +// ./bench_corpus [--rounds N] file... +// +// For every file, all engines must accept it and agree on a traversal (value +// count, string bytes, sum of numbers) before anything is timed. Workloads: +// parse (build and free a document), traverse (visit every value, convert +// every number), dump (compact), and for json_view also dump with the source +// number text. Results go to bench_corpus.csv. +#include + +#if JSON_VIEW_BENCH_BOOST + #include + #include +#endif +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using nlohmann::json; +using nlohmann::json_document; +using nlohmann::json_view; + +static volatile double g_sink; + +struct stats +{ + double num = 0; + std::size_t str = 0, nodes = 0; +}; + +static void walk(json_view v, stats& st) +{ + ++st.nodes; + switch (v.type()) + { + case json::value_t::object: + for (auto it = v.begin(); it != v.end(); ++it) + { + st.str += it.key().size(); + walk(*it, st); + } + break; + case json::value_t::array: + for (const json_view e : v) + { + walk(e, st); + } + break; + case json::value_t::string: + st.str += v.get_string().size(); + break; + case json::value_t::number_integer: + st.num += static_cast(v.get()); + break; + case json::value_t::number_unsigned: + st.num += static_cast(v.get()); + break; + case json::value_t::number_float: + st.num += v.get(); + break; + default: + break; + } +} + +static void walk(yyjson_val* v, stats& st) +{ + ++st.nodes; + switch (yyjson_get_type(v)) + { + case YYJSON_TYPE_OBJ: + { + std::size_t idx, max; + yyjson_val* k, * val; + yyjson_obj_foreach(v, idx, max, k, val) + { + st.str += yyjson_get_len(k); + walk(val, st); + } + break; + } + case YYJSON_TYPE_ARR: + { + std::size_t idx, max; + yyjson_val* val; + yyjson_arr_foreach(v, idx, max, val) + { + walk(val, st); + } + break; + } + case YYJSON_TYPE_STR: + st.str += yyjson_get_len(v); + break; + case YYJSON_TYPE_NUM: + st.num += yyjson_is_sint(v) ? static_cast(yyjson_get_sint(v)) : yyjson_is_uint(v) ? static_cast(yyjson_get_uint(v)) : yyjson_get_real(v); + break; + default: + break; + } +} + +static void walk(simdjson::dom::element e, stats& st) +{ + ++st.nodes; + switch (e.type()) + { + case simdjson::dom::element_type::OBJECT: + for (auto f : simdjson::dom::object(e)) + { + st.str += f.key.size(); + walk(f.value, st); + } + break; + case simdjson::dom::element_type::ARRAY: + for (auto c : simdjson::dom::array(e)) + { + walk(c, st); + } + break; + case simdjson::dom::element_type::STRING: + st.str += std::string_view(e).size(); + break; + case simdjson::dom::element_type::INT64: + st.num += static_cast(int64_t(e)); + break; + case simdjson::dom::element_type::UINT64: + st.num += static_cast(uint64_t(e)); + break; + case simdjson::dom::element_type::DOUBLE: + st.num += double(e); + break; + default: + break; + } +} + +#if JSON_VIEW_BENCH_BOOST +static void walk(const boost::json::value& v, stats& st) +{ + ++st.nodes; + switch (v.kind()) + { + case boost::json::kind::object: + for (const auto& kv : v.get_object()) + { + st.str += kv.key().size(); + walk(kv.value(), st); + } + break; + case boost::json::kind::array: + for (const auto& c : v.get_array()) + { + walk(c, st); + } + break; + case boost::json::kind::string: + st.str += v.get_string().size(); + break; + case boost::json::kind::int64: + st.num += static_cast(v.get_int64()); + break; + case boost::json::kind::uint64: + st.num += static_cast(v.get_uint64()); + break; + case boost::json::kind::double_: + st.num += v.get_double(); + break; + default: + break; + } +} +#endif + +static std::string slurp(const std::string& p) +{ + std::ifstream f(p, std::ios::binary); + std::stringstream ss; + ss << f.rdbuf(); + return ss.str(); +} + +static bool same(const stats& a, const stats& b) +{ + return a.nodes == b.nodes && a.str == b.str && (a.num == b.num || std::fabs(a.num - b.num) <= 1e-9 * std::fabs(a.num)); +} + +int main(int argc, char** argv) +{ + int rounds = 0; // 0: by size + std::vector files; + for (int i = 1; i < argc; ++i) + { + if (std::strcmp(argv[i], "--rounds") == 0 && i + 1 < argc) + { + rounds = std::atoi(argv[++i]); + } + else + { + files.push_back(argv[i]); + } + } + std::FILE* csv = std::fopen("bench_corpus.csv", "w"); + std::fprintf(csv, "file,bytes,workload,engine,ns\n"); + simdjson::dom::parser sj; + for (const auto& path : files) + { + const std::string s = slurp(path); + const std::string name = path.substr(path.rfind('/') + 1); + const simdjson::padded_string ps(s); + + // all engines must agree before timing + stats a, b, c, d; + const json_document doc = json_document::parse(s); + walk(doc.root(), a); + yyjson_doc* y = yyjson_read(s.data(), s.size(), 0); + auto sjr = sj.parse(ps); +#if JSON_VIEW_BENCH_BOOST + boost::json::parse_options opt; + opt.numbers = boost::json::number_precision::precise; + boost::json::monotonic_resource mr0; + const boost::json::value bv = boost::json::parse(s, &mr0, opt); +#endif + if (y == nullptr || sjr.error()) + { + std::printf("%-34s skipped (an engine rejects it)\n", name.c_str()); + yyjson_doc_free(y); + continue; + } + walk(yyjson_doc_get_root(y), b); + walk(sjr.value_unsafe(), c); +#if JSON_VIEW_BENCH_BOOST + walk(bv, d); +#else + d = a; +#endif + yyjson_doc_free(y); + const bool ok = same(a, b) && same(a, c) && same(a, d); + + const int r = rounds > 0 ? rounds : static_cast(std::max(3, std::min(60, 400000000 / (s.size() + 1)))); + struct engine + { + std::string name; + std::function fn; + }; + json_document vd = json_document::parse(s); + yyjson_doc* yd = yyjson_read(s.data(), s.size(), 0); + simdjson::dom::parser sjd; + const simdjson::dom::element se = sjd.parse(ps).value_unsafe(); + const std::vector>> workloads = + { + { + "parse", { + {"json_view", [&] { auto x = json_document::parse(s); g_sink = static_cast(x.node_count()); }}, + {"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast(yyjson_doc_get_val_count(x)); yyjson_doc_free(x); }}, + {"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }}, +#endif + } + }, + { + "traverse", { + {"json_view", [&] { auto x = json_document::parse(s); stats st; walk(x.root(), st); g_sink = st.num; }}, + {"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(x), st); g_sink = st.num; yyjson_doc_free(x); }}, + {"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); stats st; walk(v, st); g_sink = st.num; }}, +#endif + } + }, + { + "dump", { + {"json_view", [&] { std::string o = vd.root().dump(); g_sink = static_cast(o.size()); }}, + {"yyjson", [&] { std::size_t n = 0; char* o = yyjson_write(yd, 0, &n); g_sink = static_cast(n); std::free(o); }}, + {"simdjson DOM", [&] { std::string o = simdjson::to_string(se); g_sink = static_cast(o.size()); }}, + {"json_view (source numbers)", [&] { std::string o = vd.root().dump(-1, ' ', false, json_view::number_format::source); g_sink = static_cast(o.size()); }}, + } + }, + }; + std::printf("%-34s %9zu B%s\n", name.c_str(), s.size(), ok ? "" : " [ENGINES DISAGREE]"); + for (const auto& wl : workloads) + { + std::vector best(wl.second.size(), 1e300); + for (int i = 0; i < r; ++i) + { + for (std::size_t k = 0; k < wl.second.size(); ++k) + { + const auto t0 = std::chrono::steady_clock::now(); + wl.second[k].fn(); + best[k] = std::min(best[k], std::chrono::duration(std::chrono::steady_clock::now() - t0).count()); + } + } + std::printf(" %-9s", wl.first.c_str()); + for (std::size_t k = 0; k < wl.second.size(); ++k) + { + std::printf(" %s %.2f GB/s (%.2fx)", wl.second[k].name.c_str(), static_cast(s.size()) / best[k], best[k] / best[0]); + std::fprintf(csv, "%s,%zu,%s,%s,%.1f\n", name.c_str(), s.size(), wl.first.c_str(), wl.second[k].name.c_str(), best[k]); + } + std::printf("\n"); + std::fflush(stdout); + } + yyjson_doc_free(yd); + } + std::fclose(csv); +} diff --git a/tests/benchmarks/json_view/bench_view.cpp b/tests/benchmarks/json_view/bench_view.cpp new file mode 100644 index 000000000..fc1af65ee --- /dev/null +++ b/tests/benchmarks/json_view/bench_view.cpp @@ -0,0 +1,735 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// Same-feature-set benchmark: read-only JSON documents with random access. +// +// json_view nlohmann/json_view.hpp (fresh document per parse / reused) +// yyjson yyjson_read(): immutable document, random access +// simdjson DOM dom::parser (reused, as recommended): immutable, random access +// references (different feature sets): +// simdjson OD On-Demand: forward-only, lazy +// Boost.JSON owning, mutable DOM (monotonic resource) +// json::parse owning, mutable DOM (nlohmann today) +// +// Workloads: parse (build + free), traverse (visit everything, convert every +// number, touch every string and key), select (a few fields per document), +// dump (compact serialization of the parsed document). +// All engines run interleaved in every round; the best round is reported. +#include + +#if JSON_VIEW_BENCH_BOOST + #include + #include +#endif +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +using nlohmann::json; +using nlohmann::json_document; +using nlohmann::json_view; + +static volatile double g_sink; + +struct stats +{ + double num = 0; + std::size_t str = 0, nodes = 0; +}; + +// ---------------- traversal ---------------- + +static void walk(json_view v, stats& st) +{ + ++st.nodes; + switch (v.type()) + { + case json::value_t::object: + for (auto it = v.begin(); it != v.end(); ++it) + { + st.str += it.key().size(); + walk(*it, st); + } + break; + case json::value_t::array: + for (const json_view e : v) + { + walk(e, st); + } + break; + case json::value_t::string: + st.str += v.get_string().size(); + break; + case json::value_t::number_integer: + st.num += static_cast(v.get()); + break; + case json::value_t::number_unsigned: + st.num += static_cast(v.get()); + break; + case json::value_t::number_float: + st.num += v.get(); + break; + default: + break; + } +} + +static void walk(const json& j, stats& st) +{ + ++st.nodes; + switch (j.type()) + { + case json::value_t::object: + for (const auto& kv : j.get_ref()) + { + st.str += kv.first.size(); + walk(kv.second, st); + } + break; + case json::value_t::array: + for (const auto& e : j.get_ref()) + { + walk(e, st); + } + break; + case json::value_t::string: + st.str += j.get_ref().size(); + break; + case json::value_t::number_integer: + st.num += static_cast(*j.get_ptr()); + break; + case json::value_t::number_unsigned: + st.num += static_cast(*j.get_ptr()); + break; + case json::value_t::number_float: + st.num += *j.get_ptr(); + break; + default: + break; + } +} + +static void walk(yyjson_val* v, stats& st) +{ + ++st.nodes; + switch (yyjson_get_type(v)) + { + case YYJSON_TYPE_OBJ: + { + std::size_t idx, max; + yyjson_val* k, * val; + yyjson_obj_foreach(v, idx, max, k, val) + { + st.str += yyjson_get_len(k); + walk(val, st); + } + break; + } + case YYJSON_TYPE_ARR: + { + std::size_t idx, max; + yyjson_val* val; + yyjson_arr_foreach(v, idx, max, val) + { + walk(val, st); + } + break; + } + case YYJSON_TYPE_STR: + st.str += yyjson_get_len(v); + break; + case YYJSON_TYPE_NUM: + if (yyjson_is_sint(v)) + { + st.num += static_cast(yyjson_get_sint(v)); + } + else if (yyjson_is_uint(v)) + { + st.num += static_cast(yyjson_get_uint(v)); + } + else + { + st.num += yyjson_get_real(v); + } + break; + default: + break; + } +} + +static void walk(simdjson::dom::element e, stats& st) +{ + ++st.nodes; + switch (e.type()) + { + case simdjson::dom::element_type::OBJECT: + for (auto f : simdjson::dom::object(e)) + { + st.str += f.key.size(); + walk(f.value, st); + } + break; + case simdjson::dom::element_type::ARRAY: + for (auto c : simdjson::dom::array(e)) + { + walk(c, st); + } + break; + case simdjson::dom::element_type::STRING: + st.str += std::string_view(e).size(); + break; + case simdjson::dom::element_type::INT64: + st.num += static_cast(int64_t(e)); + break; + case simdjson::dom::element_type::UINT64: + st.num += static_cast(uint64_t(e)); + break; + case simdjson::dom::element_type::DOUBLE: + st.num += double(e); + break; + default: + break; + } +} + +static void walk_od(simdjson::ondemand::value v, stats& st) +{ + ++st.nodes; + switch (v.type()) + { + case simdjson::ondemand::json_type::object: + for (auto f : v.get_object()) + { + st.str += std::string_view(f.unescaped_key()).size(); + walk_od(f.value(), st); + } + break; + case simdjson::ondemand::json_type::array: + for (auto c : v.get_array()) + { + walk_od(c.value(), st); + } + break; + case simdjson::ondemand::json_type::string: + st.str += std::string_view(v.get_string()).size(); + break; + case simdjson::ondemand::json_type::number: + { + simdjson::ondemand::number n = v.get_number(); + switch (n.get_number_type()) + { + case simdjson::ondemand::number_type::signed_integer: + st.num += static_cast(n.get_int64()); + break; + case simdjson::ondemand::number_type::unsigned_integer: + st.num += static_cast(n.get_uint64()); + break; + default: + st.num += n.get_double(); + break; + } + break; + } + case simdjson::ondemand::json_type::boolean: + (void)bool(v.get_bool()); + break; + default: + (void)v.is_null(); + break; + } +} + +#if JSON_VIEW_BENCH_BOOST +static void walk(const boost::json::value& v, stats& st) +{ + ++st.nodes; + switch (v.kind()) + { + case boost::json::kind::object: + for (const auto& kv : v.get_object()) + { + st.str += kv.key().size(); + walk(kv.value(), st); + } + break; + case boost::json::kind::array: + for (const auto& c : v.get_array()) + { + walk(c, st); + } + break; + case boost::json::kind::string: + st.str += v.get_string().size(); + break; + case boost::json::kind::int64: + st.num += static_cast(v.get_int64()); + break; + case boost::json::kind::uint64: + st.num += static_cast(v.get_uint64()); + break; + case boost::json::kind::double_: + st.num += v.get_double(); + break; + default: + break; + } +} +#endif + +// ---------------- selective access ---------------- +// twitter: per status id, user.screen_name, retweet_count +// citm: per performance id, eventId, #seatCategories; #events +// canada: type, features[0].geometry.type, #coordinates +// jeopardy: per question: round == "Final Jeopardy!", len(category) +// status (one tweet): id, user.screen_name, retweet_count +// rpc: method, params.minuend, id + +static double pick(const std::string& name, json_view r) +{ + double acc = 0; + if (name == "twitter" || name == "status") + { + auto one = [&](json_view s) + { + acc += static_cast(s["id"].get()); + acc += static_cast(s["user"]["screen_name"].get_string().size()); + acc += static_cast(s["retweet_count"].get()); + }; + if (name == "twitter") + { + for (const json_view s : r["statuses"]) + { + one(s); + } + } + else + { + one(r); + } + } + else if (name == "citm_catalog") + { + for (const json_view p : r["performances"]) + { + acc += static_cast(p["id"].get() + p["eventId"].get() + p["seatCategories"].size()); + } + acc += static_cast(r["events"].size()); + } + else if (name == "canada") + { + const json_view g = r["features"][0]["geometry"]; + acc += static_cast(r["type"].get_string().size() + g["type"].get_string().size() + g["coordinates"].size()); + } + else if (name == "jeopardy") + { + for (const json_view q : r) + { + acc += q["round"].get_string() == "Final Jeopardy!" ? 1 : 0; + acc += static_cast(q["category"].get_string().size()); + } + } + else if (name == "rpc") + { + acc += static_cast(r["method"].get_string().size()); + acc += static_cast(r["params"]["minuend"].get() + r["id"].get()); + } + return acc; +} + +static double pick(const std::string& name, const json& r) +{ + double acc = 0; + if (name == "twitter" || name == "status") + { + auto one = [&](const json & s) + { + acc += static_cast(s["id"].get()); + acc += static_cast(s["user"]["screen_name"].get_ref().size()); + acc += static_cast(s["retweet_count"].get()); + }; + if (name == "twitter") + { + for (const auto& s : r["statuses"]) + { + one(s); + } + } + else + { + one(r); + } + } + else if (name == "citm_catalog") + { + for (const auto& p : r["performances"]) + { + acc += static_cast(p["id"].get() + p["eventId"].get() + p["seatCategories"].size()); + } + acc += static_cast(r["events"].size()); + } + else if (name == "canada") + { + const json& g = r["features"][0]["geometry"]; + acc += static_cast(r["type"].get_ref().size() + g["type"].get_ref().size() + g["coordinates"].size()); + } + else if (name == "jeopardy") + { + for (const auto& q : r) + { + acc += q["round"].get_ref() == "Final Jeopardy!" ? 1 : 0; + acc += static_cast(q["category"].get_ref().size()); + } + } + else if (name == "rpc") + { + acc += static_cast(r["method"].get_ref().size()); + acc += static_cast(r["params"]["minuend"].get() + r["id"].get()); + } + return acc; +} + +static double pick(const std::string& name, yyjson_val* r) +{ + double acc = 0; + auto get = [](yyjson_val * o, const char* k) + { + return yyjson_obj_get(o, k); + }; + if (name == "twitter" || name == "status") + { + auto one = [&](yyjson_val * s) + { + acc += static_cast(yyjson_get_uint(get(s, "id"))); + acc += static_cast(yyjson_get_len(get(get(s, "user"), "screen_name"))); + acc += static_cast(yyjson_get_sint(get(s, "retweet_count"))); + }; + if (name == "twitter") + { + std::size_t idx, max; + yyjson_val* s; + yyjson_arr_foreach(get(r, "statuses"), idx, max, s) + { + one(s); + } + } + else + { + one(r); + } + } + else if (name == "citm_catalog") + { + std::size_t idx, max; + yyjson_val* p; + yyjson_arr_foreach(get(r, "performances"), idx, max, p) + { + acc += static_cast(yyjson_get_uint(get(p, "id")) + yyjson_get_uint(get(p, "eventId")) + yyjson_arr_size(get(p, "seatCategories"))); + } + acc += static_cast(yyjson_obj_size(get(r, "events"))); + } + else if (name == "canada") + { + yyjson_val* g = get(yyjson_arr_get(get(r, "features"), 0), "geometry"); + acc += static_cast(yyjson_get_len(get(r, "type")) + yyjson_get_len(get(g, "type")) + yyjson_arr_size(get(g, "coordinates"))); + } + else if (name == "jeopardy") + { + std::size_t idx, max; + yyjson_val* q; + yyjson_arr_foreach(r, idx, max, q) + { + acc += yyjson_equals_str(get(q, "round"), "Final Jeopardy!") ? 1 : 0; + acc += static_cast(yyjson_get_len(get(q, "category"))); + } + } + else if (name == "rpc") + { + acc += static_cast(yyjson_get_len(get(r, "method"))); + acc += static_cast(yyjson_get_sint(get(get(r, "params"), "minuend")) + yyjson_get_sint(get(r, "id"))); + } + return acc; +} + +static double pick(const std::string& name, simdjson::dom::element r) +{ + double acc = 0; + if (name == "twitter" || name == "status") + { + auto one = [&](simdjson::dom::element s) + { + acc += static_cast(uint64_t(s["id"])); + acc += static_cast(std::string_view(s["user"]["screen_name"]).size()); + acc += static_cast(int64_t(s["retweet_count"])); + }; + if (name == "twitter") + { + for (auto s : simdjson::dom::array(r["statuses"])) + { + one(s); + } + } + else + { + one(r); + } + } + else if (name == "citm_catalog") + { + for (auto p : simdjson::dom::array(r["performances"])) + { + acc += static_cast(uint64_t(p["id"]) + uint64_t(p["eventId"]) + simdjson::dom::array(p["seatCategories"]).size()); + } + acc += static_cast(simdjson::dom::object(r["events"]).size()); + } + else if (name == "canada") + { + auto g = r["features"].at(0)["geometry"]; + acc += static_cast(std::string_view(r["type"]).size() + std::string_view(g["type"]).size() + simdjson::dom::array(g["coordinates"]).size()); + } + else if (name == "jeopardy") + { + for (auto q : simdjson::dom::array(r)) + { + acc += std::string_view(q["round"]) == "Final Jeopardy!" ? 1 : 0; + acc += static_cast(std::string_view(q["category"]).size()); + } + } + else if (name == "rpc") + { + acc += static_cast(std::string_view(r["method"]).size()); + acc += static_cast(int64_t(r["params"]["minuend"]) + int64_t(r["id"])); + } + return acc; +} + +static double pick_od(const std::string& name, simdjson::ondemand::document& d) +{ + double acc = 0; + if (name == "twitter" || name == "status") + { + auto one = [&](simdjson::ondemand::object s) + { + acc += static_cast(uint64_t(s["id"])); + acc += static_cast(std::string_view(s["user"]["screen_name"]).size()); + acc += static_cast(int64_t(s["retweet_count"])); + }; + if (name == "twitter") + { + for (auto s : d["statuses"]) + { + one(s.get_object()); + } + } + else + { + one(d.get_object()); + } + } + else if (name == "citm_catalog") + { + simdjson::ondemand::object ev = d["events"].get_object(); + acc += static_cast(ev.count_fields()); + for (auto p : d["performances"]) + { + simdjson::ondemand::object o = p.get_object(); + const auto a = uint64_t(o["eventId"]) + uint64_t(o["id"]); + simdjson::ondemand::array sc = o["seatCategories"].get_array(); + acc += static_cast(a + sc.count_elements()); + } + } + else if (name == "canada") + { + acc += static_cast(std::string_view(d["type"]).size()); + auto g = d["features"].at(0)["geometry"]; + acc += static_cast(std::string_view(g["type"]).size()); + simdjson::ondemand::array co = g["coordinates"].get_array(); + acc += static_cast(co.count_elements()); + } + else if (name == "jeopardy") + { + for (auto q : d) + { + simdjson::ondemand::object o = q.get_object(); + acc += static_cast(std::string_view(o["category"]).size()); + acc += std::string_view(o["round"]) == "Final Jeopardy!" ? 1 : 0; + } + } + else if (name == "rpc") + { + acc += static_cast(std::string_view(d["method"]).size()); + acc += static_cast(int64_t(d["params"]["minuend"])); + acc += static_cast(int64_t(d["id"])); + } + return acc; +} + +// ---------------- harness ---------------- + +static std::string slurp(const std::string& p) +{ + std::ifstream f(p, std::ios::binary); + std::stringstream ss; + ss << f.rdbuf(); + return ss.str(); +} + +struct engine +{ + std::string name; + std::function fn; +}; + +int main(int argc, char** argv) +{ + if (argc < 2) + { + std::fprintf(stderr, "usage: %s [rounds] [document]\n", argv[0]); + return 1; + } + const std::string T = std::string(argv[1]) + "/"; + const int rounds = argc > 2 ? std::atoi(argv[2]) : 30; + const std::string only = argc > 3 ? argv[3] : ""; + struct doc + { + std::string name, text; + int batch; + }; + std::vector docs; + for (const char* f : + {"nativejson-benchmark/twitter.json", "nativejson-benchmark/citm_catalog.json", "nativejson-benchmark/canada.json", "jeopardy/jeopardy.json" + }) + { + std::string n = std::string(f).substr(std::string(f).find('/') + 1); + docs.push_back({n.substr(0, n.size() - 5), slurp(T + f), 1}); + } + docs.push_back({"status", json::parse(docs[0].text)["statuses"][0].dump(), 200}); + docs.push_back({"rpc", R"({"jsonrpc": "2.0", "method": "subtract", "params": {"minuend": 42, "subtrahend": 23}, "id": 3})", 5000}); + + // correctness cross-check of the workloads + for (const auto& dc : docs) + { + stats a, b, c, dd; + walk(json::parse(dc.text), a); + auto d = json_document::parse(dc.text); + walk(d.root(), b); + yyjson_doc* y = yyjson_read(dc.text.data(), dc.text.size(), 0); + walk(yyjson_doc_get_root(y), c); + simdjson::dom::parser p; + walk(p.parse(dc.text).value(), dd); + const bool ok = a.nodes == b.nodes && a.nodes == c.nodes && a.nodes == dd.nodes && a.str == b.str && a.str == c.str && a.str == dd.str + && std::fabs(a.num - b.num) <= 1e-9 * std::fabs(a.num) && std::fabs(a.num - c.num) <= 1e-9 * std::fabs(a.num); + const double pa = pick(dc.name, json::parse(dc.text)), pb = pick(dc.name, d.root()), pc = pick(dc.name, yyjson_doc_get_root(y)), pd = pick(dc.name, p.parse(dc.text).value()); + std::printf("check %-13s traverse %s select %s\n", dc.name.c_str(), ok ? "OK" : "MISMATCH", (pa == pb && pa == pc && pa == pd) ? "OK" : "MISMATCH"); + yyjson_doc_free(y); + } + + std::FILE* csv = std::fopen("bench_view.csv", "w"); + std::fprintf(csv, "doc,bytes,workload,engine,ns\n"); + json_document reused; + simdjson::dom::parser sj; + simdjson::ondemand::parser od; + for (const auto& dc : docs) + { + if (!only.empty() && dc.name != only) + { + continue; + } + const std::string& s = dc.text; + const simdjson::padded_string ps(s); + const std::string name = dc.name; + std::vector>> workloads; + + workloads.push_back({"parse", { + {"json_view", [&] { auto d = json_document::parse(s); g_sink = static_cast(d.node_count()); }}, + {"json_view (reused)", [&] { reused.read(s); g_sink = static_cast(reused.node_count()); }}, + {"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast(yyjson_doc_get_val_count(d)); yyjson_doc_free(d); }}, + {"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }}, +#endif + {"json::parse", [&] { json j = json::parse(s); g_sink = static_cast(j.size()); }}, + }}); + workloads.push_back({"traverse", { + {"json_view", [&] { auto d = json_document::parse(s); stats st; walk(d.root(), st); g_sink = st.num; }}, + {"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(d), st); g_sink = st.num; yyjson_doc_free(d); }}, + {"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }}, + {"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); stats st; walk_od(d.get_value().value_unsafe(), st); g_sink = st.num; }}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); stats st; walk(v, st); g_sink = st.num; }}, +#endif + {"json::parse", [&] { json j = json::parse(s); stats st; walk(j, st); g_sink = st.num; }}, + }}); + workloads.push_back({"select", { + {"json_view", [&] { auto d = json_document::parse(s); g_sink = pick(name, d.root()); }}, + {"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = pick(name, yyjson_doc_get_root(d)); yyjson_doc_free(d); }}, + {"simdjson DOM", [&] { g_sink = pick(name, sj.parse(ps).value_unsafe()); }}, + {"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); g_sink = pick_od(name, d); }}, + {"json::parse", [&] { json j = json::parse(s); g_sink = pick(name, j); }}, + }}); + { + // serialization of an already parsed document + static json_document vd; + vd.read(s); + static yyjson_doc* yd = nullptr; + if (yd) + { + yyjson_doc_free(yd); + } + yd = yyjson_read(s.data(), s.size(), 0); + static simdjson::dom::parser sjd; + static simdjson::dom::element se; + se = sjd.parse(ps).value_unsafe(); + static json jd; + jd = json::parse(s); + workloads.push_back({"dump", { + {"json_view", [&] { std::string o = vd.root().dump(); g_sink = static_cast(o.size()); }}, + {"yyjson", [&] { std::size_t n = 0; char* o = yyjson_write(yd, 0, &n); g_sink = static_cast(n); std::free(o); }}, + {"simdjson DOM", [&] { std::string o = simdjson::to_string(se); g_sink = static_cast(o.size()); }}, + {"json::parse", [&] { std::string o = jd.dump(); g_sink = static_cast(o.size()); }}, + }}); + } + + for (auto& wl : workloads) + { + std::vector best(wl.second.size(), 1e300); + const int r = s.size() > 10000000 ? std::max(3, rounds / 5) : rounds; + for (int i = 0; i < r; ++i) + { + for (std::size_t k = 0; k < wl.second.size(); ++k) + { + const auto t0 = std::chrono::steady_clock::now(); + for (int b = 0; b < dc.batch; ++b) + { + wl.second[k].fn(); + } + const double ns = std::chrono::duration(std::chrono::steady_clock::now() - t0).count() / dc.batch; + best[k] = std::min(best[k], ns); + } + } + const double ref = best[0]; + std::printf("%-13s %-9s", dc.name.c_str(), wl.first.c_str()); + for (std::size_t k = 0; k < wl.second.size(); ++k) + { + const double us = best[k] / 1e3; + std::printf(" %s %s%s (%.2fx)", wl.second[k].name.c_str(), us >= 100 ? "" : "", (us >= 1000 ? std::to_string(static_cast(us)) + "us" : (std::to_string(us).substr(0, 5) + "us")).c_str(), best[k] / ref); + std::fprintf(csv, "%s,%zu,%s,%s,%.1f\n", dc.name.c_str(), s.size(), wl.first.c_str(), wl.second[k].name.c_str(), best[k]); + } + std::printf("\n"); + std::fflush(stdout); + } + } + std::fclose(csv); +} diff --git a/tests/benchmarks/json_view/compare.py b/tests/benchmarks/json_view/compare.py new file mode 100755 index 000000000..e28d419c6 --- /dev/null +++ b/tests/benchmarks/json_view/compare.py @@ -0,0 +1,295 @@ +#!/usr/bin/env python3 +# __ _____ _____ _____ +# __| | __| | | | JSON for Modern C++ (supporting code) +# | | |__ | | | | | | version 3.12.0 +# |_____|_____|_____|_|___| https://github.com/nlohmann/json +# +# SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +# SPDX-License-Identifier: MIT + +"""Compare json_view with yyjson, simdjson, Boost.JSON, and json::parse. + +Builds bench_view.cpp and bench_corpus.cpp against the include/ directory of +this checkout, runs them, and writes the results with everything needed to +reproduce them (date, commit, CPU, OS, compiler, library versions, flags) to +results/-.md and .csv next to this script. + +The other libraries come from the system (--system, the default: pkg-config +or Homebrew) or are downloaded as pinned releases and checked against their +SHA-256 (--download). Boost.JSON is optional: without Boost headers, its +columns are skipped, and the results say so. + +Only the Python 3 standard library is used; a C++17 compiler is needed. +""" + +import argparse +import datetime +import hashlib +import os +import platform +import re +import shlex +import shutil +import subprocess +import sys +import tarfile +import urllib.request + +HERE = os.path.dirname(os.path.abspath(__file__)) +REPO = os.path.abspath(os.path.join(HERE, '..', '..', '..')) + +# pinned releases for --download; the hashes are those of the archives +PINNED = { + 'yyjson': { + 'version': '0.13.0', + 'url': 'https://github.com/ibireme/yyjson/archive/refs/tags/0.13.0.tar.gz', + 'sha256': None, # TODO: pin before the first published comparison + }, + 'simdjson': { + 'version': '4.6.11', + 'url': 'https://github.com/simdjson/simdjson/archive/refs/tags/v4.6.11.tar.gz', + 'sha256': None, # TODO: pin before the first published comparison + }, + 'boost': { + 'version': '1.92.0', + 'url': 'https://archives.boost.io/release/1.92.0/source/boost_1_92_0.tar.gz', + 'sha256': None, # TODO: pin before the first published comparison + }, +} + +# the documents of bench_view.cpp, relative to the json_test_data directory +DEFAULT_CORPUS = [ + 'nativejson-benchmark/twitter.json', + 'nativejson-benchmark/citm_catalog.json', + 'nativejson-benchmark/canada.json', + 'jeopardy/jeopardy.json', +] + + +def run(cmd, **kwargs): + print('+ ' + ' '.join(shlex.quote(c) for c in cmd), flush=True) + return subprocess.run(cmd, check=True, **kwargs) + + +def output(cmd): + try: + return subprocess.run(cmd, check=True, capture_output=True, text=True).stdout.strip() + except (OSError, subprocess.CalledProcessError): + return '' + + +# --------------------------------------------------------------------------- +# libraries +# --------------------------------------------------------------------------- + +class Library: + """include directories, sources to compile, and linker flags of a library""" + + def __init__(self, name, include=None, sources=None, link=None, version=''): + self.name = name + self.include = include or [] + self.sources = sources or [] + self.link = link or [] + self.version = version + + +def header_version(path, pattern): + try: + with open(path, encoding='utf-8', errors='replace') as f: + m = re.search(pattern, f.read()) + return m.group(1) if m else '' + except OSError: + return '' + + +def library_version(name, include_dirs): + patterns = { + 'yyjson': ('yyjson.h', r'#define\s+YYJSON_VERSION_STRING\s+"([^"]+)"'), + 'simdjson': ('simdjson.h', r'#define\s+SIMDJSON_VERSION\s+"?([0-9.]+)"?'), + 'boost': (os.path.join('boost', 'version.hpp'), r'#define\s+BOOST_LIB_VERSION\s+"([^"]+)"'), + } + header, pattern = patterns[name] + for d in include_dirs: + v = header_version(os.path.join(d, header), pattern) + if v: + return v.replace('_', '.') + return '' + + +def system_library(name): + """a library found with pkg-config or Homebrew, or None""" + flags = output(['pkg-config', '--cflags', '--libs', name]).split() + if flags: + include = [f[2:] for f in flags if f.startswith('-I')] + link = [f for f in flags if f.startswith('-L') or f.startswith('-l')] + libdirs = [f[2:] for f in link if f.startswith('-L')] + link += ['-Wl,-rpath,' + d for d in libdirs] + return Library(name, include, [], link, library_version(name, include)) + prefix = output(['brew', '--prefix', name]) if shutil.which('brew') else '' + if prefix and os.path.isdir(os.path.join(prefix, 'include')): + include = [os.path.join(prefix, 'include')] + link = [] + if name != 'boost': + lib = os.path.join(prefix, 'lib') + link = ['-L' + lib, '-l' + name, '-Wl,-rpath,' + lib] + return Library(name, include, [], link, library_version(name, include)) + if name == 'boost': + for d in ['/usr/include', '/usr/local/include']: + if os.path.isfile(os.path.join(d, 'boost', 'json.hpp')): + return Library(name, [d], [], [], library_version(name, [d])) + return None + + +def download_library(name, work): + """a pinned release, downloaded and checked, or an error""" + pin = PINNED[name] + if not pin['sha256']: + sys.exit(f'error: no SHA-256 pinned for {name} {pin["version"]} yet; use --system') + archive = os.path.join(work, 'download', os.path.basename(pin['url'])) + os.makedirs(os.path.dirname(archive), exist_ok=True) + if not os.path.isfile(archive): + print(f'downloading {pin["url"]}', flush=True) + urllib.request.urlretrieve(pin['url'], archive) + with open(archive, 'rb') as f: + digest = hashlib.sha256(f.read()).hexdigest() + if digest != pin['sha256']: + sys.exit(f'error: SHA-256 of {archive} is {digest}, expected {pin["sha256"]}') + target = os.path.join(work, 'download', f'{name}-{pin["version"]}') + if not os.path.isdir(target): + with tarfile.open(archive) as t: + t.extractall(os.path.join(work, 'download')) # noqa: S202 (checked archive) + src = [os.path.join(work, 'download', d) for d in os.listdir(os.path.join(work, 'download')) + if d.lower().startswith(name) and os.path.isdir(os.path.join(work, 'download', d))][0] + if name == 'yyjson': + return Library(name, [os.path.join(src, 'src')], [os.path.join(src, 'src', 'yyjson.c')], [], pin['version']) + if name == 'simdjson': + single = os.path.join(src, 'singleheader') + return Library(name, [single], [os.path.join(single, 'simdjson.cpp')], [], pin['version']) + return Library(name, [src], [], [], pin['version']) + + +# --------------------------------------------------------------------------- +# machine description +# --------------------------------------------------------------------------- + +def cpu_model(): + if sys.platform == 'darwin': + return output(['sysctl', '-n', 'machdep.cpu.brand_string']) + try: + with open('/proc/cpuinfo', encoding='utf-8') as f: + for line in f: + if line.startswith('model name') or line.startswith('Model'): + return line.split(':', 1)[1].strip() + except OSError: + pass + return platform.processor() + + +def git_commit(): + commit = output(['git', '-C', REPO, 'rev-parse', '--short=12', 'HEAD']) + dirty = output(['git', '-C', REPO, 'status', '--porcelain', '--untracked-files=no']) + return commit + (' (with local changes)' if dirty else '') + + +# --------------------------------------------------------------------------- +# main +# --------------------------------------------------------------------------- + +def main(): + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument('--data', required=True, help='json_test_data directory (with nativejson-benchmark/ and jeopardy/)') + ap.add_argument('--download', action='store_true', help='use pinned downloads instead of system libraries') + ap.add_argument('--no-boost', action='store_true', help='skip Boost.JSON') + ap.add_argument('--native', action='store_true', help='compile for this CPU (-march=native / -mcpu=native)') + ap.add_argument('--rounds', type=int, default=30, help='rounds of bench_view (default: 30)') + ap.add_argument('--corpus', nargs='*', default=[], help='more files for bench_corpus') + ap.add_argument('--build-dir', default=os.path.join(HERE, 'build'), help='where to build (default: build/ next to this script)') + args = ap.parse_args() + + cxx = os.environ.get('CXX', 'c++') + cc = os.environ.get('CC', 'cc') + os.makedirs(args.build_dir, exist_ok=True) + + libs = {} + for name in ['yyjson', 'simdjson', 'boost']: + if name == 'boost' and args.no_boost: + continue + lib = download_library(name, args.build_dir) if args.download else system_library(name) + if lib is None and name != 'boost': + sys.exit(f'error: {name} not found; install it, or use --download') + if lib is not None: + libs[name] = lib + with_boost = 'boost' in libs + if not with_boost: + print('Boost.JSON not found: its columns are skipped', flush=True) + + flags = ['-std=c++17', '-O3', '-DNDEBUG', f'-DJSON_VIEW_BENCH_BOOST={1 if with_boost else 0}'] + if args.native: + flags.append('-mcpu=native' if platform.machine().lower() in ('arm64', 'aarch64') else '-march=native') + include = ['-I' + os.path.join(REPO, 'include')] + ['-I' + d for lib in libs.values() for d in lib.include] + link = [f for lib in libs.values() for f in lib.link] + + # C sources of downloaded libraries are compiled once + objects = [] + for lib in libs.values(): + for src in lib.sources: + obj = os.path.join(args.build_dir, os.path.basename(src) + '.o') + compiler = cc if src.endswith('.c') else cxx + run([compiler] + (['-std=c++17'] if compiler == cxx else []) + ['-O3', '-DNDEBUG', '-c', src, '-o', obj] + + ['-I' + d for d in lib.include]) + objects.append(obj) + + binaries = {} + for bench in ['bench_view', 'bench_corpus']: + exe = os.path.join(args.build_dir, bench) + run([cxx] + flags + include + [os.path.join(HERE, bench + '.cpp')] + objects + link + ['-o', exe]) + binaries[bench] = exe + + # run: bench_view on its documents, bench_corpus on those and the given files + corpus = [os.path.join(args.data, f) for f in DEFAULT_CORPUS] + args.corpus + outputs = {} + outputs['bench_view'] = run([binaries['bench_view'], args.data, str(args.rounds)], cwd=args.build_dir, + capture_output=True, text=True).stdout + outputs['bench_corpus'] = run([binaries['bench_corpus']] + corpus, cwd=args.build_dir, + capture_output=True, text=True).stdout + for name, text in outputs.items(): + print(text) + + # results with their metadata + now = datetime.datetime.now() + host = re.sub(r'[^A-Za-z0-9-]+', '-', platform.node().split('.')[0]) or 'host' + stem = os.path.join(HERE, 'results', f'{now:%Y-%m-%d}-{host}') + os.makedirs(os.path.dirname(stem), exist_ok=True) + meta = [ + ('date', f'{now:%Y-%m-%d %H:%M}'), + ('commit', git_commit()), + ('CPU', cpu_model()), + ('OS', f'{platform.system()} {platform.release()} ({platform.machine()})'), + ('compiler', output([cxx, '--version']).splitlines()[0] if output([cxx, '--version']) else cxx), + ('flags', ' '.join(flags)), + ('yyjson', libs['yyjson'].version), + ('simdjson', libs['simdjson'].version), + ('Boost.JSON', libs['boost'].version if with_boost else 'skipped (not found)'), + ('libraries from', 'pinned downloads' if args.download else 'the system'), + ('rounds', str(args.rounds)), + ] + with open(stem + '.md', 'w', encoding='utf-8') as f: + f.write(f'# json_view comparison, {now:%Y-%m-%d}\n\n') + f.write('Generated by `tests/benchmarks/json_view/compare.py`; best of the interleaved rounds.\n\n') + f.write('| | |\n|---|---|\n') + for key, value in meta: + f.write(f'| {key} | {value} |\n') + for name, text in outputs.items(): + f.write(f'\n## {name}\n\n```\n{text.rstrip()}\n```\n') + with open(stem + '.csv', 'w', encoding='utf-8') as out: + out.write(''.join(f'# {key}: {value}\n' for key, value in meta)) + for name in ['bench_view', 'bench_corpus']: + path = os.path.join(args.build_dir, name + '.csv') + if os.path.isfile(path): + with open(path, encoding='utf-8') as f: + out.write(f'# {name}\n' + f.read()) + print(f'results: {stem}.md, {stem}.csv') + + +if __name__ == '__main__': + main()