diff --git a/tests/benchmarks/json_view/README.md b/tests/benchmarks/json_view/README.md index 24d0fd200..bb8727284 100644 --- a/tests/benchmarks/json_view/README.md +++ b/tests/benchmarks/json_view/README.md @@ -60,7 +60,10 @@ outputs are checked to describe the same value. Before anything is timed, all engines must accept each document and agree on the traversal: the number of values, the bytes of all strings and keys, and the sum of all numbers. All engines run interleaved in every round, and the best -round is reported, as time and as a factor of the `json_view` time (below 1 means faster than `json_view`). +round is reported, as time and as a factor of the `json_view` time (below 1 means faster than `json_view`). Each timed +call follows an untimed call of the same engine: otherwise the engine after `json::parse` pays for the allocator +cleaning up the tens of thousands of nodes `json::parse` just freed (with glibc, this made `json_view` look 1.7 times +slower on citm_catalog traverse). The engines do not all offer the same features, which the numbers should be read with: @@ -68,11 +71,18 @@ The engines do not all offer the same features, which the numbers should be read |---|---|---|---|---| | `json_view` | immutable index into the text | yes | no | a fresh document per parse; "reused" parses into the same document | | yyjson | immutable (`yyjson_read`) | yes | via a mutable copy | | -| simdjson DOM | immutable, parser reused | yes | no | | +| simdjson DOM | immutable, parser reused | yes | no | "fresh" uses a new parser per parse | | simdjson On-Demand | none: forward-only, lazy | no | no | only traverse and select | | Boost.JSON | owning, mutable DOM | yes | yes | monotonic resource | | `json::parse` | owning, mutable DOM | yes | yes | | +Reusing memory matters as much as the parser. simdjson DOM reuses its parser, so it writes into memory it already +touched; a fresh `json_view` document or yyjson document gets new memory for every parse. On Linux, glibc returns large +blocks to the system when they are freed, so every fresh parse of a large document pays a page fault per 4 KiB page: +on x86-64 Linux, a fresh `json_view` parse of jeopardy took about twice as long as a reused one. On macOS on Apple +silicon, with 16 KiB pages, the difference is much smaller. Compare "json_view (reused)" with "simdjson DOM", and the +fresh `json_view` with "simdjson DOM (fresh)" and yyjson. + ## Published results Results are only published with the file `compare.py` wrote, which names the machine and the versions; see diff --git a/tests/benchmarks/json_view/bench_corpus.cpp b/tests/benchmarks/json_view/bench_corpus.cpp index 7d4f5b2f9..739c4f46e 100644 --- a/tests/benchmarks/json_view/bench_corpus.cpp +++ b/tests/benchmarks/json_view/bench_corpus.cpp @@ -29,6 +29,7 @@ #include #include #include +#include #include #include #include @@ -195,6 +196,11 @@ static void walk(const boost::json::value& v, stats& st) static std::string slurp(const std::string& p) { std::ifstream f(p, std::ios::binary); + if (!f) + { + std::fprintf(stderr, "cannot open %s\n", p.c_str()); + std::exit(1); + } std::stringstream ss; ss << f.rdbuf(); return ss.str(); @@ -222,6 +228,7 @@ int main(int argc, char** argv) } std::FILE* csv = std::fopen("bench_corpus.csv", "w"); std::fprintf(csv, "file,bytes,workload,engine,ns\n"); + json_document reused; simdjson::dom::parser sj; for (const auto& path : files) { @@ -272,8 +279,10 @@ int main(int argc, char** argv) { "parse", { {"json_view", [&] { auto x = json_document::parse(s); g_sink = static_cast(x.node_count()); }}, + {"json_view (reused)", [&] { reused.read(s); g_sink = static_cast(reused.node_count()); }}, {"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast(yyjson_doc_get_val_count(x)); yyjson_doc_free(x); }}, {"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }}, + {"simdjson DOM (fresh)", [&] { simdjson::dom::parser p; auto e = p.parse(ps).value_unsafe(); g_sink = e.is_object(); }}, #if JSON_VIEW_BENCH_BOOST {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }}, #endif @@ -306,6 +315,9 @@ int main(int argc, char** argv) { for (std::size_t k = 0; k < wl.second.size(); ++k) { + // an untimed call first: whatever the previous engine left to the allocator + // (e.g. thousands of freed json nodes) is cleaned up here, not in the timing + wl.second[k].fn(); const auto t0 = std::chrono::steady_clock::now(); wl.second[k].fn(); best[k] = std::min(best[k], std::chrono::duration(std::chrono::steady_clock::now() - t0).count()); diff --git a/tests/benchmarks/json_view/bench_edit.cpp b/tests/benchmarks/json_view/bench_edit.cpp index 3e75ed754..d5b35f912 100644 --- a/tests/benchmarks/json_view/bench_edit.cpp +++ b/tests/benchmarks/json_view/bench_edit.cpp @@ -479,6 +479,11 @@ static std::string edit_boost(const std::string& name, const std::string& s, boo static std::string slurp(const std::string& p) { std::ifstream f(p, std::ios::binary); + if (!f) + { + std::fprintf(stderr, "cannot open %s\n", p.c_str()); + std::exit(1); + } std::stringstream ss; ss << f.rdbuf(); return ss.str(); @@ -550,6 +555,9 @@ int main(int argc, char** argv) { for (std::size_t k = 0; k < engines.size(); ++k) { + // an untimed call first: whatever the previous engine left to the allocator + // (e.g. thousands of freed json nodes) is cleaned up here, not in the timing + g_sink = engines[k].second(dc.name, dc.text, update).size(); const auto t0 = std::chrono::steady_clock::now(); for (int b = 0; b < dc.batch; ++b) { diff --git a/tests/benchmarks/json_view/bench_view.cpp b/tests/benchmarks/json_view/bench_view.cpp index fc1af65ee..4f4234db5 100644 --- a/tests/benchmarks/json_view/bench_view.cpp +++ b/tests/benchmarks/json_view/bench_view.cpp @@ -10,7 +10,8 @@ // // json_view nlohmann/json_view.hpp (fresh document per parse / reused) // yyjson yyjson_read(): immutable document, random access -// simdjson DOM dom::parser (reused, as recommended): immutable, random access +// simdjson DOM dom::parser (reused, as recommended; "fresh": a new parser +// per parse): immutable, random access // references (different feature sets): // simdjson OD On-Demand: forward-only, lazy // Boost.JSON owning, mutable DOM (monotonic resource) @@ -19,7 +20,8 @@ // Workloads: parse (build + free), traverse (visit everything, convert every // number, touch every string and key), select (a few fields per document), // dump (compact serialization of the parsed document). -// All engines run interleaved in every round; the best round is reported. +// All engines run interleaved in every round, each timed call after an untimed +// one of the same engine; the best round is reported. #include #if JSON_VIEW_BENCH_BOOST @@ -33,6 +35,7 @@ #include #include #include +#include #include #include #include @@ -581,6 +584,11 @@ static double pick_od(const std::string& name, simdjson::ondemand::document& d) static std::string slurp(const std::string& p) { std::ifstream f(p, std::ios::binary); + if (!f) + { + std::fprintf(stderr, "cannot open %s\n", p.c_str()); + std::exit(1); + } std::stringstream ss; ss << f.rdbuf(); return ss.str(); @@ -657,6 +665,7 @@ int main(int argc, char** argv) {"json_view (reused)", [&] { reused.read(s); g_sink = static_cast(reused.node_count()); }}, {"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast(yyjson_doc_get_val_count(d)); yyjson_doc_free(d); }}, {"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }}, + {"simdjson DOM (fresh)", [&] { simdjson::dom::parser p; auto e = p.parse(ps).value_unsafe(); g_sink = e.is_object(); }}, #if JSON_VIEW_BENCH_BOOST {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }}, #endif @@ -664,6 +673,7 @@ int main(int argc, char** argv) }}); workloads.push_back({"traverse", { {"json_view", [&] { auto d = json_document::parse(s); stats st; walk(d.root(), st); g_sink = st.num; }}, + {"json_view (reused)", [&] { reused.read(s); stats st; walk(reused.root(), st); g_sink = st.num; }}, {"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(d), st); g_sink = st.num; yyjson_doc_free(d); }}, {"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }}, {"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); stats st; walk_od(d.get_value().value_unsafe(), st); g_sink = st.num; }}, @@ -674,6 +684,7 @@ int main(int argc, char** argv) }}); workloads.push_back({"select", { {"json_view", [&] { auto d = json_document::parse(s); g_sink = pick(name, d.root()); }}, + {"json_view (reused)", [&] { reused.read(s); g_sink = pick(name, reused.root()); }}, {"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = pick(name, yyjson_doc_get_root(d)); yyjson_doc_free(d); }}, {"simdjson DOM", [&] { g_sink = pick(name, sj.parse(ps).value_unsafe()); }}, {"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); g_sink = pick_od(name, d); }}, @@ -710,6 +721,9 @@ int main(int argc, char** argv) { for (std::size_t k = 0; k < wl.second.size(); ++k) { + // an untimed call first: whatever the previous engine left to the allocator + // (e.g. thousands of freed json nodes) is cleaned up here, not in the timing + wl.second[k].fn(); const auto t0 = std::chrono::steady_clock::now(); for (int b = 0; b < dc.batch; ++b) { diff --git a/tests/benchmarks/json_view/compare.py b/tests/benchmarks/json_view/compare.py index 17bf474c6..32a28b115 100755 --- a/tests/benchmarks/json_view/compare.py +++ b/tests/benchmarks/json_view/compare.py @@ -154,11 +154,14 @@ def download_library(name, work): if not os.path.isfile(archive): print(f'downloading {pin["url"]}', flush=True) # the URLs are the https constants in PINNED, and the SHA-256 is checked below - urllib.request.urlretrieve(pin['url'], archive) # nosec B310 + # (into a .part file first, so that an interrupted download is not kept) + urllib.request.urlretrieve(pin['url'], archive + '.part') # nosec B310 + os.replace(archive + '.part', archive) with open(archive, 'rb') as f: digest = hashlib.sha256(f.read()).hexdigest() if digest != pin['sha256']: - sys.exit(f'error: SHA-256 of {archive} is {digest}, expected {pin["sha256"]}') + os.remove(archive) # downloaded again by the next run + sys.exit(f'error: SHA-256 of {archive} is {digest}, expected {pin["sha256"]} (removed)') src = os.path.join(work, 'download', pin['dir']) if not os.path.isdir(src): with tarfile.open(archive) as t: @@ -214,6 +217,10 @@ def main(): ap.add_argument('--corpus', nargs='*', default=[], help='more files for bench_corpus') ap.add_argument('--build-dir', default=os.path.join(HERE, 'build'), help='where to build (default: build/ next to this script)') args = ap.parse_args() + # the benchmarks run in the build directory: make the paths absolute + args.data = os.path.abspath(args.data) + args.corpus = [os.path.abspath(f) for f in args.corpus] + args.build_dir = os.path.abspath(args.build_dir) cxx = os.environ.get('CXX', 'c++') cc = os.environ.get('CC', 'cc')