From d2514a46f7d94bd09ce2110df22127163b8967ee Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 11 Sep 2026 13:15:09 +0200 Subject: [PATCH 1/2] Add benchmarks for the binary readers (#5510) The benchmark suite covered parsing JSON text, dumping, and serializing to CBOR, but only one binary read: FromMsgpack. Nothing measured from_cbor, from_ubjson, from_bjdata or from_bson, so a change to binary_reader.hpp had no baseline to be compared against. Add read benchmarks for every format, in the two shapes that matter: from a contiguous buffer, which is what most callers pass, and from a FILE*, which is what FromMsgpack already measures and which compiles to different code. FromMsgpack itself is left untouched so its numbers stay comparable across releases. The input is derived at setup time by serializing a parsed test file, because the test data repository ships JSON only. The test files are wide and shallow, but the readers' cost is per container, so add three value shapes they do not cover -- deeply nested containers, many sibling containers, and one flat array of numbers -- plus an indefinite-length CBOR string, a form the writer never emits and which therefore has to be assembled by hand. UBJSON and BJData are also captured in their size- and type-annotated form, which the readers handle in a separate code path. BSON requires an object at the top level, so it cannot reuse the array-rooted test files; it is captured on the object-rooted ones, and the shapes are wrapped in an object so every format measures the same value. The setup marks the benchmark as skipped rather than letting the exception escape if that requirement is ever violated. Signed-off-by: Niels Lohmann --- tests/benchmarks/src/benchmarks.cpp | 319 ++++++++++++++++++++++++++++ 1 file changed, 319 insertions(+) diff --git a/tests/benchmarks/src/benchmarks.cpp b/tests/benchmarks/src/benchmarks.cpp index 9522df5a4..f949f3f12 100644 --- a/tests/benchmarks/src/benchmarks.cpp +++ b/tests/benchmarks/src/benchmarks.cpp @@ -252,4 +252,323 @@ static void BinaryToCbor(benchmark::State& state) } BENCHMARK(BinaryToCbor)->RangeMultiplier(2)->Range(8, 8 << 12); +////////////////////////////////////////////////////////////////////////////// +// parse binary formats +////////////////////////////////////////////////////////////////////////////// + +// Only MessagePack had a read benchmark (FromMsgpack above, left untouched so +// its numbers stay comparable across releases). The benchmarks below cover the +// other formats, and read from a contiguous buffer as well as from a FILE*: +// most callers pass a container, and the two adapters compile to different +// code. The test data repository ships JSON only, so the input for each is +// derived at setup time by serializing a parsed test file. + +/// binary format to benchmark; the _optimized variants add UBJSON/BJData size +/// and type annotations, which the readers handle in a separate code path +enum class binary_format +{ + cbor, + msgpack, + ubjson, + ubjson_optimized, + bjdata, + bjdata_optimized, + bson +}; + +static std::vector to_binary(const json& j, const binary_format format) +{ + switch (format) + { + case binary_format::cbor: + return json::to_cbor(j); + case binary_format::msgpack: + return json::to_msgpack(j); + case binary_format::ubjson: + return json::to_ubjson(j); + case binary_format::ubjson_optimized: + return json::to_ubjson(j, true, true); + case binary_format::bjdata: + return json::to_bjdata(j); + case binary_format::bjdata_optimized: + return json::to_bjdata(j, true, true); + case binary_format::bson: + default: + return json::to_bson(j); + } +} + +static json from_binary(const std::vector& bytes, const binary_format format) +{ + switch (format) + { + case binary_format::cbor: + return json::from_cbor(bytes); + case binary_format::msgpack: + return json::from_msgpack(bytes); + case binary_format::ubjson: + case binary_format::ubjson_optimized: + return json::from_ubjson(bytes); + case binary_format::bjdata: + case binary_format::bjdata_optimized: + return json::from_bjdata(bytes); + case binary_format::bson: + default: + return json::from_bson(bytes); + } +} + +static json from_binary(std::FILE* file, const binary_format format) +{ + switch (format) + { + case binary_format::cbor: + return json::from_cbor(file); + case binary_format::msgpack: + return json::from_msgpack(file); + case binary_format::ubjson: + case binary_format::ubjson_optimized: + return json::from_ubjson(file); + case binary_format::bjdata: + case binary_format::bjdata_optimized: + return json::from_bjdata(file); + case binary_format::bson: + default: + return json::from_bson(file); + } +} + +/*! +@brief serialize a parsed test file to @a format + +Returns an empty vector and marks the benchmark as skipped if the file cannot +be represented in the format, rather than letting the exception escape: BSON +requires an object at the top level, and several test files are arrays. +*/ +static std::vector binary_input(benchmark::State& state, const char* filename, const binary_format format) +{ + std::ifstream f(filename); + std::string const str((std::istreambuf_iterator(f)), std::istreambuf_iterator()); + const json j = json::parse(str); + + if (format == binary_format::bson && !j.is_object()) + { + state.SkipWithError("BSON requires an object at the top level"); + return {}; + } + + return to_binary(j, format); +} + +static void FromBinaryBuffer(benchmark::State& state, const char* filename, const binary_format format) +{ + const std::vector bytes = binary_input(state, filename, format); + if (bytes.empty()) + { + return; + } + + for (auto _ : state) + { + // the value is destroyed outside the timed section, because destroying + // a large DOM is not what this benchmark measures + state.PauseTiming(); + auto* j = new json(); + state.ResumeTiming(); + + *j = from_binary(bytes, format); + + state.PauseTiming(); + delete j; + state.ResumeTiming(); + } + + state.SetBytesProcessed(state.iterations() * bytes.size()); +} + +BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / floats, TEST_DATA_DIRECTORY "/regression/floats.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / signed_ints, TEST_DATA_DIRECTORY "/regression/signed_ints.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson_optimized / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson_optimized); +BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson_optimized / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson_optimized); +BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bjdata); +BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata); +BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata_optimized / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bjdata_optimized); +BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata_optimized / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata_optimized); +// BSON requires an object at the top level, so the array-rooted test files +// (jeopardy and the regression files) cannot be captured here +BENCHMARK_CAPTURE(FromBinaryBuffer, bson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bson); +BENCHMARK_CAPTURE(FromBinaryBuffer, bson / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::bson); +BENCHMARK_CAPTURE(FromBinaryBuffer, bson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bson); + +static void FromBinaryFile(benchmark::State& state, const char* filename, const binary_format format) +{ + const std::vector bytes = binary_input(state, filename, format); + if (bytes.empty()) + { + return; + } + + const char* tmp = "benchmark_input.bin"; + std::ofstream o(tmp, std::ios::binary); + o.write(reinterpret_cast(bytes.data()), static_cast(bytes.size())); + o.flush(); + o.close(); + + for (auto _ : state) + { + state.PauseTiming(); + auto* j = new json(); + auto* file = std::fopen(tmp, "rb"); + state.ResumeTiming(); + + *j = from_binary(file, format); + + state.PauseTiming(); + std::fclose(file); + delete j; + state.ResumeTiming(); + } + + state.SetBytesProcessed(state.iterations() * bytes.size()); +} + +BENCHMARK_CAPTURE(FromBinaryFile, cbor / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryFile, cbor / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryFile, ubjson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryFile, ubjson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryFile, bjdata / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata); +BENCHMARK_CAPTURE(FromBinaryFile, bson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bson); + +////////////////////////////////////////////////////////////////////////////// +// parse binary formats: value shapes +////////////////////////////////////////////////////////////////////////////// + +// The test files above are wide and shallow, but the readers' cost is per +// container, so these cover the shapes that stress the container handling +// itself. Every shape is wrapped in an object so that BSON, which requires an +// object at the top level, measures the same value as the other formats. + +/// deeply nested arrays: one container per level, no other work +static json make_nested() +{ + json nested = json::array(); + json* p = &nested; + for (std::size_t i = 1; i < 1000; ++i) + { + p->push_back(json::array()); + p = &p->operator[](0); + } + + json j = json::object(); + j["data"] = std::move(nested); + return j; +} + +/// many sibling containers: maximum container churn, minimum nesting +static json make_containers() +{ + json data = json::array(); + for (std::size_t i = 0; i < 100000; ++i) + { + data.push_back(json::array({1, 2})); + } + + json j = json::object(); + j["data"] = std::move(data); + return j; +} + +/// one flat array of numbers: the scalar decoding path, which must not move +static json make_scalars() +{ + json data = json::array(); + for (std::size_t i = 0; i < 1000000; ++i) + { + data.push_back(i); + } + + json j = json::object(); + j["data"] = std::move(data); + return j; +} + +static void FromBinaryShape(benchmark::State& state, json (*build)(), const binary_format format) +{ + const std::vector bytes = to_binary(build(), format); + + for (auto _ : state) + { + state.PauseTiming(); + auto* j = new json(); + state.ResumeTiming(); + + *j = from_binary(bytes, format); + + state.PauseTiming(); + delete j; + state.ResumeTiming(); + } + + state.SetBytesProcessed(state.iterations() * bytes.size()); +} + +BENCHMARK_CAPTURE(FromBinaryShape, nested / cbor, make_nested, binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryShape, nested / msgpack, make_nested, binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryShape, nested / ubjson, make_nested, binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryShape, nested / bjdata, make_nested, binary_format::bjdata); +BENCHMARK_CAPTURE(FromBinaryShape, nested / bson, make_nested, binary_format::bson); +BENCHMARK_CAPTURE(FromBinaryShape, containers / cbor, make_containers, binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryShape, containers / msgpack, make_containers, binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryShape, containers / ubjson, make_containers, binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryShape, containers / ubjson_optimized, make_containers, binary_format::ubjson_optimized); +BENCHMARK_CAPTURE(FromBinaryShape, containers / bjdata, make_containers, binary_format::bjdata); +BENCHMARK_CAPTURE(FromBinaryShape, containers / bson, make_containers, binary_format::bson); +// BSON names every array element, so a large array measures key generation +// rather than scalar decoding and is left out here +BENCHMARK_CAPTURE(FromBinaryShape, scalars / cbor, make_scalars, binary_format::cbor); +BENCHMARK_CAPTURE(FromBinaryShape, scalars / msgpack, make_scalars, binary_format::msgpack); +BENCHMARK_CAPTURE(FromBinaryShape, scalars / ubjson, make_scalars, binary_format::ubjson); +BENCHMARK_CAPTURE(FromBinaryShape, scalars / bjdata, make_scalars, binary_format::bjdata); + +/*! +@brief parse an indefinite-length CBOR string + +The writer never emits this form, so the input is assembled by hand: 0x7F +opens the string, each chunk is a one-character string, and 0xFF closes it. +*/ +static void FromCborChunkedString(benchmark::State& state, const std::size_t chunks) +{ + std::vector bytes; + bytes.reserve(2 * chunks + 2); + bytes.push_back(0x7F); + for (std::size_t i = 0; i < chunks; ++i) + { + bytes.push_back(0x61); // string of length 1 + bytes.push_back(0x61); // 'a' + } + bytes.push_back(0xFF); + + for (auto _ : state) + { + json j = json::from_cbor(bytes); + benchmark::DoNotOptimize(j); + } + + state.SetBytesProcessed(state.iterations() * bytes.size()); +} + +BENCHMARK_CAPTURE(FromCborChunkedString, 10000 chunks, 10000); + BENCHMARK_MAIN(); From aa391dc0a56f8409e2e7aca6e7c9a9d766d44ce8 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 11 Sep 2026 17:55:46 +0200 Subject: [PATCH 2/2] Fix two develop CI regressions: dump() nodiscard warning and binary-reader const-correctness (#5520) * Discard dump()'s [[nodiscard]] return value in an exception-only check CHECK_THROWS_WITH_AS(j.dump(), ...) called dump() only to trigger and catch the exception, but never used the return value. dump() is warn_unused_result, so GCC's pedantic build (-Werror --all-warnings) rejected it as -Werror=unused-result, breaking ci_test_gcc. Wrapped in utils::ignore_return_value(), matching every other such call in this file. Signed-off-by: Niels Lohmann * Mark container_frame top as const in CBOR/UBJSON readers clang-tidy's misc-const-correctness flagged these on PR #5520's CI: the BSON sibling copy was already const, but these two were left mutable even though only container_stack.back().remaining is ever written. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/detail/input/binary_reader.hpp | 4 ++-- single_include/nlohmann/json.hpp | 4 ++-- tests/src/unit-serialization.cpp | 2 +- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index efa93070d..e0343fd0b 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -1434,7 +1434,7 @@ class binary_reader // a copy, not a reference: it must stay valid across the // pop_back() below, which destroys the container_stack element // it would otherwise alias - container_frame top = container_stack.back(); + const container_frame top = container_stack.back(); bool at_end = false; if (top.remaining != npos) @@ -2211,7 +2211,7 @@ class binary_reader // would otherwise alias. for (;;) { - container_frame top = container_stack.back(); + const container_frame top = container_stack.back(); if (top.remaining != npos) { diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 150118a1c..3b2233a57 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -13383,7 +13383,7 @@ class binary_reader // a copy, not a reference: it must stay valid across the // pop_back() below, which destroys the container_stack element // it would otherwise alias - container_frame top = container_stack.back(); + const container_frame top = container_stack.back(); bool at_end = false; if (top.remaining != npos) @@ -14160,7 +14160,7 @@ class binary_reader // would otherwise alias. for (;;) { - container_frame top = container_stack.back(); + const container_frame top = container_stack.back(); if (top.remaining != npos) { diff --git a/tests/src/unit-serialization.cpp b/tests/src/unit-serialization.cpp index eddf59f2c..511108c64 100644 --- a/tests/src/unit-serialization.cpp +++ b/tests/src/unit-serialization.cpp @@ -469,7 +469,7 @@ TEST_CASE("serialization of strings (bulk fast path)") SECTION("invalid UTF-8 handling is unaffected by the fast path") { const json j = std::string("valid\xff" "more"); - CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0xFF", json::type_error&); + CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0xFF", json::type_error&); CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"valid\xef\xbf\xbd" "more\""); CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"valid\\ufffdmore\""); CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"validmore\"");