From af91eee2cc770a369d578273cca7ea277008f82b Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Wed, 16 Sep 2026 20:11:06 +0200 Subject: [PATCH 1/9] Bump the codeql-action group with 4 updates (#5536) Bumps the codeql-action group with 4 updates: [github/codeql-action/init](https://github.com/github/codeql-action), [github/codeql-action/autobuild](https://github.com/github/codeql-action), [github/codeql-action/analyze](https://github.com/github/codeql-action) and [github/codeql-action/upload-sarif](https://github.com/github/codeql-action). Updates `github/codeql-action/init` from 4.37.9 to 4.38.0 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/cdf488f595d80d6e07e03d4674febd5ab45fa938...b96794f015dfd88f77b49b1c93e0fa7110f94c63) Updates `github/codeql-action/autobuild` from 4.37.9 to 4.38.0 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/cdf488f595d80d6e07e03d4674febd5ab45fa938...b96794f015dfd88f77b49b1c93e0fa7110f94c63) Updates `github/codeql-action/analyze` from 4.37.9 to 4.38.0 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/cdf488f595d80d6e07e03d4674febd5ab45fa938...b96794f015dfd88f77b49b1c93e0fa7110f94c63) Updates `github/codeql-action/upload-sarif` from 4.37.9 to 4.38.0 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/cdf488f595d80d6e07e03d4674febd5ab45fa938...b96794f015dfd88f77b49b1c93e0fa7110f94c63) --- updated-dependencies: - dependency-name: github/codeql-action/init dependency-version: 4.38.0 dependency-type: direct:production update-type: version-update:semver-minor dependency-group: codeql-action - dependency-name: github/codeql-action/autobuild dependency-version: 4.38.0 dependency-type: direct:production update-type: version-update:semver-minor dependency-group: codeql-action - dependency-name: github/codeql-action/analyze dependency-version: 4.38.0 dependency-type: direct:production update-type: version-update:semver-minor dependency-group: codeql-action - dependency-name: github/codeql-action/upload-sarif dependency-version: 4.38.0 dependency-type: direct:production update-type: version-update:semver-minor dependency-group: codeql-action ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- .github/workflows/codeql-analysis.yml | 6 +++--- .github/workflows/flawfinder.yml | 2 +- .github/workflows/scorecards.yml | 2 +- .github/workflows/semgrep.yml | 2 +- 4 files changed, 6 insertions(+), 6 deletions(-) diff --git a/.github/workflows/codeql-analysis.yml b/.github/workflows/codeql-analysis.yml index c5497ef9c..1c2e73cd1 100644 --- a/.github/workflows/codeql-analysis.yml +++ b/.github/workflows/codeql-analysis.yml @@ -38,14 +38,14 @@ jobs: # Initializes the CodeQL tools for scanning. - name: Initialize CodeQL - uses: github/codeql-action/init@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/init@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: languages: c-cpp # Autobuild attempts to build any compiled languages (C/C++, C#, or Java). # If this step fails, then you should remove it and run the build manually (see below) - name: Autobuild - uses: github/codeql-action/autobuild@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/autobuild@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 - name: Perform CodeQL Analysis - uses: github/codeql-action/analyze@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/analyze@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 diff --git a/.github/workflows/flawfinder.yml b/.github/workflows/flawfinder.yml index 7c4d22b3c..2c8befd40 100644 --- a/.github/workflows/flawfinder.yml +++ b/.github/workflows/flawfinder.yml @@ -43,6 +43,6 @@ jobs: output: 'flawfinder_results.sarif' - name: Upload analysis results to GitHub Security tab - uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: sarif_file: ${{github.workspace}}/flawfinder_results.sarif diff --git a/.github/workflows/scorecards.yml b/.github/workflows/scorecards.yml index de919b1ab..113da079a 100644 --- a/.github/workflows/scorecards.yml +++ b/.github/workflows/scorecards.yml @@ -76,6 +76,6 @@ jobs: # Upload the results to GitHub's code scanning dashboard. - name: "Upload to code-scanning" - uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: sarif_file: results.sarif diff --git a/.github/workflows/semgrep.yml b/.github/workflows/semgrep.yml index 38de00932..dc326db55 100644 --- a/.github/workflows/semgrep.yml +++ b/.github/workflows/semgrep.yml @@ -61,7 +61,7 @@ jobs: # Upload SARIF file generated in previous step - name: Upload SARIF file - uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: sarif_file: semgrep.sarif if: always() From ff65f688f723e8aea56905f77fdcf2c0761bf779 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:11:19 +0200 Subject: [PATCH 2/9] Cover binary values in the indentation regression tests (#5533) #5285's indentation regression test didn't exercise json::binary, which serializes as an object but always writes its byte array compactly (dump_byte()). #5186 had covered this case before it was closed as superseded; port just that coverage here. Signed-off-by: Niels Lohmann --- tests/src/unit-serialization.cpp | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/tests/src/unit-serialization.cpp b/tests/src/unit-serialization.cpp index 511108c64..00b305a75 100644 --- a/tests/src/unit-serialization.cpp +++ b/tests/src/unit-serialization.cpp @@ -522,6 +522,19 @@ TEST_CASE("indentation is written straight into the write buffer") CHECK(json::parse(out) == j); } + SECTION("binary values are indented the same way") + { + // a binary value is serialized as an object with "bytes" and + // "subtype" keys; the byte array itself is always written compactly + // (see dump_byte()), so only the surrounding object's indentation + // goes through put_indent() + const json j = json::binary({1, 2, 3}, 128); + CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"bytes\": [1, 2, 3],\n" + + std::string(2000, ' ') + "\"subtype\": 128\n}"); + CHECK(j.dump(2000, '\t') == "{\n" + std::string(2000, '\t') + "\"bytes\": [1, 2, 3],\n" + + std::string(2000, '\t') + "\"subtype\": 128\n}"); + } + SECTION("indentation is unchanged for ordinary widths") { const json j = {{"a", {1, 2}}, {"b", nullptr}}; From f58db1c9e81b2c03cb97152657b513bd53b50dac Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:15:20 +0200 Subject: [PATCH 3/9] Add test coverage for ordered_json/alt_json across binary formats and patch/diff/flatten APIs (#5480) * Add test coverage for ordered_json/alt_json across binary formats and patch/diff/flatten APIs Closes a test-coverage gap from #5421: ordered_json (and the alt_string-based basic_json specialization from unit-alt-string.cpp) were never round-tripped through the binary formats (CBOR/MessagePack/UBJSON/BSON/BJData), nor through flatten()/unflatten(), diff()/patch()/patch_inplace(), or merge_patch(). Also adds a std::formatter spot-check, mirroring the precedent set by the format_as() ADL-deduction test. Signed-off-by: Niels Lohmann * Pass alt_string's std::string constructor argument by value (clang-tidy modernize-pass-by-value) Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- tests/src/unit-ordered_json2.cpp | 489 +++++++++++++++++++++++++++++++ tests/src/unit-std-format.cpp | 13 + 2 files changed, 502 insertions(+) create mode 100644 tests/src/unit-ordered_json2.cpp diff --git a/tests/src/unit-ordered_json2.cpp b/tests/src/unit-ordered_json2.cpp new file mode 100644 index 000000000..83eb7668d --- /dev/null +++ b/tests/src/unit-ordered_json2.cpp @@ -0,0 +1,489 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-FileCopyrightText: 2018 Vitaliy Manushkin +// SPDX-License-Identifier: MIT + +// This file closes a test-coverage gap described in GitHub issue #5421: +// nlohmann::ordered_json (and other non-default basic_json specializations, +// such as the alt_string-based one from unit-alt-string.cpp) were never +// exercised through the binary formats (CBOR/MessagePack/UBJSON/BSON/BJData) +// or through flatten()/unflatten()/diff()/patch()/merge_patch(). + +#include "doctest_compatibility.h" + +#include + +#include +#include +#include +#include + +using nlohmann::json; +using nlohmann::ordered_json; + +///////////////////////////////////////////////////////////////////////////// +// alt_json: a second, independent copy of the custom-string_t basic_json +// specialization defined in unit-alt-string.cpp. +// +// It is duplicated here (rather than shared via a header) because every +// unit-*.cpp file in this test suite is compiled into its own standalone +// executable (see tests/CMakeLists.txt), so there is no ODR concern in +// having the same class name defined in multiple translation units. +// +// Two members had to be added relative to the original alt_string +// (a constructor from std::string, and a find(char, pos) overload) because +// the original type was never used with the binary writers/readers before +// this file: BSON's array/document writer converts std::to_string() results +// and checks for embedded NUL characters via find(char), and the UBJSON/BSON +// high-precision-number path constructs the SAX string_t argument from a +// std::string. Neither path is exercised anywhere else in the test suite for +// this type, which is presumably why the gap was never noticed. +///////////////////////////////////////////////////////////////////////////// + +class alt_string; +bool operator<(const char* op1, const alt_string& op2) noexcept; // NOLINT(misc-use-internal-linkage) +void int_to_string(alt_string& target, std::size_t value); // NOLINT(misc-use-internal-linkage) + +class alt_string +{ + public: + using value_type = std::string::value_type; + + static constexpr auto npos = (std::numeric_limits::max)(); + + alt_string(const char* str): str_impl(str) {} + alt_string(const char* str, std::size_t count): str_impl(str, count) {} + alt_string(std::string str): str_impl(std::move(str)) {} + alt_string(size_t count, char chr): str_impl(count, chr) {} + alt_string() = default; + + alt_string& append(char ch) + { + str_impl.push_back(ch); + return *this; + } + + alt_string& append(const alt_string& str) + { + str_impl.append(str.str_impl); + return *this; + } + + alt_string& append(const char* s, std::size_t length) + { + str_impl.append(s, length); + return *this; + } + + void push_back(char c) + { + str_impl.push_back(c); + } + + template + bool operator==(const op_type& op) const + { + return str_impl == op; + } + + bool operator==(const alt_string& op) const + { + return str_impl == op.str_impl; + } + + template + bool operator!=(const op_type& op) const + { + return str_impl != op; + } + + bool operator!=(const alt_string& op) const + { + return str_impl != op.str_impl; + } + + std::size_t size() const noexcept + { + return str_impl.size(); + } + + void resize(std::size_t n) + { + str_impl.resize(n); + } + + void resize(std::size_t n, char c) + { + str_impl.resize(n, c); + } + + template + bool operator<(const op_type& op) const noexcept + { + return str_impl < op; + } + + bool operator<(const alt_string& op) const noexcept + { + return str_impl < op.str_impl; + } + + const char* c_str() const + { + return str_impl.c_str(); + } + + char& operator[](std::size_t index) + { + return str_impl[index]; + } + + const char& operator[](std::size_t index) const + { + return str_impl[index]; + } + + char& back() + { + return str_impl.back(); + } + + const char& back() const + { + return str_impl.back(); + } + + void clear() + { + str_impl.clear(); + } + + const value_type* data() const + { + return str_impl.data(); + } + + bool empty() const + { + return str_impl.empty(); + } + + std::size_t find(const alt_string& str, std::size_t pos = 0) const + { + return str_impl.find(str.str_impl, pos); + } + + // needed by binary_writer's BSON support, which probes string keys for + // embedded NUL characters via find(char) + std::size_t find(char c, std::size_t pos = 0) const + { + return str_impl.find(c, pos); + } + + std::size_t find_first_of(char c, std::size_t pos = 0) const + { + return str_impl.find_first_of(c, pos); + } + + alt_string substr(std::size_t pos = 0, std::size_t count = npos) const + { + const std::string s = str_impl.substr(pos, count); + return {s.data(), s.size()}; + } + + alt_string& replace(std::size_t pos, std::size_t count, const alt_string& str) + { + str_impl.replace(pos, count, str.str_impl); + return *this; + } + + void reserve(std::size_t new_cap = 0) + { + str_impl.reserve(new_cap); + } + + private: + std::string str_impl {}; // NOLINT(readability-redundant-member-init) + + friend bool operator<(const char* /*op1*/, const alt_string& /*op2*/) noexcept; +}; + +void int_to_string(alt_string& target, std::size_t value) +{ + target = std::to_string(value).c_str(); +} + +using alt_json = nlohmann::basic_json < + std::map, + std::vector, + alt_string, + bool, + std::int64_t, + std::uint64_t, + double, + std::allocator, + nlohmann::adl_serializer >; + +bool operator<(const char* op1, const alt_string& op2) noexcept +{ + return op1 < op2.str_impl; +} + +namespace +{ + +// collects the object keys of j, in iteration order +std::vector collect_keys(const ordered_json& j) +{ + std::vector result; + for (auto it = j.cbegin(); it != j.cend(); ++it) + { + result.push_back(it.key()); + } + return result; +} + +// a nested object/array value with keys inserted in non-alphabetical order, +// used to check both round-trip equality and (for ordered_json) that +// insertion order survives a trip through a binary format +ordered_json make_rich_ordered_json() +{ + ordered_json j; + j["zebra"] = 1; + j["apple"] = ordered_json::array({1, 2, 3}); + j["mango"]["z_nested"] = true; + j["mango"]["a_nested"] = nullptr; + j["banana"] = "some text"; + j["cherry"] = 3.14; + return j; +} + +alt_json make_rich_alt_json() +{ + alt_json j; + j["zebra"] = 1; + j["apple"] = alt_json::array({1, 2, 3}); + j["mango"]["z_nested"] = true; + j["mango"]["a_nested"] = nullptr; + j["banana"] = "some text"; + j["cherry"] = 3.14; + return j; +} + +} // namespace + +TEST_CASE("ordered_json across binary formats") +{ + const ordered_json original = make_rich_ordered_json(); + const std::vector original_keys = collect_keys(original); + const std::vector original_mango_keys = collect_keys(original["mango"]); + + SECTION("CBOR") + { + const auto bytes = ordered_json::to_cbor(original); + const auto restored = ordered_json::from_cbor(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("MessagePack") + { + const auto bytes = ordered_json::to_msgpack(original); + const auto restored = ordered_json::from_msgpack(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("UBJSON") + { + const auto bytes = ordered_json::to_ubjson(original); + const auto restored = ordered_json::from_ubjson(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("BSON") + { + const auto bytes = ordered_json::to_bson(original); + const auto restored = ordered_json::from_bson(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("BJData") + { + const auto bytes = ordered_json::to_bjdata(original); + const auto restored = ordered_json::from_bjdata(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } +} + +TEST_CASE("alt_json (custom string_t) across binary formats") +{ + const alt_json original = make_rich_alt_json(); + + SECTION("CBOR") + { + const auto bytes = alt_json::to_cbor(original); + const auto restored = alt_json::from_cbor(bytes); + CHECK(restored == original); + } + + SECTION("MessagePack") + { + const auto bytes = alt_json::to_msgpack(original); + const auto restored = alt_json::from_msgpack(bytes); + CHECK(restored == original); + } + + SECTION("UBJSON") + { + const auto bytes = alt_json::to_ubjson(original); + const auto restored = alt_json::from_ubjson(bytes); + CHECK(restored == original); + } + + SECTION("BSON") + { + const auto bytes = alt_json::to_bson(original); + const auto restored = alt_json::from_bson(bytes); + CHECK(restored == original); + } + + SECTION("BJData") + { + const auto bytes = alt_json::to_bjdata(original); + const auto restored = alt_json::from_bjdata(bytes); + CHECK(restored == original); + } +} + +TEST_CASE("ordered_json operator== is sensitive to key order") +{ + // Unlike nlohmann::json (whose object_t is a std::map, so equality never + // depends on insertion order), ordered_json's object_t (ordered_map) is a + // std::vector> under the hood, and does not define its + // own operator==: it inherits std::vector's element-wise comparison. As a + // result, two ordered_json objects holding the very same key/value pairs + // in different insertion order compare *unequal*. This is the property + // that makes the round-trip `CHECK(restored == original)` checks above a + // meaningful order-preservation check by themselves (the explicit + // collect_keys() comparisons make that check explicit/readable, and + // guard against this operator== behavior ever changing). + ordered_json a; + a["x"] = 1; + a["y"] = 2; + + ordered_json b; + b["y"] = 2; + b["x"] = 1; + + CHECK(a.size() == b.size()); + CHECK(a["x"] == b["x"]); + CHECK(a["y"] == b["y"]); + CHECK_FALSE(a == b); +} + +TEST_CASE("duplicate keys in a binary-encoded object") +{ + // CBOR encoding of a map with two entries under the same key "a": {"a": 1, "a": 2} + const std::vector cbor_bytes + { + 0xA2, 0x61, 'a', 0x01, 0x61, 'a', 0x02 + }; + + // Both json (std::map, via operator[]) and ordered_json (ordered_map, via + // operator[]) build binary-decoded objects by looking up/creating the + // entry for each incoming key and then assigning the value into it. This + // means a repeated key does *not* produce two entries in either case; + // instead, the *first* occurrence's position is kept (relevant only for + // ordered_json) while the *last* occurrence's value wins (for both) -- + // this matches operator[]'s "assign the referenced slot" semantics, and + // is worth noting because it differs from the initializer-list + // construction path (`ordered_json{{"a",1},{"a",2}}`), which builds + // through insert()/emplace() and therefore keeps the *first* value, not + // the last (see the "There are no dup keys..." case in + // unit-ordered_json.cpp). + const auto j = json::from_cbor(cbor_bytes); + const auto oj = ordered_json::from_cbor(cbor_bytes); + + CHECK(j.size() == 1); + CHECK(oj.size() == 1); + CHECK(j["a"] == 2); + CHECK(oj["a"] == 2); + CHECK(j == json(oj)); +} + +TEST_CASE("ordered_json through flatten/unflatten") +{ + const ordered_json original = make_rich_ordered_json(); + const std::vector original_keys = collect_keys(original); + const std::vector original_mango_keys = collect_keys(original["mango"]); + + const ordered_json flat = original.flatten(); + const ordered_json unflattened = flat.unflatten(); + + CHECK(unflattened == original); + // flatten() walks the value depth-first in iteration order and + // unflatten() re-inserts each flattened key via operator[] in the flat + // object's iteration order, so for ordered_json the original key order + // (both top-level and nested) is preserved end-to-end. + CHECK(collect_keys(unflattened) == original_keys); + CHECK(collect_keys(unflattened["mango"]) == original_mango_keys); +} + +TEST_CASE("ordered_json through diff/patch/patch_inplace") +{ + ordered_json original; + original["one"] = 1; + original["two"] = 2; + original["three"] = 3; + + ordered_json target = original; + target["one"] = 100; // replace + target.erase("two"); // remove + target["four"] = 4; // add + + const ordered_json patch = ordered_json::diff(original, target); + + SECTION("patch") + { + const ordered_json patched = original.patch(patch); + CHECK(patched == target); + } + + SECTION("patch_inplace") + { + ordered_json copy = original; + copy.patch_inplace(patch); + CHECK(copy == target); + } +} + +TEST_CASE("ordered_json through merge_patch") +{ + ordered_json original; + original["a"] = 1; + original["b"] = 2; + + const ordered_json patch = {{"b", nullptr}, {"c", 3}}; + + original.merge_patch(patch); + + ordered_json expected; + expected["a"] = 1; + expected["c"] = 3; + + CHECK(original == expected); + CHECK(collect_keys(original) == collect_keys(expected)); +} diff --git a/tests/src/unit-std-format.cpp b/tests/src/unit-std-format.cpp index f7a364884..d18ccbd32 100644 --- a/tests/src/unit-std-format.cpp +++ b/tests/src/unit-std-format.cpp @@ -17,6 +17,7 @@ #include using json = nlohmann::json; +using ordered_json = nlohmann::ordered_json; // JSON_HAS_CPP_20 (do not remove; see note at top of file) #if JSON_HAS_STD_FORMAT @@ -93,4 +94,16 @@ TEST_CASE("std::formatter") } } +TEST_CASE("std::formatter") +{ + // spot-check a non-default basic_json instantiation, since the formatter + // is written against the generic NLOHMANN_BASIC_JSON_TPL_DECLARATION + // template and must actually instantiate (and behave correctly) for + // template arguments other than nlohmann::json + const ordered_json j = {{"foo", 1}, {"bar", {1, 2, 3}}}; + CHECK(std::format("{}", j) == j.dump()); + CHECK(std::format("{:#}", j) == j.dump(4)); + CHECK(std::format("{:2}", j) == j.dump(2)); +} + #endif From 75efd6b1c316c68b7240876cd536fed038b9560e Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:15:46 +0200 Subject: [PATCH 4/9] Test suite: cover untested macro configs, std::formatter branches, patch_inplace, and fix a duplicate TEST_CASE name (#5492) * test: cover JSON_NO_IO, JSON_THROW/TRY/CATCH_USER, JSON_SKIP_LIBRARY_VERSION_CHECK, and JSON_DisableEnumSerialization in CI (#5423) These four supported configuration macros were never actually compiled anywhere in the test matrix: - JSON_NO_IO and the JSON_THROW_USER/JSON_TRY_USER/JSON_CATCH_USER trio are exercised together in a new tests/src/unit-no_io_and_user_exceptions.cpp, which is automatically picked up by the existing unit-*.cpp test glob and thus built across the whole standard test matrix. - JSON_SKIP_LIBRARY_VERSION_CHECK is exercised by a new, dedicated tests/src/skip_library_version_check.cpp, compiled directly by the new ci_test_skiplibraryversioncheck target in cmake/ci.cmake: the scenario it simulates (mixing two differently-versioned inclusions of the library) unavoidably triggers the compiler's own "macro redefined" warning, which would fail under the library's own -Weverything/-Werror unit test matrix for a reason unrelated to the macro under test. - JSON_DisableEnumSerialization already had #if-guarded tests in several unit-*.cpp files (from #4384), but no CMake target ever actually set the JSON_DisableEnumSerialization CMake option, so that guarded code was never compiled. Add ci_test_disableenumserialization, mirroring the existing ci_test_noimplicitconversions/ci_test_noglobaludls targets. Building the full test suite with this option on surfaced one real, narrow gap: get() on std::vector (used by unit-regression2.cpp's custom BinaryType tests) relies on std::byte being handled via enum serialization, so add the same #if-guard convention to the two affected SECTIONs there. Both new CI targets are added to the ci_cmake_options matrix in .github/workflows/ubuntu.yml, alongside the existing ci_test_* targets. Signed-off-by: Niels Lohmann * test: cover multi-digit widths and bare alignment in std::formatter (#5423) Every existing std::formatter spec with a width used a single digit (e.g. "{:2}"), so the width-parsing loop's accumulation of a second/third digit was never exercised; add multi-digit width cases. Likewise, every existing spec with an alignment character also had an explicit fill character, so the bare-alignment branch (e.g. "{:<}", with no fill) was never exercised; add cases asserting it keeps the default space indent character. Signed-off-by: Niels Lohmann * test: add coverage for patch_inplace() (#5423) patch_inplace() had no unit test at all. Add a happy-path case mirroring an existing patch() example, and -- more importantly -- pin its distinguishing contract versus patch(): when a multi-operation JSON Patch fails partway through, patch_inplace() (which mutates the document directly, operation by operation) leaves whatever operations already succeeded applied, whereas patch() (which applies the patch to an internal copy that is discarded on exception) leaves the original completely untouched either way. Verified empirically against the current implementation before writing the assertions. Signed-off-by: Niels Lohmann * test: fix duplicate TEST_CASE name in unit-no-mem-leak-on-adl-serialize.cpp (#5423) Two distinct TEST_CASEs were both named "check_for_mem_leak_on_adl_to_json-2". doctest allows duplicate names, so both still ran, but it makes --test-case= filtering and reporting ambiguous. Rename the second one to "-3", continuing the existing "-1"/"-2" sequence. Signed-off-by: Niels Lohmann * test: add direct coverage for the std::u8string to_json overload (#5423) The ADL to_json overload for std::basic_string was only ever reached indirectly, via std::filesystem::path::u8string(). Add a test that constructs a json value directly from a std::u8string, gated the same way as the overload itself (include/nlohmann/detail/conversions/to_json.hpp): behind both the std::filesystem::path feature guard and __cpp_lib_char8_t, since the overload only exists when both are satisfied. Signed-off-by: Niels Lohmann * test: verify move semantics of byte_container_with_subtype's rvalue constructors (#5423) The two rvalue-reference constructors were never distinguished from their const-lvalue-reference twins by any test. Add a "move semantics" section that constructs from an rvalue std::vector, checks the resulting container keeps the exact same buffer address as the source (a stronger check than just observing the source ended up empty, since a copy-then-clear could do that too), and confirms the source vector was left empty. Signed-off-by: Niels Lohmann * Guard patch_inplace() partial-application test against JSON_NOEXCEPTION The "distinguishing contract vs patch(): partial application on failure" test relies on doc.patch_inplace(patch) actually throwing so the partially-applied state can be observed right after the throw point. Under ci_test_noexceptions, JSON_THROW() calls std::abort() instead of throwing, and doctest's --no-throw test filter (which that CI job passes) makes CHECK_THROWS_AS() a no-op that never even evaluates its expression -- so patch_inplace() is never called and the follow-up assertions fail against the untouched original document. Guard the whole SECTION with #if !defined(JSON_NOEXCEPTION), following the same convention already used elsewhere in the test suite (e.g. unit-class_parser.cpp) for exception-dependent tests. Signed-off-by: Niels Lohmann * Fix MSVC C2220 in the std::u8string conversion test MSVC's C5321 ("nonstandard extension used: encoding '\xNN' as a multi-byte utf-8 character") is promoted to a hard error by our MSVC CI configs. It fires because the test composed a non-ASCII UTF-8 sequence inside a u8"" literal using raw \x byte escapes; MSVC treats that as nonstandard and suggests using \u universal-character-names instead, which every compiler agrees on and which compiles down to the exact same encoded bytes. Signed-off-by: Niels Lohmann * Guard the JSON_THROW_USER test against JSON_NOEXCEPTION and GCC's -Wunused-result Two independent CI configurations failed to build/run this new test: - ci_test_noexceptions runs the whole suite with -DJSON_NOEXCEPTION and doctest's "--no-throw" filter, which compiles CHECK_THROWS_AS() down to a no-op that never even invokes the guarded expression. Since this test's whole point is to observe json_throw_user_call_count after json::parse()/at() actually throw, it can't be meaningfully run under that filter (our JSON_THROW_USER override still throws real exceptions regardless of JSON_NOEXCEPTION, but the assertion never gets a chance to run). Guard the TEST_CASE with #if !defined(JSON_NOEXCEPTION), mirroring the existing precedent in unit-json_patch.cpp. - ci_test_gcc and ci_test_standards_gcc(11) failed with -Werror=unused-result on the discarded json::parse() return value. json::parse() is marked warn_unused_result, and unlike a real [[nodiscard]] attribute, GCC does not consider that satisfied by doctest's (void)-cast around the expression in C++11 mode. Assign the result to a discarded local instead, matching the established `json _ = json::parse(...)` idiom already used throughout unit-class_parser.cpp. Signed-off-by: Niels Lohmann * Suppress a clang-tidy false positive on an intentional defensive copy performance-unnecessary-copy-initialization suggests copy_for_patch could be a reference since it's never modified -- but the copy is the point: it guards against a hypothetical regression where patch() mutates its receiver, which a reference could never catch (the follow-up assertion would just compare `original` to itself). Signed-off-by: Niels Lohmann * Fix clang-tidy findings in the JSON_NO_IO/JSON_THROW_USER test - bugprone-macro-parentheses: wrap the JSON_THROW_USER macro argument in parentheses at the throw site. - modernize-raw-string-literal: switch two escaped JSON string literals to raw string literals. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/workflows/ubuntu.yml | 2 +- cmake/ci.cmake | 34 +++++++ tests/src/skip_library_version_check.cpp | 61 ++++++++++++ .../src/unit-byte_container_with_subtype.cpp | 33 +++++++ tests/src/unit-conversions.cpp | 34 +++++++ tests/src/unit-json_patch.cpp | 96 +++++++++++++++++++ .../src/unit-no-mem-leak-on-adl-serialize.cpp | 2 +- tests/src/unit-no_io_and_user_exceptions.cpp | 91 ++++++++++++++++++ tests/src/unit-regression3.cpp | 10 ++ tests/src/unit-std-format.cpp | 17 ++++ 10 files changed, 378 insertions(+), 2 deletions(-) create mode 100644 tests/src/skip_library_version_check.cpp create mode 100644 tests/src/unit-no_io_and_user_exceptions.cpp diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index 69a3cbc45..7ac4cfcb0 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -100,7 +100,7 @@ jobs: container: ubuntu:focal strategy: matrix: - target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_simdutf] + target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_skiplibraryversioncheck, ci_test_simdutf] steps: - name: Install build-essential run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev diff --git a/cmake/ci.cmake b/cmake/ci.cmake index 18fef2075..7d085fb2b 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -260,6 +260,40 @@ add_custom_target(ci_test_noglobaludls COMMENT "Compile and test with global UDLs disabled" ) +############################################################################### +# Disable enum serialization. +############################################################################### + +add_custom_target(ci_test_disableenumserialization + COMMAND ${CMAKE_COMMAND} + -DCMAKE_BUILD_TYPE=Debug -GNinja + -DJSON_BuildTests=ON -DJSON_FastTests=ON -DJSON_DisableEnumSerialization=ON + -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_disableenumserialization + COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_disableenumserialization + COMMAND cd ${PROJECT_BINARY_DIR}/build_disableenumserialization && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure + COMMENT "Compile and test with enum serialization disabled" +) + +############################################################################### +# Skip the multiple-inclusion library version check. +############################################################################### + +# tests/src/skip_library_version_check.cpp deliberately simulates a scenario +# (mixing two differently-versioned inclusions of the library in one +# translation unit) that unavoidably triggers the compiler's own "macro +# redefined" warning, so -- unlike the ci_test_* targets above -- it is +# compiled directly here, with a modest warning set, instead of being folded +# into the library's own -Weverything/-Werror unit test matrix. +add_custom_target(ci_test_skiplibraryversioncheck + COMMAND ${CMAKE_COMMAND} -E make_directory ${PROJECT_BINARY_DIR}/skip_library_version_check + COMMAND ${CMAKE_CXX_COMPILER} -std=c++11 -Wall -Wextra + -I${PROJECT_SOURCE_DIR}/include + ${PROJECT_SOURCE_DIR}/tests/src/skip_library_version_check.cpp + -o ${PROJECT_BINARY_DIR}/skip_library_version_check/skip_library_version_check + COMMAND ${PROJECT_BINARY_DIR}/skip_library_version_check/skip_library_version_check + COMMENT "Compile and run a translation unit simulating a mismatched library version, with JSON_SKIP_LIBRARY_VERSION_CHECK defined" +) + ############################################################################### # Coverage. ############################################################################### diff --git a/tests/src/skip_library_version_check.cpp b/tests/src/skip_library_version_check.cpp new file mode 100644 index 000000000..ddaa4415c --- /dev/null +++ b/tests/src/skip_library_version_check.cpp @@ -0,0 +1,61 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// Standalone compile-and-run check for the JSON_SKIP_LIBRARY_VERSION_CHECK +// configuration macro, which (per #5423) was never exercised anywhere in the +// test matrix. +// +// include/nlohmann/detail/abi_macros.hpp normally emits a #warning if +// NLOHMANN_JSON_VERSION_MAJOR/MINOR/PATCH are already defined (as they would +// be by an earlier inclusion of a different version of the library) with +// values that mismatch the version about to be defined -- unless +// JSON_SKIP_LIBRARY_VERSION_CHECK is defined, in which case the check (and +// that #warning) is skipped. +// +// This file deliberately is not named tests/src/unit-*.cpp: it is compiled +// directly (with a modest, non-strict warning set) by the dedicated +// ci_test_skiplibraryversioncheck target in cmake/ci.cmake, rather than being +// folded into the library's own -Weverything/-Werror unit test matrix. That +// is because the scenario simulated here -- mixing two different, already +// differently-versioned inclusions of the library in one translation unit -- +// unavoidably also triggers the *compiler's own* "macro redefined" warning, +// independent of (and unaffected by) JSON_SKIP_LIBRARY_VERSION_CHECK, which +// only ever silences the library's own #warning. Building this file under +// -Weverything -Werror would therefore fail for a reason unrelated to the +// macro under test. +#define NLOHMANN_JSON_VERSION_MAJOR 0 +#define NLOHMANN_JSON_VERSION_MINOR 0 +#define NLOHMANN_JSON_VERSION_PATCH 0 + +#define JSON_SKIP_LIBRARY_VERSION_CHECK 1 + +#include + +int main() +{ + // reaching this point at all already proves that the mismatched, + // pre-defined version macros above did not stop compilation -- which is + // exactly what JSON_SKIP_LIBRARY_VERSION_CHECK is for. The library must + // also still be fully usable. + const nlohmann::json j = {{"a", 1}, {"b", {1, 2, 3}}}; + if (j.dump() != "{\"a\":1,\"b\":[1,2,3]}") + { + return 1; + } + + // include/nlohmann/detail/abi_macros.hpp unconditionally (re)defines the + // version macros to the library's real, current version right after the + // (here, skipped) mismatch check, regardless of the deliberately wrong + // stand-in values defined above. + if (NLOHMANN_JSON_VERSION_MAJOR == 0 && NLOHMANN_JSON_VERSION_MINOR == 0 && NLOHMANN_JSON_VERSION_PATCH == 0) + { + return 1; + } + + return 0; +} diff --git a/tests/src/unit-byte_container_with_subtype.cpp b/tests/src/unit-byte_container_with_subtype.cpp index 3983ba95f..2e448ac7d 100644 --- a/tests/src/unit-byte_container_with_subtype.cpp +++ b/tests/src/unit-byte_container_with_subtype.cpp @@ -42,6 +42,39 @@ TEST_CASE("byte_container_with_subtype") CHECK(container.subtype() == static_cast(-1)); } + SECTION("move semantics") + { + // the rvalue-reference constructor (without a subtype) must actually move + // the passed-in container rather than copy it; comparing the buffer address + // before and after is a stronger check than just observing the source is + // empty afterward, since a copy-then-clear could also leave it empty + { + std::vector bytes = {{0xCA, 0xFE, 0xBA, 0xBE}}; + const auto* const data_ptr = bytes.data(); + + nlohmann::byte_container_with_subtype> container(std::move(bytes)); + + CHECK(container.size() == 4); + CHECK(container.data() == data_ptr); + CHECK(!container.has_subtype()); + CHECK(bytes.empty()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved) + } + + // same check for the rvalue-reference constructor that also takes a subtype + { + std::vector bytes = {{0xCA, 0xFE, 0xBA, 0xBE}}; + const auto* const data_ptr = bytes.data(); + + nlohmann::byte_container_with_subtype> container(std::move(bytes), 42); + + CHECK(container.size() == 4); + CHECK(container.data() == data_ptr); + CHECK(container.has_subtype()); + CHECK(container.subtype() == 42); + CHECK(bytes.empty()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved) + } + } + SECTION("comparisons") { std::vector const bytes = {{0xCA, 0xFE, 0xBA, 0xBE}}; diff --git a/tests/src/unit-conversions.cpp b/tests/src/unit-conversions.cpp index 4975854c0..90d972f71 100644 --- a/tests/src/unit-conversions.cpp +++ b/tests/src/unit-conversions.cpp @@ -1792,6 +1792,40 @@ TEST_CASE("std::filesystem::path") } #endif +// the ADL to_json overload for std::u8string only exists under the same guard +// as std::filesystem::path support (it is otherwise only reached indirectly, +// via std::filesystem::path::u8string()) -- mirror both #if conditions from +// include/nlohmann/detail/conversions/to_json.hpp exactly +#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM +#if defined(__cpp_lib_char8_t) +TEST_CASE("std::u8string") +{ + SECTION("ascii") + { + const std::u8string s = u8"Path"; + json const j = s; + + CHECK(j.template get() == "Path"); + } + + SECTION("utf-8") + { + // use \u universal-character-names (rather than raw \x byte escapes + // or literal non-ASCII source bytes) to compose the multi-byte UTF-8 + // encoding -- MSVC treats \x escapes used that way inside a u8 + // literal as a nonstandard extension (warning C5321), which some of + // our CI configs promote to an error; \u is portable and produces + // the exact same encoded bytes without depending on the source + // file's encoding + const std::u8string s = u8"P\u011B\u0161ina"; + json const j = s; + + CHECK(j.template get() == "P\xc4\x9b\xc5\xa1ina"); + } +} +#endif +#endif + TEST_CASE("std::optional") { SECTION("null") diff --git a/tests/src/unit-json_patch.cpp b/tests/src/unit-json_patch.cpp index 257e455aa..7731c7d92 100644 --- a/tests/src/unit-json_patch.cpp +++ b/tests/src/unit-json_patch.cpp @@ -672,6 +672,102 @@ TEST_CASE("JSON patch") } } + SECTION("patch_inplace") + { + SECTION("happy path: patch_inplace mirrors patch() on success") + { + // mirrors "A.5. Replacing a Value" above, but applies the patch with + // patch_inplace() to a mutable copy instead of using patch()'s + // returned copy + json doc = R"( + { + "baz": "qux", + "foo": "bar" + } + )"_json; + + json const patch = R"( + [ + { "op": "replace", "path": "/baz", "value": "boo" } + ] + )"_json; + + json const expected = R"( + { + "baz": "boo", + "foo": "bar" + } + )"_json; + + doc.patch_inplace(patch); + CHECK(doc == expected); + } + + // this test relies on the "test" operation actually throwing so the + // partial-application state can be observed right after the throw + // point; under JSON_NOEXCEPTION, JSON_THROW() calls std::abort() + // instead (there is no C++ exception to throw), and doctest's + // CHECK_THROWS_AS() is compiled out to a no-op that never even + // invokes the given expression (see doctest's "--no-throw" test + // filter, which ci_test_noexceptions passes) -- so patch()/ + // patch_inplace() would never be called at all and the follow-up + // state assertions below would fail against the untouched original +#if !defined(JSON_NOEXCEPTION) + SECTION("distinguishing contract vs patch(): partial application on failure") + { + // Unlike patch(), which is all-or-nothing because it applies the + // patch to an internal copy that is simply discarded when an + // exception is thrown (leaving the original untouched no matter + // what), patch_inplace() mutates the document it is called on + // directly and immediately, operation by operation. So if a JSON + // Patch fails partway through, whatever operations already + // succeeded remain applied -- the document is left in a partially + // patched state. This is empirically verified current behavior, + // not just documented intent, and is pinned here as such. + json const original = R"( + { + "baz": "qux", + "foo": "bar" + } + )"_json; + + // the first operation ("replace") succeeds; the second ("test") + // fails because the value at "/baz" no longer (and never did) + // equal "not boo" + json const patch = R"( + [ + { "op": "replace", "path": "/baz", "value": "boo" }, + { "op": "test", "path": "/baz", "value": "not boo" } + ] + )"_json; + + // patch() never modifies the object it is called on -- it always + // operates on (and returns) a separate copy, so the original is + // left completely untouched, regardless of success or failure. + // copy_for_patch is intentionally a real copy, not a reference + // to `original`: the whole point of this check is to catch a + // hypothetical future regression where patch() *does* mutate its + // receiver. Using a reference here would make the assertion + // below compare `original` to itself -- trivially true even if + // such a bug existed -- which is exactly what a static analyzer + // can't see when it suggests "this copy is never modified, use + // a reference instead". + json copy_for_patch = original; // NOLINT(performance-unnecessary-copy-initialization) + CHECK_THROWS_AS(copy_for_patch.patch(patch), json::other_error&); + CHECK(copy_for_patch == original); + + // patch_inplace(), in contrast, already applied the successful + // "replace" operation to the document before the "test" operation + // threw -- that change is not rolled back + json doc = original; + CHECK_THROWS_AS(doc.patch_inplace(patch), json::other_error&); + CHECK(doc != original); + CHECK(doc.at("baz") == "boo"); + CHECK(doc.at("foo") == "bar"); + } +#endif // !defined(JSON_NOEXCEPTION) + } + SECTION("errors") { SECTION("unknown operation") diff --git a/tests/src/unit-no-mem-leak-on-adl-serialize.cpp b/tests/src/unit-no-mem-leak-on-adl-serialize.cpp index 469fc2c75..cfbdff008 100644 --- a/tests/src/unit-no-mem-leak-on-adl-serialize.cpp +++ b/tests/src/unit-no-mem-leak-on-adl-serialize.cpp @@ -70,7 +70,7 @@ TEST_CASE("check_for_mem_leak_on_adl_to_json-2") } } -TEST_CASE("check_for_mem_leak_on_adl_to_json-2") +TEST_CASE("check_for_mem_leak_on_adl_to_json-3") { try { diff --git a/tests/src/unit-no_io_and_user_exceptions.cpp b/tests/src/unit-no_io_and_user_exceptions.cpp new file mode 100644 index 000000000..667d114e7 --- /dev/null +++ b/tests/src/unit-no_io_and_user_exceptions.cpp @@ -0,0 +1,91 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// This translation unit is a dedicated, small compile-and-run check for two +// configuration macros that (per #5423) were never exercised anywhere in the +// test matrix: +// - JSON_NO_IO, which removes the library's / support +// (operator<<, operator>>, and the stream-based overloads of dump()/parse()) +// - the JSON_THROW_USER / JSON_TRY_USER / JSON_CATCH_USER trio, which lets a +// user replace the library's internal exception handling +// +// Both macros are about excluding/replacing a facility the library would +// otherwise pull in on its own, and defining one has no bearing on the other, +// so -- to keep the test matrix small -- they are exercised together in a +// single dedicated file instead of two. +// +// JSON_NO_IO requires this file itself to never rely on /; +// only string-based parsing/dumping is used below. +#define JSON_NO_IO 1 + +// The user-supplied exception macros below are a *conforming* replacement: +// they simply forward to the real throw/try/catch keywords (via a counter so +// the test can assert each macro was actually invoked, not just defined), so +// every exception-related behavior the library relies on internally -- +// including rethrowing std::out_of_range as json::out_of_range in at() -- +// keeps working exactly as it would with the library's own default macros. +static int json_throw_user_call_count = 0; // NOLINT(cppcoreguidelines-avoid-non-const-global-variables) + +#define JSON_THROW_USER(exception) do { ++json_throw_user_call_count; throw (exception); } while (false) // NOLINT(cppcoreguidelines-macro-usage) +#define JSON_TRY_USER try // NOLINT(cppcoreguidelines-macro-usage) +#define JSON_CATCH_USER(exception) catch (exception) // NOLINT(cppcoreguidelines-macro-usage) + +#include "doctest_compatibility.h" + +#include +using json = nlohmann::json; + +TEST_CASE("JSON_NO_IO") +{ + // everything that does not touch / must keep working: + // parsing from and dumping to std::string + const json j = json::parse(R"({"a":[1,2,3],"b":true})"); + CHECK(j.dump() == R"({"a":[1,2,3],"b":true})"); + CHECK(j.at("a").size() == 3); + CHECK(j.at("b").get() == true); +} + +// this test relies on CHECK_THROWS_AS() actually invoking the guarded +// expression so json_throw_user_call_count gets bumped and can be observed +// afterwards; doctest's "--no-throw" test filter (which ci_test_noexceptions +// passes, together with a global -DJSON_NOEXCEPTION added to CMAKE_CXX_FLAGS +// for every translation unit in that build, this file included) compiles +// CHECK_THROWS_AS() out to a no-op that never even invokes the given +// expression -- so json::parse()/at() below would never be called at all and +// the call-count assertions would fail even though our JSON_THROW_USER +// override (which always really throws, regardless of JSON_NOEXCEPTION) would +// have worked fine on its own +#if !defined(JSON_NOEXCEPTION) +TEST_CASE("JSON_THROW_USER, JSON_TRY_USER, JSON_CATCH_USER") +{ + json_throw_user_call_count = 0; + + // json::parse() is [[nodiscard]] (JSON_HEDLEY_WARN_UNUSED_RESULT); under + // GCC in C++11 mode that expands to __attribute__((warn_unused_result)), + // which -- unlike a [[nodiscard]] attribute proper -- GCC does not + // consider satisfied by doctest's CHECK_THROWS_AS() wrapping the + // expression in a (void) cast, so the discarded return value would still + // be flagged under -Werror=unused-result; assign it to discard it instead, + // matching the established `json _ = json::parse(...)` pattern used + // elsewhere in the test suite (see unit-class_parser.cpp) + json _; // NOLINT(readability-identifier-naming) + + // a parse error goes through JSON_THROW directly, i.e., through our + // JSON_THROW_USER override + CHECK_THROWS_AS(_ = json::parse("this is not JSON"), json::parse_error&); + CHECK(json_throw_user_call_count > 0); + + // at() on an out-of-range array index internally catches std::out_of_range + // (JSON_TRY_USER/JSON_CATCH_USER) and rethrows it as json::out_of_range + // (JSON_THROW_USER again), so this exercises all three macros together + const int count_before = json_throw_user_call_count; + const json arr = json::array({1, 2, 3}); + CHECK_THROWS_AS(arr.at(10), json::out_of_range&); + CHECK(json_throw_user_call_count > count_before); +} +#endif diff --git a/tests/src/unit-regression3.cpp b/tests/src/unit-regression3.cpp index 11c6a7da8..a5be9ec4b 100644 --- a/tests/src/unit-regression3.cpp +++ b/tests/src/unit-regression3.cpp @@ -18,6 +18,14 @@ // for some reason including this after the json header leads to linker errors with VS 2017... #include +// skip tests if JSON_DisableEnumSerialization=ON (#4384): std::byte is a +// scoped enum, so get() (needed below to get>() +// from a plain JSON array, not just from an already-binary value) relies on +// enum serialization being enabled +#if defined(JSON_DISABLE_ENUM_SERIALIZATION) && (JSON_DISABLE_ENUM_SERIALIZATION == 1) + #define SKIP_TESTS_FOR_ENUM_SERIALIZATION +#endif + #define JSON_TESTS_PRIVATE #include using json = nlohmann::json; @@ -466,6 +474,7 @@ TEST_CASE("regression tests 3") CHECK((decoded == json_4804::array())); } +#ifndef SKIP_TESTS_FOR_ENUM_SERIALIZATION SECTION("discussion #4209 - custom BinaryType direct assignment and round-tripping") { // Test that assigning a custom BinaryType directly creates a binary value, not an array @@ -499,6 +508,7 @@ TEST_CASE("regression tests 3") CHECK(extracted[1] == std::byte{2}); CHECK(extracted[2] == std::byte{3}); } +#endif SECTION("issue #5046 - implicit conversion of return json to std::optional no longer implicit") { diff --git a/tests/src/unit-std-format.cpp b/tests/src/unit-std-format.cpp index d18ccbd32..58cbbf5cc 100644 --- a/tests/src/unit-std-format.cpp +++ b/tests/src/unit-std-format.cpp @@ -53,6 +53,23 @@ TEST_CASE("std::formatter") CHECK(std::format("{:2}", j) == j.dump(2)); CHECK(std::format("{:#2}", j) == j.dump(2)); CHECK(std::format("{:8}", j) == j.dump(8)); + // multi-digit widths must accumulate every digit, not just the first + CHECK(std::format("{:12}", j) == j.dump(12)); + CHECK(std::format("{:#12}", j) == j.dump(12)); + CHECK(std::format("{:10}", j) == j.dump(10)); + } + + SECTION("bare alignment with no fill character defaults to a space indent character") + { + const json j = {{"foo", 1}, {"bar", {1, 2, 3}}}; + // without a preceding fill character, the alignment character itself must not + // be mistaken for the indent character -- the default space is kept + CHECK(std::format("{:<}", j) == j.dump()); + CHECK(std::format("{:>}", j) == j.dump()); + CHECK(std::format("{:^}", j) == j.dump()); + CHECK(std::format("{:<3}", j) == j.dump(3, ' ')); + CHECK(std::format("{:>3}", j) == j.dump(3, ' ')); + CHECK(std::format("{:^3}", j) == j.dump(3, ' ')); } SECTION("fill-and-align sets the indent character, like dump(indent, indent_char)") From 29ba5973b697c014b64a2ec051f1d75e70f0f6ab Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:16:40 +0200 Subject: [PATCH 5/9] Add missing diagnostic-positions test coverage (lifetime, input adapters, SAX) (#5482) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Add missing diagnostic-positions test coverage (lifetime, input adapters, SAX) Building on the merged unit-class_parser.cpp from #5417, add characterization tests (regression protection for existing behavior, not a behavior change) for JSON_DIAGNOSTIC_POSITIONS: - value lifetime: copy ctor copies positions recursively, move ctor resets the moved-from value to npos, and mutating a parsed document (operator[], push_back, erase) leaves the parent's stale span and siblings' positions untouched while new values get npos. - input adapters: wide-string input positions count transcoded UTF-8 bytes (not wide characters), BOM-prefixed input's start_pos() reflects the skipped 3-byte BOM, istringstream/ifstream/iterator-pair inputs report consistent (non-npos) positions, and binary formats (CBOR, MessagePack, UBJSON, BSON) always report npos. - a user-constructed json_sax_dom_parser with no lexer (as used when driving json::sax_parse() directly) reports npos for every value, since it has no m_lexer_ref to source positions from. While characterizing swap(), found that basic_json::swap() (and the friend swap() that forwards to it) does not swap start_position/end_position, unlike copy-assignment's operator=(basic_json), which does as part of its copy-and-swap implementation. This looks like a real inconsistency/bug, but per the scope of this test-only change it is only pinned (not fixed) here; see the comment at the "swap() does NOT exchange positions" section. Fixes #5420 Signed-off-by: Niels Lohmann * Fix MSVC source-encoding portability in the wide-string position test Use é escapes instead of a literal UTF-8-encoded 'é' inside the L"" literal, so the wide string's content does not depend on the compiler's assumed source character set (MSVC without /utf-8 decodes raw non-ASCII source bytes using the system code page rather than as UTF-8, which was producing a wstring of unexpected length/content and failing the ws.size()/end_pos() assertions on Windows CI). Also reworded a comment that unintentionally embedded the literal substring "TODO check", which clang-tidy's google-readability-todo check flags regardless of quoting context. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- tests/src/unit-class_parser.cpp | 320 ++++++++++++++++++++++++++++++++ 1 file changed, 320 insertions(+) diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index af76e93cc..5b4af321c 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -17,6 +17,8 @@ using nlohmann::json; #include #include +#include +#include #include #include #include @@ -2445,3 +2447,321 @@ TEST_CASE("last-read diagnostics are identical across input adapters") } } #endif // !defined(JSON_NOEXCEPTION) + +// this test characterizes the current (documented-by-example, not otherwise +// specified) behavior of JSON_DIAGNOSTIC_POSITIONS positions with respect to +// value lifetime (copy/move/swap/mutation), the various input adapters, and +// user-driven SAX usage. It is regression protection, not a behavior +// specification: if any of these checks fail after a change to json.hpp, +// that change deliberately altered observable behavior and the test (and +// this comment) should be updated accordingly, rather than "fixed" blindly. +#if JSON_DIAGNOSTIC_POSITIONS +TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") +{ + SECTION("value lifetime") + { + SECTION("copy constructor copies positions, recursively") + { + // basic_json(const basic_json&) (json.hpp, around line 1192) copies + // start_position/end_position for the value itself; nested values + // are copied via their own copy constructor (through the copied + // object/array container), so positions are preserved throughout + // the whole tree. + const std::string s = R"({"a":1,"b":[1,2,3]})"; + const json a = json::parse(s); + const json b = a; // NOLINT(performance-unnecessary-copy-initialization) + + CHECK(b.start_pos() == a.start_pos()); + CHECK(b.end_pos() == a.end_pos()); + CHECK(b["b"].start_pos() == a["b"].start_pos()); + CHECK(b["b"].end_pos() == a["b"].end_pos()); + CHECK(b["b"][0].start_pos() == a["b"][0].start_pos()); + CHECK(b["b"][0].end_pos() == a["b"][0].end_pos()); + + // sanity: the positions are meaningful (not all npos) + CHECK(b.start_pos() == 0); + CHECK(b.end_pos() == s.size()); + } + + SECTION("move constructor resets the moved-from value to npos") + { + // basic_json(basic_json&&) (json.hpp, around line 1265) copies + // other's start_position/end_position into *this and then resets + // other's to npos (see the cppcheck-suppress[accessForwarded] + // annotation there, which flags this reset as worth a second + // look). Only the top-level moved-from value is affected; its + // (moved-away) children are gone along with it. + const std::string s = R"({"a":1,"b":[1,2,3]})"; + json a = json::parse(s); + const auto a_start = a.start_pos(); + const auto a_end = a.end_pos(); + const auto nested_start = a["b"].start_pos(); + const auto nested_end = a["b"].end_pos(); + + const json b(std::move(a)); + + // the destination retains the original positions, recursively + CHECK(b.start_pos() == a_start); + CHECK(b.end_pos() == a_end); + CHECK(b["b"].start_pos() == nested_start); + CHECK(b["b"].end_pos() == nested_end); + + // the moved-from value is reset to a null and reports npos + CHECK(a.is_null()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) + CHECK(a.start_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) + CHECK(a.end_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) + } + + SECTION("swap() does NOT exchange positions (likely a real bug, see below)") + { + // NOTE (characterizing, not fixing, for #5420): basic_json::swap() + // (json.hpp, around line 3540, and the friend swap() that forwards + // to it) swaps m_data.m_type and m_data.m_value but -- unlike + // copy-assignment's operator=(basic_json) (json.hpp, around line + // 1291), which swaps start_position/end_position as part of its + // copy-and-swap implementation -- it never touches + // start_position/end_position. So after swap(a, b), the *values* + // of a and b are exchanged, but their *positions* are not: each + // ends up with its own original position describing the other's + // new content. This looks like an oversight/inconsistency rather + // than intended behavior, and is flagged to the maintainer; this + // test only pins the current (surprising) behavior so a fix (or a + // deliberate decision to keep it) shows up here as an intentional + // change rather than a silent regression. + json a = json::parse(R"({"a":1})"); + json b = json::parse(R"([1,2,3,4,5])"); + const auto a_start = a.start_pos(); + const auto a_end = a.end_pos(); + const auto b_start = b.start_pos(); + const auto b_end = b.end_pos(); + // both start at 0 (root values start right away), but their + // lengths (and thus end positions) differ, which is enough to + // tell after the swap whether positions actually moved with + // the values + CHECK(a_end != b_end); + + using std::swap; + swap(a, b); + + // values were exchanged as expected ... + CHECK(a == json::parse(R"([1,2,3,4,5])")); + CHECK(b == json::parse(R"({"a":1})")); + + // ... but positions were NOT: each variable kept its own + // original position, now describing the other's content + CHECK(a.start_pos() == a_start); + CHECK(a.end_pos() == a_end); + CHECK(b.start_pos() == b_start); + CHECK(b.end_pos() == b_end); + } + + SECTION("mutating a parsed document leaves positions of unrelated values untouched") + { + // Positions are recorded once, during parsing, and are not + // recomputed on mutation. As a consequence, after a mutation the + // parent's own recorded span may no longer describe its current + // (serialized) content -- it still describes what was originally + // parsed. This is characterized here as current behavior, not + // asserted to be desirable or specified. + SECTION("operator[] adding a new object key") + { + const std::string s = R"({"a":1})"; + json j = json::parse(s); + const auto root_start = j.start_pos(); + const auto root_end = j.end_pos(); + const auto a_start = j["a"].start_pos(); + const auto a_end = j["a"].end_pos(); + + j["c"] = 42; + + // the newly-added value was never parsed, so it has no position + CHECK(j["c"].start_pos() == std::string::npos); + CHECK(j["c"].end_pos() == std::string::npos); + + // the existing sibling's position is unaffected + CHECK(j["a"].start_pos() == a_start); + CHECK(j["a"].end_pos() == a_end); + + // the parent's own recorded span is left as-is (now stale: + // it still reflects the original, shorter `{"a":1}` string) + CHECK(j.start_pos() == root_start); + CHECK(j.end_pos() == root_end); + } + + SECTION("push_back on a parsed array") + { + const std::string s = R"([1,2,3])"; + json j = json::parse(s); + const auto root_start = j.start_pos(); + const auto root_end = j.end_pos(); + const auto first_start = j[0].start_pos(); + + j.push_back(4); + + CHECK(j.back().start_pos() == std::string::npos); + CHECK(j.back().end_pos() == std::string::npos); + CHECK(j[0].start_pos() == first_start); + CHECK(j.start_pos() == root_start); + CHECK(j.end_pos() == root_end); + } + + SECTION("erase on a parsed array shifts elements but keeps their own positions") + { + const std::string s = R"([1,2,3])"; + json j = json::parse(s); + const auto second_start = j[1].start_pos(); + const auto third_start = j[2].start_pos(); + const auto root_start = j.start_pos(); + const auto root_end = j.end_pos(); + + j.erase(0); + + // remaining elements moved down an index, but each one still + // reports the position it had *before* the erase (i.e. its + // position in the original source string, not a + // recalculated one) + CHECK(j[0].start_pos() == second_start); + CHECK(j[1].start_pos() == third_start); + + // the parent's own recorded span is again left as-is + CHECK(j.start_pos() == root_start); + CHECK(j.end_pos() == root_end); + } + } + } + + SECTION("input adapters") + { + SECTION("wide string input: positions count transcoded UTF-8 bytes, not wide characters") + { + // 'é' (U+00E9) is a single code unit in a wchar_t/UTF-16 string, but + // transcodes to 2 bytes in UTF-8; the lexer only ever sees the + // transcoded UTF-8 byte stream, so reported positions are byte + // offsets into that UTF-8 stream, not indices into the original + // std::wstring. + // é (rather than a literal 'é' byte sequence in this source + // file) so the wide-string literal's meaning does not depend on + // the compiler's assumed source character set (MSVC, without + // /utf-8, would otherwise decode the raw UTF-8 bytes using the + // system code page instead of as UTF-8) + const std::wstring ws = L"{\"a\":\"\u00e9\u00e9\"}"; + CHECK(ws.size() == 10); // 10 wide characters + + const json j = json::parse(ws); + CHECK(j.start_pos() == 0); + // the transcoded UTF-8 form is 2 bytes longer than the wide string, + // because each of the two 'é' characters becomes 2 UTF-8 bytes + CHECK(j.end_pos() == 12); + CHECK(j.end_pos() != ws.size()); + + const json& a = j["a"]; + CHECK(a.start_pos() == 5); + CHECK(a.end_pos() == 11); + } + + SECTION("BOM-prefixed input: start_pos() reflects the skipped 3-byte BOM") + { + const std::string s = "\xEF\xBB\xBF{\"a\":1}"; + const json j = json::parse(s); + + // the lexer silently skips the BOM before parsing the value, so + // the root value's recorded span starts right after it + CHECK(j.start_pos() == 3); + CHECK(j.end_pos() == s.size()); + } + + SECTION("std::istringstream: positions are consistent, not npos") + { + const std::string s = R"({"a":1,"b":2})"; + std::istringstream ss(s); + const json j = json::parse(ss); + + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == s.size()); + CHECK(j["a"].start_pos() == 5); + } + + SECTION("std::ifstream: positions are consistent, not npos") + { + const std::string s = R"({"a":1,"b":2})"; + { + std::ofstream file("unit-class_parser_diagnostic_positions.tmp"); + file << s; + } + + { + std::ifstream f("unit-class_parser_diagnostic_positions.tmp"); + const json j = json::parse(f); + + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == s.size()); + CHECK(j["a"].start_pos() == 5); + } + + static_cast(std::remove("unit-class_parser_diagnostic_positions.tmp")); + } + + SECTION("iterator-pair input: positions are consistent, not npos") + { + const std::string s = R"({"a":1,"b":2})"; + const json j = json::parse(s.begin(), s.end()); + + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == s.size()); + CHECK(j["a"].start_pos() == 5); + } + + SECTION("binary formats have no text positions") + { + // binary formats (CBOR, MessagePack, UBJSON, BSON, BJData) are + // parsed via detail::binary_reader, which never sets + // start_position/end_position on the values it produces (they + // have no notion of a text offset), so every value's position + // stays at its default of npos. + const json src = json::parse(R"({"a":1,"b":[1,2]})"); + + const json from_cbor = json::from_cbor(json::to_cbor(src)); + CHECK(from_cbor.start_pos() == std::string::npos); + CHECK(from_cbor.end_pos() == std::string::npos); + CHECK(from_cbor["a"].start_pos() == std::string::npos); + CHECK(from_cbor["b"][0].start_pos() == std::string::npos); + + const json from_msgpack = json::from_msgpack(json::to_msgpack(src)); + CHECK(from_msgpack.start_pos() == std::string::npos); + CHECK(from_msgpack.end_pos() == std::string::npos); + + const json from_ubjson = json::from_ubjson(json::to_ubjson(src)); + CHECK(from_ubjson.start_pos() == std::string::npos); + CHECK(from_ubjson.end_pos() == std::string::npos); + + const json from_bson_val = json::from_bson(json::to_bson(src)); + CHECK(from_bson_val.start_pos() == std::string::npos); + CHECK(from_bson_val.end_pos() == std::string::npos); + } + } + + SECTION("user-driven SAX consumers with no lexer report npos") + { + // json::parse() internally wires up its json_sax_dom_parser with a + // pointer to its own lexer (see parser.hpp), which is how positions + // get set at all. A user who constructs a json_sax_dom_parser + // directly (e.g. to drive it via json::sax_parse()) and does not + // supply a lexer pointer gets a consumer with m_lexer_ref == nullptr; + // every "if (m_lexer_ref)" guard in json_sax.hpp is then skipped, so + // every value it produces keeps its default, unset position (npos). + // This was previously true but silently unasserted (operator== + // ignores positions), see #5420. + json result; + nlohmann::detail::json_sax_dom_parser sdp(result); + const std::string s = R"({"a":1,"b":[1,2,3]})"; + CHECK(json::sax_parse(s, &sdp)); + + CHECK(result.start_pos() == std::string::npos); + CHECK(result.end_pos() == std::string::npos); + CHECK(result["a"].start_pos() == std::string::npos); + CHECK(result["a"].end_pos() == std::string::npos); + CHECK(result["b"][0].start_pos() == std::string::npos); + CHECK(result["b"][0].end_pos() == std::string::npos); + } +} +#endif From e8e1ba0db9ed3d760da1e9d55ce3dc4f44cc597e Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:16:43 +0200 Subject: [PATCH 6/9] Fix dead ill-formed-fourth-byte UTF-8 test sections (byte3/byte4 typo) (#5499) * Fix dead ill-formed-fourth-byte UTF-8 test sections (byte3/byte4 typo) The "ill-formed: wrong fourth byte" SECTIONs in unit-unicode3.cpp, unit-unicode4.cpp, and unit-unicode5.cpp guarded their loop with a check on byte3 instead of byte4. Since the enclosing loop already restricts byte3 to its valid range, the guard was always true and the section's "continue" fired unconditionally, so check_utf8string()/check_utf8dump() were never actually invoked for a malformed fourth byte. Fixing the guard naively (byte3 -> byte4) would also have swept the full byte2 x byte3 combinatorics for every byte4 value, adding millions of redundant iterations: the lexer validates continuation bytes strictly in sequence with early exit (see next_byte_in_range() in lexer.hpp), so once byte2/byte3 are within their valid range, the byte4 outcome does not depend on which valid byte2/byte3 values were chosen. Instead, byte2 and byte3 are now held to a small hedge of representative valid prefixes (range corners plus a midpoint) while byte4 is still swept exhaustively over its full 0x00-0xFF range, since that is the actual property under test. Also fixed the garbled "skip fourth second byte" comment in unit-unicode3.cpp. Verified offline: before the fix, the "wrong fourth byte" subcase executes 0 assertions in all three files (proving it was dead code); after the fix, it executes 11520 (unicode3), 34560 (unicode4), and 11520 (unicode5) assertions, and a deliberately reintroduced bug in the lexer's byte4 range check causes it to fail (proving it is now meaningful). Total per-file assertion counts grow by the same small amounts, not by millions, and all other sections in these files still pass unchanged. Fixes #5416 Signed-off-by: Niels Lohmann * Keep full byte2 x byte3 combinatorics in the wrong-fourth-byte sections The maintainer wants exhaustive coverage of every byte combination here rather than the representative-prefix reduction, matching the style of the sibling "wrong second/third byte" sections in the same files. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- tests/src/unit-unicode3.cpp | 4 ++-- tests/src/unit-unicode4.cpp | 2 +- tests/src/unit-unicode5.cpp | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/src/unit-unicode3.cpp b/tests/src/unit-unicode3.cpp index d5627d8cc..12c12eea4 100644 --- a/tests/src/unit-unicode3.cpp +++ b/tests/src/unit-unicode3.cpp @@ -306,8 +306,8 @@ TEST_CASE("Unicode (3/5)" * doctest::skip()) { for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { - // skip fourth second byte - if (0x80 <= byte3 && byte3 <= 0xBF) + // skip correct fourth byte + if (0x80 <= byte4 && byte4 <= 0xBF) { continue; } diff --git a/tests/src/unit-unicode4.cpp b/tests/src/unit-unicode4.cpp index f15a1499f..43cf7095e 100644 --- a/tests/src/unit-unicode4.cpp +++ b/tests/src/unit-unicode4.cpp @@ -307,7 +307,7 @@ TEST_CASE("Unicode (4/5)" * doctest::skip()) for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { // skip correct fourth byte - if (0x80 <= byte3 && byte3 <= 0xBF) + if (0x80 <= byte4 && byte4 <= 0xBF) { continue; } diff --git a/tests/src/unit-unicode5.cpp b/tests/src/unit-unicode5.cpp index e35801823..bc0312820 100644 --- a/tests/src/unit-unicode5.cpp +++ b/tests/src/unit-unicode5.cpp @@ -307,7 +307,7 @@ TEST_CASE("Unicode (5/5)" * doctest::skip()) for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { // skip correct fourth byte - if (0x80 <= byte3 && byte3 <= 0xBF) + if (0x80 <= byte4 && byte4 <= 0xBF) { continue; } From 502e9d66f64feb378b7a89b17fc4a40d4c63a45b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:20:21 +0200 Subject: [PATCH 7/9] Fix to_bjdata() emitting the Draft-3-only 'B' marker in default Draft-2 mode (#5479) * Fix to_bjdata() emitting the Draft-3-only 'B' marker in default Draft-2 mode _ArrayType_ = "byte" mapped unconditionally to the BJData type marker 'B', regardless of the requested bjdata_version. 'B' is defined only by BJData Draft 3; with the default version (draft2), this produced a stream that is invalid for Draft 2 and, unlike every other _ArrayType_, round-tripped back as a binary value instead of the original annotated object. Only accept "byte" / emit 'B' when bjdata_version selects Draft 3. Under Draft 2, fall back to the same plain-object encoding used elsewhere in this function for other invalid-annotation cases, so the value round-trips correctly. Fixes #5404. Signed-off-by: Niels Lohmann * Future-proof the Draft-3-only 'B' marker gate @gregmarr pointed out that dtype == 'B' && bjdata_version != draft3 only future-proofs by accident, since bjdata_version_t currently has exactly two values. Compare with < instead, so a later draft that keeps the 'B' marker valid does not need this gate revisited. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .../nlohmann/detail/output/binary_writer.hpp | 10 ++++ single_include/nlohmann/json.hpp | 10 ++++ tests/src/unit-bjdata.cpp | 49 ++++++++++++++++++- 3 files changed, 67 insertions(+), 2 deletions(-) diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index e9ccd23b5..aaa638801 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -1707,6 +1707,16 @@ class binary_writer } CharType dtype = it->second; + // the 'B' (byte) marker is only defined from BJData Draft 3 onward; + // emitting it under an earlier draft would produce a stream that an + // earlier-draft reader rejects, so such an object falls back to a + // plain object encoding instead (see the "Binary values" section of + // the BJData documentation) + if (dtype == 'B' && bjdata_version < bjdata_version_t::draft3) + { + return true; + } + key = "_ArraySize_"; // the dimensions are written verbatim as the header length below, so a // value that is not an array cannot produce a valid one: null emits 'Z' diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 35443e141..4253fcdcf 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -20384,6 +20384,16 @@ class binary_writer } CharType dtype = it->second; + // the 'B' (byte) marker is only defined from BJData Draft 3 onward; + // emitting it under an earlier draft would produce a stream that an + // earlier-draft reader rejects, so such an object falls back to a + // plain object encoding instead (see the "Binary values" section of + // the BJData documentation) + if (dtype == 'B' && bjdata_version < bjdata_version_t::draft3) + { + return true; + } + key = "_ArraySize_"; // the dimensions are written verbatim as the header length below, so a // value that is not an array cannot produce a valid one: null emits 'Z' diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 9338e663e..a53bd17ce 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -2586,7 +2586,12 @@ TEST_CASE("BJData") CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d); CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D); CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C); - CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true) == v_B); + // v_B uses the Draft-3-only 'B' marker, so it round-trips only when + // Draft 3 is explicitly selected (see GitHub issue #5404); the + // default Draft 2 falls back to a plain object instead, covered by + // the "ndarray with _ArrayType_ "byte" is gated by the BJData draft + // version" section below + CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true, json::bjdata_version_t::draft3) == v_B); } SECTION("ndarray with data not matching _ArrayType_ is written as an object") @@ -2629,8 +2634,10 @@ TEST_CASE("BJData") // the C++ API stores an int literal as number_integer, so _ArrayType_ // names the wire type rather than the storage. Both storages have to // produce the same typed array for every type. + // "byte" is checked separately below since it additionally requires + // BJData Draft 3 to be selected explicitly (see GitHub issue #5404). for (const char* type : - {"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char", "byte" + {"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char" }) { CAPTURE(type); @@ -2641,6 +2648,14 @@ TEST_CASE("BJData") CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}))); } + { + const std::string text = R"({"_ArrayType_":"byte","_ArraySize_":[2,3],"_ArrayData_":[1,2,3,4,5,6]})"; + const auto from_text = json::to_bjdata(json::parse(text), true, true, json::bjdata_version_t::draft3); + CHECK(from_text.at(0) == '['); + CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}), + true, true, json::bjdata_version_t::draft3)); + } + // negative values under a signed type behave the same way const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})")); CHECK(from_neg.at(0) == '['); @@ -2823,6 +2838,36 @@ TEST_CASE("BJData") CHECK(out_single_ok.at(0) == '['); CHECK(json::from_bjdata(out_single_ok) == json({1.5f})); } + + SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version") + { + // the 'B' (byte) marker used by _ArrayType_ "byte" is only defined + // by BJData Draft 3; Draft 2 (the default) has no such marker, so + // emitting it unconditionally produced a stream that a Draft 2 + // reader could not parse as intended (see GitHub issue #5404). + // Two dimensions are used so that a successfully written ndarray + // round-trips back into the annotated object (a single dimension + // is, by the BJData ndarray convention, read back as a plain + // binary value rather than the annotated object, same as every + // other single-dimension ndarray of a non-"byte" type is read + // back as a plain array instead of the annotated object). + json const j_byte = json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + + // default (Draft 2): falls back to a plain object and round-trips + const auto out_draft2 = json::to_bjdata(j_byte); + CHECK(out_draft2.at(0) == '{'); + CHECK(json::from_bjdata(out_draft2) == j_byte); + + // explicit Draft 2: same as the default + const auto out_draft2_explicit = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft2); + CHECK(out_draft2_explicit.at(0) == '{'); + CHECK(json::from_bjdata(out_draft2_explicit) == j_byte); + + // Draft 3 explicitly selected: still uses the compact 'B' ndarray encoding + const auto out_draft3 = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft3); + CHECK(out_draft3 == std::vector({'[', '$', 'B', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6})); + CHECK(json::from_bjdata(out_draft3) == j_byte); + } } } From c41152e62092c16cf4fc4625becacc3eb9d9319f Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:20:22 +0200 Subject: [PATCH 8/9] Fall back to plain-object encoding when to_bjdata()'s _ArrayType_ annotation is not a string (#5494) * Fall back to plain-object encoding when _ArrayType_ is not a string write_bjdata_ndarray() looked up _ArrayType_ by calling get() directly, which throws type_error.302 when the annotation is not a string (e.g. a number, null, boolean, array, or object). Per the documented BJData ndarray contract, an object only qualifies for the compact ndarray encoding if _ArrayType_ names a known type; anything else must fall back to plain-object encoding, the same way an unknown type-name string already does. Add an is_string() check before the get() call so a non-string _ArrayType_ takes the existing "unrecognized type name" fallback path instead of throwing. Fixes #5398. Signed-off-by: Niels Lohmann * Relax the BJData fuzzer's round-trip check from byte-exact to value-exact Fixing #5398 lets to_bjdata() proceed past the object it used to reject, which exposed a pre-existing, unrelated round-trip quirk to the fuzzer: a binary_t value serialized through the non-optimized ("$U#"-less) array encoding is parsed back as a plain array of numbers, since from_bjdata() has no way to tell "array of uint8 numbers" apart from "array of bytes" without that optimized header. Re-serializing that plain array then goes through the generic smallest-type writer, which - unrelated to this PR, and long predating it - prefers the 'i' (int8) marker over 'U' (uint8) for values that fit both, so the re-encoded bytes can differ from the original even though both decode to the same value. This is not introduced by the #5398 fix; the same divergence reproduces from a bare json::binary_t value with no _ArrayType_ annotation involved at all, on the commit immediately preceding it. A general fix would mean changing the shared UBJSON/BJData smallest-type selection that hundreds of existing tests pin to 'i' for small positive integers, which is out of scope and too risky for this PR. Update fuzzer-parse_bjdata.cpp's round-trip assertions to check that re-serializing is value-stable (from_bjdata(to_bjdata(j)) == j) rather than byte-exact, matching the guarantee BJData actually provides, and add a regression test in unit-bjdata.cpp using the exact OSS-Fuzz input that documents the behavior. Signed-off-by: Niels Lohmann * Compare dump()s instead of json values in the BJData fuzzer's round-trip check The value-stability assertion added to fix the earlier OSS-Fuzz crash (json::from_bjdata(to_bjdata(j2)) == j2) itself broke on a NaN payload: IEEE 754 NaN is never equal to itself, so operator== reports two structurally-identical trees containing a non-finite double as different -- not a round-trip bug, just NaN's ordinary non-reflexivity. dump() serializes any non-finite double the same deterministic way (as JSON null, since JSON cannot represent NaN or Infinity), so comparing dumps is stable under exactly the values that break operator==. Verified against both the original OSS-Fuzz crash input and the new one (0x68 0x68 0x7c, which decodes to a NaN), plus a local 2.5M-case random-input sweep with no failures. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .../nlohmann/detail/output/binary_writer.hpp | 10 +++ single_include/nlohmann/json.hpp | 10 +++ tests/src/fuzzer-parse_bjdata.cpp | 38 ++++++++- tests/src/unit-bjdata.cpp | 77 +++++++++++++++++++ 4 files changed, 131 insertions(+), 4 deletions(-) diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index aaa638801..4bd173257 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -1698,6 +1698,16 @@ class binary_writer }; string_t key = "_ArrayType_"; + // the type name is looked up as a string below; a non-string + // annotation (e.g. a number, null, or an array) cannot name a known + // dtype, so it is treated the same as an unrecognized type name and + // falls back to a plain object encoding instead of throwing + // type_error.302 out of get() + if (!value.at(key).is_string()) + { + return true; + } + // use get() instead of static_cast to avoid an // ambiguous conversion under explicit instantiation on C++17 (see #4825) auto it = bjdtype.find(value.at(key).template get()); diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4253fcdcf..213236b51 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -20375,6 +20375,16 @@ class binary_writer }; string_t key = "_ArrayType_"; + // the type name is looked up as a string below; a non-string + // annotation (e.g. a number, null, or an array) cannot name a known + // dtype, so it is treated the same as an unrecognized type name and + // falls back to a plain object encoding instead of throwing + // type_error.302 out of get() + if (!value.at(key).is_string()) + { + return true; + } + // use get() instead of static_cast to avoid an // ambiguous conversion under explicit instantiation on C++17 (see #4825) auto it = bjdtype.find(value.at(key).template get()); diff --git a/tests/src/fuzzer-parse_bjdata.cpp b/tests/src/fuzzer-parse_bjdata.cpp index 1d1d56a5c..a88479933 100644 --- a/tests/src/fuzzer-parse_bjdata.cpp +++ b/tests/src/fuzzer-parse_bjdata.cpp @@ -21,6 +21,27 @@ array data, it performs the following steps: - j4 = from_bjdata(vec3) - assert(j1 == j4) +Re-serializing j2/j3/j4 with the same use_size/use_type settings is checked +for value-stability rather than byte-exact stability: from_bjdata(to_bjdata(j2)) +must equal j2 (and likewise for j3, j4). Byte-exact stability does not hold in +general, because a BJData value can lose type fidelity across a round trip +(e.g. a binary_t value serialized without the optimized "$U#" array header is +parsed back as a plain array of numbers, see #5398 and the discussion on +PR #5494) - the numeric value is preserved, but the writer's smallest-type +selection for the now-plain numbers may legitimately pick a different, but +equally valid, single-byte type marker than the dedicated binary-data writer +would have. Both encodings are valid BJData and both decode to the same +value, so this is not treated as a round-trip failure here. + +"Value-stable" is checked by comparing dump()s rather than with operator== +directly: a BJData/UBJSON payload can decode to a non-finite double (NaN or ++-Infinity), and IEEE 754 NaN is never equal to itself, so operator== would +report two structurally-identical trees as different whenever a NaN is +involved -- not a round-trip bug, just NaN's ordinary (non-)reflexivity. +dump() serializes any non-finite double the same deterministic way (as JSON +`null`, since JSON itself cannot represent NaN/Infinity), so comparing +dumps is stable under exactly the same values that break operator==. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ @@ -31,6 +52,13 @@ drivers. using json = nlohmann::json; +// value-stable comparison for the round-trip checks below; see the note +// above on why this compares dump()s rather than the json values directly +static bool is_value_stable(const json& lhs, const json& rhs) +{ + return lhs.dump() == rhs.dump(); +} + // see http://llvm.org/docs/LibFuzzer.html extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { @@ -56,10 +84,12 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) json const j3 = json::from_bjdata(vec3); json const j4 = json::from_bjdata(vec4); - // serializations must match - assert(json::to_bjdata(j2, false, false) == vec2); - assert(json::to_bjdata(j3, true, false) == vec3); - assert(json::to_bjdata(j4, true, true) == vec4); + // re-serializing must be value-stable (see the notes above on + // why byte-exact stability is not guaranteed in general, and + // why this compares dump()s rather than the values directly) + assert(is_value_stable(json::from_bjdata(json::to_bjdata(j2, false, false)), j2)); + assert(is_value_stable(json::from_bjdata(json::to_bjdata(j3, true, false)), j3)); + assert(is_value_stable(json::from_bjdata(json::to_bjdata(j4, true, true)), j4)); } catch (const json::parse_error&) { diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index a53bd17ce..334259fb7 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -2746,6 +2746,83 @@ TEST_CASE("BJData") CHECK(json::from_bjdata(json::to_bjdata(j_size), true, true) == j_size); } + SECTION("ndarray whose _ArrayType_ is not a string stays as object") + { + // the type name is looked up as a string below the annotation + // check; a non-string _ArrayType_ cannot name a known dtype, + // so calling get() on it would throw type_error.302 + // instead of falling back like an unrecognized type name + // already does (see GitHub issue #5398) + json const j_number = json({{"_ArrayType_", 1}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_number = json::to_bjdata(j_number); + CHECK(out_number.at(0) == '{'); + CHECK(json::from_bjdata(out_number) == j_number); + + json const j_null = json({{"_ArrayType_", nullptr}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_null = json::to_bjdata(j_null); + CHECK(out_null.at(0) == '{'); + CHECK(json::from_bjdata(out_null) == j_null); + + json const j_bool = json({{"_ArrayType_", true}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_bool = json::to_bjdata(j_bool); + CHECK(out_bool.at(0) == '{'); + CHECK(json::from_bjdata(out_bool) == j_bool); + + json const j_array = json({{"_ArrayType_", {"uint8"}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_array = json::to_bjdata(j_array); + CHECK(out_array.at(0) == '{'); + CHECK(json::from_bjdata(out_array) == j_array); + + json const j_object = json({{"_ArrayType_", {{"a", 1}}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_object = json::to_bjdata(j_object); + CHECK(out_object.at(0) == '{'); + CHECK(json::from_bjdata(out_object) == j_object); + } + + SECTION("re-serializing a value containing a plain-array-of-bytes is value-stable but not byte-stable") + { + // OSS-Fuzz found this input (an array whose first element is a + // binary_t byte, followed by an object whose _ArrayType_ is + // not a string) while exercising the fix for #5398 above: once + // the fix stops to_bjdata() from throwing type_error.302 for + // the third element, serialization proceeds far enough to + // reach a pre-existing, unrelated round-trip quirk in how a + // single-byte binary_t value is re-encoded. + std::vector const input + { + 0x5b, 0x5b, 0x24, 0x42, 0x23, 0x5b, 0x69, 0x01, 0x5d, 0x5b, 0x5b, 0x5d, 0x7b, 0x55, 0x0b, + 0x5f, 0x41, 0x72, 0x72, 0x61, 0x79, 0x44, 0x61, 0x74, 0x61, 0x5f, 0x54, 0x55, 0x0b, 0x5f, + 0x41, 0x72, 0x72, 0x61, 0x79, 0x53, 0x69, 0x7a, 0x65, 0x5f, 0x5a, 0x55, 0x0b, 0x5f, 0x41, + 0x72, 0x72, 0x61, 0x79, 0x54, 0x79, 0x70, 0x65, 0x5f, 0x54, 0x7d, 0x5d + }; + json const j1 = json::from_bjdata(input); + + // to_bjdata() must not throw (this is what #5398 fixes) + std::vector vec2; + CHECK_NOTHROW(vec2 = json::to_bjdata(j1, false, false)); + + // parsing back a plain (non-optimized) array of bytes cannot + // recover that it used to be a binary_t: from_bjdata() has no + // way to distinguish "array of uint8 numbers" from "array of + // bytes" unless the compact "$U#" array header is used, so + // the binary_t collapses into a plain JSON array + json const j2 = json::from_bjdata(vec2); + CHECK(j1 != j2); + CHECK(j2 == json({{91}, json::array(), {{"_ArrayData_", true}, {"_ArraySize_", nullptr}, {"_ArrayType_", true}}})); + + // re-serializing j2 no longer goes through the dedicated + // binary_t writer (which always uses the 'U' marker for raw + // bytes); the now-plain number 91 goes through the generic + // smallest-type writer instead, which - like the rest of the + // UBJSON/BJData writer, and unchanged by this fix - prefers + // the 'i' (int8) marker over 'U' (uint8) for values that fit + // both. Both markers are valid BJData and both decode back to + // 91, so this is not byte-for-byte identical to vec2, but it + // is value-stable: parsing it again reproduces j2 exactly. + std::vector const vec3 = json::to_bjdata(j2, false, false); + CHECK(json::from_bjdata(vec3) == j2); + } + SECTION("ndarray whose dimensions overflow stays as object") { // the product of the dimensions wraps around std::size_t to 0 From 663013ce641af95e2ea1abe50ac785b5e05c410b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 22 Sep 2026 21:35:00 +0200 Subject: [PATCH 9/9] Fix incorrect diagnostic-positions test assertion for swap() (#5539) The characterization test added in #5482 asserted that basic_json::swap() does NOT exchange start_position/end_position, based on a misreading of the code cited for #5420. In fact swap() (json.hpp, around line 3637) does swap start_position/end_position along with the value, consistent with copy-assignment. The test's assumption was backwards, so it failed on every CI job across every branch/PR since the commit landed. Correct the assertions to match the actual (and correct) behavior: positions are exchanged together with values. Signed-off-by: Niels Lohmann Co-authored-by: Claude Sonnet 5 --- tests/src/unit-class_parser.cpp | 35 +++++++++++++-------------------- 1 file changed, 14 insertions(+), 21 deletions(-) diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index 5b4af321c..e22c4cacf 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -2512,22 +2512,15 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") CHECK(a.end_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) } - SECTION("swap() does NOT exchange positions (likely a real bug, see below)") + SECTION("swap() exchanges positions along with values") { - // NOTE (characterizing, not fixing, for #5420): basic_json::swap() - // (json.hpp, around line 3540, and the friend swap() that forwards - // to it) swaps m_data.m_type and m_data.m_value but -- unlike - // copy-assignment's operator=(basic_json) (json.hpp, around line - // 1291), which swaps start_position/end_position as part of its - // copy-and-swap implementation -- it never touches - // start_position/end_position. So after swap(a, b), the *values* - // of a and b are exchanged, but their *positions* are not: each - // ends up with its own original position describing the other's - // new content. This looks like an oversight/inconsistency rather - // than intended behavior, and is flagged to the maintainer; this - // test only pins the current (surprising) behavior so a fix (or a - // deliberate decision to keep it) shows up here as an intentional - // change rather than a silent regression. + // basic_json::swap() (json.hpp, around line 3626, and the friend + // swap() that forwards to it) swaps start_position/end_position + // together with m_data.m_type and m_data.m_value, so after + // swap(a, b) each variable's position describes its own new + // content, consistent with copy-assignment's + // operator=(basic_json) (json.hpp, around line 1291), which also + // swaps positions as part of its copy-and-swap implementation. json a = json::parse(R"({"a":1})"); json b = json::parse(R"([1,2,3,4,5])"); const auto a_start = a.start_pos(); @@ -2547,12 +2540,12 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") CHECK(a == json::parse(R"([1,2,3,4,5])")); CHECK(b == json::parse(R"({"a":1})")); - // ... but positions were NOT: each variable kept its own - // original position, now describing the other's content - CHECK(a.start_pos() == a_start); - CHECK(a.end_pos() == a_end); - CHECK(b.start_pos() == b_start); - CHECK(b.end_pos() == b_end); + // ... and so were positions: each variable now carries the + // other's original position, describing its own new content + CHECK(a.start_pos() == b_start); + CHECK(a.end_pos() == b_end); + CHECK(b.start_pos() == a_start); + CHECK(b.end_pos() == a_end); } SECTION("mutating a parsed document leaves positions of unrelated values untouched")