diff --git a/.github/workflows/codeql-analysis.yml b/.github/workflows/codeql-analysis.yml index c5497ef9c..1c2e73cd1 100644 --- a/.github/workflows/codeql-analysis.yml +++ b/.github/workflows/codeql-analysis.yml @@ -38,14 +38,14 @@ jobs: # Initializes the CodeQL tools for scanning. - name: Initialize CodeQL - uses: github/codeql-action/init@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/init@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: languages: c-cpp # Autobuild attempts to build any compiled languages (C/C++, C#, or Java). # If this step fails, then you should remove it and run the build manually (see below) - name: Autobuild - uses: github/codeql-action/autobuild@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/autobuild@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 - name: Perform CodeQL Analysis - uses: github/codeql-action/analyze@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/analyze@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 diff --git a/.github/workflows/flawfinder.yml b/.github/workflows/flawfinder.yml index 7c4d22b3c..2c8befd40 100644 --- a/.github/workflows/flawfinder.yml +++ b/.github/workflows/flawfinder.yml @@ -43,6 +43,6 @@ jobs: output: 'flawfinder_results.sarif' - name: Upload analysis results to GitHub Security tab - uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: sarif_file: ${{github.workspace}}/flawfinder_results.sarif diff --git a/.github/workflows/scorecards.yml b/.github/workflows/scorecards.yml index de919b1ab..113da079a 100644 --- a/.github/workflows/scorecards.yml +++ b/.github/workflows/scorecards.yml @@ -76,6 +76,6 @@ jobs: # Upload the results to GitHub's code scanning dashboard. - name: "Upload to code-scanning" - uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: sarif_file: results.sarif diff --git a/.github/workflows/semgrep.yml b/.github/workflows/semgrep.yml index 38de00932..dc326db55 100644 --- a/.github/workflows/semgrep.yml +++ b/.github/workflows/semgrep.yml @@ -61,7 +61,7 @@ jobs: # Upload SARIF file generated in previous step - name: Upload SARIF file - uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: sarif_file: semgrep.sarif if: always() diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index ea912ee2e..212fc8268 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -100,7 +100,7 @@ jobs: container: ubuntu:focal strategy: matrix: - target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_simdutf, ci_test_no_thread_local] + target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_skiplibraryversioncheck, ci_test_simdutf, ci_test_no_thread_local] steps: - name: Install build-essential run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev diff --git a/cmake/ci.cmake b/cmake/ci.cmake index e681b5d0b..28293c1cd 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -260,6 +260,40 @@ add_custom_target(ci_test_noglobaludls COMMENT "Compile and test with global UDLs disabled" ) +############################################################################### +# Disable enum serialization. +############################################################################### + +add_custom_target(ci_test_disableenumserialization + COMMAND ${CMAKE_COMMAND} + -DCMAKE_BUILD_TYPE=Debug -GNinja + -DJSON_BuildTests=ON -DJSON_FastTests=ON -DJSON_DisableEnumSerialization=ON + -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_disableenumserialization + COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_disableenumserialization + COMMAND cd ${PROJECT_BINARY_DIR}/build_disableenumserialization && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure + COMMENT "Compile and test with enum serialization disabled" +) + +############################################################################### +# Skip the multiple-inclusion library version check. +############################################################################### + +# tests/src/skip_library_version_check.cpp deliberately simulates a scenario +# (mixing two differently-versioned inclusions of the library in one +# translation unit) that unavoidably triggers the compiler's own "macro +# redefined" warning, so -- unlike the ci_test_* targets above -- it is +# compiled directly here, with a modest warning set, instead of being folded +# into the library's own -Weverything/-Werror unit test matrix. +add_custom_target(ci_test_skiplibraryversioncheck + COMMAND ${CMAKE_COMMAND} -E make_directory ${PROJECT_BINARY_DIR}/skip_library_version_check + COMMAND ${CMAKE_CXX_COMPILER} -std=c++11 -Wall -Wextra + -I${PROJECT_SOURCE_DIR}/include + ${PROJECT_SOURCE_DIR}/tests/src/skip_library_version_check.cpp + -o ${PROJECT_BINARY_DIR}/skip_library_version_check/skip_library_version_check + COMMAND ${PROJECT_BINARY_DIR}/skip_library_version_check/skip_library_version_check + COMMENT "Compile and run a translation unit simulating a mismatched library version, with JSON_SKIP_LIBRARY_VERSION_CHECK defined" +) + ############################################################################### # Disable thread-local storage. ############################################################################### diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index e9ccd23b5..4bd173257 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -1698,6 +1698,16 @@ class binary_writer }; string_t key = "_ArrayType_"; + // the type name is looked up as a string below; a non-string + // annotation (e.g. a number, null, or an array) cannot name a known + // dtype, so it is treated the same as an unrecognized type name and + // falls back to a plain object encoding instead of throwing + // type_error.302 out of get() + if (!value.at(key).is_string()) + { + return true; + } + // use get() instead of static_cast to avoid an // ambiguous conversion under explicit instantiation on C++17 (see #4825) auto it = bjdtype.find(value.at(key).template get()); @@ -1707,6 +1717,16 @@ class binary_writer } CharType dtype = it->second; + // the 'B' (byte) marker is only defined from BJData Draft 3 onward; + // emitting it under an earlier draft would produce a stream that an + // earlier-draft reader rejects, so such an object falls back to a + // plain object encoding instead (see the "Binary values" section of + // the BJData documentation) + if (dtype == 'B' && bjdata_version < bjdata_version_t::draft3) + { + return true; + } + key = "_ArraySize_"; // the dimensions are written verbatim as the header length below, so a // value that is not an array cannot produce a valid one: null emits 'Z' diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 1d7367433..24d5019c4 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -20384,6 +20384,16 @@ class binary_writer }; string_t key = "_ArrayType_"; + // the type name is looked up as a string below; a non-string + // annotation (e.g. a number, null, or an array) cannot name a known + // dtype, so it is treated the same as an unrecognized type name and + // falls back to a plain object encoding instead of throwing + // type_error.302 out of get() + if (!value.at(key).is_string()) + { + return true; + } + // use get() instead of static_cast to avoid an // ambiguous conversion under explicit instantiation on C++17 (see #4825) auto it = bjdtype.find(value.at(key).template get()); @@ -20393,6 +20403,16 @@ class binary_writer } CharType dtype = it->second; + // the 'B' (byte) marker is only defined from BJData Draft 3 onward; + // emitting it under an earlier draft would produce a stream that an + // earlier-draft reader rejects, so such an object falls back to a + // plain object encoding instead (see the "Binary values" section of + // the BJData documentation) + if (dtype == 'B' && bjdata_version < bjdata_version_t::draft3) + { + return true; + } + key = "_ArraySize_"; // the dimensions are written verbatim as the header length below, so a // value that is not an array cannot produce a valid one: null emits 'Z' diff --git a/tests/src/fuzzer-parse_bjdata.cpp b/tests/src/fuzzer-parse_bjdata.cpp index 1d1d56a5c..a88479933 100644 --- a/tests/src/fuzzer-parse_bjdata.cpp +++ b/tests/src/fuzzer-parse_bjdata.cpp @@ -21,6 +21,27 @@ array data, it performs the following steps: - j4 = from_bjdata(vec3) - assert(j1 == j4) +Re-serializing j2/j3/j4 with the same use_size/use_type settings is checked +for value-stability rather than byte-exact stability: from_bjdata(to_bjdata(j2)) +must equal j2 (and likewise for j3, j4). Byte-exact stability does not hold in +general, because a BJData value can lose type fidelity across a round trip +(e.g. a binary_t value serialized without the optimized "$U#" array header is +parsed back as a plain array of numbers, see #5398 and the discussion on +PR #5494) - the numeric value is preserved, but the writer's smallest-type +selection for the now-plain numbers may legitimately pick a different, but +equally valid, single-byte type marker than the dedicated binary-data writer +would have. Both encodings are valid BJData and both decode to the same +value, so this is not treated as a round-trip failure here. + +"Value-stable" is checked by comparing dump()s rather than with operator== +directly: a BJData/UBJSON payload can decode to a non-finite double (NaN or ++-Infinity), and IEEE 754 NaN is never equal to itself, so operator== would +report two structurally-identical trees as different whenever a NaN is +involved -- not a round-trip bug, just NaN's ordinary (non-)reflexivity. +dump() serializes any non-finite double the same deterministic way (as JSON +`null`, since JSON itself cannot represent NaN/Infinity), so comparing +dumps is stable under exactly the same values that break operator==. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ @@ -31,6 +52,13 @@ drivers. using json = nlohmann::json; +// value-stable comparison for the round-trip checks below; see the note +// above on why this compares dump()s rather than the json values directly +static bool is_value_stable(const json& lhs, const json& rhs) +{ + return lhs.dump() == rhs.dump(); +} + // see http://llvm.org/docs/LibFuzzer.html extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { @@ -56,10 +84,12 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) json const j3 = json::from_bjdata(vec3); json const j4 = json::from_bjdata(vec4); - // serializations must match - assert(json::to_bjdata(j2, false, false) == vec2); - assert(json::to_bjdata(j3, true, false) == vec3); - assert(json::to_bjdata(j4, true, true) == vec4); + // re-serializing must be value-stable (see the notes above on + // why byte-exact stability is not guaranteed in general, and + // why this compares dump()s rather than the values directly) + assert(is_value_stable(json::from_bjdata(json::to_bjdata(j2, false, false)), j2)); + assert(is_value_stable(json::from_bjdata(json::to_bjdata(j3, true, false)), j3)); + assert(is_value_stable(json::from_bjdata(json::to_bjdata(j4, true, true)), j4)); } catch (const json::parse_error&) { diff --git a/tests/src/skip_library_version_check.cpp b/tests/src/skip_library_version_check.cpp new file mode 100644 index 000000000..ddaa4415c --- /dev/null +++ b/tests/src/skip_library_version_check.cpp @@ -0,0 +1,61 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// Standalone compile-and-run check for the JSON_SKIP_LIBRARY_VERSION_CHECK +// configuration macro, which (per #5423) was never exercised anywhere in the +// test matrix. +// +// include/nlohmann/detail/abi_macros.hpp normally emits a #warning if +// NLOHMANN_JSON_VERSION_MAJOR/MINOR/PATCH are already defined (as they would +// be by an earlier inclusion of a different version of the library) with +// values that mismatch the version about to be defined -- unless +// JSON_SKIP_LIBRARY_VERSION_CHECK is defined, in which case the check (and +// that #warning) is skipped. +// +// This file deliberately is not named tests/src/unit-*.cpp: it is compiled +// directly (with a modest, non-strict warning set) by the dedicated +// ci_test_skiplibraryversioncheck target in cmake/ci.cmake, rather than being +// folded into the library's own -Weverything/-Werror unit test matrix. That +// is because the scenario simulated here -- mixing two different, already +// differently-versioned inclusions of the library in one translation unit -- +// unavoidably also triggers the *compiler's own* "macro redefined" warning, +// independent of (and unaffected by) JSON_SKIP_LIBRARY_VERSION_CHECK, which +// only ever silences the library's own #warning. Building this file under +// -Weverything -Werror would therefore fail for a reason unrelated to the +// macro under test. +#define NLOHMANN_JSON_VERSION_MAJOR 0 +#define NLOHMANN_JSON_VERSION_MINOR 0 +#define NLOHMANN_JSON_VERSION_PATCH 0 + +#define JSON_SKIP_LIBRARY_VERSION_CHECK 1 + +#include + +int main() +{ + // reaching this point at all already proves that the mismatched, + // pre-defined version macros above did not stop compilation -- which is + // exactly what JSON_SKIP_LIBRARY_VERSION_CHECK is for. The library must + // also still be fully usable. + const nlohmann::json j = {{"a", 1}, {"b", {1, 2, 3}}}; + if (j.dump() != "{\"a\":1,\"b\":[1,2,3]}") + { + return 1; + } + + // include/nlohmann/detail/abi_macros.hpp unconditionally (re)defines the + // version macros to the library's real, current version right after the + // (here, skipped) mismatch check, regardless of the deliberately wrong + // stand-in values defined above. + if (NLOHMANN_JSON_VERSION_MAJOR == 0 && NLOHMANN_JSON_VERSION_MINOR == 0 && NLOHMANN_JSON_VERSION_PATCH == 0) + { + return 1; + } + + return 0; +} diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 9338e663e..334259fb7 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -2586,7 +2586,12 @@ TEST_CASE("BJData") CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d); CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D); CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C); - CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true) == v_B); + // v_B uses the Draft-3-only 'B' marker, so it round-trips only when + // Draft 3 is explicitly selected (see GitHub issue #5404); the + // default Draft 2 falls back to a plain object instead, covered by + // the "ndarray with _ArrayType_ "byte" is gated by the BJData draft + // version" section below + CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true, json::bjdata_version_t::draft3) == v_B); } SECTION("ndarray with data not matching _ArrayType_ is written as an object") @@ -2629,8 +2634,10 @@ TEST_CASE("BJData") // the C++ API stores an int literal as number_integer, so _ArrayType_ // names the wire type rather than the storage. Both storages have to // produce the same typed array for every type. + // "byte" is checked separately below since it additionally requires + // BJData Draft 3 to be selected explicitly (see GitHub issue #5404). for (const char* type : - {"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char", "byte" + {"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char" }) { CAPTURE(type); @@ -2641,6 +2648,14 @@ TEST_CASE("BJData") CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}))); } + { + const std::string text = R"({"_ArrayType_":"byte","_ArraySize_":[2,3],"_ArrayData_":[1,2,3,4,5,6]})"; + const auto from_text = json::to_bjdata(json::parse(text), true, true, json::bjdata_version_t::draft3); + CHECK(from_text.at(0) == '['); + CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}), + true, true, json::bjdata_version_t::draft3)); + } + // negative values under a signed type behave the same way const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})")); CHECK(from_neg.at(0) == '['); @@ -2731,6 +2746,83 @@ TEST_CASE("BJData") CHECK(json::from_bjdata(json::to_bjdata(j_size), true, true) == j_size); } + SECTION("ndarray whose _ArrayType_ is not a string stays as object") + { + // the type name is looked up as a string below the annotation + // check; a non-string _ArrayType_ cannot name a known dtype, + // so calling get() on it would throw type_error.302 + // instead of falling back like an unrecognized type name + // already does (see GitHub issue #5398) + json const j_number = json({{"_ArrayType_", 1}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_number = json::to_bjdata(j_number); + CHECK(out_number.at(0) == '{'); + CHECK(json::from_bjdata(out_number) == j_number); + + json const j_null = json({{"_ArrayType_", nullptr}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_null = json::to_bjdata(j_null); + CHECK(out_null.at(0) == '{'); + CHECK(json::from_bjdata(out_null) == j_null); + + json const j_bool = json({{"_ArrayType_", true}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_bool = json::to_bjdata(j_bool); + CHECK(out_bool.at(0) == '{'); + CHECK(json::from_bjdata(out_bool) == j_bool); + + json const j_array = json({{"_ArrayType_", {"uint8"}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_array = json::to_bjdata(j_array); + CHECK(out_array.at(0) == '{'); + CHECK(json::from_bjdata(out_array) == j_array); + + json const j_object = json({{"_ArrayType_", {{"a", 1}}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_object = json::to_bjdata(j_object); + CHECK(out_object.at(0) == '{'); + CHECK(json::from_bjdata(out_object) == j_object); + } + + SECTION("re-serializing a value containing a plain-array-of-bytes is value-stable but not byte-stable") + { + // OSS-Fuzz found this input (an array whose first element is a + // binary_t byte, followed by an object whose _ArrayType_ is + // not a string) while exercising the fix for #5398 above: once + // the fix stops to_bjdata() from throwing type_error.302 for + // the third element, serialization proceeds far enough to + // reach a pre-existing, unrelated round-trip quirk in how a + // single-byte binary_t value is re-encoded. + std::vector const input + { + 0x5b, 0x5b, 0x24, 0x42, 0x23, 0x5b, 0x69, 0x01, 0x5d, 0x5b, 0x5b, 0x5d, 0x7b, 0x55, 0x0b, + 0x5f, 0x41, 0x72, 0x72, 0x61, 0x79, 0x44, 0x61, 0x74, 0x61, 0x5f, 0x54, 0x55, 0x0b, 0x5f, + 0x41, 0x72, 0x72, 0x61, 0x79, 0x53, 0x69, 0x7a, 0x65, 0x5f, 0x5a, 0x55, 0x0b, 0x5f, 0x41, + 0x72, 0x72, 0x61, 0x79, 0x54, 0x79, 0x70, 0x65, 0x5f, 0x54, 0x7d, 0x5d + }; + json const j1 = json::from_bjdata(input); + + // to_bjdata() must not throw (this is what #5398 fixes) + std::vector vec2; + CHECK_NOTHROW(vec2 = json::to_bjdata(j1, false, false)); + + // parsing back a plain (non-optimized) array of bytes cannot + // recover that it used to be a binary_t: from_bjdata() has no + // way to distinguish "array of uint8 numbers" from "array of + // bytes" unless the compact "$U#" array header is used, so + // the binary_t collapses into a plain JSON array + json const j2 = json::from_bjdata(vec2); + CHECK(j1 != j2); + CHECK(j2 == json({{91}, json::array(), {{"_ArrayData_", true}, {"_ArraySize_", nullptr}, {"_ArrayType_", true}}})); + + // re-serializing j2 no longer goes through the dedicated + // binary_t writer (which always uses the 'U' marker for raw + // bytes); the now-plain number 91 goes through the generic + // smallest-type writer instead, which - like the rest of the + // UBJSON/BJData writer, and unchanged by this fix - prefers + // the 'i' (int8) marker over 'U' (uint8) for values that fit + // both. Both markers are valid BJData and both decode back to + // 91, so this is not byte-for-byte identical to vec2, but it + // is value-stable: parsing it again reproduces j2 exactly. + std::vector const vec3 = json::to_bjdata(j2, false, false); + CHECK(json::from_bjdata(vec3) == j2); + } + SECTION("ndarray whose dimensions overflow stays as object") { // the product of the dimensions wraps around std::size_t to 0 @@ -2823,6 +2915,36 @@ TEST_CASE("BJData") CHECK(out_single_ok.at(0) == '['); CHECK(json::from_bjdata(out_single_ok) == json({1.5f})); } + + SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version") + { + // the 'B' (byte) marker used by _ArrayType_ "byte" is only defined + // by BJData Draft 3; Draft 2 (the default) has no such marker, so + // emitting it unconditionally produced a stream that a Draft 2 + // reader could not parse as intended (see GitHub issue #5404). + // Two dimensions are used so that a successfully written ndarray + // round-trips back into the annotated object (a single dimension + // is, by the BJData ndarray convention, read back as a plain + // binary value rather than the annotated object, same as every + // other single-dimension ndarray of a non-"byte" type is read + // back as a plain array instead of the annotated object). + json const j_byte = json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + + // default (Draft 2): falls back to a plain object and round-trips + const auto out_draft2 = json::to_bjdata(j_byte); + CHECK(out_draft2.at(0) == '{'); + CHECK(json::from_bjdata(out_draft2) == j_byte); + + // explicit Draft 2: same as the default + const auto out_draft2_explicit = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft2); + CHECK(out_draft2_explicit.at(0) == '{'); + CHECK(json::from_bjdata(out_draft2_explicit) == j_byte); + + // Draft 3 explicitly selected: still uses the compact 'B' ndarray encoding + const auto out_draft3 = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft3); + CHECK(out_draft3 == std::vector({'[', '$', 'B', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6})); + CHECK(json::from_bjdata(out_draft3) == j_byte); + } } } diff --git a/tests/src/unit-byte_container_with_subtype.cpp b/tests/src/unit-byte_container_with_subtype.cpp index 3983ba95f..2e448ac7d 100644 --- a/tests/src/unit-byte_container_with_subtype.cpp +++ b/tests/src/unit-byte_container_with_subtype.cpp @@ -42,6 +42,39 @@ TEST_CASE("byte_container_with_subtype") CHECK(container.subtype() == static_cast(-1)); } + SECTION("move semantics") + { + // the rvalue-reference constructor (without a subtype) must actually move + // the passed-in container rather than copy it; comparing the buffer address + // before and after is a stronger check than just observing the source is + // empty afterward, since a copy-then-clear could also leave it empty + { + std::vector bytes = {{0xCA, 0xFE, 0xBA, 0xBE}}; + const auto* const data_ptr = bytes.data(); + + nlohmann::byte_container_with_subtype> container(std::move(bytes)); + + CHECK(container.size() == 4); + CHECK(container.data() == data_ptr); + CHECK(!container.has_subtype()); + CHECK(bytes.empty()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved) + } + + // same check for the rvalue-reference constructor that also takes a subtype + { + std::vector bytes = {{0xCA, 0xFE, 0xBA, 0xBE}}; + const auto* const data_ptr = bytes.data(); + + nlohmann::byte_container_with_subtype> container(std::move(bytes), 42); + + CHECK(container.size() == 4); + CHECK(container.data() == data_ptr); + CHECK(container.has_subtype()); + CHECK(container.subtype() == 42); + CHECK(bytes.empty()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved) + } + } + SECTION("comparisons") { std::vector const bytes = {{0xCA, 0xFE, 0xBA, 0xBE}}; diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index af76e93cc..e22c4cacf 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -17,6 +17,8 @@ using nlohmann::json; #include #include +#include +#include #include #include #include @@ -2445,3 +2447,314 @@ TEST_CASE("last-read diagnostics are identical across input adapters") } } #endif // !defined(JSON_NOEXCEPTION) + +// this test characterizes the current (documented-by-example, not otherwise +// specified) behavior of JSON_DIAGNOSTIC_POSITIONS positions with respect to +// value lifetime (copy/move/swap/mutation), the various input adapters, and +// user-driven SAX usage. It is regression protection, not a behavior +// specification: if any of these checks fail after a change to json.hpp, +// that change deliberately altered observable behavior and the test (and +// this comment) should be updated accordingly, rather than "fixed" blindly. +#if JSON_DIAGNOSTIC_POSITIONS +TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") +{ + SECTION("value lifetime") + { + SECTION("copy constructor copies positions, recursively") + { + // basic_json(const basic_json&) (json.hpp, around line 1192) copies + // start_position/end_position for the value itself; nested values + // are copied via their own copy constructor (through the copied + // object/array container), so positions are preserved throughout + // the whole tree. + const std::string s = R"({"a":1,"b":[1,2,3]})"; + const json a = json::parse(s); + const json b = a; // NOLINT(performance-unnecessary-copy-initialization) + + CHECK(b.start_pos() == a.start_pos()); + CHECK(b.end_pos() == a.end_pos()); + CHECK(b["b"].start_pos() == a["b"].start_pos()); + CHECK(b["b"].end_pos() == a["b"].end_pos()); + CHECK(b["b"][0].start_pos() == a["b"][0].start_pos()); + CHECK(b["b"][0].end_pos() == a["b"][0].end_pos()); + + // sanity: the positions are meaningful (not all npos) + CHECK(b.start_pos() == 0); + CHECK(b.end_pos() == s.size()); + } + + SECTION("move constructor resets the moved-from value to npos") + { + // basic_json(basic_json&&) (json.hpp, around line 1265) copies + // other's start_position/end_position into *this and then resets + // other's to npos (see the cppcheck-suppress[accessForwarded] + // annotation there, which flags this reset as worth a second + // look). Only the top-level moved-from value is affected; its + // (moved-away) children are gone along with it. + const std::string s = R"({"a":1,"b":[1,2,3]})"; + json a = json::parse(s); + const auto a_start = a.start_pos(); + const auto a_end = a.end_pos(); + const auto nested_start = a["b"].start_pos(); + const auto nested_end = a["b"].end_pos(); + + const json b(std::move(a)); + + // the destination retains the original positions, recursively + CHECK(b.start_pos() == a_start); + CHECK(b.end_pos() == a_end); + CHECK(b["b"].start_pos() == nested_start); + CHECK(b["b"].end_pos() == nested_end); + + // the moved-from value is reset to a null and reports npos + CHECK(a.is_null()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) + CHECK(a.start_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) + CHECK(a.end_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) + } + + SECTION("swap() exchanges positions along with values") + { + // basic_json::swap() (json.hpp, around line 3626, and the friend + // swap() that forwards to it) swaps start_position/end_position + // together with m_data.m_type and m_data.m_value, so after + // swap(a, b) each variable's position describes its own new + // content, consistent with copy-assignment's + // operator=(basic_json) (json.hpp, around line 1291), which also + // swaps positions as part of its copy-and-swap implementation. + json a = json::parse(R"({"a":1})"); + json b = json::parse(R"([1,2,3,4,5])"); + const auto a_start = a.start_pos(); + const auto a_end = a.end_pos(); + const auto b_start = b.start_pos(); + const auto b_end = b.end_pos(); + // both start at 0 (root values start right away), but their + // lengths (and thus end positions) differ, which is enough to + // tell after the swap whether positions actually moved with + // the values + CHECK(a_end != b_end); + + using std::swap; + swap(a, b); + + // values were exchanged as expected ... + CHECK(a == json::parse(R"([1,2,3,4,5])")); + CHECK(b == json::parse(R"({"a":1})")); + + // ... and so were positions: each variable now carries the + // other's original position, describing its own new content + CHECK(a.start_pos() == b_start); + CHECK(a.end_pos() == b_end); + CHECK(b.start_pos() == a_start); + CHECK(b.end_pos() == a_end); + } + + SECTION("mutating a parsed document leaves positions of unrelated values untouched") + { + // Positions are recorded once, during parsing, and are not + // recomputed on mutation. As a consequence, after a mutation the + // parent's own recorded span may no longer describe its current + // (serialized) content -- it still describes what was originally + // parsed. This is characterized here as current behavior, not + // asserted to be desirable or specified. + SECTION("operator[] adding a new object key") + { + const std::string s = R"({"a":1})"; + json j = json::parse(s); + const auto root_start = j.start_pos(); + const auto root_end = j.end_pos(); + const auto a_start = j["a"].start_pos(); + const auto a_end = j["a"].end_pos(); + + j["c"] = 42; + + // the newly-added value was never parsed, so it has no position + CHECK(j["c"].start_pos() == std::string::npos); + CHECK(j["c"].end_pos() == std::string::npos); + + // the existing sibling's position is unaffected + CHECK(j["a"].start_pos() == a_start); + CHECK(j["a"].end_pos() == a_end); + + // the parent's own recorded span is left as-is (now stale: + // it still reflects the original, shorter `{"a":1}` string) + CHECK(j.start_pos() == root_start); + CHECK(j.end_pos() == root_end); + } + + SECTION("push_back on a parsed array") + { + const std::string s = R"([1,2,3])"; + json j = json::parse(s); + const auto root_start = j.start_pos(); + const auto root_end = j.end_pos(); + const auto first_start = j[0].start_pos(); + + j.push_back(4); + + CHECK(j.back().start_pos() == std::string::npos); + CHECK(j.back().end_pos() == std::string::npos); + CHECK(j[0].start_pos() == first_start); + CHECK(j.start_pos() == root_start); + CHECK(j.end_pos() == root_end); + } + + SECTION("erase on a parsed array shifts elements but keeps their own positions") + { + const std::string s = R"([1,2,3])"; + json j = json::parse(s); + const auto second_start = j[1].start_pos(); + const auto third_start = j[2].start_pos(); + const auto root_start = j.start_pos(); + const auto root_end = j.end_pos(); + + j.erase(0); + + // remaining elements moved down an index, but each one still + // reports the position it had *before* the erase (i.e. its + // position in the original source string, not a + // recalculated one) + CHECK(j[0].start_pos() == second_start); + CHECK(j[1].start_pos() == third_start); + + // the parent's own recorded span is again left as-is + CHECK(j.start_pos() == root_start); + CHECK(j.end_pos() == root_end); + } + } + } + + SECTION("input adapters") + { + SECTION("wide string input: positions count transcoded UTF-8 bytes, not wide characters") + { + // 'é' (U+00E9) is a single code unit in a wchar_t/UTF-16 string, but + // transcodes to 2 bytes in UTF-8; the lexer only ever sees the + // transcoded UTF-8 byte stream, so reported positions are byte + // offsets into that UTF-8 stream, not indices into the original + // std::wstring. + // é (rather than a literal 'é' byte sequence in this source + // file) so the wide-string literal's meaning does not depend on + // the compiler's assumed source character set (MSVC, without + // /utf-8, would otherwise decode the raw UTF-8 bytes using the + // system code page instead of as UTF-8) + const std::wstring ws = L"{\"a\":\"\u00e9\u00e9\"}"; + CHECK(ws.size() == 10); // 10 wide characters + + const json j = json::parse(ws); + CHECK(j.start_pos() == 0); + // the transcoded UTF-8 form is 2 bytes longer than the wide string, + // because each of the two 'é' characters becomes 2 UTF-8 bytes + CHECK(j.end_pos() == 12); + CHECK(j.end_pos() != ws.size()); + + const json& a = j["a"]; + CHECK(a.start_pos() == 5); + CHECK(a.end_pos() == 11); + } + + SECTION("BOM-prefixed input: start_pos() reflects the skipped 3-byte BOM") + { + const std::string s = "\xEF\xBB\xBF{\"a\":1}"; + const json j = json::parse(s); + + // the lexer silently skips the BOM before parsing the value, so + // the root value's recorded span starts right after it + CHECK(j.start_pos() == 3); + CHECK(j.end_pos() == s.size()); + } + + SECTION("std::istringstream: positions are consistent, not npos") + { + const std::string s = R"({"a":1,"b":2})"; + std::istringstream ss(s); + const json j = json::parse(ss); + + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == s.size()); + CHECK(j["a"].start_pos() == 5); + } + + SECTION("std::ifstream: positions are consistent, not npos") + { + const std::string s = R"({"a":1,"b":2})"; + { + std::ofstream file("unit-class_parser_diagnostic_positions.tmp"); + file << s; + } + + { + std::ifstream f("unit-class_parser_diagnostic_positions.tmp"); + const json j = json::parse(f); + + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == s.size()); + CHECK(j["a"].start_pos() == 5); + } + + static_cast(std::remove("unit-class_parser_diagnostic_positions.tmp")); + } + + SECTION("iterator-pair input: positions are consistent, not npos") + { + const std::string s = R"({"a":1,"b":2})"; + const json j = json::parse(s.begin(), s.end()); + + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == s.size()); + CHECK(j["a"].start_pos() == 5); + } + + SECTION("binary formats have no text positions") + { + // binary formats (CBOR, MessagePack, UBJSON, BSON, BJData) are + // parsed via detail::binary_reader, which never sets + // start_position/end_position on the values it produces (they + // have no notion of a text offset), so every value's position + // stays at its default of npos. + const json src = json::parse(R"({"a":1,"b":[1,2]})"); + + const json from_cbor = json::from_cbor(json::to_cbor(src)); + CHECK(from_cbor.start_pos() == std::string::npos); + CHECK(from_cbor.end_pos() == std::string::npos); + CHECK(from_cbor["a"].start_pos() == std::string::npos); + CHECK(from_cbor["b"][0].start_pos() == std::string::npos); + + const json from_msgpack = json::from_msgpack(json::to_msgpack(src)); + CHECK(from_msgpack.start_pos() == std::string::npos); + CHECK(from_msgpack.end_pos() == std::string::npos); + + const json from_ubjson = json::from_ubjson(json::to_ubjson(src)); + CHECK(from_ubjson.start_pos() == std::string::npos); + CHECK(from_ubjson.end_pos() == std::string::npos); + + const json from_bson_val = json::from_bson(json::to_bson(src)); + CHECK(from_bson_val.start_pos() == std::string::npos); + CHECK(from_bson_val.end_pos() == std::string::npos); + } + } + + SECTION("user-driven SAX consumers with no lexer report npos") + { + // json::parse() internally wires up its json_sax_dom_parser with a + // pointer to its own lexer (see parser.hpp), which is how positions + // get set at all. A user who constructs a json_sax_dom_parser + // directly (e.g. to drive it via json::sax_parse()) and does not + // supply a lexer pointer gets a consumer with m_lexer_ref == nullptr; + // every "if (m_lexer_ref)" guard in json_sax.hpp is then skipped, so + // every value it produces keeps its default, unset position (npos). + // This was previously true but silently unasserted (operator== + // ignores positions), see #5420. + json result; + nlohmann::detail::json_sax_dom_parser sdp(result); + const std::string s = R"({"a":1,"b":[1,2,3]})"; + CHECK(json::sax_parse(s, &sdp)); + + CHECK(result.start_pos() == std::string::npos); + CHECK(result.end_pos() == std::string::npos); + CHECK(result["a"].start_pos() == std::string::npos); + CHECK(result["a"].end_pos() == std::string::npos); + CHECK(result["b"][0].start_pos() == std::string::npos); + CHECK(result["b"][0].end_pos() == std::string::npos); + } +} +#endif diff --git a/tests/src/unit-conversions.cpp b/tests/src/unit-conversions.cpp index 4975854c0..90d972f71 100644 --- a/tests/src/unit-conversions.cpp +++ b/tests/src/unit-conversions.cpp @@ -1792,6 +1792,40 @@ TEST_CASE("std::filesystem::path") } #endif +// the ADL to_json overload for std::u8string only exists under the same guard +// as std::filesystem::path support (it is otherwise only reached indirectly, +// via std::filesystem::path::u8string()) -- mirror both #if conditions from +// include/nlohmann/detail/conversions/to_json.hpp exactly +#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM +#if defined(__cpp_lib_char8_t) +TEST_CASE("std::u8string") +{ + SECTION("ascii") + { + const std::u8string s = u8"Path"; + json const j = s; + + CHECK(j.template get() == "Path"); + } + + SECTION("utf-8") + { + // use \u universal-character-names (rather than raw \x byte escapes + // or literal non-ASCII source bytes) to compose the multi-byte UTF-8 + // encoding -- MSVC treats \x escapes used that way inside a u8 + // literal as a nonstandard extension (warning C5321), which some of + // our CI configs promote to an error; \u is portable and produces + // the exact same encoded bytes without depending on the source + // file's encoding + const std::u8string s = u8"P\u011B\u0161ina"; + json const j = s; + + CHECK(j.template get() == "P\xc4\x9b\xc5\xa1ina"); + } +} +#endif +#endif + TEST_CASE("std::optional") { SECTION("null") diff --git a/tests/src/unit-json_patch.cpp b/tests/src/unit-json_patch.cpp index 257e455aa..7731c7d92 100644 --- a/tests/src/unit-json_patch.cpp +++ b/tests/src/unit-json_patch.cpp @@ -672,6 +672,102 @@ TEST_CASE("JSON patch") } } + SECTION("patch_inplace") + { + SECTION("happy path: patch_inplace mirrors patch() on success") + { + // mirrors "A.5. Replacing a Value" above, but applies the patch with + // patch_inplace() to a mutable copy instead of using patch()'s + // returned copy + json doc = R"( + { + "baz": "qux", + "foo": "bar" + } + )"_json; + + json const patch = R"( + [ + { "op": "replace", "path": "/baz", "value": "boo" } + ] + )"_json; + + json const expected = R"( + { + "baz": "boo", + "foo": "bar" + } + )"_json; + + doc.patch_inplace(patch); + CHECK(doc == expected); + } + + // this test relies on the "test" operation actually throwing so the + // partial-application state can be observed right after the throw + // point; under JSON_NOEXCEPTION, JSON_THROW() calls std::abort() + // instead (there is no C++ exception to throw), and doctest's + // CHECK_THROWS_AS() is compiled out to a no-op that never even + // invokes the given expression (see doctest's "--no-throw" test + // filter, which ci_test_noexceptions passes) -- so patch()/ + // patch_inplace() would never be called at all and the follow-up + // state assertions below would fail against the untouched original +#if !defined(JSON_NOEXCEPTION) + SECTION("distinguishing contract vs patch(): partial application on failure") + { + // Unlike patch(), which is all-or-nothing because it applies the + // patch to an internal copy that is simply discarded when an + // exception is thrown (leaving the original untouched no matter + // what), patch_inplace() mutates the document it is called on + // directly and immediately, operation by operation. So if a JSON + // Patch fails partway through, whatever operations already + // succeeded remain applied -- the document is left in a partially + // patched state. This is empirically verified current behavior, + // not just documented intent, and is pinned here as such. + json const original = R"( + { + "baz": "qux", + "foo": "bar" + } + )"_json; + + // the first operation ("replace") succeeds; the second ("test") + // fails because the value at "/baz" no longer (and never did) + // equal "not boo" + json const patch = R"( + [ + { "op": "replace", "path": "/baz", "value": "boo" }, + { "op": "test", "path": "/baz", "value": "not boo" } + ] + )"_json; + + // patch() never modifies the object it is called on -- it always + // operates on (and returns) a separate copy, so the original is + // left completely untouched, regardless of success or failure. + // copy_for_patch is intentionally a real copy, not a reference + // to `original`: the whole point of this check is to catch a + // hypothetical future regression where patch() *does* mutate its + // receiver. Using a reference here would make the assertion + // below compare `original` to itself -- trivially true even if + // such a bug existed -- which is exactly what a static analyzer + // can't see when it suggests "this copy is never modified, use + // a reference instead". + json copy_for_patch = original; // NOLINT(performance-unnecessary-copy-initialization) + CHECK_THROWS_AS(copy_for_patch.patch(patch), json::other_error&); + CHECK(copy_for_patch == original); + + // patch_inplace(), in contrast, already applied the successful + // "replace" operation to the document before the "test" operation + // threw -- that change is not rolled back + json doc = original; + CHECK_THROWS_AS(doc.patch_inplace(patch), json::other_error&); + CHECK(doc != original); + CHECK(doc.at("baz") == "boo"); + CHECK(doc.at("foo") == "bar"); + } +#endif // !defined(JSON_NOEXCEPTION) + } + SECTION("errors") { SECTION("unknown operation") diff --git a/tests/src/unit-no-mem-leak-on-adl-serialize.cpp b/tests/src/unit-no-mem-leak-on-adl-serialize.cpp index 469fc2c75..cfbdff008 100644 --- a/tests/src/unit-no-mem-leak-on-adl-serialize.cpp +++ b/tests/src/unit-no-mem-leak-on-adl-serialize.cpp @@ -70,7 +70,7 @@ TEST_CASE("check_for_mem_leak_on_adl_to_json-2") } } -TEST_CASE("check_for_mem_leak_on_adl_to_json-2") +TEST_CASE("check_for_mem_leak_on_adl_to_json-3") { try { diff --git a/tests/src/unit-no_io_and_user_exceptions.cpp b/tests/src/unit-no_io_and_user_exceptions.cpp new file mode 100644 index 000000000..667d114e7 --- /dev/null +++ b/tests/src/unit-no_io_and_user_exceptions.cpp @@ -0,0 +1,91 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// This translation unit is a dedicated, small compile-and-run check for two +// configuration macros that (per #5423) were never exercised anywhere in the +// test matrix: +// - JSON_NO_IO, which removes the library's / support +// (operator<<, operator>>, and the stream-based overloads of dump()/parse()) +// - the JSON_THROW_USER / JSON_TRY_USER / JSON_CATCH_USER trio, which lets a +// user replace the library's internal exception handling +// +// Both macros are about excluding/replacing a facility the library would +// otherwise pull in on its own, and defining one has no bearing on the other, +// so -- to keep the test matrix small -- they are exercised together in a +// single dedicated file instead of two. +// +// JSON_NO_IO requires this file itself to never rely on /; +// only string-based parsing/dumping is used below. +#define JSON_NO_IO 1 + +// The user-supplied exception macros below are a *conforming* replacement: +// they simply forward to the real throw/try/catch keywords (via a counter so +// the test can assert each macro was actually invoked, not just defined), so +// every exception-related behavior the library relies on internally -- +// including rethrowing std::out_of_range as json::out_of_range in at() -- +// keeps working exactly as it would with the library's own default macros. +static int json_throw_user_call_count = 0; // NOLINT(cppcoreguidelines-avoid-non-const-global-variables) + +#define JSON_THROW_USER(exception) do { ++json_throw_user_call_count; throw (exception); } while (false) // NOLINT(cppcoreguidelines-macro-usage) +#define JSON_TRY_USER try // NOLINT(cppcoreguidelines-macro-usage) +#define JSON_CATCH_USER(exception) catch (exception) // NOLINT(cppcoreguidelines-macro-usage) + +#include "doctest_compatibility.h" + +#include +using json = nlohmann::json; + +TEST_CASE("JSON_NO_IO") +{ + // everything that does not touch / must keep working: + // parsing from and dumping to std::string + const json j = json::parse(R"({"a":[1,2,3],"b":true})"); + CHECK(j.dump() == R"({"a":[1,2,3],"b":true})"); + CHECK(j.at("a").size() == 3); + CHECK(j.at("b").get() == true); +} + +// this test relies on CHECK_THROWS_AS() actually invoking the guarded +// expression so json_throw_user_call_count gets bumped and can be observed +// afterwards; doctest's "--no-throw" test filter (which ci_test_noexceptions +// passes, together with a global -DJSON_NOEXCEPTION added to CMAKE_CXX_FLAGS +// for every translation unit in that build, this file included) compiles +// CHECK_THROWS_AS() out to a no-op that never even invokes the given +// expression -- so json::parse()/at() below would never be called at all and +// the call-count assertions would fail even though our JSON_THROW_USER +// override (which always really throws, regardless of JSON_NOEXCEPTION) would +// have worked fine on its own +#if !defined(JSON_NOEXCEPTION) +TEST_CASE("JSON_THROW_USER, JSON_TRY_USER, JSON_CATCH_USER") +{ + json_throw_user_call_count = 0; + + // json::parse() is [[nodiscard]] (JSON_HEDLEY_WARN_UNUSED_RESULT); under + // GCC in C++11 mode that expands to __attribute__((warn_unused_result)), + // which -- unlike a [[nodiscard]] attribute proper -- GCC does not + // consider satisfied by doctest's CHECK_THROWS_AS() wrapping the + // expression in a (void) cast, so the discarded return value would still + // be flagged under -Werror=unused-result; assign it to discard it instead, + // matching the established `json _ = json::parse(...)` pattern used + // elsewhere in the test suite (see unit-class_parser.cpp) + json _; // NOLINT(readability-identifier-naming) + + // a parse error goes through JSON_THROW directly, i.e., through our + // JSON_THROW_USER override + CHECK_THROWS_AS(_ = json::parse("this is not JSON"), json::parse_error&); + CHECK(json_throw_user_call_count > 0); + + // at() on an out-of-range array index internally catches std::out_of_range + // (JSON_TRY_USER/JSON_CATCH_USER) and rethrows it as json::out_of_range + // (JSON_THROW_USER again), so this exercises all three macros together + const int count_before = json_throw_user_call_count; + const json arr = json::array({1, 2, 3}); + CHECK_THROWS_AS(arr.at(10), json::out_of_range&); + CHECK(json_throw_user_call_count > count_before); +} +#endif diff --git a/tests/src/unit-ordered_json2.cpp b/tests/src/unit-ordered_json2.cpp new file mode 100644 index 000000000..83eb7668d --- /dev/null +++ b/tests/src/unit-ordered_json2.cpp @@ -0,0 +1,489 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-FileCopyrightText: 2018 Vitaliy Manushkin +// SPDX-License-Identifier: MIT + +// This file closes a test-coverage gap described in GitHub issue #5421: +// nlohmann::ordered_json (and other non-default basic_json specializations, +// such as the alt_string-based one from unit-alt-string.cpp) were never +// exercised through the binary formats (CBOR/MessagePack/UBJSON/BSON/BJData) +// or through flatten()/unflatten()/diff()/patch()/merge_patch(). + +#include "doctest_compatibility.h" + +#include + +#include +#include +#include +#include + +using nlohmann::json; +using nlohmann::ordered_json; + +///////////////////////////////////////////////////////////////////////////// +// alt_json: a second, independent copy of the custom-string_t basic_json +// specialization defined in unit-alt-string.cpp. +// +// It is duplicated here (rather than shared via a header) because every +// unit-*.cpp file in this test suite is compiled into its own standalone +// executable (see tests/CMakeLists.txt), so there is no ODR concern in +// having the same class name defined in multiple translation units. +// +// Two members had to be added relative to the original alt_string +// (a constructor from std::string, and a find(char, pos) overload) because +// the original type was never used with the binary writers/readers before +// this file: BSON's array/document writer converts std::to_string() results +// and checks for embedded NUL characters via find(char), and the UBJSON/BSON +// high-precision-number path constructs the SAX string_t argument from a +// std::string. Neither path is exercised anywhere else in the test suite for +// this type, which is presumably why the gap was never noticed. +///////////////////////////////////////////////////////////////////////////// + +class alt_string; +bool operator<(const char* op1, const alt_string& op2) noexcept; // NOLINT(misc-use-internal-linkage) +void int_to_string(alt_string& target, std::size_t value); // NOLINT(misc-use-internal-linkage) + +class alt_string +{ + public: + using value_type = std::string::value_type; + + static constexpr auto npos = (std::numeric_limits::max)(); + + alt_string(const char* str): str_impl(str) {} + alt_string(const char* str, std::size_t count): str_impl(str, count) {} + alt_string(std::string str): str_impl(std::move(str)) {} + alt_string(size_t count, char chr): str_impl(count, chr) {} + alt_string() = default; + + alt_string& append(char ch) + { + str_impl.push_back(ch); + return *this; + } + + alt_string& append(const alt_string& str) + { + str_impl.append(str.str_impl); + return *this; + } + + alt_string& append(const char* s, std::size_t length) + { + str_impl.append(s, length); + return *this; + } + + void push_back(char c) + { + str_impl.push_back(c); + } + + template + bool operator==(const op_type& op) const + { + return str_impl == op; + } + + bool operator==(const alt_string& op) const + { + return str_impl == op.str_impl; + } + + template + bool operator!=(const op_type& op) const + { + return str_impl != op; + } + + bool operator!=(const alt_string& op) const + { + return str_impl != op.str_impl; + } + + std::size_t size() const noexcept + { + return str_impl.size(); + } + + void resize(std::size_t n) + { + str_impl.resize(n); + } + + void resize(std::size_t n, char c) + { + str_impl.resize(n, c); + } + + template + bool operator<(const op_type& op) const noexcept + { + return str_impl < op; + } + + bool operator<(const alt_string& op) const noexcept + { + return str_impl < op.str_impl; + } + + const char* c_str() const + { + return str_impl.c_str(); + } + + char& operator[](std::size_t index) + { + return str_impl[index]; + } + + const char& operator[](std::size_t index) const + { + return str_impl[index]; + } + + char& back() + { + return str_impl.back(); + } + + const char& back() const + { + return str_impl.back(); + } + + void clear() + { + str_impl.clear(); + } + + const value_type* data() const + { + return str_impl.data(); + } + + bool empty() const + { + return str_impl.empty(); + } + + std::size_t find(const alt_string& str, std::size_t pos = 0) const + { + return str_impl.find(str.str_impl, pos); + } + + // needed by binary_writer's BSON support, which probes string keys for + // embedded NUL characters via find(char) + std::size_t find(char c, std::size_t pos = 0) const + { + return str_impl.find(c, pos); + } + + std::size_t find_first_of(char c, std::size_t pos = 0) const + { + return str_impl.find_first_of(c, pos); + } + + alt_string substr(std::size_t pos = 0, std::size_t count = npos) const + { + const std::string s = str_impl.substr(pos, count); + return {s.data(), s.size()}; + } + + alt_string& replace(std::size_t pos, std::size_t count, const alt_string& str) + { + str_impl.replace(pos, count, str.str_impl); + return *this; + } + + void reserve(std::size_t new_cap = 0) + { + str_impl.reserve(new_cap); + } + + private: + std::string str_impl {}; // NOLINT(readability-redundant-member-init) + + friend bool operator<(const char* /*op1*/, const alt_string& /*op2*/) noexcept; +}; + +void int_to_string(alt_string& target, std::size_t value) +{ + target = std::to_string(value).c_str(); +} + +using alt_json = nlohmann::basic_json < + std::map, + std::vector, + alt_string, + bool, + std::int64_t, + std::uint64_t, + double, + std::allocator, + nlohmann::adl_serializer >; + +bool operator<(const char* op1, const alt_string& op2) noexcept +{ + return op1 < op2.str_impl; +} + +namespace +{ + +// collects the object keys of j, in iteration order +std::vector collect_keys(const ordered_json& j) +{ + std::vector result; + for (auto it = j.cbegin(); it != j.cend(); ++it) + { + result.push_back(it.key()); + } + return result; +} + +// a nested object/array value with keys inserted in non-alphabetical order, +// used to check both round-trip equality and (for ordered_json) that +// insertion order survives a trip through a binary format +ordered_json make_rich_ordered_json() +{ + ordered_json j; + j["zebra"] = 1; + j["apple"] = ordered_json::array({1, 2, 3}); + j["mango"]["z_nested"] = true; + j["mango"]["a_nested"] = nullptr; + j["banana"] = "some text"; + j["cherry"] = 3.14; + return j; +} + +alt_json make_rich_alt_json() +{ + alt_json j; + j["zebra"] = 1; + j["apple"] = alt_json::array({1, 2, 3}); + j["mango"]["z_nested"] = true; + j["mango"]["a_nested"] = nullptr; + j["banana"] = "some text"; + j["cherry"] = 3.14; + return j; +} + +} // namespace + +TEST_CASE("ordered_json across binary formats") +{ + const ordered_json original = make_rich_ordered_json(); + const std::vector original_keys = collect_keys(original); + const std::vector original_mango_keys = collect_keys(original["mango"]); + + SECTION("CBOR") + { + const auto bytes = ordered_json::to_cbor(original); + const auto restored = ordered_json::from_cbor(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("MessagePack") + { + const auto bytes = ordered_json::to_msgpack(original); + const auto restored = ordered_json::from_msgpack(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("UBJSON") + { + const auto bytes = ordered_json::to_ubjson(original); + const auto restored = ordered_json::from_ubjson(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("BSON") + { + const auto bytes = ordered_json::to_bson(original); + const auto restored = ordered_json::from_bson(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("BJData") + { + const auto bytes = ordered_json::to_bjdata(original); + const auto restored = ordered_json::from_bjdata(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } +} + +TEST_CASE("alt_json (custom string_t) across binary formats") +{ + const alt_json original = make_rich_alt_json(); + + SECTION("CBOR") + { + const auto bytes = alt_json::to_cbor(original); + const auto restored = alt_json::from_cbor(bytes); + CHECK(restored == original); + } + + SECTION("MessagePack") + { + const auto bytes = alt_json::to_msgpack(original); + const auto restored = alt_json::from_msgpack(bytes); + CHECK(restored == original); + } + + SECTION("UBJSON") + { + const auto bytes = alt_json::to_ubjson(original); + const auto restored = alt_json::from_ubjson(bytes); + CHECK(restored == original); + } + + SECTION("BSON") + { + const auto bytes = alt_json::to_bson(original); + const auto restored = alt_json::from_bson(bytes); + CHECK(restored == original); + } + + SECTION("BJData") + { + const auto bytes = alt_json::to_bjdata(original); + const auto restored = alt_json::from_bjdata(bytes); + CHECK(restored == original); + } +} + +TEST_CASE("ordered_json operator== is sensitive to key order") +{ + // Unlike nlohmann::json (whose object_t is a std::map, so equality never + // depends on insertion order), ordered_json's object_t (ordered_map) is a + // std::vector> under the hood, and does not define its + // own operator==: it inherits std::vector's element-wise comparison. As a + // result, two ordered_json objects holding the very same key/value pairs + // in different insertion order compare *unequal*. This is the property + // that makes the round-trip `CHECK(restored == original)` checks above a + // meaningful order-preservation check by themselves (the explicit + // collect_keys() comparisons make that check explicit/readable, and + // guard against this operator== behavior ever changing). + ordered_json a; + a["x"] = 1; + a["y"] = 2; + + ordered_json b; + b["y"] = 2; + b["x"] = 1; + + CHECK(a.size() == b.size()); + CHECK(a["x"] == b["x"]); + CHECK(a["y"] == b["y"]); + CHECK_FALSE(a == b); +} + +TEST_CASE("duplicate keys in a binary-encoded object") +{ + // CBOR encoding of a map with two entries under the same key "a": {"a": 1, "a": 2} + const std::vector cbor_bytes + { + 0xA2, 0x61, 'a', 0x01, 0x61, 'a', 0x02 + }; + + // Both json (std::map, via operator[]) and ordered_json (ordered_map, via + // operator[]) build binary-decoded objects by looking up/creating the + // entry for each incoming key and then assigning the value into it. This + // means a repeated key does *not* produce two entries in either case; + // instead, the *first* occurrence's position is kept (relevant only for + // ordered_json) while the *last* occurrence's value wins (for both) -- + // this matches operator[]'s "assign the referenced slot" semantics, and + // is worth noting because it differs from the initializer-list + // construction path (`ordered_json{{"a",1},{"a",2}}`), which builds + // through insert()/emplace() and therefore keeps the *first* value, not + // the last (see the "There are no dup keys..." case in + // unit-ordered_json.cpp). + const auto j = json::from_cbor(cbor_bytes); + const auto oj = ordered_json::from_cbor(cbor_bytes); + + CHECK(j.size() == 1); + CHECK(oj.size() == 1); + CHECK(j["a"] == 2); + CHECK(oj["a"] == 2); + CHECK(j == json(oj)); +} + +TEST_CASE("ordered_json through flatten/unflatten") +{ + const ordered_json original = make_rich_ordered_json(); + const std::vector original_keys = collect_keys(original); + const std::vector original_mango_keys = collect_keys(original["mango"]); + + const ordered_json flat = original.flatten(); + const ordered_json unflattened = flat.unflatten(); + + CHECK(unflattened == original); + // flatten() walks the value depth-first in iteration order and + // unflatten() re-inserts each flattened key via operator[] in the flat + // object's iteration order, so for ordered_json the original key order + // (both top-level and nested) is preserved end-to-end. + CHECK(collect_keys(unflattened) == original_keys); + CHECK(collect_keys(unflattened["mango"]) == original_mango_keys); +} + +TEST_CASE("ordered_json through diff/patch/patch_inplace") +{ + ordered_json original; + original["one"] = 1; + original["two"] = 2; + original["three"] = 3; + + ordered_json target = original; + target["one"] = 100; // replace + target.erase("two"); // remove + target["four"] = 4; // add + + const ordered_json patch = ordered_json::diff(original, target); + + SECTION("patch") + { + const ordered_json patched = original.patch(patch); + CHECK(patched == target); + } + + SECTION("patch_inplace") + { + ordered_json copy = original; + copy.patch_inplace(patch); + CHECK(copy == target); + } +} + +TEST_CASE("ordered_json through merge_patch") +{ + ordered_json original; + original["a"] = 1; + original["b"] = 2; + + const ordered_json patch = {{"b", nullptr}, {"c", 3}}; + + original.merge_patch(patch); + + ordered_json expected; + expected["a"] = 1; + expected["c"] = 3; + + CHECK(original == expected); + CHECK(collect_keys(original) == collect_keys(expected)); +} diff --git a/tests/src/unit-regression3.cpp b/tests/src/unit-regression3.cpp index 11c6a7da8..a5be9ec4b 100644 --- a/tests/src/unit-regression3.cpp +++ b/tests/src/unit-regression3.cpp @@ -18,6 +18,14 @@ // for some reason including this after the json header leads to linker errors with VS 2017... #include +// skip tests if JSON_DisableEnumSerialization=ON (#4384): std::byte is a +// scoped enum, so get() (needed below to get>() +// from a plain JSON array, not just from an already-binary value) relies on +// enum serialization being enabled +#if defined(JSON_DISABLE_ENUM_SERIALIZATION) && (JSON_DISABLE_ENUM_SERIALIZATION == 1) + #define SKIP_TESTS_FOR_ENUM_SERIALIZATION +#endif + #define JSON_TESTS_PRIVATE #include using json = nlohmann::json; @@ -466,6 +474,7 @@ TEST_CASE("regression tests 3") CHECK((decoded == json_4804::array())); } +#ifndef SKIP_TESTS_FOR_ENUM_SERIALIZATION SECTION("discussion #4209 - custom BinaryType direct assignment and round-tripping") { // Test that assigning a custom BinaryType directly creates a binary value, not an array @@ -499,6 +508,7 @@ TEST_CASE("regression tests 3") CHECK(extracted[1] == std::byte{2}); CHECK(extracted[2] == std::byte{3}); } +#endif SECTION("issue #5046 - implicit conversion of return json to std::optional no longer implicit") { diff --git a/tests/src/unit-serialization.cpp b/tests/src/unit-serialization.cpp index 511108c64..00b305a75 100644 --- a/tests/src/unit-serialization.cpp +++ b/tests/src/unit-serialization.cpp @@ -522,6 +522,19 @@ TEST_CASE("indentation is written straight into the write buffer") CHECK(json::parse(out) == j); } + SECTION("binary values are indented the same way") + { + // a binary value is serialized as an object with "bytes" and + // "subtype" keys; the byte array itself is always written compactly + // (see dump_byte()), so only the surrounding object's indentation + // goes through put_indent() + const json j = json::binary({1, 2, 3}, 128); + CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"bytes\": [1, 2, 3],\n" + + std::string(2000, ' ') + "\"subtype\": 128\n}"); + CHECK(j.dump(2000, '\t') == "{\n" + std::string(2000, '\t') + "\"bytes\": [1, 2, 3],\n" + + std::string(2000, '\t') + "\"subtype\": 128\n}"); + } + SECTION("indentation is unchanged for ordinary widths") { const json j = {{"a", {1, 2}}, {"b", nullptr}}; diff --git a/tests/src/unit-std-format.cpp b/tests/src/unit-std-format.cpp index f7a364884..58cbbf5cc 100644 --- a/tests/src/unit-std-format.cpp +++ b/tests/src/unit-std-format.cpp @@ -17,6 +17,7 @@ #include using json = nlohmann::json; +using ordered_json = nlohmann::ordered_json; // JSON_HAS_CPP_20 (do not remove; see note at top of file) #if JSON_HAS_STD_FORMAT @@ -52,6 +53,23 @@ TEST_CASE("std::formatter") CHECK(std::format("{:2}", j) == j.dump(2)); CHECK(std::format("{:#2}", j) == j.dump(2)); CHECK(std::format("{:8}", j) == j.dump(8)); + // multi-digit widths must accumulate every digit, not just the first + CHECK(std::format("{:12}", j) == j.dump(12)); + CHECK(std::format("{:#12}", j) == j.dump(12)); + CHECK(std::format("{:10}", j) == j.dump(10)); + } + + SECTION("bare alignment with no fill character defaults to a space indent character") + { + const json j = {{"foo", 1}, {"bar", {1, 2, 3}}}; + // without a preceding fill character, the alignment character itself must not + // be mistaken for the indent character -- the default space is kept + CHECK(std::format("{:<}", j) == j.dump()); + CHECK(std::format("{:>}", j) == j.dump()); + CHECK(std::format("{:^}", j) == j.dump()); + CHECK(std::format("{:<3}", j) == j.dump(3, ' ')); + CHECK(std::format("{:>3}", j) == j.dump(3, ' ')); + CHECK(std::format("{:^3}", j) == j.dump(3, ' ')); } SECTION("fill-and-align sets the indent character, like dump(indent, indent_char)") @@ -93,4 +111,16 @@ TEST_CASE("std::formatter") } } +TEST_CASE("std::formatter") +{ + // spot-check a non-default basic_json instantiation, since the formatter + // is written against the generic NLOHMANN_BASIC_JSON_TPL_DECLARATION + // template and must actually instantiate (and behave correctly) for + // template arguments other than nlohmann::json + const ordered_json j = {{"foo", 1}, {"bar", {1, 2, 3}}}; + CHECK(std::format("{}", j) == j.dump()); + CHECK(std::format("{:#}", j) == j.dump(4)); + CHECK(std::format("{:2}", j) == j.dump(2)); +} + #endif diff --git a/tests/src/unit-unicode3.cpp b/tests/src/unit-unicode3.cpp index d5627d8cc..12c12eea4 100644 --- a/tests/src/unit-unicode3.cpp +++ b/tests/src/unit-unicode3.cpp @@ -306,8 +306,8 @@ TEST_CASE("Unicode (3/5)" * doctest::skip()) { for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { - // skip fourth second byte - if (0x80 <= byte3 && byte3 <= 0xBF) + // skip correct fourth byte + if (0x80 <= byte4 && byte4 <= 0xBF) { continue; } diff --git a/tests/src/unit-unicode4.cpp b/tests/src/unit-unicode4.cpp index f15a1499f..43cf7095e 100644 --- a/tests/src/unit-unicode4.cpp +++ b/tests/src/unit-unicode4.cpp @@ -307,7 +307,7 @@ TEST_CASE("Unicode (4/5)" * doctest::skip()) for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { // skip correct fourth byte - if (0x80 <= byte3 && byte3 <= 0xBF) + if (0x80 <= byte4 && byte4 <= 0xBF) { continue; } diff --git a/tests/src/unit-unicode5.cpp b/tests/src/unit-unicode5.cpp index e35801823..bc0312820 100644 --- a/tests/src/unit-unicode5.cpp +++ b/tests/src/unit-unicode5.cpp @@ -307,7 +307,7 @@ TEST_CASE("Unicode (5/5)" * doctest::skip()) for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { // skip correct fourth byte - if (0x80 <= byte3 && byte3 <= 0xBF) + if (0x80 <= byte4 && byte4 <= 0xBF) { continue; }