Merge remote-tracking branch 'origin/develop' into pr5389-merge

# Conflicts:
#	.github/workflows/ubuntu.yml
#	cmake/ci.cmake
This commit is contained in:
Niels Lohmann
2026-09-22 21:47:03 +02:00
24 changed files with 1414 additions and 18 deletions
+3 -3
View File
@@ -38,14 +38,14 @@ jobs:
# Initializes the CodeQL tools for scanning.
- name: Initialize CodeQL
uses: github/codeql-action/init@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9
uses: github/codeql-action/init@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0
with:
languages: c-cpp
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
# If this step fails, then you should remove it and run the build manually (see below)
- name: Autobuild
uses: github/codeql-action/autobuild@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9
uses: github/codeql-action/autobuild@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0
- name: Perform CodeQL Analysis
uses: github/codeql-action/analyze@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9
uses: github/codeql-action/analyze@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0
+1 -1
View File
@@ -43,6 +43,6 @@ jobs:
output: 'flawfinder_results.sarif'
- name: Upload analysis results to GitHub Security tab
uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9
uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0
with:
sarif_file: ${{github.workspace}}/flawfinder_results.sarif
+1 -1
View File
@@ -76,6 +76,6 @@ jobs:
# Upload the results to GitHub's code scanning dashboard.
- name: "Upload to code-scanning"
uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9
uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0
with:
sarif_file: results.sarif
+1 -1
View File
@@ -61,7 +61,7 @@ jobs:
# Upload SARIF file generated in previous step
- name: Upload SARIF file
uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9
uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0
with:
sarif_file: semgrep.sarif
if: always()
+1 -1
View File
@@ -100,7 +100,7 @@ jobs:
container: ubuntu:focal
strategy:
matrix:
target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_simdutf, ci_test_no_thread_local]
target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_skiplibraryversioncheck, ci_test_simdutf, ci_test_no_thread_local]
steps:
- name: Install build-essential
run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev
+34
View File
@@ -260,6 +260,40 @@ add_custom_target(ci_test_noglobaludls
COMMENT "Compile and test with global UDLs disabled"
)
###############################################################################
# Disable enum serialization.
###############################################################################
add_custom_target(ci_test_disableenumserialization
COMMAND ${CMAKE_COMMAND}
-DCMAKE_BUILD_TYPE=Debug -GNinja
-DJSON_BuildTests=ON -DJSON_FastTests=ON -DJSON_DisableEnumSerialization=ON
-S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_disableenumserialization
COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_disableenumserialization
COMMAND cd ${PROJECT_BINARY_DIR}/build_disableenumserialization && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure
COMMENT "Compile and test with enum serialization disabled"
)
###############################################################################
# Skip the multiple-inclusion library version check.
###############################################################################
# tests/src/skip_library_version_check.cpp deliberately simulates a scenario
# (mixing two differently-versioned inclusions of the library in one
# translation unit) that unavoidably triggers the compiler's own "macro
# redefined" warning, so -- unlike the ci_test_* targets above -- it is
# compiled directly here, with a modest warning set, instead of being folded
# into the library's own -Weverything/-Werror unit test matrix.
add_custom_target(ci_test_skiplibraryversioncheck
COMMAND ${CMAKE_COMMAND} -E make_directory ${PROJECT_BINARY_DIR}/skip_library_version_check
COMMAND ${CMAKE_CXX_COMPILER} -std=c++11 -Wall -Wextra
-I${PROJECT_SOURCE_DIR}/include
${PROJECT_SOURCE_DIR}/tests/src/skip_library_version_check.cpp
-o ${PROJECT_BINARY_DIR}/skip_library_version_check/skip_library_version_check
COMMAND ${PROJECT_BINARY_DIR}/skip_library_version_check/skip_library_version_check
COMMENT "Compile and run a translation unit simulating a mismatched library version, with JSON_SKIP_LIBRARY_VERSION_CHECK defined"
)
###############################################################################
# Disable thread-local storage.
###############################################################################
@@ -1698,6 +1698,16 @@ class binary_writer
};
string_t key = "_ArrayType_";
// the type name is looked up as a string below; a non-string
// annotation (e.g. a number, null, or an array) cannot name a known
// dtype, so it is treated the same as an unrecognized type name and
// falls back to a plain object encoding instead of throwing
// type_error.302 out of get<string_t>()
if (!value.at(key).is_string())
{
return true;
}
// use get<string_t>() instead of static_cast<string_t> to avoid an
// ambiguous conversion under explicit instantiation on C++17 (see #4825)
auto it = bjdtype.find(value.at(key).template get<string_t>());
@@ -1707,6 +1717,16 @@ class binary_writer
}
CharType dtype = it->second;
// the 'B' (byte) marker is only defined from BJData Draft 3 onward;
// emitting it under an earlier draft would produce a stream that an
// earlier-draft reader rejects, so such an object falls back to a
// plain object encoding instead (see the "Binary values" section of
// the BJData documentation)
if (dtype == 'B' && bjdata_version < bjdata_version_t::draft3)
{
return true;
}
key = "_ArraySize_";
// the dimensions are written verbatim as the header length below, so a
// value that is not an array cannot produce a valid one: null emits 'Z'
+20
View File
@@ -20384,6 +20384,16 @@ class binary_writer
};
string_t key = "_ArrayType_";
// the type name is looked up as a string below; a non-string
// annotation (e.g. a number, null, or an array) cannot name a known
// dtype, so it is treated the same as an unrecognized type name and
// falls back to a plain object encoding instead of throwing
// type_error.302 out of get<string_t>()
if (!value.at(key).is_string())
{
return true;
}
// use get<string_t>() instead of static_cast<string_t> to avoid an
// ambiguous conversion under explicit instantiation on C++17 (see #4825)
auto it = bjdtype.find(value.at(key).template get<string_t>());
@@ -20393,6 +20403,16 @@ class binary_writer
}
CharType dtype = it->second;
// the 'B' (byte) marker is only defined from BJData Draft 3 onward;
// emitting it under an earlier draft would produce a stream that an
// earlier-draft reader rejects, so such an object falls back to a
// plain object encoding instead (see the "Binary values" section of
// the BJData documentation)
if (dtype == 'B' && bjdata_version < bjdata_version_t::draft3)
{
return true;
}
key = "_ArraySize_";
// the dimensions are written verbatim as the header length below, so a
// value that is not an array cannot produce a valid one: null emits 'Z'
+34 -4
View File
@@ -21,6 +21,27 @@ array data, it performs the following steps:
- j4 = from_bjdata(vec3)
- assert(j1 == j4)
Re-serializing j2/j3/j4 with the same use_size/use_type settings is checked
for value-stability rather than byte-exact stability: from_bjdata(to_bjdata(j2))
must equal j2 (and likewise for j3, j4). Byte-exact stability does not hold in
general, because a BJData value can lose type fidelity across a round trip
(e.g. a binary_t value serialized without the optimized "$U#" array header is
parsed back as a plain array of numbers, see #5398 and the discussion on
PR #5494) - the numeric value is preserved, but the writer's smallest-type
selection for the now-plain numbers may legitimately pick a different, but
equally valid, single-byte type marker than the dedicated binary-data writer
would have. Both encodings are valid BJData and both decode to the same
value, so this is not treated as a round-trip failure here.
"Value-stable" is checked by comparing dump()s rather than with operator==
directly: a BJData/UBJSON payload can decode to a non-finite double (NaN or
+-Infinity), and IEEE 754 NaN is never equal to itself, so operator== would
report two structurally-identical trees as different whenever a NaN is
involved -- not a round-trip bug, just NaN's ordinary (non-)reflexivity.
dump() serializes any non-finite double the same deterministic way (as JSON
`null`, since JSON itself cannot represent NaN/Infinity), so comparing
dumps is stable under exactly the same values that break operator==.
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
drivers.
*/
@@ -31,6 +52,13 @@ drivers.
using json = nlohmann::json;
// value-stable comparison for the round-trip checks below; see the note
// above on why this compares dump()s rather than the json values directly
static bool is_value_stable(const json& lhs, const json& rhs)
{
return lhs.dump() == rhs.dump();
}
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
{
@@ -56,10 +84,12 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
json const j3 = json::from_bjdata(vec3);
json const j4 = json::from_bjdata(vec4);
// serializations must match
assert(json::to_bjdata(j2, false, false) == vec2);
assert(json::to_bjdata(j3, true, false) == vec3);
assert(json::to_bjdata(j4, true, true) == vec4);
// re-serializing must be value-stable (see the notes above on
// why byte-exact stability is not guaranteed in general, and
// why this compares dump()s rather than the values directly)
assert(is_value_stable(json::from_bjdata(json::to_bjdata(j2, false, false)), j2));
assert(is_value_stable(json::from_bjdata(json::to_bjdata(j3, true, false)), j3));
assert(is_value_stable(json::from_bjdata(json::to_bjdata(j4, true, true)), j4));
}
catch (const json::parse_error&)
{
+61
View File
@@ -0,0 +1,61 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
// Standalone compile-and-run check for the JSON_SKIP_LIBRARY_VERSION_CHECK
// configuration macro, which (per #5423) was never exercised anywhere in the
// test matrix.
//
// include/nlohmann/detail/abi_macros.hpp normally emits a #warning if
// NLOHMANN_JSON_VERSION_MAJOR/MINOR/PATCH are already defined (as they would
// be by an earlier inclusion of a different version of the library) with
// values that mismatch the version about to be defined -- unless
// JSON_SKIP_LIBRARY_VERSION_CHECK is defined, in which case the check (and
// that #warning) is skipped.
//
// This file deliberately is not named tests/src/unit-*.cpp: it is compiled
// directly (with a modest, non-strict warning set) by the dedicated
// ci_test_skiplibraryversioncheck target in cmake/ci.cmake, rather than being
// folded into the library's own -Weverything/-Werror unit test matrix. That
// is because the scenario simulated here -- mixing two different, already
// differently-versioned inclusions of the library in one translation unit --
// unavoidably also triggers the *compiler's own* "macro redefined" warning,
// independent of (and unaffected by) JSON_SKIP_LIBRARY_VERSION_CHECK, which
// only ever silences the library's own #warning. Building this file under
// -Weverything -Werror would therefore fail for a reason unrelated to the
// macro under test.
#define NLOHMANN_JSON_VERSION_MAJOR 0
#define NLOHMANN_JSON_VERSION_MINOR 0
#define NLOHMANN_JSON_VERSION_PATCH 0
#define JSON_SKIP_LIBRARY_VERSION_CHECK 1
#include <nlohmann/json.hpp>
int main()
{
// reaching this point at all already proves that the mismatched,
// pre-defined version macros above did not stop compilation -- which is
// exactly what JSON_SKIP_LIBRARY_VERSION_CHECK is for. The library must
// also still be fully usable.
const nlohmann::json j = {{"a", 1}, {"b", {1, 2, 3}}};
if (j.dump() != "{\"a\":1,\"b\":[1,2,3]}")
{
return 1;
}
// include/nlohmann/detail/abi_macros.hpp unconditionally (re)defines the
// version macros to the library's real, current version right after the
// (here, skipped) mismatch check, regardless of the deliberately wrong
// stand-in values defined above.
if (NLOHMANN_JSON_VERSION_MAJOR == 0 && NLOHMANN_JSON_VERSION_MINOR == 0 && NLOHMANN_JSON_VERSION_PATCH == 0)
{
return 1;
}
return 0;
}
+124 -2
View File
@@ -2586,7 +2586,12 @@ TEST_CASE("BJData")
CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d);
CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D);
CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C);
CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true) == v_B);
// v_B uses the Draft-3-only 'B' marker, so it round-trips only when
// Draft 3 is explicitly selected (see GitHub issue #5404); the
// default Draft 2 falls back to a plain object instead, covered by
// the "ndarray with _ArrayType_ "byte" is gated by the BJData draft
// version" section below
CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true, json::bjdata_version_t::draft3) == v_B);
}
SECTION("ndarray with data not matching _ArrayType_ is written as an object")
@@ -2629,8 +2634,10 @@ TEST_CASE("BJData")
// the C++ API stores an int literal as number_integer, so _ArrayType_
// names the wire type rather than the storage. Both storages have to
// produce the same typed array for every type.
// "byte" is checked separately below since it additionally requires
// BJData Draft 3 to be selected explicitly (see GitHub issue #5404).
for (const char* type :
{"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char", "byte"
{"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char"
})
{
CAPTURE(type);
@@ -2641,6 +2648,14 @@ TEST_CASE("BJData")
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}})));
}
{
const std::string text = R"({"_ArrayType_":"byte","_ArraySize_":[2,3],"_ArrayData_":[1,2,3,4,5,6]})";
const auto from_text = json::to_bjdata(json::parse(text), true, true, json::bjdata_version_t::draft3);
CHECK(from_text.at(0) == '[');
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}),
true, true, json::bjdata_version_t::draft3));
}
// negative values under a signed type behave the same way
const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})"));
CHECK(from_neg.at(0) == '[');
@@ -2731,6 +2746,83 @@ TEST_CASE("BJData")
CHECK(json::from_bjdata(json::to_bjdata(j_size), true, true) == j_size);
}
SECTION("ndarray whose _ArrayType_ is not a string stays as object")
{
// the type name is looked up as a string below the annotation
// check; a non-string _ArrayType_ cannot name a known dtype,
// so calling get<string_t>() on it would throw type_error.302
// instead of falling back like an unrecognized type name
// already does (see GitHub issue #5398)
json const j_number = json({{"_ArrayType_", 1}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
const auto out_number = json::to_bjdata(j_number);
CHECK(out_number.at(0) == '{');
CHECK(json::from_bjdata(out_number) == j_number);
json const j_null = json({{"_ArrayType_", nullptr}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
const auto out_null = json::to_bjdata(j_null);
CHECK(out_null.at(0) == '{');
CHECK(json::from_bjdata(out_null) == j_null);
json const j_bool = json({{"_ArrayType_", true}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
const auto out_bool = json::to_bjdata(j_bool);
CHECK(out_bool.at(0) == '{');
CHECK(json::from_bjdata(out_bool) == j_bool);
json const j_array = json({{"_ArrayType_", {"uint8"}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
const auto out_array = json::to_bjdata(j_array);
CHECK(out_array.at(0) == '{');
CHECK(json::from_bjdata(out_array) == j_array);
json const j_object = json({{"_ArrayType_", {{"a", 1}}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
const auto out_object = json::to_bjdata(j_object);
CHECK(out_object.at(0) == '{');
CHECK(json::from_bjdata(out_object) == j_object);
}
SECTION("re-serializing a value containing a plain-array-of-bytes is value-stable but not byte-stable")
{
// OSS-Fuzz found this input (an array whose first element is a
// binary_t byte, followed by an object whose _ArrayType_ is
// not a string) while exercising the fix for #5398 above: once
// the fix stops to_bjdata() from throwing type_error.302 for
// the third element, serialization proceeds far enough to
// reach a pre-existing, unrelated round-trip quirk in how a
// single-byte binary_t value is re-encoded.
std::vector<std::uint8_t> const input
{
0x5b, 0x5b, 0x24, 0x42, 0x23, 0x5b, 0x69, 0x01, 0x5d, 0x5b, 0x5b, 0x5d, 0x7b, 0x55, 0x0b,
0x5f, 0x41, 0x72, 0x72, 0x61, 0x79, 0x44, 0x61, 0x74, 0x61, 0x5f, 0x54, 0x55, 0x0b, 0x5f,
0x41, 0x72, 0x72, 0x61, 0x79, 0x53, 0x69, 0x7a, 0x65, 0x5f, 0x5a, 0x55, 0x0b, 0x5f, 0x41,
0x72, 0x72, 0x61, 0x79, 0x54, 0x79, 0x70, 0x65, 0x5f, 0x54, 0x7d, 0x5d
};
json const j1 = json::from_bjdata(input);
// to_bjdata() must not throw (this is what #5398 fixes)
std::vector<std::uint8_t> vec2;
CHECK_NOTHROW(vec2 = json::to_bjdata(j1, false, false));
// parsing back a plain (non-optimized) array of bytes cannot
// recover that it used to be a binary_t: from_bjdata() has no
// way to distinguish "array of uint8 numbers" from "array of
// bytes" unless the compact "$U#" array header is used, so
// the binary_t collapses into a plain JSON array
json const j2 = json::from_bjdata(vec2);
CHECK(j1 != j2);
CHECK(j2 == json({{91}, json::array(), {{"_ArrayData_", true}, {"_ArraySize_", nullptr}, {"_ArrayType_", true}}}));
// re-serializing j2 no longer goes through the dedicated
// binary_t writer (which always uses the 'U' marker for raw
// bytes); the now-plain number 91 goes through the generic
// smallest-type writer instead, which - like the rest of the
// UBJSON/BJData writer, and unchanged by this fix - prefers
// the 'i' (int8) marker over 'U' (uint8) for values that fit
// both. Both markers are valid BJData and both decode back to
// 91, so this is not byte-for-byte identical to vec2, but it
// is value-stable: parsing it again reproduces j2 exactly.
std::vector<std::uint8_t> const vec3 = json::to_bjdata(j2, false, false);
CHECK(json::from_bjdata(vec3) == j2);
}
SECTION("ndarray whose dimensions overflow stays as object")
{
// the product of the dimensions wraps around std::size_t to 0
@@ -2823,6 +2915,36 @@ TEST_CASE("BJData")
CHECK(out_single_ok.at(0) == '[');
CHECK(json::from_bjdata(out_single_ok) == json({1.5f}));
}
SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version")
{
// the 'B' (byte) marker used by _ArrayType_ "byte" is only defined
// by BJData Draft 3; Draft 2 (the default) has no such marker, so
// emitting it unconditionally produced a stream that a Draft 2
// reader could not parse as intended (see GitHub issue #5404).
// Two dimensions are used so that a successfully written ndarray
// round-trips back into the annotated object (a single dimension
// is, by the BJData ndarray convention, read back as a plain
// binary value rather than the annotated object, same as every
// other single-dimension ndarray of a non-"byte" type is read
// back as a plain array instead of the annotated object).
json const j_byte = json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}});
// default (Draft 2): falls back to a plain object and round-trips
const auto out_draft2 = json::to_bjdata(j_byte);
CHECK(out_draft2.at(0) == '{');
CHECK(json::from_bjdata(out_draft2) == j_byte);
// explicit Draft 2: same as the default
const auto out_draft2_explicit = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft2);
CHECK(out_draft2_explicit.at(0) == '{');
CHECK(json::from_bjdata(out_draft2_explicit) == j_byte);
// Draft 3 explicitly selected: still uses the compact 'B' ndarray encoding
const auto out_draft3 = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft3);
CHECK(out_draft3 == std::vector<uint8_t>({'[', '$', 'B', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6}));
CHECK(json::from_bjdata(out_draft3) == j_byte);
}
}
}
@@ -42,6 +42,39 @@ TEST_CASE("byte_container_with_subtype")
CHECK(container.subtype() == static_cast<subtype_type>(-1));
}
SECTION("move semantics")
{
// the rvalue-reference constructor (without a subtype) must actually move
// the passed-in container rather than copy it; comparing the buffer address
// before and after is a stronger check than just observing the source is
// empty afterward, since a copy-then-clear could also leave it empty
{
std::vector<std::uint8_t> bytes = {{0xCA, 0xFE, 0xBA, 0xBE}};
const auto* const data_ptr = bytes.data();
nlohmann::byte_container_with_subtype<std::vector<std::uint8_t>> container(std::move(bytes));
CHECK(container.size() == 4);
CHECK(container.data() == data_ptr);
CHECK(!container.has_subtype());
CHECK(bytes.empty()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved)
}
// same check for the rvalue-reference constructor that also takes a subtype
{
std::vector<std::uint8_t> bytes = {{0xCA, 0xFE, 0xBA, 0xBE}};
const auto* const data_ptr = bytes.data();
nlohmann::byte_container_with_subtype<std::vector<std::uint8_t>> container(std::move(bytes), 42);
CHECK(container.size() == 4);
CHECK(container.data() == data_ptr);
CHECK(container.has_subtype());
CHECK(container.subtype() == 42);
CHECK(bytes.empty()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved)
}
}
SECTION("comparisons")
{
std::vector<std::uint8_t> const bytes = {{0xCA, 0xFE, 0xBA, 0xBE}};
+313
View File
@@ -17,6 +17,8 @@ using nlohmann::json;
#include <valarray>
#include <algorithm>
#include <cstdio>
#include <fstream>
#include <list>
#include <sstream>
#include <string>
@@ -2445,3 +2447,314 @@ TEST_CASE("last-read diagnostics are identical across input adapters")
}
}
#endif // !defined(JSON_NOEXCEPTION)
// this test characterizes the current (documented-by-example, not otherwise
// specified) behavior of JSON_DIAGNOSTIC_POSITIONS positions with respect to
// value lifetime (copy/move/swap/mutation), the various input adapters, and
// user-driven SAX usage. It is regression protection, not a behavior
// specification: if any of these checks fail after a change to json.hpp,
// that change deliberately altered observable behavior and the test (and
// this comment) should be updated accordingly, rather than "fixed" blindly.
#if JSON_DIAGNOSTIC_POSITIONS
TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX")
{
SECTION("value lifetime")
{
SECTION("copy constructor copies positions, recursively")
{
// basic_json(const basic_json&) (json.hpp, around line 1192) copies
// start_position/end_position for the value itself; nested values
// are copied via their own copy constructor (through the copied
// object/array container), so positions are preserved throughout
// the whole tree.
const std::string s = R"({"a":1,"b":[1,2,3]})";
const json a = json::parse(s);
const json b = a; // NOLINT(performance-unnecessary-copy-initialization)
CHECK(b.start_pos() == a.start_pos());
CHECK(b.end_pos() == a.end_pos());
CHECK(b["b"].start_pos() == a["b"].start_pos());
CHECK(b["b"].end_pos() == a["b"].end_pos());
CHECK(b["b"][0].start_pos() == a["b"][0].start_pos());
CHECK(b["b"][0].end_pos() == a["b"][0].end_pos());
// sanity: the positions are meaningful (not all npos)
CHECK(b.start_pos() == 0);
CHECK(b.end_pos() == s.size());
}
SECTION("move constructor resets the moved-from value to npos")
{
// basic_json(basic_json&&) (json.hpp, around line 1265) copies
// other's start_position/end_position into *this and then resets
// other's to npos (see the cppcheck-suppress[accessForwarded]
// annotation there, which flags this reset as worth a second
// look). Only the top-level moved-from value is affected; its
// (moved-away) children are gone along with it.
const std::string s = R"({"a":1,"b":[1,2,3]})";
json a = json::parse(s);
const auto a_start = a.start_pos();
const auto a_end = a.end_pos();
const auto nested_start = a["b"].start_pos();
const auto nested_end = a["b"].end_pos();
const json b(std::move(a));
// the destination retains the original positions, recursively
CHECK(b.start_pos() == a_start);
CHECK(b.end_pos() == a_end);
CHECK(b["b"].start_pos() == nested_start);
CHECK(b["b"].end_pos() == nested_end);
// the moved-from value is reset to a null and reports npos
CHECK(a.is_null()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move)
CHECK(a.start_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move)
CHECK(a.end_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move)
}
SECTION("swap() exchanges positions along with values")
{
// basic_json::swap() (json.hpp, around line 3626, and the friend
// swap() that forwards to it) swaps start_position/end_position
// together with m_data.m_type and m_data.m_value, so after
// swap(a, b) each variable's position describes its own new
// content, consistent with copy-assignment's
// operator=(basic_json) (json.hpp, around line 1291), which also
// swaps positions as part of its copy-and-swap implementation.
json a = json::parse(R"({"a":1})");
json b = json::parse(R"([1,2,3,4,5])");
const auto a_start = a.start_pos();
const auto a_end = a.end_pos();
const auto b_start = b.start_pos();
const auto b_end = b.end_pos();
// both start at 0 (root values start right away), but their
// lengths (and thus end positions) differ, which is enough to
// tell after the swap whether positions actually moved with
// the values
CHECK(a_end != b_end);
using std::swap;
swap(a, b);
// values were exchanged as expected ...
CHECK(a == json::parse(R"([1,2,3,4,5])"));
CHECK(b == json::parse(R"({"a":1})"));
// ... and so were positions: each variable now carries the
// other's original position, describing its own new content
CHECK(a.start_pos() == b_start);
CHECK(a.end_pos() == b_end);
CHECK(b.start_pos() == a_start);
CHECK(b.end_pos() == a_end);
}
SECTION("mutating a parsed document leaves positions of unrelated values untouched")
{
// Positions are recorded once, during parsing, and are not
// recomputed on mutation. As a consequence, after a mutation the
// parent's own recorded span may no longer describe its current
// (serialized) content -- it still describes what was originally
// parsed. This is characterized here as current behavior, not
// asserted to be desirable or specified.
SECTION("operator[] adding a new object key")
{
const std::string s = R"({"a":1})";
json j = json::parse(s);
const auto root_start = j.start_pos();
const auto root_end = j.end_pos();
const auto a_start = j["a"].start_pos();
const auto a_end = j["a"].end_pos();
j["c"] = 42;
// the newly-added value was never parsed, so it has no position
CHECK(j["c"].start_pos() == std::string::npos);
CHECK(j["c"].end_pos() == std::string::npos);
// the existing sibling's position is unaffected
CHECK(j["a"].start_pos() == a_start);
CHECK(j["a"].end_pos() == a_end);
// the parent's own recorded span is left as-is (now stale:
// it still reflects the original, shorter `{"a":1}` string)
CHECK(j.start_pos() == root_start);
CHECK(j.end_pos() == root_end);
}
SECTION("push_back on a parsed array")
{
const std::string s = R"([1,2,3])";
json j = json::parse(s);
const auto root_start = j.start_pos();
const auto root_end = j.end_pos();
const auto first_start = j[0].start_pos();
j.push_back(4);
CHECK(j.back().start_pos() == std::string::npos);
CHECK(j.back().end_pos() == std::string::npos);
CHECK(j[0].start_pos() == first_start);
CHECK(j.start_pos() == root_start);
CHECK(j.end_pos() == root_end);
}
SECTION("erase on a parsed array shifts elements but keeps their own positions")
{
const std::string s = R"([1,2,3])";
json j = json::parse(s);
const auto second_start = j[1].start_pos();
const auto third_start = j[2].start_pos();
const auto root_start = j.start_pos();
const auto root_end = j.end_pos();
j.erase(0);
// remaining elements moved down an index, but each one still
// reports the position it had *before* the erase (i.e. its
// position in the original source string, not a
// recalculated one)
CHECK(j[0].start_pos() == second_start);
CHECK(j[1].start_pos() == third_start);
// the parent's own recorded span is again left as-is
CHECK(j.start_pos() == root_start);
CHECK(j.end_pos() == root_end);
}
}
}
SECTION("input adapters")
{
SECTION("wide string input: positions count transcoded UTF-8 bytes, not wide characters")
{
// 'é' (U+00E9) is a single code unit in a wchar_t/UTF-16 string, but
// transcodes to 2 bytes in UTF-8; the lexer only ever sees the
// transcoded UTF-8 byte stream, so reported positions are byte
// offsets into that UTF-8 stream, not indices into the original
// std::wstring.
// é (rather than a literal 'é' byte sequence in this source
// file) so the wide-string literal's meaning does not depend on
// the compiler's assumed source character set (MSVC, without
// /utf-8, would otherwise decode the raw UTF-8 bytes using the
// system code page instead of as UTF-8)
const std::wstring ws = L"{\"a\":\"\u00e9\u00e9\"}";
CHECK(ws.size() == 10); // 10 wide characters
const json j = json::parse(ws);
CHECK(j.start_pos() == 0);
// the transcoded UTF-8 form is 2 bytes longer than the wide string,
// because each of the two 'é' characters becomes 2 UTF-8 bytes
CHECK(j.end_pos() == 12);
CHECK(j.end_pos() != ws.size());
const json& a = j["a"];
CHECK(a.start_pos() == 5);
CHECK(a.end_pos() == 11);
}
SECTION("BOM-prefixed input: start_pos() reflects the skipped 3-byte BOM")
{
const std::string s = "\xEF\xBB\xBF{\"a\":1}";
const json j = json::parse(s);
// the lexer silently skips the BOM before parsing the value, so
// the root value's recorded span starts right after it
CHECK(j.start_pos() == 3);
CHECK(j.end_pos() == s.size());
}
SECTION("std::istringstream: positions are consistent, not npos")
{
const std::string s = R"({"a":1,"b":2})";
std::istringstream ss(s);
const json j = json::parse(ss);
CHECK(j.start_pos() == 0);
CHECK(j.end_pos() == s.size());
CHECK(j["a"].start_pos() == 5);
}
SECTION("std::ifstream: positions are consistent, not npos")
{
const std::string s = R"({"a":1,"b":2})";
{
std::ofstream file("unit-class_parser_diagnostic_positions.tmp");
file << s;
}
{
std::ifstream f("unit-class_parser_diagnostic_positions.tmp");
const json j = json::parse(f);
CHECK(j.start_pos() == 0);
CHECK(j.end_pos() == s.size());
CHECK(j["a"].start_pos() == 5);
}
static_cast<void>(std::remove("unit-class_parser_diagnostic_positions.tmp"));
}
SECTION("iterator-pair input: positions are consistent, not npos")
{
const std::string s = R"({"a":1,"b":2})";
const json j = json::parse(s.begin(), s.end());
CHECK(j.start_pos() == 0);
CHECK(j.end_pos() == s.size());
CHECK(j["a"].start_pos() == 5);
}
SECTION("binary formats have no text positions")
{
// binary formats (CBOR, MessagePack, UBJSON, BSON, BJData) are
// parsed via detail::binary_reader, which never sets
// start_position/end_position on the values it produces (they
// have no notion of a text offset), so every value's position
// stays at its default of npos.
const json src = json::parse(R"({"a":1,"b":[1,2]})");
const json from_cbor = json::from_cbor(json::to_cbor(src));
CHECK(from_cbor.start_pos() == std::string::npos);
CHECK(from_cbor.end_pos() == std::string::npos);
CHECK(from_cbor["a"].start_pos() == std::string::npos);
CHECK(from_cbor["b"][0].start_pos() == std::string::npos);
const json from_msgpack = json::from_msgpack(json::to_msgpack(src));
CHECK(from_msgpack.start_pos() == std::string::npos);
CHECK(from_msgpack.end_pos() == std::string::npos);
const json from_ubjson = json::from_ubjson(json::to_ubjson(src));
CHECK(from_ubjson.start_pos() == std::string::npos);
CHECK(from_ubjson.end_pos() == std::string::npos);
const json from_bson_val = json::from_bson(json::to_bson(src));
CHECK(from_bson_val.start_pos() == std::string::npos);
CHECK(from_bson_val.end_pos() == std::string::npos);
}
}
SECTION("user-driven SAX consumers with no lexer report npos")
{
// json::parse() internally wires up its json_sax_dom_parser with a
// pointer to its own lexer (see parser.hpp), which is how positions
// get set at all. A user who constructs a json_sax_dom_parser
// directly (e.g. to drive it via json::sax_parse()) and does not
// supply a lexer pointer gets a consumer with m_lexer_ref == nullptr;
// every "if (m_lexer_ref)" guard in json_sax.hpp is then skipped, so
// every value it produces keeps its default, unset position (npos).
// This was previously true but silently unasserted (operator==
// ignores positions), see #5420.
json result;
nlohmann::detail::json_sax_dom_parser<json, nlohmann::detail::string_input_adapter_type> sdp(result);
const std::string s = R"({"a":1,"b":[1,2,3]})";
CHECK(json::sax_parse(s, &sdp));
CHECK(result.start_pos() == std::string::npos);
CHECK(result.end_pos() == std::string::npos);
CHECK(result["a"].start_pos() == std::string::npos);
CHECK(result["a"].end_pos() == std::string::npos);
CHECK(result["b"][0].start_pos() == std::string::npos);
CHECK(result["b"][0].end_pos() == std::string::npos);
}
}
#endif
+34
View File
@@ -1792,6 +1792,40 @@ TEST_CASE("std::filesystem::path")
}
#endif
// the ADL to_json overload for std::u8string only exists under the same guard
// as std::filesystem::path support (it is otherwise only reached indirectly,
// via std::filesystem::path::u8string()) -- mirror both #if conditions from
// include/nlohmann/detail/conversions/to_json.hpp exactly
#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM
#if defined(__cpp_lib_char8_t)
TEST_CASE("std::u8string")
{
SECTION("ascii")
{
const std::u8string s = u8"Path";
json const j = s;
CHECK(j.template get<std::string>() == "Path");
}
SECTION("utf-8")
{
// use \u universal-character-names (rather than raw \x byte escapes
// or literal non-ASCII source bytes) to compose the multi-byte UTF-8
// encoding -- MSVC treats \x escapes used that way inside a u8
// literal as a nonstandard extension (warning C5321), which some of
// our CI configs promote to an error; \u is portable and produces
// the exact same encoded bytes without depending on the source
// file's encoding
const std::u8string s = u8"P\u011B\u0161ina";
json const j = s;
CHECK(j.template get<std::string>() == "P\xc4\x9b\xc5\xa1ina");
}
}
#endif
#endif
TEST_CASE("std::optional")
{
SECTION("null")
+96
View File
@@ -672,6 +672,102 @@ TEST_CASE("JSON patch")
}
}
SECTION("patch_inplace")
{
SECTION("happy path: patch_inplace mirrors patch() on success")
{
// mirrors "A.5. Replacing a Value" above, but applies the patch with
// patch_inplace() to a mutable copy instead of using patch()'s
// returned copy
json doc = R"(
{
"baz": "qux",
"foo": "bar"
}
)"_json;
json const patch = R"(
[
{ "op": "replace", "path": "/baz", "value": "boo" }
]
)"_json;
json const expected = R"(
{
"baz": "boo",
"foo": "bar"
}
)"_json;
doc.patch_inplace(patch);
CHECK(doc == expected);
}
// this test relies on the "test" operation actually throwing so the
// partial-application state can be observed right after the throw
// point; under JSON_NOEXCEPTION, JSON_THROW() calls std::abort()
// instead (there is no C++ exception to throw), and doctest's
// CHECK_THROWS_AS() is compiled out to a no-op that never even
// invokes the given expression (see doctest's "--no-throw" test
// filter, which ci_test_noexceptions passes) -- so patch()/
// patch_inplace() would never be called at all and the follow-up
// state assertions below would fail against the untouched original
#if !defined(JSON_NOEXCEPTION)
SECTION("distinguishing contract vs patch(): partial application on failure")
{
// Unlike patch(), which is all-or-nothing because it applies the
// patch to an internal copy that is simply discarded when an
// exception is thrown (leaving the original untouched no matter
// what), patch_inplace() mutates the document it is called on
// directly and immediately, operation by operation. So if a JSON
// Patch fails partway through, whatever operations already
// succeeded remain applied -- the document is left in a partially
// patched state. This is empirically verified current behavior,
// not just documented intent, and is pinned here as such.
json const original = R"(
{
"baz": "qux",
"foo": "bar"
}
)"_json;
// the first operation ("replace") succeeds; the second ("test")
// fails because the value at "/baz" no longer (and never did)
// equal "not boo"
json const patch = R"(
[
{ "op": "replace", "path": "/baz", "value": "boo" },
{ "op": "test", "path": "/baz", "value": "not boo" }
]
)"_json;
// patch() never modifies the object it is called on -- it always
// operates on (and returns) a separate copy, so the original is
// left completely untouched, regardless of success or failure.
// copy_for_patch is intentionally a real copy, not a reference
// to `original`: the whole point of this check is to catch a
// hypothetical future regression where patch() *does* mutate its
// receiver. Using a reference here would make the assertion
// below compare `original` to itself -- trivially true even if
// such a bug existed -- which is exactly what a static analyzer
// can't see when it suggests "this copy is never modified, use
// a reference instead".
json copy_for_patch = original; // NOLINT(performance-unnecessary-copy-initialization)
CHECK_THROWS_AS(copy_for_patch.patch(patch), json::other_error&);
CHECK(copy_for_patch == original);
// patch_inplace(), in contrast, already applied the successful
// "replace" operation to the document before the "test" operation
// threw -- that change is not rolled back
json doc = original;
CHECK_THROWS_AS(doc.patch_inplace(patch), json::other_error&);
CHECK(doc != original);
CHECK(doc.at("baz") == "boo");
CHECK(doc.at("foo") == "bar");
}
#endif // !defined(JSON_NOEXCEPTION)
}
SECTION("errors")
{
SECTION("unknown operation")
@@ -70,7 +70,7 @@ TEST_CASE("check_for_mem_leak_on_adl_to_json-2")
}
}
TEST_CASE("check_for_mem_leak_on_adl_to_json-2")
TEST_CASE("check_for_mem_leak_on_adl_to_json-3")
{
try
{
@@ -0,0 +1,91 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
// This translation unit is a dedicated, small compile-and-run check for two
// configuration macros that (per #5423) were never exercised anywhere in the
// test matrix:
// - JSON_NO_IO, which removes the library's <istream>/<ostream> support
// (operator<<, operator>>, and the stream-based overloads of dump()/parse())
// - the JSON_THROW_USER / JSON_TRY_USER / JSON_CATCH_USER trio, which lets a
// user replace the library's internal exception handling
//
// Both macros are about excluding/replacing a facility the library would
// otherwise pull in on its own, and defining one has no bearing on the other,
// so -- to keep the test matrix small -- they are exercised together in a
// single dedicated file instead of two.
//
// JSON_NO_IO requires this file itself to never rely on <iostream>/<sstream>;
// only string-based parsing/dumping is used below.
#define JSON_NO_IO 1
// The user-supplied exception macros below are a *conforming* replacement:
// they simply forward to the real throw/try/catch keywords (via a counter so
// the test can assert each macro was actually invoked, not just defined), so
// every exception-related behavior the library relies on internally --
// including rethrowing std::out_of_range as json::out_of_range in at() --
// keeps working exactly as it would with the library's own default macros.
static int json_throw_user_call_count = 0; // NOLINT(cppcoreguidelines-avoid-non-const-global-variables)
#define JSON_THROW_USER(exception) do { ++json_throw_user_call_count; throw (exception); } while (false) // NOLINT(cppcoreguidelines-macro-usage)
#define JSON_TRY_USER try // NOLINT(cppcoreguidelines-macro-usage)
#define JSON_CATCH_USER(exception) catch (exception) // NOLINT(cppcoreguidelines-macro-usage)
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
using json = nlohmann::json;
TEST_CASE("JSON_NO_IO")
{
// everything that does not touch <istream>/<ostream> must keep working:
// parsing from and dumping to std::string
const json j = json::parse(R"({"a":[1,2,3],"b":true})");
CHECK(j.dump() == R"({"a":[1,2,3],"b":true})");
CHECK(j.at("a").size() == 3);
CHECK(j.at("b").get<bool>() == true);
}
// this test relies on CHECK_THROWS_AS() actually invoking the guarded
// expression so json_throw_user_call_count gets bumped and can be observed
// afterwards; doctest's "--no-throw" test filter (which ci_test_noexceptions
// passes, together with a global -DJSON_NOEXCEPTION added to CMAKE_CXX_FLAGS
// for every translation unit in that build, this file included) compiles
// CHECK_THROWS_AS() out to a no-op that never even invokes the given
// expression -- so json::parse()/at() below would never be called at all and
// the call-count assertions would fail even though our JSON_THROW_USER
// override (which always really throws, regardless of JSON_NOEXCEPTION) would
// have worked fine on its own
#if !defined(JSON_NOEXCEPTION)
TEST_CASE("JSON_THROW_USER, JSON_TRY_USER, JSON_CATCH_USER")
{
json_throw_user_call_count = 0;
// json::parse() is [[nodiscard]] (JSON_HEDLEY_WARN_UNUSED_RESULT); under
// GCC in C++11 mode that expands to __attribute__((warn_unused_result)),
// which -- unlike a [[nodiscard]] attribute proper -- GCC does not
// consider satisfied by doctest's CHECK_THROWS_AS() wrapping the
// expression in a (void) cast, so the discarded return value would still
// be flagged under -Werror=unused-result; assign it to discard it instead,
// matching the established `json _ = json::parse(...)` pattern used
// elsewhere in the test suite (see unit-class_parser.cpp)
json _; // NOLINT(readability-identifier-naming)
// a parse error goes through JSON_THROW directly, i.e., through our
// JSON_THROW_USER override
CHECK_THROWS_AS(_ = json::parse("this is not JSON"), json::parse_error&);
CHECK(json_throw_user_call_count > 0);
// at() on an out-of-range array index internally catches std::out_of_range
// (JSON_TRY_USER/JSON_CATCH_USER) and rethrows it as json::out_of_range
// (JSON_THROW_USER again), so this exercises all three macros together
const int count_before = json_throw_user_call_count;
const json arr = json::array({1, 2, 3});
CHECK_THROWS_AS(arr.at(10), json::out_of_range&);
CHECK(json_throw_user_call_count > count_before);
}
#endif
+489
View File
@@ -0,0 +1,489 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-FileCopyrightText: 2018 Vitaliy Manushkin <agri@akamo.info>
// SPDX-License-Identifier: MIT
// This file closes a test-coverage gap described in GitHub issue #5421:
// nlohmann::ordered_json (and other non-default basic_json specializations,
// such as the alt_string-based one from unit-alt-string.cpp) were never
// exercised through the binary formats (CBOR/MessagePack/UBJSON/BSON/BJData)
// or through flatten()/unflatten()/diff()/patch()/merge_patch().
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
#include <cstdint>
#include <string>
#include <utility>
#include <vector>
using nlohmann::json;
using nlohmann::ordered_json;
/////////////////////////////////////////////////////////////////////////////
// alt_json: a second, independent copy of the custom-string_t basic_json
// specialization defined in unit-alt-string.cpp.
//
// It is duplicated here (rather than shared via a header) because every
// unit-*.cpp file in this test suite is compiled into its own standalone
// executable (see tests/CMakeLists.txt), so there is no ODR concern in
// having the same class name defined in multiple translation units.
//
// Two members had to be added relative to the original alt_string
// (a constructor from std::string, and a find(char, pos) overload) because
// the original type was never used with the binary writers/readers before
// this file: BSON's array/document writer converts std::to_string() results
// and checks for embedded NUL characters via find(char), and the UBJSON/BSON
// high-precision-number path constructs the SAX string_t argument from a
// std::string. Neither path is exercised anywhere else in the test suite for
// this type, which is presumably why the gap was never noticed.
/////////////////////////////////////////////////////////////////////////////
class alt_string;
bool operator<(const char* op1, const alt_string& op2) noexcept; // NOLINT(misc-use-internal-linkage)
void int_to_string(alt_string& target, std::size_t value); // NOLINT(misc-use-internal-linkage)
class alt_string
{
public:
using value_type = std::string::value_type;
static constexpr auto npos = (std::numeric_limits<std::size_t>::max)();
alt_string(const char* str): str_impl(str) {}
alt_string(const char* str, std::size_t count): str_impl(str, count) {}
alt_string(std::string str): str_impl(std::move(str)) {}
alt_string(size_t count, char chr): str_impl(count, chr) {}
alt_string() = default;
alt_string& append(char ch)
{
str_impl.push_back(ch);
return *this;
}
alt_string& append(const alt_string& str)
{
str_impl.append(str.str_impl);
return *this;
}
alt_string& append(const char* s, std::size_t length)
{
str_impl.append(s, length);
return *this;
}
void push_back(char c)
{
str_impl.push_back(c);
}
template <typename op_type>
bool operator==(const op_type& op) const
{
return str_impl == op;
}
bool operator==(const alt_string& op) const
{
return str_impl == op.str_impl;
}
template <typename op_type>
bool operator!=(const op_type& op) const
{
return str_impl != op;
}
bool operator!=(const alt_string& op) const
{
return str_impl != op.str_impl;
}
std::size_t size() const noexcept
{
return str_impl.size();
}
void resize(std::size_t n)
{
str_impl.resize(n);
}
void resize(std::size_t n, char c)
{
str_impl.resize(n, c);
}
template <typename op_type>
bool operator<(const op_type& op) const noexcept
{
return str_impl < op;
}
bool operator<(const alt_string& op) const noexcept
{
return str_impl < op.str_impl;
}
const char* c_str() const
{
return str_impl.c_str();
}
char& operator[](std::size_t index)
{
return str_impl[index];
}
const char& operator[](std::size_t index) const
{
return str_impl[index];
}
char& back()
{
return str_impl.back();
}
const char& back() const
{
return str_impl.back();
}
void clear()
{
str_impl.clear();
}
const value_type* data() const
{
return str_impl.data();
}
bool empty() const
{
return str_impl.empty();
}
std::size_t find(const alt_string& str, std::size_t pos = 0) const
{
return str_impl.find(str.str_impl, pos);
}
// needed by binary_writer's BSON support, which probes string keys for
// embedded NUL characters via find(char)
std::size_t find(char c, std::size_t pos = 0) const
{
return str_impl.find(c, pos);
}
std::size_t find_first_of(char c, std::size_t pos = 0) const
{
return str_impl.find_first_of(c, pos);
}
alt_string substr(std::size_t pos = 0, std::size_t count = npos) const
{
const std::string s = str_impl.substr(pos, count);
return {s.data(), s.size()};
}
alt_string& replace(std::size_t pos, std::size_t count, const alt_string& str)
{
str_impl.replace(pos, count, str.str_impl);
return *this;
}
void reserve(std::size_t new_cap = 0)
{
str_impl.reserve(new_cap);
}
private:
std::string str_impl {}; // NOLINT(readability-redundant-member-init)
friend bool operator<(const char* /*op1*/, const alt_string& /*op2*/) noexcept;
};
void int_to_string(alt_string& target, std::size_t value)
{
target = std::to_string(value).c_str();
}
using alt_json = nlohmann::basic_json <
std::map,
std::vector,
alt_string,
bool,
std::int64_t,
std::uint64_t,
double,
std::allocator,
nlohmann::adl_serializer >;
bool operator<(const char* op1, const alt_string& op2) noexcept
{
return op1 < op2.str_impl;
}
namespace
{
// collects the object keys of j, in iteration order
std::vector<std::string> collect_keys(const ordered_json& j)
{
std::vector<std::string> result;
for (auto it = j.cbegin(); it != j.cend(); ++it)
{
result.push_back(it.key());
}
return result;
}
// a nested object/array value with keys inserted in non-alphabetical order,
// used to check both round-trip equality and (for ordered_json) that
// insertion order survives a trip through a binary format
ordered_json make_rich_ordered_json()
{
ordered_json j;
j["zebra"] = 1;
j["apple"] = ordered_json::array({1, 2, 3});
j["mango"]["z_nested"] = true;
j["mango"]["a_nested"] = nullptr;
j["banana"] = "some text";
j["cherry"] = 3.14;
return j;
}
alt_json make_rich_alt_json()
{
alt_json j;
j["zebra"] = 1;
j["apple"] = alt_json::array({1, 2, 3});
j["mango"]["z_nested"] = true;
j["mango"]["a_nested"] = nullptr;
j["banana"] = "some text";
j["cherry"] = 3.14;
return j;
}
} // namespace
TEST_CASE("ordered_json across binary formats")
{
const ordered_json original = make_rich_ordered_json();
const std::vector<std::string> original_keys = collect_keys(original);
const std::vector<std::string> original_mango_keys = collect_keys(original["mango"]);
SECTION("CBOR")
{
const auto bytes = ordered_json::to_cbor(original);
const auto restored = ordered_json::from_cbor(bytes);
CHECK(restored == original);
CHECK(collect_keys(restored) == original_keys);
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
}
SECTION("MessagePack")
{
const auto bytes = ordered_json::to_msgpack(original);
const auto restored = ordered_json::from_msgpack(bytes);
CHECK(restored == original);
CHECK(collect_keys(restored) == original_keys);
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
}
SECTION("UBJSON")
{
const auto bytes = ordered_json::to_ubjson(original);
const auto restored = ordered_json::from_ubjson(bytes);
CHECK(restored == original);
CHECK(collect_keys(restored) == original_keys);
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
}
SECTION("BSON")
{
const auto bytes = ordered_json::to_bson(original);
const auto restored = ordered_json::from_bson(bytes);
CHECK(restored == original);
CHECK(collect_keys(restored) == original_keys);
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
}
SECTION("BJData")
{
const auto bytes = ordered_json::to_bjdata(original);
const auto restored = ordered_json::from_bjdata(bytes);
CHECK(restored == original);
CHECK(collect_keys(restored) == original_keys);
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
}
}
TEST_CASE("alt_json (custom string_t) across binary formats")
{
const alt_json original = make_rich_alt_json();
SECTION("CBOR")
{
const auto bytes = alt_json::to_cbor(original);
const auto restored = alt_json::from_cbor(bytes);
CHECK(restored == original);
}
SECTION("MessagePack")
{
const auto bytes = alt_json::to_msgpack(original);
const auto restored = alt_json::from_msgpack(bytes);
CHECK(restored == original);
}
SECTION("UBJSON")
{
const auto bytes = alt_json::to_ubjson(original);
const auto restored = alt_json::from_ubjson(bytes);
CHECK(restored == original);
}
SECTION("BSON")
{
const auto bytes = alt_json::to_bson(original);
const auto restored = alt_json::from_bson(bytes);
CHECK(restored == original);
}
SECTION("BJData")
{
const auto bytes = alt_json::to_bjdata(original);
const auto restored = alt_json::from_bjdata(bytes);
CHECK(restored == original);
}
}
TEST_CASE("ordered_json operator== is sensitive to key order")
{
// Unlike nlohmann::json (whose object_t is a std::map, so equality never
// depends on insertion order), ordered_json's object_t (ordered_map) is a
// std::vector<std::pair<Key, T>> under the hood, and does not define its
// own operator==: it inherits std::vector's element-wise comparison. As a
// result, two ordered_json objects holding the very same key/value pairs
// in different insertion order compare *unequal*. This is the property
// that makes the round-trip `CHECK(restored == original)` checks above a
// meaningful order-preservation check by themselves (the explicit
// collect_keys() comparisons make that check explicit/readable, and
// guard against this operator== behavior ever changing).
ordered_json a;
a["x"] = 1;
a["y"] = 2;
ordered_json b;
b["y"] = 2;
b["x"] = 1;
CHECK(a.size() == b.size());
CHECK(a["x"] == b["x"]);
CHECK(a["y"] == b["y"]);
CHECK_FALSE(a == b);
}
TEST_CASE("duplicate keys in a binary-encoded object")
{
// CBOR encoding of a map with two entries under the same key "a": {"a": 1, "a": 2}
const std::vector<std::uint8_t> cbor_bytes
{
0xA2, 0x61, 'a', 0x01, 0x61, 'a', 0x02
};
// Both json (std::map, via operator[]) and ordered_json (ordered_map, via
// operator[]) build binary-decoded objects by looking up/creating the
// entry for each incoming key and then assigning the value into it. This
// means a repeated key does *not* produce two entries in either case;
// instead, the *first* occurrence's position is kept (relevant only for
// ordered_json) while the *last* occurrence's value wins (for both) --
// this matches operator[]'s "assign the referenced slot" semantics, and
// is worth noting because it differs from the initializer-list
// construction path (`ordered_json{{"a",1},{"a",2}}`), which builds
// through insert()/emplace() and therefore keeps the *first* value, not
// the last (see the "There are no dup keys..." case in
// unit-ordered_json.cpp).
const auto j = json::from_cbor(cbor_bytes);
const auto oj = ordered_json::from_cbor(cbor_bytes);
CHECK(j.size() == 1);
CHECK(oj.size() == 1);
CHECK(j["a"] == 2);
CHECK(oj["a"] == 2);
CHECK(j == json(oj));
}
TEST_CASE("ordered_json through flatten/unflatten")
{
const ordered_json original = make_rich_ordered_json();
const std::vector<std::string> original_keys = collect_keys(original);
const std::vector<std::string> original_mango_keys = collect_keys(original["mango"]);
const ordered_json flat = original.flatten();
const ordered_json unflattened = flat.unflatten();
CHECK(unflattened == original);
// flatten() walks the value depth-first in iteration order and
// unflatten() re-inserts each flattened key via operator[] in the flat
// object's iteration order, so for ordered_json the original key order
// (both top-level and nested) is preserved end-to-end.
CHECK(collect_keys(unflattened) == original_keys);
CHECK(collect_keys(unflattened["mango"]) == original_mango_keys);
}
TEST_CASE("ordered_json through diff/patch/patch_inplace")
{
ordered_json original;
original["one"] = 1;
original["two"] = 2;
original["three"] = 3;
ordered_json target = original;
target["one"] = 100; // replace
target.erase("two"); // remove
target["four"] = 4; // add
const ordered_json patch = ordered_json::diff(original, target);
SECTION("patch")
{
const ordered_json patched = original.patch(patch);
CHECK(patched == target);
}
SECTION("patch_inplace")
{
ordered_json copy = original;
copy.patch_inplace(patch);
CHECK(copy == target);
}
}
TEST_CASE("ordered_json through merge_patch")
{
ordered_json original;
original["a"] = 1;
original["b"] = 2;
const ordered_json patch = {{"b", nullptr}, {"c", 3}};
original.merge_patch(patch);
ordered_json expected;
expected["a"] = 1;
expected["c"] = 3;
CHECK(original == expected);
CHECK(collect_keys(original) == collect_keys(expected));
}
+10
View File
@@ -18,6 +18,14 @@
// for some reason including this after the json header leads to linker errors with VS 2017...
#include <locale>
// skip tests if JSON_DisableEnumSerialization=ON (#4384): std::byte is a
// scoped enum, so get<std::byte>() (needed below to get<std::vector<std::byte>>()
// from a plain JSON array, not just from an already-binary value) relies on
// enum serialization being enabled
#if defined(JSON_DISABLE_ENUM_SERIALIZATION) && (JSON_DISABLE_ENUM_SERIALIZATION == 1)
#define SKIP_TESTS_FOR_ENUM_SERIALIZATION
#endif
#define JSON_TESTS_PRIVATE
#include <nlohmann/json.hpp>
using json = nlohmann::json;
@@ -466,6 +474,7 @@ TEST_CASE("regression tests 3")
CHECK((decoded == json_4804::array()));
}
#ifndef SKIP_TESTS_FOR_ENUM_SERIALIZATION
SECTION("discussion #4209 - custom BinaryType direct assignment and round-tripping")
{
// Test that assigning a custom BinaryType directly creates a binary value, not an array
@@ -499,6 +508,7 @@ TEST_CASE("regression tests 3")
CHECK(extracted[1] == std::byte{2});
CHECK(extracted[2] == std::byte{3});
}
#endif
SECTION("issue #5046 - implicit conversion of return json to std::optional no longer implicit")
{
+13
View File
@@ -522,6 +522,19 @@ TEST_CASE("indentation is written straight into the write buffer")
CHECK(json::parse(out) == j);
}
SECTION("binary values are indented the same way")
{
// a binary value is serialized as an object with "bytes" and
// "subtype" keys; the byte array itself is always written compactly
// (see dump_byte()), so only the surrounding object's indentation
// goes through put_indent()
const json j = json::binary({1, 2, 3}, 128);
CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"bytes\": [1, 2, 3],\n"
+ std::string(2000, ' ') + "\"subtype\": 128\n}");
CHECK(j.dump(2000, '\t') == "{\n" + std::string(2000, '\t') + "\"bytes\": [1, 2, 3],\n"
+ std::string(2000, '\t') + "\"subtype\": 128\n}");
}
SECTION("indentation is unchanged for ordinary widths")
{
const json j = {{"a", {1, 2}}, {"b", nullptr}};
+30
View File
@@ -17,6 +17,7 @@
#include <nlohmann/json.hpp>
using json = nlohmann::json;
using ordered_json = nlohmann::ordered_json;
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
#if JSON_HAS_STD_FORMAT
@@ -52,6 +53,23 @@ TEST_CASE("std::formatter<nlohmann::json>")
CHECK(std::format("{:2}", j) == j.dump(2));
CHECK(std::format("{:#2}", j) == j.dump(2));
CHECK(std::format("{:8}", j) == j.dump(8));
// multi-digit widths must accumulate every digit, not just the first
CHECK(std::format("{:12}", j) == j.dump(12));
CHECK(std::format("{:#12}", j) == j.dump(12));
CHECK(std::format("{:10}", j) == j.dump(10));
}
SECTION("bare alignment with no fill character defaults to a space indent character")
{
const json j = {{"foo", 1}, {"bar", {1, 2, 3}}};
// without a preceding fill character, the alignment character itself must not
// be mistaken for the indent character -- the default space is kept
CHECK(std::format("{:<}", j) == j.dump());
CHECK(std::format("{:>}", j) == j.dump());
CHECK(std::format("{:^}", j) == j.dump());
CHECK(std::format("{:<3}", j) == j.dump(3, ' '));
CHECK(std::format("{:>3}", j) == j.dump(3, ' '));
CHECK(std::format("{:^3}", j) == j.dump(3, ' '));
}
SECTION("fill-and-align sets the indent character, like dump(indent, indent_char)")
@@ -93,4 +111,16 @@ TEST_CASE("std::formatter<nlohmann::json>")
}
}
TEST_CASE("std::formatter<nlohmann::ordered_json>")
{
// spot-check a non-default basic_json instantiation, since the formatter
// is written against the generic NLOHMANN_BASIC_JSON_TPL_DECLARATION
// template and must actually instantiate (and behave correctly) for
// template arguments other than nlohmann::json
const ordered_json j = {{"foo", 1}, {"bar", {1, 2, 3}}};
CHECK(std::format("{}", j) == j.dump());
CHECK(std::format("{:#}", j) == j.dump(4));
CHECK(std::format("{:2}", j) == j.dump(2));
}
#endif
+2 -2
View File
@@ -306,8 +306,8 @@ TEST_CASE("Unicode (3/5)" * doctest::skip())
{
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip fourth second byte
if (0x80 <= byte3 && byte3 <= 0xBF)
// skip correct fourth byte
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
+1 -1
View File
@@ -307,7 +307,7 @@ TEST_CASE("Unicode (4/5)" * doctest::skip())
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip correct fourth byte
if (0x80 <= byte3 && byte3 <= 0xBF)
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
+1 -1
View File
@@ -307,7 +307,7 @@ TEST_CASE("Unicode (5/5)" * doctest::skip())
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip correct fourth byte
if (0x80 <= byte3 && byte3 <= 0xBF)
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}