mirror of
https://github.com/nlohmann/json.git
synced 2026-09-07 08:47:57 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2e3393b45c | ||
|
|
0236475eef |
@@ -1647,20 +1647,6 @@ class binary_writer
|
||||
return 'D'; // float 64
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief checks whether a JSON number fits into @a TargetType
|
||||
@param[in] el a JSON number of either the signed or unsigned integer kind
|
||||
@return whether @a el's value can be represented by @a TargetType without
|
||||
wrapping, regardless of which of the two kinds it is stored as
|
||||
*/
|
||||
template<typename TargetType>
|
||||
static bool bjdata_ndarray_value_in_range(const BasicJsonType& el)
|
||||
{
|
||||
return el.is_number_unsigned()
|
||||
? value_in_range_of<TargetType>(el.template get<std::uint64_t>())
|
||||
: value_in_range_of<TargetType>(el.template get<std::int64_t>());
|
||||
}
|
||||
|
||||
/*!
|
||||
@return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid
|
||||
*/
|
||||
@@ -1672,16 +1658,6 @@ class binary_writer
|
||||
};
|
||||
|
||||
string_t key = "_ArrayType_";
|
||||
// the type name is looked up as a string below; a non-string
|
||||
// annotation (e.g. a number, null, or an array) cannot name a known
|
||||
// dtype, so it is treated the same as an unrecognized type name and
|
||||
// falls back to a plain object encoding instead of throwing
|
||||
// type_error.302 out of get<string_t>()
|
||||
if (!value.at(key).is_string())
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// use get<string_t>() instead of static_cast<string_t> to avoid an
|
||||
// ambiguous conversion under explicit instantiation on C++17 (see #4825)
|
||||
auto it = bjdtype.find(value.at(key).template get<string_t>());
|
||||
@@ -1691,16 +1667,6 @@ class binary_writer
|
||||
}
|
||||
CharType dtype = it->second;
|
||||
|
||||
// the 'B' (byte) marker is only defined by BJData Draft 3; emitting it
|
||||
// under the default Draft 2 mode would produce a stream that Draft 2
|
||||
// readers reject, so such an object falls back to a plain object
|
||||
// encoding instead (see the "Binary values" section of the BJData
|
||||
// documentation)
|
||||
if (dtype == 'B' && bjdata_version != bjdata_version_t::draft3)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
key = "_ArraySize_";
|
||||
// the dimensions are written verbatim as the header length below, so a
|
||||
// value that is not an array cannot produce a valid one: null emits 'Z'
|
||||
@@ -1765,60 +1731,6 @@ class binary_writer
|
||||
}
|
||||
}
|
||||
|
||||
// every element is cast to the (possibly narrower) C++ type matching
|
||||
// dtype below; a value that does not fit that type would silently
|
||||
// wrap (integers) or overflow to infinity (the "single" precision
|
||||
// float) instead of being reported, so such an object falls back to
|
||||
// a plain object encoding as well
|
||||
for (const auto& el : value.at(key))
|
||||
{
|
||||
bool in_range = true;
|
||||
switch (dtype)
|
||||
{
|
||||
case 'U':
|
||||
case 'C':
|
||||
case 'B':
|
||||
in_range = bjdata_ndarray_value_in_range<std::uint8_t>(el);
|
||||
break;
|
||||
case 'i':
|
||||
in_range = bjdata_ndarray_value_in_range<std::int8_t>(el);
|
||||
break;
|
||||
case 'u':
|
||||
in_range = bjdata_ndarray_value_in_range<std::uint16_t>(el);
|
||||
break;
|
||||
case 'I':
|
||||
in_range = bjdata_ndarray_value_in_range<std::int16_t>(el);
|
||||
break;
|
||||
case 'm':
|
||||
in_range = bjdata_ndarray_value_in_range<std::uint32_t>(el);
|
||||
break;
|
||||
case 'l':
|
||||
in_range = bjdata_ndarray_value_in_range<std::int32_t>(el);
|
||||
break;
|
||||
case 'M':
|
||||
in_range = bjdata_ndarray_value_in_range<std::uint64_t>(el);
|
||||
break;
|
||||
case 'L':
|
||||
in_range = bjdata_ndarray_value_in_range<std::int64_t>(el);
|
||||
break;
|
||||
case 'd':
|
||||
{
|
||||
const auto dval = el.template get<double>();
|
||||
in_range = !std::isfinite(dval) ||
|
||||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
|
||||
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
// 'D' (double) already spans the full range of number_float_t
|
||||
break;
|
||||
}
|
||||
if (!in_range)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
oa->write_character('[');
|
||||
oa->write_character('$');
|
||||
oa->write_character(dtype);
|
||||
|
||||
@@ -18655,20 +18655,6 @@ class binary_writer
|
||||
return 'D'; // float 64
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief checks whether a JSON number fits into @a TargetType
|
||||
@param[in] el a JSON number of either the signed or unsigned integer kind
|
||||
@return whether @a el's value can be represented by @a TargetType without
|
||||
wrapping, regardless of which of the two kinds it is stored as
|
||||
*/
|
||||
template<typename TargetType>
|
||||
static bool bjdata_ndarray_value_in_range(const BasicJsonType& el)
|
||||
{
|
||||
return el.is_number_unsigned()
|
||||
? value_in_range_of<TargetType>(el.template get<std::uint64_t>())
|
||||
: value_in_range_of<TargetType>(el.template get<std::int64_t>());
|
||||
}
|
||||
|
||||
/*!
|
||||
@return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid
|
||||
*/
|
||||
@@ -18680,16 +18666,6 @@ class binary_writer
|
||||
};
|
||||
|
||||
string_t key = "_ArrayType_";
|
||||
// the type name is looked up as a string below; a non-string
|
||||
// annotation (e.g. a number, null, or an array) cannot name a known
|
||||
// dtype, so it is treated the same as an unrecognized type name and
|
||||
// falls back to a plain object encoding instead of throwing
|
||||
// type_error.302 out of get<string_t>()
|
||||
if (!value.at(key).is_string())
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// use get<string_t>() instead of static_cast<string_t> to avoid an
|
||||
// ambiguous conversion under explicit instantiation on C++17 (see #4825)
|
||||
auto it = bjdtype.find(value.at(key).template get<string_t>());
|
||||
@@ -18699,16 +18675,6 @@ class binary_writer
|
||||
}
|
||||
CharType dtype = it->second;
|
||||
|
||||
// the 'B' (byte) marker is only defined by BJData Draft 3; emitting it
|
||||
// under the default Draft 2 mode would produce a stream that Draft 2
|
||||
// readers reject, so such an object falls back to a plain object
|
||||
// encoding instead (see the "Binary values" section of the BJData
|
||||
// documentation)
|
||||
if (dtype == 'B' && bjdata_version != bjdata_version_t::draft3)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
key = "_ArraySize_";
|
||||
// the dimensions are written verbatim as the header length below, so a
|
||||
// value that is not an array cannot produce a valid one: null emits 'Z'
|
||||
@@ -18773,60 +18739,6 @@ class binary_writer
|
||||
}
|
||||
}
|
||||
|
||||
// every element is cast to the (possibly narrower) C++ type matching
|
||||
// dtype below; a value that does not fit that type would silently
|
||||
// wrap (integers) or overflow to infinity (the "single" precision
|
||||
// float) instead of being reported, so such an object falls back to
|
||||
// a plain object encoding as well
|
||||
for (const auto& el : value.at(key))
|
||||
{
|
||||
bool in_range = true;
|
||||
switch (dtype)
|
||||
{
|
||||
case 'U':
|
||||
case 'C':
|
||||
case 'B':
|
||||
in_range = bjdata_ndarray_value_in_range<std::uint8_t>(el);
|
||||
break;
|
||||
case 'i':
|
||||
in_range = bjdata_ndarray_value_in_range<std::int8_t>(el);
|
||||
break;
|
||||
case 'u':
|
||||
in_range = bjdata_ndarray_value_in_range<std::uint16_t>(el);
|
||||
break;
|
||||
case 'I':
|
||||
in_range = bjdata_ndarray_value_in_range<std::int16_t>(el);
|
||||
break;
|
||||
case 'm':
|
||||
in_range = bjdata_ndarray_value_in_range<std::uint32_t>(el);
|
||||
break;
|
||||
case 'l':
|
||||
in_range = bjdata_ndarray_value_in_range<std::int32_t>(el);
|
||||
break;
|
||||
case 'M':
|
||||
in_range = bjdata_ndarray_value_in_range<std::uint64_t>(el);
|
||||
break;
|
||||
case 'L':
|
||||
in_range = bjdata_ndarray_value_in_range<std::int64_t>(el);
|
||||
break;
|
||||
case 'd':
|
||||
{
|
||||
const auto dval = el.template get<double>();
|
||||
in_range = !std::isfinite(dval) ||
|
||||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
|
||||
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
|
||||
break;
|
||||
}
|
||||
default:
|
||||
// 'D' (double) already spans the full range of number_float_t
|
||||
break;
|
||||
}
|
||||
if (!in_range)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
oa->write_character('[');
|
||||
oa->write_character('$');
|
||||
oa->write_character(dtype);
|
||||
|
||||
@@ -21,27 +21,6 @@ array data, it performs the following steps:
|
||||
- j4 = from_bjdata(vec3)
|
||||
- assert(j1 == j4)
|
||||
|
||||
Re-serializing j2/j3/j4 with the same use_size/use_type settings is checked
|
||||
for value-stability rather than byte-exact stability: from_bjdata(to_bjdata(j2))
|
||||
must equal j2 (and likewise for j3, j4). Byte-exact stability does not hold in
|
||||
general, because a BJData value can lose type fidelity across a round trip
|
||||
(e.g. a binary_t value serialized without the optimized "$U#" array header is
|
||||
parsed back as a plain array of numbers, see #5398 and the discussion on
|
||||
PR #5494) - the numeric value is preserved, but the writer's smallest-type
|
||||
selection for the now-plain numbers may legitimately pick a different, but
|
||||
equally valid, single-byte type marker than the dedicated binary-data writer
|
||||
would have. Both encodings are valid BJData and both decode to the same
|
||||
value, so this is not treated as a round-trip failure here.
|
||||
|
||||
"Value-stable" is checked by comparing dump()s rather than with operator==
|
||||
directly: a BJData/UBJSON payload can decode to a non-finite double (NaN or
|
||||
+-Infinity), and IEEE 754 NaN is never equal to itself, so operator== would
|
||||
report two structurally-identical trees as different whenever a NaN is
|
||||
involved -- not a round-trip bug, just NaN's ordinary (non-)reflexivity.
|
||||
dump() serializes any non-finite double the same deterministic way (as JSON
|
||||
`null`, since JSON itself cannot represent NaN/Infinity), so comparing
|
||||
dumps is stable under exactly the same values that break operator==.
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -52,13 +31,6 @@ drivers.
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// value-stable comparison for the round-trip checks below; see the note
|
||||
// above on why this compares dump()s rather than the json values directly
|
||||
static bool is_value_stable(const json& lhs, const json& rhs)
|
||||
{
|
||||
return lhs.dump() == rhs.dump();
|
||||
}
|
||||
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
@@ -84,12 +56,10 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
json const j3 = json::from_bjdata(vec3);
|
||||
json const j4 = json::from_bjdata(vec4);
|
||||
|
||||
// re-serializing must be value-stable (see the notes above on
|
||||
// why byte-exact stability is not guaranteed in general, and
|
||||
// why this compares dump()s rather than the values directly)
|
||||
assert(is_value_stable(json::from_bjdata(json::to_bjdata(j2, false, false)), j2));
|
||||
assert(is_value_stable(json::from_bjdata(json::to_bjdata(j3, true, false)), j3));
|
||||
assert(is_value_stable(json::from_bjdata(json::to_bjdata(j4, true, true)), j4));
|
||||
// serializations must match
|
||||
assert(json::to_bjdata(j2, false, false) == vec2);
|
||||
assert(json::to_bjdata(j3, true, false) == vec3);
|
||||
assert(json::to_bjdata(j4, true, true) == vec4);
|
||||
}
|
||||
catch (const json::parse_error&)
|
||||
{
|
||||
|
||||
+2
-171
@@ -2586,12 +2586,7 @@ TEST_CASE("BJData")
|
||||
CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d);
|
||||
CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D);
|
||||
CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C);
|
||||
// v_B uses the Draft-3-only 'B' marker, so it round-trips only when
|
||||
// Draft 3 is explicitly selected (see GitHub issue #5404); the
|
||||
// default Draft 2 falls back to a plain object instead, covered by
|
||||
// the "ndarray with _ArrayType_ "byte" is gated by the BJData draft
|
||||
// version" section below
|
||||
CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true, json::bjdata_version_t::draft3) == v_B);
|
||||
CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true) == v_B);
|
||||
}
|
||||
|
||||
SECTION("ndarray with data not matching _ArrayType_ is written as an object")
|
||||
@@ -2634,10 +2629,8 @@ TEST_CASE("BJData")
|
||||
// the C++ API stores an int literal as number_integer, so _ArrayType_
|
||||
// names the wire type rather than the storage. Both storages have to
|
||||
// produce the same typed array for every type.
|
||||
// "byte" is checked separately below since it additionally requires
|
||||
// BJData Draft 3 to be selected explicitly (see GitHub issue #5404).
|
||||
for (const char* type :
|
||||
{"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char"
|
||||
{"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char", "byte"
|
||||
})
|
||||
{
|
||||
CAPTURE(type);
|
||||
@@ -2648,14 +2641,6 @@ TEST_CASE("BJData")
|
||||
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}})));
|
||||
}
|
||||
|
||||
{
|
||||
const std::string text = R"({"_ArrayType_":"byte","_ArraySize_":[2,3],"_ArrayData_":[1,2,3,4,5,6]})";
|
||||
const auto from_text = json::to_bjdata(json::parse(text), true, true, json::bjdata_version_t::draft3);
|
||||
CHECK(from_text.at(0) == '[');
|
||||
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}),
|
||||
true, true, json::bjdata_version_t::draft3));
|
||||
}
|
||||
|
||||
// negative values under a signed type behave the same way
|
||||
const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})"));
|
||||
CHECK(from_neg.at(0) == '[');
|
||||
@@ -2746,83 +2731,6 @@ TEST_CASE("BJData")
|
||||
CHECK(json::from_bjdata(json::to_bjdata(j_size), true, true) == j_size);
|
||||
}
|
||||
|
||||
SECTION("ndarray whose _ArrayType_ is not a string stays as object")
|
||||
{
|
||||
// the type name is looked up as a string below the annotation
|
||||
// check; a non-string _ArrayType_ cannot name a known dtype,
|
||||
// so calling get<string_t>() on it would throw type_error.302
|
||||
// instead of falling back like an unrecognized type name
|
||||
// already does (see GitHub issue #5398)
|
||||
json const j_number = json({{"_ArrayType_", 1}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
|
||||
const auto out_number = json::to_bjdata(j_number);
|
||||
CHECK(out_number.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_number) == j_number);
|
||||
|
||||
json const j_null = json({{"_ArrayType_", nullptr}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
|
||||
const auto out_null = json::to_bjdata(j_null);
|
||||
CHECK(out_null.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_null) == j_null);
|
||||
|
||||
json const j_bool = json({{"_ArrayType_", true}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
|
||||
const auto out_bool = json::to_bjdata(j_bool);
|
||||
CHECK(out_bool.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_bool) == j_bool);
|
||||
|
||||
json const j_array = json({{"_ArrayType_", {"uint8"}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
|
||||
const auto out_array = json::to_bjdata(j_array);
|
||||
CHECK(out_array.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_array) == j_array);
|
||||
|
||||
json const j_object = json({{"_ArrayType_", {{"a", 1}}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}});
|
||||
const auto out_object = json::to_bjdata(j_object);
|
||||
CHECK(out_object.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_object) == j_object);
|
||||
}
|
||||
|
||||
SECTION("re-serializing a value containing a plain-array-of-bytes is value-stable but not byte-stable")
|
||||
{
|
||||
// OSS-Fuzz found this input (an array whose first element is a
|
||||
// binary_t byte, followed by an object whose _ArrayType_ is
|
||||
// not a string) while exercising the fix for #5398 above: once
|
||||
// the fix stops to_bjdata() from throwing type_error.302 for
|
||||
// the third element, serialization proceeds far enough to
|
||||
// reach a pre-existing, unrelated round-trip quirk in how a
|
||||
// single-byte binary_t value is re-encoded.
|
||||
std::vector<std::uint8_t> const input
|
||||
{
|
||||
0x5b, 0x5b, 0x24, 0x42, 0x23, 0x5b, 0x69, 0x01, 0x5d, 0x5b, 0x5b, 0x5d, 0x7b, 0x55, 0x0b,
|
||||
0x5f, 0x41, 0x72, 0x72, 0x61, 0x79, 0x44, 0x61, 0x74, 0x61, 0x5f, 0x54, 0x55, 0x0b, 0x5f,
|
||||
0x41, 0x72, 0x72, 0x61, 0x79, 0x53, 0x69, 0x7a, 0x65, 0x5f, 0x5a, 0x55, 0x0b, 0x5f, 0x41,
|
||||
0x72, 0x72, 0x61, 0x79, 0x54, 0x79, 0x70, 0x65, 0x5f, 0x54, 0x7d, 0x5d
|
||||
};
|
||||
json const j1 = json::from_bjdata(input);
|
||||
|
||||
// to_bjdata() must not throw (this is what #5398 fixes)
|
||||
std::vector<std::uint8_t> vec2;
|
||||
CHECK_NOTHROW(vec2 = json::to_bjdata(j1, false, false));
|
||||
|
||||
// parsing back a plain (non-optimized) array of bytes cannot
|
||||
// recover that it used to be a binary_t: from_bjdata() has no
|
||||
// way to distinguish "array of uint8 numbers" from "array of
|
||||
// bytes" unless the compact "$U#" array header is used, so
|
||||
// the binary_t collapses into a plain JSON array
|
||||
json const j2 = json::from_bjdata(vec2);
|
||||
CHECK(j1 != j2);
|
||||
CHECK(j2 == json({{91}, json::array(), {{"_ArrayData_", true}, {"_ArraySize_", nullptr}, {"_ArrayType_", true}}}));
|
||||
|
||||
// re-serializing j2 no longer goes through the dedicated
|
||||
// binary_t writer (which always uses the 'U' marker for raw
|
||||
// bytes); the now-plain number 91 goes through the generic
|
||||
// smallest-type writer instead, which - like the rest of the
|
||||
// UBJSON/BJData writer, and unchanged by this fix - prefers
|
||||
// the 'i' (int8) marker over 'U' (uint8) for values that fit
|
||||
// both. Both markers are valid BJData and both decode back to
|
||||
// 91, so this is not byte-for-byte identical to vec2, but it
|
||||
// is value-stable: parsing it again reproduces j2 exactly.
|
||||
std::vector<std::uint8_t> const vec3 = json::to_bjdata(j2, false, false);
|
||||
CHECK(json::from_bjdata(vec3) == j2);
|
||||
}
|
||||
|
||||
SECTION("ndarray whose dimensions overflow stays as object")
|
||||
{
|
||||
// the product of the dimensions wraps around std::size_t to 0
|
||||
@@ -2868,83 +2776,6 @@ TEST_CASE("BJData")
|
||||
CHECK(out_num.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_num) == j_num);
|
||||
}
|
||||
|
||||
SECTION("ndarray with out-of-range _ArrayData_ elements stays as object")
|
||||
{
|
||||
// each element is cast to the (possibly narrower) C++ type
|
||||
// named by _ArrayType_ before being written; a value that
|
||||
// does not fit that type would silently wrap instead of
|
||||
// being reported, so such an object falls back to a plain
|
||||
// object encoding that still round-trips (see GitHub issue #5403)
|
||||
|
||||
// an unsigned element that does not fit uint8
|
||||
json const j_uint8 = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 256}}});
|
||||
const auto out_uint8 = json::to_bjdata(j_uint8);
|
||||
CHECK(out_uint8.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_uint8) == j_uint8);
|
||||
|
||||
// a signed element that does not fit int8
|
||||
json const j_int8 = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 200}}});
|
||||
const auto out_int8 = json::to_bjdata(j_int8);
|
||||
CHECK(out_int8.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_int8) == j_int8);
|
||||
|
||||
// a negative element is likewise out of range for an
|
||||
// unsigned _ArrayType_
|
||||
json const j_uint16_neg = json({{"_ArrayType_", "uint16"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, -1}}});
|
||||
const auto out_uint16_neg = json::to_bjdata(j_uint16_neg);
|
||||
CHECK(out_uint16_neg.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_uint16_neg) == j_uint16_neg);
|
||||
|
||||
// a double element that overflows to infinity when narrowed
|
||||
// to the "single" (float) precision named by _ArrayType_
|
||||
json const j_single = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1.5, 1e40}}});
|
||||
const auto out_single = json::to_bjdata(j_single);
|
||||
CHECK(out_single.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_single) == j_single);
|
||||
|
||||
// in-range boundary values still use the compact ndarray encoding
|
||||
json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {0, 255}}});
|
||||
CHECK(json::to_bjdata(j_uint8_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, ']', 0, 255}));
|
||||
|
||||
json const j_int8_ok = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {-128, 127}}});
|
||||
CHECK(json::to_bjdata(j_int8_ok) == std::vector<uint8_t>({'[', '$', 'i', '#', '[', 'i', 2, ']', 0x80, 0x7F}));
|
||||
|
||||
json const j_single_ok = json({{"_ArrayType_", "single"}, {"_ArraySize_", {1}}, {"_ArrayData_", {1.5}}});
|
||||
const auto out_single_ok = json::to_bjdata(j_single_ok);
|
||||
CHECK(out_single_ok.at(0) == '[');
|
||||
CHECK(json::from_bjdata(out_single_ok) == json({1.5f}));
|
||||
}
|
||||
|
||||
SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version")
|
||||
{
|
||||
// the 'B' (byte) marker used by _ArrayType_ "byte" is only defined
|
||||
// by BJData Draft 3; Draft 2 (the default) has no such marker, so
|
||||
// emitting it unconditionally produced a stream that a Draft 2
|
||||
// reader could not parse as intended (see GitHub issue #5404).
|
||||
// Two dimensions are used so that a successfully written ndarray
|
||||
// round-trips back into the annotated object (a single dimension
|
||||
// is, by the BJData ndarray convention, read back as a plain
|
||||
// binary value rather than the annotated object, same as every
|
||||
// other single-dimension ndarray of a non-"byte" type is read
|
||||
// back as a plain array instead of the annotated object).
|
||||
json const j_byte = json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}});
|
||||
|
||||
// default (Draft 2): falls back to a plain object and round-trips
|
||||
const auto out_draft2 = json::to_bjdata(j_byte);
|
||||
CHECK(out_draft2.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_draft2) == j_byte);
|
||||
|
||||
// explicit Draft 2: same as the default
|
||||
const auto out_draft2_explicit = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft2);
|
||||
CHECK(out_draft2_explicit.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_draft2_explicit) == j_byte);
|
||||
|
||||
// Draft 3 explicitly selected: still uses the compact 'B' ndarray encoding
|
||||
const auto out_draft3 = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft3);
|
||||
CHECK(out_draft3 == std::vector<uint8_t>({'[', '$', 'B', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6}));
|
||||
CHECK(json::from_bjdata(out_draft3) == j_byte);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,489 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-FileCopyrightText: 2018 Vitaliy Manushkin <agri@akamo.info>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
// This file closes a test-coverage gap described in GitHub issue #5421:
|
||||
// nlohmann::ordered_json (and other non-default basic_json specializations,
|
||||
// such as the alt_string-based one from unit-alt-string.cpp) were never
|
||||
// exercised through the binary formats (CBOR/MessagePack/UBJSON/BSON/BJData)
|
||||
// or through flatten()/unflatten()/diff()/patch()/merge_patch().
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
using nlohmann::json;
|
||||
using nlohmann::ordered_json;
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////
|
||||
// alt_json: a second, independent copy of the custom-string_t basic_json
|
||||
// specialization defined in unit-alt-string.cpp.
|
||||
//
|
||||
// It is duplicated here (rather than shared via a header) because every
|
||||
// unit-*.cpp file in this test suite is compiled into its own standalone
|
||||
// executable (see tests/CMakeLists.txt), so there is no ODR concern in
|
||||
// having the same class name defined in multiple translation units.
|
||||
//
|
||||
// Two members had to be added relative to the original alt_string
|
||||
// (a constructor from std::string, and a find(char, pos) overload) because
|
||||
// the original type was never used with the binary writers/readers before
|
||||
// this file: BSON's array/document writer converts std::to_string() results
|
||||
// and checks for embedded NUL characters via find(char), and the UBJSON/BSON
|
||||
// high-precision-number path constructs the SAX string_t argument from a
|
||||
// std::string. Neither path is exercised anywhere else in the test suite for
|
||||
// this type, which is presumably why the gap was never noticed.
|
||||
/////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
class alt_string;
|
||||
bool operator<(const char* op1, const alt_string& op2) noexcept; // NOLINT(misc-use-internal-linkage)
|
||||
void int_to_string(alt_string& target, std::size_t value); // NOLINT(misc-use-internal-linkage)
|
||||
|
||||
class alt_string
|
||||
{
|
||||
public:
|
||||
using value_type = std::string::value_type;
|
||||
|
||||
static constexpr auto npos = (std::numeric_limits<std::size_t>::max)();
|
||||
|
||||
alt_string(const char* str): str_impl(str) {}
|
||||
alt_string(const char* str, std::size_t count): str_impl(str, count) {}
|
||||
alt_string(std::string str): str_impl(std::move(str)) {}
|
||||
alt_string(size_t count, char chr): str_impl(count, chr) {}
|
||||
alt_string() = default;
|
||||
|
||||
alt_string& append(char ch)
|
||||
{
|
||||
str_impl.push_back(ch);
|
||||
return *this;
|
||||
}
|
||||
|
||||
alt_string& append(const alt_string& str)
|
||||
{
|
||||
str_impl.append(str.str_impl);
|
||||
return *this;
|
||||
}
|
||||
|
||||
alt_string& append(const char* s, std::size_t length)
|
||||
{
|
||||
str_impl.append(s, length);
|
||||
return *this;
|
||||
}
|
||||
|
||||
void push_back(char c)
|
||||
{
|
||||
str_impl.push_back(c);
|
||||
}
|
||||
|
||||
template <typename op_type>
|
||||
bool operator==(const op_type& op) const
|
||||
{
|
||||
return str_impl == op;
|
||||
}
|
||||
|
||||
bool operator==(const alt_string& op) const
|
||||
{
|
||||
return str_impl == op.str_impl;
|
||||
}
|
||||
|
||||
template <typename op_type>
|
||||
bool operator!=(const op_type& op) const
|
||||
{
|
||||
return str_impl != op;
|
||||
}
|
||||
|
||||
bool operator!=(const alt_string& op) const
|
||||
{
|
||||
return str_impl != op.str_impl;
|
||||
}
|
||||
|
||||
std::size_t size() const noexcept
|
||||
{
|
||||
return str_impl.size();
|
||||
}
|
||||
|
||||
void resize(std::size_t n)
|
||||
{
|
||||
str_impl.resize(n);
|
||||
}
|
||||
|
||||
void resize(std::size_t n, char c)
|
||||
{
|
||||
str_impl.resize(n, c);
|
||||
}
|
||||
|
||||
template <typename op_type>
|
||||
bool operator<(const op_type& op) const noexcept
|
||||
{
|
||||
return str_impl < op;
|
||||
}
|
||||
|
||||
bool operator<(const alt_string& op) const noexcept
|
||||
{
|
||||
return str_impl < op.str_impl;
|
||||
}
|
||||
|
||||
const char* c_str() const
|
||||
{
|
||||
return str_impl.c_str();
|
||||
}
|
||||
|
||||
char& operator[](std::size_t index)
|
||||
{
|
||||
return str_impl[index];
|
||||
}
|
||||
|
||||
const char& operator[](std::size_t index) const
|
||||
{
|
||||
return str_impl[index];
|
||||
}
|
||||
|
||||
char& back()
|
||||
{
|
||||
return str_impl.back();
|
||||
}
|
||||
|
||||
const char& back() const
|
||||
{
|
||||
return str_impl.back();
|
||||
}
|
||||
|
||||
void clear()
|
||||
{
|
||||
str_impl.clear();
|
||||
}
|
||||
|
||||
const value_type* data() const
|
||||
{
|
||||
return str_impl.data();
|
||||
}
|
||||
|
||||
bool empty() const
|
||||
{
|
||||
return str_impl.empty();
|
||||
}
|
||||
|
||||
std::size_t find(const alt_string& str, std::size_t pos = 0) const
|
||||
{
|
||||
return str_impl.find(str.str_impl, pos);
|
||||
}
|
||||
|
||||
// needed by binary_writer's BSON support, which probes string keys for
|
||||
// embedded NUL characters via find(char)
|
||||
std::size_t find(char c, std::size_t pos = 0) const
|
||||
{
|
||||
return str_impl.find(c, pos);
|
||||
}
|
||||
|
||||
std::size_t find_first_of(char c, std::size_t pos = 0) const
|
||||
{
|
||||
return str_impl.find_first_of(c, pos);
|
||||
}
|
||||
|
||||
alt_string substr(std::size_t pos = 0, std::size_t count = npos) const
|
||||
{
|
||||
const std::string s = str_impl.substr(pos, count);
|
||||
return {s.data(), s.size()};
|
||||
}
|
||||
|
||||
alt_string& replace(std::size_t pos, std::size_t count, const alt_string& str)
|
||||
{
|
||||
str_impl.replace(pos, count, str.str_impl);
|
||||
return *this;
|
||||
}
|
||||
|
||||
void reserve(std::size_t new_cap = 0)
|
||||
{
|
||||
str_impl.reserve(new_cap);
|
||||
}
|
||||
|
||||
private:
|
||||
std::string str_impl {}; // NOLINT(readability-redundant-member-init)
|
||||
|
||||
friend bool operator<(const char* /*op1*/, const alt_string& /*op2*/) noexcept;
|
||||
};
|
||||
|
||||
void int_to_string(alt_string& target, std::size_t value)
|
||||
{
|
||||
target = std::to_string(value).c_str();
|
||||
}
|
||||
|
||||
using alt_json = nlohmann::basic_json <
|
||||
std::map,
|
||||
std::vector,
|
||||
alt_string,
|
||||
bool,
|
||||
std::int64_t,
|
||||
std::uint64_t,
|
||||
double,
|
||||
std::allocator,
|
||||
nlohmann::adl_serializer >;
|
||||
|
||||
bool operator<(const char* op1, const alt_string& op2) noexcept
|
||||
{
|
||||
return op1 < op2.str_impl;
|
||||
}
|
||||
|
||||
namespace
|
||||
{
|
||||
|
||||
// collects the object keys of j, in iteration order
|
||||
std::vector<std::string> collect_keys(const ordered_json& j)
|
||||
{
|
||||
std::vector<std::string> result;
|
||||
for (auto it = j.cbegin(); it != j.cend(); ++it)
|
||||
{
|
||||
result.push_back(it.key());
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// a nested object/array value with keys inserted in non-alphabetical order,
|
||||
// used to check both round-trip equality and (for ordered_json) that
|
||||
// insertion order survives a trip through a binary format
|
||||
ordered_json make_rich_ordered_json()
|
||||
{
|
||||
ordered_json j;
|
||||
j["zebra"] = 1;
|
||||
j["apple"] = ordered_json::array({1, 2, 3});
|
||||
j["mango"]["z_nested"] = true;
|
||||
j["mango"]["a_nested"] = nullptr;
|
||||
j["banana"] = "some text";
|
||||
j["cherry"] = 3.14;
|
||||
return j;
|
||||
}
|
||||
|
||||
alt_json make_rich_alt_json()
|
||||
{
|
||||
alt_json j;
|
||||
j["zebra"] = 1;
|
||||
j["apple"] = alt_json::array({1, 2, 3});
|
||||
j["mango"]["z_nested"] = true;
|
||||
j["mango"]["a_nested"] = nullptr;
|
||||
j["banana"] = "some text";
|
||||
j["cherry"] = 3.14;
|
||||
return j;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("ordered_json across binary formats")
|
||||
{
|
||||
const ordered_json original = make_rich_ordered_json();
|
||||
const std::vector<std::string> original_keys = collect_keys(original);
|
||||
const std::vector<std::string> original_mango_keys = collect_keys(original["mango"]);
|
||||
|
||||
SECTION("CBOR")
|
||||
{
|
||||
const auto bytes = ordered_json::to_cbor(original);
|
||||
const auto restored = ordered_json::from_cbor(bytes);
|
||||
CHECK(restored == original);
|
||||
CHECK(collect_keys(restored) == original_keys);
|
||||
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
|
||||
}
|
||||
|
||||
SECTION("MessagePack")
|
||||
{
|
||||
const auto bytes = ordered_json::to_msgpack(original);
|
||||
const auto restored = ordered_json::from_msgpack(bytes);
|
||||
CHECK(restored == original);
|
||||
CHECK(collect_keys(restored) == original_keys);
|
||||
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
|
||||
}
|
||||
|
||||
SECTION("UBJSON")
|
||||
{
|
||||
const auto bytes = ordered_json::to_ubjson(original);
|
||||
const auto restored = ordered_json::from_ubjson(bytes);
|
||||
CHECK(restored == original);
|
||||
CHECK(collect_keys(restored) == original_keys);
|
||||
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
|
||||
}
|
||||
|
||||
SECTION("BSON")
|
||||
{
|
||||
const auto bytes = ordered_json::to_bson(original);
|
||||
const auto restored = ordered_json::from_bson(bytes);
|
||||
CHECK(restored == original);
|
||||
CHECK(collect_keys(restored) == original_keys);
|
||||
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
|
||||
}
|
||||
|
||||
SECTION("BJData")
|
||||
{
|
||||
const auto bytes = ordered_json::to_bjdata(original);
|
||||
const auto restored = ordered_json::from_bjdata(bytes);
|
||||
CHECK(restored == original);
|
||||
CHECK(collect_keys(restored) == original_keys);
|
||||
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("alt_json (custom string_t) across binary formats")
|
||||
{
|
||||
const alt_json original = make_rich_alt_json();
|
||||
|
||||
SECTION("CBOR")
|
||||
{
|
||||
const auto bytes = alt_json::to_cbor(original);
|
||||
const auto restored = alt_json::from_cbor(bytes);
|
||||
CHECK(restored == original);
|
||||
}
|
||||
|
||||
SECTION("MessagePack")
|
||||
{
|
||||
const auto bytes = alt_json::to_msgpack(original);
|
||||
const auto restored = alt_json::from_msgpack(bytes);
|
||||
CHECK(restored == original);
|
||||
}
|
||||
|
||||
SECTION("UBJSON")
|
||||
{
|
||||
const auto bytes = alt_json::to_ubjson(original);
|
||||
const auto restored = alt_json::from_ubjson(bytes);
|
||||
CHECK(restored == original);
|
||||
}
|
||||
|
||||
SECTION("BSON")
|
||||
{
|
||||
const auto bytes = alt_json::to_bson(original);
|
||||
const auto restored = alt_json::from_bson(bytes);
|
||||
CHECK(restored == original);
|
||||
}
|
||||
|
||||
SECTION("BJData")
|
||||
{
|
||||
const auto bytes = alt_json::to_bjdata(original);
|
||||
const auto restored = alt_json::from_bjdata(bytes);
|
||||
CHECK(restored == original);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("ordered_json operator== is sensitive to key order")
|
||||
{
|
||||
// Unlike nlohmann::json (whose object_t is a std::map, so equality never
|
||||
// depends on insertion order), ordered_json's object_t (ordered_map) is a
|
||||
// std::vector<std::pair<Key, T>> under the hood, and does not define its
|
||||
// own operator==: it inherits std::vector's element-wise comparison. As a
|
||||
// result, two ordered_json objects holding the very same key/value pairs
|
||||
// in different insertion order compare *unequal*. This is the property
|
||||
// that makes the round-trip `CHECK(restored == original)` checks above a
|
||||
// meaningful order-preservation check by themselves (the explicit
|
||||
// collect_keys() comparisons make that check explicit/readable, and
|
||||
// guard against this operator== behavior ever changing).
|
||||
ordered_json a;
|
||||
a["x"] = 1;
|
||||
a["y"] = 2;
|
||||
|
||||
ordered_json b;
|
||||
b["y"] = 2;
|
||||
b["x"] = 1;
|
||||
|
||||
CHECK(a.size() == b.size());
|
||||
CHECK(a["x"] == b["x"]);
|
||||
CHECK(a["y"] == b["y"]);
|
||||
CHECK_FALSE(a == b);
|
||||
}
|
||||
|
||||
TEST_CASE("duplicate keys in a binary-encoded object")
|
||||
{
|
||||
// CBOR encoding of a map with two entries under the same key "a": {"a": 1, "a": 2}
|
||||
const std::vector<std::uint8_t> cbor_bytes
|
||||
{
|
||||
0xA2, 0x61, 'a', 0x01, 0x61, 'a', 0x02
|
||||
};
|
||||
|
||||
// Both json (std::map, via operator[]) and ordered_json (ordered_map, via
|
||||
// operator[]) build binary-decoded objects by looking up/creating the
|
||||
// entry for each incoming key and then assigning the value into it. This
|
||||
// means a repeated key does *not* produce two entries in either case;
|
||||
// instead, the *first* occurrence's position is kept (relevant only for
|
||||
// ordered_json) while the *last* occurrence's value wins (for both) --
|
||||
// this matches operator[]'s "assign the referenced slot" semantics, and
|
||||
// is worth noting because it differs from the initializer-list
|
||||
// construction path (`ordered_json{{"a",1},{"a",2}}`), which builds
|
||||
// through insert()/emplace() and therefore keeps the *first* value, not
|
||||
// the last (see the "There are no dup keys..." case in
|
||||
// unit-ordered_json.cpp).
|
||||
const auto j = json::from_cbor(cbor_bytes);
|
||||
const auto oj = ordered_json::from_cbor(cbor_bytes);
|
||||
|
||||
CHECK(j.size() == 1);
|
||||
CHECK(oj.size() == 1);
|
||||
CHECK(j["a"] == 2);
|
||||
CHECK(oj["a"] == 2);
|
||||
CHECK(j == json(oj));
|
||||
}
|
||||
|
||||
TEST_CASE("ordered_json through flatten/unflatten")
|
||||
{
|
||||
const ordered_json original = make_rich_ordered_json();
|
||||
const std::vector<std::string> original_keys = collect_keys(original);
|
||||
const std::vector<std::string> original_mango_keys = collect_keys(original["mango"]);
|
||||
|
||||
const ordered_json flat = original.flatten();
|
||||
const ordered_json unflattened = flat.unflatten();
|
||||
|
||||
CHECK(unflattened == original);
|
||||
// flatten() walks the value depth-first in iteration order and
|
||||
// unflatten() re-inserts each flattened key via operator[] in the flat
|
||||
// object's iteration order, so for ordered_json the original key order
|
||||
// (both top-level and nested) is preserved end-to-end.
|
||||
CHECK(collect_keys(unflattened) == original_keys);
|
||||
CHECK(collect_keys(unflattened["mango"]) == original_mango_keys);
|
||||
}
|
||||
|
||||
TEST_CASE("ordered_json through diff/patch/patch_inplace")
|
||||
{
|
||||
ordered_json original;
|
||||
original["one"] = 1;
|
||||
original["two"] = 2;
|
||||
original["three"] = 3;
|
||||
|
||||
ordered_json target = original;
|
||||
target["one"] = 100; // replace
|
||||
target.erase("two"); // remove
|
||||
target["four"] = 4; // add
|
||||
|
||||
const ordered_json patch = ordered_json::diff(original, target);
|
||||
|
||||
SECTION("patch")
|
||||
{
|
||||
const ordered_json patched = original.patch(patch);
|
||||
CHECK(patched == target);
|
||||
}
|
||||
|
||||
SECTION("patch_inplace")
|
||||
{
|
||||
ordered_json copy = original;
|
||||
copy.patch_inplace(patch);
|
||||
CHECK(copy == target);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("ordered_json through merge_patch")
|
||||
{
|
||||
ordered_json original;
|
||||
original["a"] = 1;
|
||||
original["b"] = 2;
|
||||
|
||||
const ordered_json patch = {{"b", nullptr}, {"c", 3}};
|
||||
|
||||
original.merge_patch(patch);
|
||||
|
||||
ordered_json expected;
|
||||
expected["a"] = 1;
|
||||
expected["c"] = 3;
|
||||
|
||||
CHECK(original == expected);
|
||||
CHECK(collect_keys(original) == collect_keys(expected));
|
||||
}
|
||||
@@ -17,6 +17,7 @@
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
using json = nlohmann::json;
|
||||
using ordered_json = nlohmann::ordered_json;
|
||||
|
||||
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
|
||||
#if JSON_HAS_STD_FORMAT
|
||||
@@ -93,4 +94,16 @@ TEST_CASE("std::formatter<nlohmann::json>")
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("std::formatter<nlohmann::ordered_json>")
|
||||
{
|
||||
// spot-check a non-default basic_json instantiation, since the formatter
|
||||
// is written against the generic NLOHMANN_BASIC_JSON_TPL_DECLARATION
|
||||
// template and must actually instantiate (and behave correctly) for
|
||||
// template arguments other than nlohmann::json
|
||||
const ordered_json j = {{"foo", 1}, {"bar", {1, 2, 3}}};
|
||||
CHECK(std::format("{}", j) == j.dump());
|
||||
CHECK(std::format("{:#}", j) == j.dump(4));
|
||||
CHECK(std::format("{:2}", j) == j.dump(2));
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
Reference in New Issue
Block a user