Compare commits

..
Author SHA1 Message Date
Niels Lohmann 50e392ab1a Fix to_bjdata() emitting the Draft-3-only 'B' marker in default Draft-2 mode
_ArrayType_ = "byte" mapped unconditionally to the BJData type marker
'B', regardless of the requested bjdata_version. 'B' is defined only by
BJData Draft 3; with the default version (draft2), this produced a
stream that is invalid for Draft 2 and, unlike every other
_ArrayType_, round-tripped back as a binary value instead of the
original annotated object.

Only accept "byte" / emit 'B' when bjdata_version selects Draft 3.
Under Draft 2, fall back to the same plain-object encoding used
elsewhere in this function for other invalid-annotation cases, so the
value round-trips correctly.

Fixes #5404.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-05 20:52:54 +02:00
Niels Lohmann d2c2db92a9 Fix to_bjdata() silently truncating out-of-range _ArrayData_ elements
write_bjdata_ndarray() validated that each _ArrayData_ element matched
the number kind (integer vs. float) named by _ArrayType_, but not its
range. An element that did not fit the target C++ type (e.g. 256 for
"uint8") was silently wrapped by the static_cast used to write it, or,
for "single", silently overflowed to infinity.

Range-check each element against the type named by _ArrayType_ before
writing it, reusing the existing fallback path that already encodes
the annotated object as a plain object for other invalid-annotation
cases in this function.

Fixes #5403.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-05 20:43:55 +02:00
4 changed files with 253 additions and 144 deletions
@@ -1647,6 +1647,20 @@ class binary_writer
return 'D'; // float 64
}
/*!
@brief checks whether a JSON number fits into @a TargetType
@param[in] el a JSON number of either the signed or unsigned integer kind
@return whether @a el's value can be represented by @a TargetType without
wrapping, regardless of which of the two kinds it is stored as
*/
template<typename TargetType>
static bool bjdata_ndarray_value_in_range(const BasicJsonType& el)
{
return el.is_number_unsigned()
? value_in_range_of<TargetType>(el.template get<std::uint64_t>())
: value_in_range_of<TargetType>(el.template get<std::int64_t>());
}
/*!
@return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid
*/
@@ -1667,6 +1681,16 @@ class binary_writer
}
CharType dtype = it->second;
// the 'B' (byte) marker is only defined by BJData Draft 3; emitting it
// under the default Draft 2 mode would produce a stream that Draft 2
// readers reject, so such an object falls back to a plain object
// encoding instead (see the "Binary values" section of the BJData
// documentation)
if (dtype == 'B' && bjdata_version != bjdata_version_t::draft3)
{
return true;
}
key = "_ArraySize_";
// the dimensions are written verbatim as the header length below, so a
// value that is not an array cannot produce a valid one: null emits 'Z'
@@ -1731,6 +1755,60 @@ class binary_writer
}
}
// every element is cast to the (possibly narrower) C++ type matching
// dtype below; a value that does not fit that type would silently
// wrap (integers) or overflow to infinity (the "single" precision
// float) instead of being reported, so such an object falls back to
// a plain object encoding as well
for (const auto& el : value.at(key))
{
bool in_range = true;
switch (dtype)
{
case 'U':
case 'C':
case 'B':
in_range = bjdata_ndarray_value_in_range<std::uint8_t>(el);
break;
case 'i':
in_range = bjdata_ndarray_value_in_range<std::int8_t>(el);
break;
case 'u':
in_range = bjdata_ndarray_value_in_range<std::uint16_t>(el);
break;
case 'I':
in_range = bjdata_ndarray_value_in_range<std::int16_t>(el);
break;
case 'm':
in_range = bjdata_ndarray_value_in_range<std::uint32_t>(el);
break;
case 'l':
in_range = bjdata_ndarray_value_in_range<std::int32_t>(el);
break;
case 'M':
in_range = bjdata_ndarray_value_in_range<std::uint64_t>(el);
break;
case 'L':
in_range = bjdata_ndarray_value_in_range<std::int64_t>(el);
break;
case 'd':
{
const auto dval = el.template get<double>();
in_range = !std::isfinite(dval) ||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
break;
}
default:
// 'D' (double) already spans the full range of number_float_t
break;
}
if (!in_range)
{
return true;
}
}
oa->write_character('[');
oa->write_character('$');
oa->write_character(dtype);
+78
View File
@@ -18655,6 +18655,20 @@ class binary_writer
return 'D'; // float 64
}
/*!
@brief checks whether a JSON number fits into @a TargetType
@param[in] el a JSON number of either the signed or unsigned integer kind
@return whether @a el's value can be represented by @a TargetType without
wrapping, regardless of which of the two kinds it is stored as
*/
template<typename TargetType>
static bool bjdata_ndarray_value_in_range(const BasicJsonType& el)
{
return el.is_number_unsigned()
? value_in_range_of<TargetType>(el.template get<std::uint64_t>())
: value_in_range_of<TargetType>(el.template get<std::int64_t>());
}
/*!
@return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid
*/
@@ -18675,6 +18689,16 @@ class binary_writer
}
CharType dtype = it->second;
// the 'B' (byte) marker is only defined by BJData Draft 3; emitting it
// under the default Draft 2 mode would produce a stream that Draft 2
// readers reject, so such an object falls back to a plain object
// encoding instead (see the "Binary values" section of the BJData
// documentation)
if (dtype == 'B' && bjdata_version != bjdata_version_t::draft3)
{
return true;
}
key = "_ArraySize_";
// the dimensions are written verbatim as the header length below, so a
// value that is not an array cannot produce a valid one: null emits 'Z'
@@ -18739,6 +18763,60 @@ class binary_writer
}
}
// every element is cast to the (possibly narrower) C++ type matching
// dtype below; a value that does not fit that type would silently
// wrap (integers) or overflow to infinity (the "single" precision
// float) instead of being reported, so such an object falls back to
// a plain object encoding as well
for (const auto& el : value.at(key))
{
bool in_range = true;
switch (dtype)
{
case 'U':
case 'C':
case 'B':
in_range = bjdata_ndarray_value_in_range<std::uint8_t>(el);
break;
case 'i':
in_range = bjdata_ndarray_value_in_range<std::int8_t>(el);
break;
case 'u':
in_range = bjdata_ndarray_value_in_range<std::uint16_t>(el);
break;
case 'I':
in_range = bjdata_ndarray_value_in_range<std::int16_t>(el);
break;
case 'm':
in_range = bjdata_ndarray_value_in_range<std::uint32_t>(el);
break;
case 'l':
in_range = bjdata_ndarray_value_in_range<std::int32_t>(el);
break;
case 'M':
in_range = bjdata_ndarray_value_in_range<std::uint64_t>(el);
break;
case 'L':
in_range = bjdata_ndarray_value_in_range<std::int64_t>(el);
break;
case 'd':
{
const auto dval = el.template get<double>();
in_range = !std::isfinite(dval) ||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
break;
}
default:
// 'D' (double) already spans the full range of number_float_t
break;
}
if (!in_range)
{
return true;
}
}
oa->write_character('[');
oa->write_character('$');
oa->write_character(dtype);
+94 -2
View File
@@ -2586,7 +2586,12 @@ TEST_CASE("BJData")
CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d);
CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D);
CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C);
CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true) == v_B);
// v_B uses the Draft-3-only 'B' marker, so it round-trips only when
// Draft 3 is explicitly selected (see GitHub issue #5404); the
// default Draft 2 falls back to a plain object instead, covered by
// the "ndarray with _ArrayType_ "byte" is gated by the BJData draft
// version" section below
CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true, json::bjdata_version_t::draft3) == v_B);
}
SECTION("ndarray with data not matching _ArrayType_ is written as an object")
@@ -2629,8 +2634,10 @@ TEST_CASE("BJData")
// the C++ API stores an int literal as number_integer, so _ArrayType_
// names the wire type rather than the storage. Both storages have to
// produce the same typed array for every type.
// "byte" is checked separately below since it additionally requires
// BJData Draft 3 to be selected explicitly (see GitHub issue #5404).
for (const char* type :
{"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char", "byte"
{"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char"
})
{
CAPTURE(type);
@@ -2641,6 +2648,14 @@ TEST_CASE("BJData")
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}})));
}
{
const std::string text = R"({"_ArrayType_":"byte","_ArraySize_":[2,3],"_ArrayData_":[1,2,3,4,5,6]})";
const auto from_text = json::to_bjdata(json::parse(text), true, true, json::bjdata_version_t::draft3);
CHECK(from_text.at(0) == '[');
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}),
true, true, json::bjdata_version_t::draft3));
}
// negative values under a signed type behave the same way
const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})"));
CHECK(from_neg.at(0) == '[');
@@ -2776,6 +2791,83 @@ TEST_CASE("BJData")
CHECK(out_num.at(0) == '{');
CHECK(json::from_bjdata(out_num) == j_num);
}
SECTION("ndarray with out-of-range _ArrayData_ elements stays as object")
{
// each element is cast to the (possibly narrower) C++ type
// named by _ArrayType_ before being written; a value that
// does not fit that type would silently wrap instead of
// being reported, so such an object falls back to a plain
// object encoding that still round-trips (see GitHub issue #5403)
// an unsigned element that does not fit uint8
json const j_uint8 = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 256}}});
const auto out_uint8 = json::to_bjdata(j_uint8);
CHECK(out_uint8.at(0) == '{');
CHECK(json::from_bjdata(out_uint8) == j_uint8);
// a signed element that does not fit int8
json const j_int8 = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 200}}});
const auto out_int8 = json::to_bjdata(j_int8);
CHECK(out_int8.at(0) == '{');
CHECK(json::from_bjdata(out_int8) == j_int8);
// a negative element is likewise out of range for an
// unsigned _ArrayType_
json const j_uint16_neg = json({{"_ArrayType_", "uint16"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, -1}}});
const auto out_uint16_neg = json::to_bjdata(j_uint16_neg);
CHECK(out_uint16_neg.at(0) == '{');
CHECK(json::from_bjdata(out_uint16_neg) == j_uint16_neg);
// a double element that overflows to infinity when narrowed
// to the "single" (float) precision named by _ArrayType_
json const j_single = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1.5, 1e40}}});
const auto out_single = json::to_bjdata(j_single);
CHECK(out_single.at(0) == '{');
CHECK(json::from_bjdata(out_single) == j_single);
// in-range boundary values still use the compact ndarray encoding
json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {0, 255}}});
CHECK(json::to_bjdata(j_uint8_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, ']', 0, 255}));
json const j_int8_ok = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {-128, 127}}});
CHECK(json::to_bjdata(j_int8_ok) == std::vector<uint8_t>({'[', '$', 'i', '#', '[', 'i', 2, ']', 0x80, 0x7F}));
json const j_single_ok = json({{"_ArrayType_", "single"}, {"_ArraySize_", {1}}, {"_ArrayData_", {1.5}}});
const auto out_single_ok = json::to_bjdata(j_single_ok);
CHECK(out_single_ok.at(0) == '[');
CHECK(json::from_bjdata(out_single_ok) == json({1.5f}));
}
SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version")
{
// the 'B' (byte) marker used by _ArrayType_ "byte" is only defined
// by BJData Draft 3; Draft 2 (the default) has no such marker, so
// emitting it unconditionally produced a stream that a Draft 2
// reader could not parse as intended (see GitHub issue #5404).
// Two dimensions are used so that a successfully written ndarray
// round-trips back into the annotated object (a single dimension
// is, by the BJData ndarray convention, read back as a plain
// binary value rather than the annotated object, same as every
// other single-dimension ndarray of a non-"byte" type is read
// back as a plain array instead of the annotated object).
json const j_byte = json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}});
// default (Draft 2): falls back to a plain object and round-trips
const auto out_draft2 = json::to_bjdata(j_byte);
CHECK(out_draft2.at(0) == '{');
CHECK(json::from_bjdata(out_draft2) == j_byte);
// explicit Draft 2: same as the default
const auto out_draft2_explicit = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft2);
CHECK(out_draft2_explicit.at(0) == '{');
CHECK(json::from_bjdata(out_draft2_explicit) == j_byte);
// Draft 3 explicitly selected: still uses the compact 'B' ndarray encoding
const auto out_draft3 = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft3);
CHECK(out_draft3 == std::vector<uint8_t>({'[', '$', 'B', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6}));
CHECK(json::from_bjdata(out_draft3) == j_byte);
}
}
}
+3 -142
View File
@@ -38,54 +38,6 @@ class huge_binary_t : public std::vector<std::uint8_t>
using huge_binary_json = nlohmann::basic_json <
std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t,
double, std::allocator, nlohmann::adl_serializer, huge_binary_t, void >;
// a string type that can be made to report a size beyond INT32_MAX without
// allocating that much memory, so BSON length overflow can be tested for
// strings and (embedded) documents as well, following the same idea as
// huge_binary_t.
//
// Unlike huge_binary_t (which is only ever used as the BSON *value* type),
// this type doubles as basic_json's StringType and is therefore also used
// for *object keys* (e.g. "s" or "nested" below). Only the designated test
// value is meant to lie about its size - if every huge_string_t (including
// keys) reported a huge size, the running totals computed while walking the
// BSON document (see calc_bson_object_size & friends in binary_writer.hpp)
// would need more than 32 bits, and on platforms where std::size_t is only
// 32 bits wide that arithmetic would silently wrap around, producing wrong
// (or even unguarded) lengths. The fake size is therefore opt-in via
// as_huge(), and plain strings - in particular object keys - keep reporting
// their real, small size.
class huge_string_t : public std::string
{
public:
using std::string::string;
huge_string_t(const std::string& s) : std::string(s) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions)
// returns a copy of @a s whose size() pretends to be huge
static huge_string_t as_huge(const std::string& s)
{
huge_string_t result(s);
result.pretend_huge = true;
return result;
}
size_type size() const noexcept
{
if (pretend_huge)
{
// one byte more than the BSON length field can represent
return static_cast<size_type>((std::numeric_limits<std::int32_t>::max)()) + 1;
}
return std::string::size();
}
private:
bool pretend_huge = false;
};
using huge_string_json = nlohmann::basic_json <
std::map, std::vector, huge_string_t, bool, std::int64_t, std::uint64_t,
double, std::allocator, nlohmann::adl_serializer, std::vector<std::uint8_t>, void >;
} // namespace
TEST_CASE("BSON")
@@ -153,36 +105,10 @@ TEST_CASE("BSON")
SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON")
{
// out_of_range.412 is thrown from a single shared helper
// (to_bson_length) that guards the BSON length fields of binary
// values, strings, and (embedded) documents alike
SECTION("binary")
{
huge_binary_json j;
j["b"] = huge_binary_json::binary(huge_binary_t{});
huge_binary_json j;
j["b"] = huge_binary_json::binary(huge_binary_t{});
CHECK_THROWS_WITH_AS(huge_binary_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483661 exceeds maximum of 2147483647", huge_binary_json::out_of_range&);
}
SECTION("string")
{
huge_string_json j;
j["s"] = huge_string_t::as_huge("value");
CHECK_THROWS_WITH_AS(huge_string_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483661 exceeds maximum of 2147483647", huge_string_json::out_of_range&);
}
SECTION("document")
{
// an oversized string nested one level deep makes the
// *embedded* document's own length exceed INT32_MAX as well
huge_string_json nested;
nested["s"] = huge_string_t::as_huge("value");
huge_string_json j;
j["nested"] = nested;
CHECK_THROWS_WITH_AS(huge_string_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483674 exceeds maximum of 2147483647", huge_string_json::out_of_range&);
}
CHECK_THROWS_WITH_AS(huge_binary_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483661 exceeds maximum of 2147483647", huge_binary_json::out_of_range&);
}
SECTION("string length must be at least 1")
@@ -267,23 +193,6 @@ TEST_CASE("BSON")
CHECK(json::from_bson(result, true, false) == j);
}
SECTION("non-empty object with bool from a non-0/1 byte (lenient parsing)")
{
// documented lenient behavior (see gh-5333): any non-zero byte
// is accepted as `true`, not just 0x01
std::vector<std::uint8_t> const input =
{
0x0D, 0x00, 0x00, 0x00, // size (little endian)
0x08, // entry: boolean
'e', 'n', 't', 'r', 'y', '\x00',
0x02, // value = 0x02 (neither 0x00 nor 0x01)
0x00 // end marker
};
const json expected = { { "entry", true } };
CHECK(json::from_bson(input) == expected);
}
SECTION("non-empty object with double")
{
json const j =
@@ -590,29 +499,6 @@ TEST_CASE("BSON")
CHECK(json::from_bson(result, true, false) == j);
}
SECTION("array elements with non-conforming keys (lenient parsing)")
{
// documented lenient behavior (see gh-5333): BSON array element
// keys are not checked against the required decimal sequence
// "0", "1", "2", ... - elements are taken in encoded order
std::vector<std::uint8_t> const input =
{
0x26, 0x00, 0x00, 0x00, // size (little endian)
0x04, 'e', 'n', 't', 'r', 'y', '\x00', // entry: embedded array
0x1A, 0x00, 0x00, 0x00, // size (little endian)
0x10, '5', 0x00, 0x0A, 0x00, 0x00, 0x00, // key "5" (bogus) -> 10
0x10, 'x', 0x00, 0x14, 0x00, 0x00, 0x00, // key "x" (non-numeric) -> 20
0x10, '1', 0x00, 0x1E, 0x00, 0x00, 0x00, // key "1" (out of order) -> 30
0x00, // end marker (embedded array)
0x00 // end marker
};
const json expected = { { "entry", json::array({10, 20, 30}) } };
CHECK(json::from_bson(input) == expected);
}
SECTION("non-empty object with binary member")
{
const size_t N = 10;
@@ -708,31 +594,6 @@ TEST_CASE("BSON")
CHECK(json::from_bson(result, true, false) == j);
}
SECTION("binary member with subtype 0x02 (old binary) keeps its inner length prefix (lenient parsing)")
{
// documented lenient behavior (see gh-5333): the payload for
// binary subtype 0x02 ("old binary") is returned as-is,
// including its own inner 4-byte length prefix; it is not
// stripped or reinterpreted
std::vector<std::uint8_t> const input =
{
0x17, 0x00, 0x00, 0x00, // size (little endian)
0x05, 'e', 'n', 't', 'r', 'y', '\x00', // entry: binary
0x06, 0x00, 0x00, 0x00, // size of binary (little endian)
0x02, // "old binary" subtype
0x02, 0x00, 0x00, 0x00, // inner length prefix (part of the old-binary payload)
0x68, 0x69, // payload ('h', 'i')
0x00 // end marker
};
// the inner length prefix is part of the (unmodified) payload
const std::vector<std::uint8_t> expected_payload = {0x02, 0x00, 0x00, 0x00, 0x68, 0x69};
const json expected = { { "entry", json::binary(expected_payload, 0x02) } };
CHECK(json::from_bson(input) == expected);
}
SECTION("Some more complex document")
{
json const j =