Compare commits

..
Author SHA1 Message Date
Niels Lohmann d2c2db92a9 Fix to_bjdata() silently truncating out-of-range _ArrayData_ elements
write_bjdata_ndarray() validated that each _ArrayData_ element matched
the number kind (integer vs. float) named by _ArrayType_, but not its
range. An element that did not fit the target C++ type (e.g. 256 for
"uint8") was silently wrapped by the static_cast used to write it, or,
for "single", silently overflowed to infinity.

Range-check each element against the type named by _ArrayType_ before
writing it, reusing the existing fallback path that already encodes
the annotated object as a plain object for other invalid-annotation
cases in this function.

Fixes #5403.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-05 20:43:55 +02:00
6 changed files with 211 additions and 354 deletions
@@ -1647,6 +1647,20 @@ class binary_writer
return 'D'; // float 64
}
/*!
@brief checks whether a JSON number fits into @a TargetType
@param[in] el a JSON number of either the signed or unsigned integer kind
@return whether @a el's value can be represented by @a TargetType without
wrapping, regardless of which of the two kinds it is stored as
*/
template<typename TargetType>
static bool bjdata_ndarray_value_in_range(const BasicJsonType& el)
{
return el.is_number_unsigned()
? value_in_range_of<TargetType>(el.template get<std::uint64_t>())
: value_in_range_of<TargetType>(el.template get<std::int64_t>());
}
/*!
@return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid
*/
@@ -1731,6 +1745,60 @@ class binary_writer
}
}
// every element is cast to the (possibly narrower) C++ type matching
// dtype below; a value that does not fit that type would silently
// wrap (integers) or overflow to infinity (the "single" precision
// float) instead of being reported, so such an object falls back to
// a plain object encoding as well
for (const auto& el : value.at(key))
{
bool in_range = true;
switch (dtype)
{
case 'U':
case 'C':
case 'B':
in_range = bjdata_ndarray_value_in_range<std::uint8_t>(el);
break;
case 'i':
in_range = bjdata_ndarray_value_in_range<std::int8_t>(el);
break;
case 'u':
in_range = bjdata_ndarray_value_in_range<std::uint16_t>(el);
break;
case 'I':
in_range = bjdata_ndarray_value_in_range<std::int16_t>(el);
break;
case 'm':
in_range = bjdata_ndarray_value_in_range<std::uint32_t>(el);
break;
case 'l':
in_range = bjdata_ndarray_value_in_range<std::int32_t>(el);
break;
case 'M':
in_range = bjdata_ndarray_value_in_range<std::uint64_t>(el);
break;
case 'L':
in_range = bjdata_ndarray_value_in_range<std::int64_t>(el);
break;
case 'd':
{
const auto dval = el.template get<double>();
in_range = !std::isfinite(dval) ||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
break;
}
default:
// 'D' (double) already spans the full range of number_float_t
break;
}
if (!in_range)
{
return true;
}
}
oa->write_character('[');
oa->write_character('$');
oa->write_character(dtype);
+17 -124
View File
@@ -3573,7 +3573,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
{
using std::swap;
swap(*(m_data.m_value.array), other);
set_parents();
}
else
{
@@ -3590,7 +3589,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
{
using std::swap;
swap(*(m_data.m_value.object), other);
set_parents();
}
else
{
@@ -5159,139 +5157,34 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
case value_t::object:
{
// first pass: record, for every source key, whether it is
// common to both objects (in source's iteration order) or
// was deleted (i.e., in source but not in target) -- this is
// a by-product of the target.find() call already needed to
// tell the two cases apart, so it adds no extra lookups. The
// "remove" ops themselves are emitted later, interleaved
// with the recursive per-key diffs in the fast path below,
// to match source's original iteration order (as the
// original, pre-reordering-aware implementation did) instead
// of grouping all removes before all recursive diffs.
std::vector<typename object_t::key_type> common_keys_source_order;
// first pass: traverse this object's elements
for (auto it = source.cbegin(); it != source.cend(); ++it)
{
// escape the key name to be used in a JSON patch
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
if (target.find(it.key()) != target.end())
{
common_keys_source_order.push_back(it.key());
// recursive call to compare object values at key it
auto temp_diff = diff(it.value(), target[it.key()], path_key);
result.insert(result.end(), temp_diff.begin(), temp_diff.end());
}
else
{
// found a key that is not in o -> remove it
result.push_back(object(
{
{"op", "remove"}, {"path", path_key}
}));
}
}
// second pass: find keys that were added (i.e., in target but
// not in source), and record the keys common to both, in
// target's iteration order -- again a by-product of the
// source.find() call already needed to detect added keys. At
// the same time, determine whether every added key comes
// after every common key in target's order (a precondition
// for the fast path below, which only ever appends new keys
// at the very end): for an object_t whose iteration order is
// a pure function of the key set (e.g. the default std::map,
// which always iterates in sorted key order), the order
// check further below is always true and this whole
// mechanism is effectively a no-op; it only matters for a
// reorderable object_t such as the one backing `ordered_json`.
// patch ops for keys that were added (i.e., in target but not
// in source); built here so the fast path below can reuse
// them without a second source.find() per target key. Only
// used by the fast path -- the slow (reordering) path
// rebuilds "add" ops for every key itself.
std::vector<typename object_t::key_type> common_keys_target_order;
basic_json added_ops(value_t::array);
bool new_keys_form_suffix = true;
bool seen_new_key = false;
// second pass: traverse other object's elements
for (auto it = target.cbegin(); it != target.cend(); ++it)
{
if (source.find(it.key()) == source.end())
{
seen_new_key = true;
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
added_ops.push_back(
{
{"op", "add"}, {"path", path_key},
{"value", it.value()}
});
}
else
{
common_keys_target_order.push_back(it.key());
if (seen_new_key)
{
new_keys_form_suffix = false;
}
}
}
if (common_keys_source_order == common_keys_target_order && new_keys_form_suffix)
{
// fast path: order of common keys already matches (or the
// object_t's iteration order does not depend on
// insertion history), so a plain per-key recursive diff
// is correct and minimal, as before. common_keys_source_order
// is, by construction, the subsequence of source's keys
// that are common to both objects, in source's iteration
// order -- so it can be walked in lockstep with `source`
// using a cheap key comparison instead of another lookup.
// Deleted keys (those source keys not in common_keys_source_order)
// are interleaved here too, in source's original order, to
// match the historical (pre-reordering-aware) output order.
auto common_it = common_keys_source_order.cbegin();
for (auto it = source.cbegin(); it != source.cend(); ++it)
{
if (common_it != common_keys_source_order.cend() && it.key() == *common_it)
{
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
auto temp_diff = diff(it.value(), target[it.key()], path_key);
result.insert(result.end(), temp_diff.begin(), temp_diff.end());
++common_it;
}
else
{
// found a key that is not in target -> remove it
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
result.push_back(object(
{
{"op", "remove"}, {"path", path_key}
}));
}
}
// append the "add" ops for brand-new keys collected above
// during the pass over target -- no second source.find()
// per target key needed
result.insert(result.end(), added_ops.begin(), added_ops.end());
}
else
{
// slow path: the common keys are in a different relative
// order in source and target (only possible for a
// reorderable object_t like ordered_map). Building a
// minimal reordering patch is a nontrivial (LCS-like)
// problem; instead, remove every source key -- both
// deleted keys (which must be removed regardless) and
// common keys (removed so they can be re-added in
// target's order) -- and re-add every key that should
// remain, with its final target value, in target's
// order. basic_json::patch()'s "add" operation on an
// object uses operator[], which appends at the end for a
// vector-backed insertion-ordered map when the key does
// not already exist -- so removing a key and then adding
// it moves it to the end, fixing its position.
for (auto it = source.cbegin(); it != source.cend(); ++it)
{
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
result.push_back(object(
{
{"op", "remove"}, {"path", path_key}
}));
}
// add every key that is either common (just removed
// above) or brand new, in target's iteration order, so
// that the final order after applying the patch matches
// target exactly
for (auto it = target.cbegin(); it != target.cend(); ++it)
{
// found a key that is not in this -> add it
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
result.push_back(
{
+85 -124
View File
@@ -18655,6 +18655,20 @@ class binary_writer
return 'D'; // float 64
}
/*!
@brief checks whether a JSON number fits into @a TargetType
@param[in] el a JSON number of either the signed or unsigned integer kind
@return whether @a el's value can be represented by @a TargetType without
wrapping, regardless of which of the two kinds it is stored as
*/
template<typename TargetType>
static bool bjdata_ndarray_value_in_range(const BasicJsonType& el)
{
return el.is_number_unsigned()
? value_in_range_of<TargetType>(el.template get<std::uint64_t>())
: value_in_range_of<TargetType>(el.template get<std::int64_t>());
}
/*!
@return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid
*/
@@ -18739,6 +18753,60 @@ class binary_writer
}
}
// every element is cast to the (possibly narrower) C++ type matching
// dtype below; a value that does not fit that type would silently
// wrap (integers) or overflow to infinity (the "single" precision
// float) instead of being reported, so such an object falls back to
// a plain object encoding as well
for (const auto& el : value.at(key))
{
bool in_range = true;
switch (dtype)
{
case 'U':
case 'C':
case 'B':
in_range = bjdata_ndarray_value_in_range<std::uint8_t>(el);
break;
case 'i':
in_range = bjdata_ndarray_value_in_range<std::int8_t>(el);
break;
case 'u':
in_range = bjdata_ndarray_value_in_range<std::uint16_t>(el);
break;
case 'I':
in_range = bjdata_ndarray_value_in_range<std::int16_t>(el);
break;
case 'm':
in_range = bjdata_ndarray_value_in_range<std::uint32_t>(el);
break;
case 'l':
in_range = bjdata_ndarray_value_in_range<std::int32_t>(el);
break;
case 'M':
in_range = bjdata_ndarray_value_in_range<std::uint64_t>(el);
break;
case 'L':
in_range = bjdata_ndarray_value_in_range<std::int64_t>(el);
break;
case 'd':
{
const auto dval = el.template get<double>();
in_range = !std::isfinite(dval) ||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
break;
}
default:
// 'D' (double) already spans the full range of number_float_t
break;
}
if (!in_range)
{
return true;
}
}
oa->write_character('[');
oa->write_character('$');
oa->write_character(dtype);
@@ -25001,7 +25069,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
{
using std::swap;
swap(*(m_data.m_value.array), other);
set_parents();
}
else
{
@@ -25018,7 +25085,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
{
using std::swap;
swap(*(m_data.m_value.object), other);
set_parents();
}
else
{
@@ -26587,139 +26653,34 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
case value_t::object:
{
// first pass: record, for every source key, whether it is
// common to both objects (in source's iteration order) or
// was deleted (i.e., in source but not in target) -- this is
// a by-product of the target.find() call already needed to
// tell the two cases apart, so it adds no extra lookups. The
// "remove" ops themselves are emitted later, interleaved
// with the recursive per-key diffs in the fast path below,
// to match source's original iteration order (as the
// original, pre-reordering-aware implementation did) instead
// of grouping all removes before all recursive diffs.
std::vector<typename object_t::key_type> common_keys_source_order;
// first pass: traverse this object's elements
for (auto it = source.cbegin(); it != source.cend(); ++it)
{
// escape the key name to be used in a JSON patch
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
if (target.find(it.key()) != target.end())
{
common_keys_source_order.push_back(it.key());
// recursive call to compare object values at key it
auto temp_diff = diff(it.value(), target[it.key()], path_key);
result.insert(result.end(), temp_diff.begin(), temp_diff.end());
}
else
{
// found a key that is not in o -> remove it
result.push_back(object(
{
{"op", "remove"}, {"path", path_key}
}));
}
}
// second pass: find keys that were added (i.e., in target but
// not in source), and record the keys common to both, in
// target's iteration order -- again a by-product of the
// source.find() call already needed to detect added keys. At
// the same time, determine whether every added key comes
// after every common key in target's order (a precondition
// for the fast path below, which only ever appends new keys
// at the very end): for an object_t whose iteration order is
// a pure function of the key set (e.g. the default std::map,
// which always iterates in sorted key order), the order
// check further below is always true and this whole
// mechanism is effectively a no-op; it only matters for a
// reorderable object_t such as the one backing `ordered_json`.
// patch ops for keys that were added (i.e., in target but not
// in source); built here so the fast path below can reuse
// them without a second source.find() per target key. Only
// used by the fast path -- the slow (reordering) path
// rebuilds "add" ops for every key itself.
std::vector<typename object_t::key_type> common_keys_target_order;
basic_json added_ops(value_t::array);
bool new_keys_form_suffix = true;
bool seen_new_key = false;
// second pass: traverse other object's elements
for (auto it = target.cbegin(); it != target.cend(); ++it)
{
if (source.find(it.key()) == source.end())
{
seen_new_key = true;
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
added_ops.push_back(
{
{"op", "add"}, {"path", path_key},
{"value", it.value()}
});
}
else
{
common_keys_target_order.push_back(it.key());
if (seen_new_key)
{
new_keys_form_suffix = false;
}
}
}
if (common_keys_source_order == common_keys_target_order && new_keys_form_suffix)
{
// fast path: order of common keys already matches (or the
// object_t's iteration order does not depend on
// insertion history), so a plain per-key recursive diff
// is correct and minimal, as before. common_keys_source_order
// is, by construction, the subsequence of source's keys
// that are common to both objects, in source's iteration
// order -- so it can be walked in lockstep with `source`
// using a cheap key comparison instead of another lookup.
// Deleted keys (those source keys not in common_keys_source_order)
// are interleaved here too, in source's original order, to
// match the historical (pre-reordering-aware) output order.
auto common_it = common_keys_source_order.cbegin();
for (auto it = source.cbegin(); it != source.cend(); ++it)
{
if (common_it != common_keys_source_order.cend() && it.key() == *common_it)
{
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
auto temp_diff = diff(it.value(), target[it.key()], path_key);
result.insert(result.end(), temp_diff.begin(), temp_diff.end());
++common_it;
}
else
{
// found a key that is not in target -> remove it
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
result.push_back(object(
{
{"op", "remove"}, {"path", path_key}
}));
}
}
// append the "add" ops for brand-new keys collected above
// during the pass over target -- no second source.find()
// per target key needed
result.insert(result.end(), added_ops.begin(), added_ops.end());
}
else
{
// slow path: the common keys are in a different relative
// order in source and target (only possible for a
// reorderable object_t like ordered_map). Building a
// minimal reordering patch is a nontrivial (LCS-like)
// problem; instead, remove every source key -- both
// deleted keys (which must be removed regardless) and
// common keys (removed so they can be re-added in
// target's order) -- and re-add every key that should
// remain, with its final target value, in target's
// order. basic_json::patch()'s "add" operation on an
// object uses operator[], which appends at the end for a
// vector-backed insertion-ordered map when the key does
// not already exist -- so removing a key and then adding
// it moves it to the end, fixing its position.
for (auto it = source.cbegin(); it != source.cend(); ++it)
{
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
result.push_back(object(
{
{"op", "remove"}, {"path", path_key}
}));
}
// add every key that is either common (just removed
// above) or brand new, in target's iteration order, so
// that the final order after applying the patch matches
// target exactly
for (auto it = target.cbegin(); it != target.cend(); ++it)
{
// found a key that is not in this -> add it
const auto path_key = detail::concat<string_t>(path, '/', detail::escape(it.key()));
result.push_back(
{
+47
View File
@@ -2776,6 +2776,53 @@ TEST_CASE("BJData")
CHECK(out_num.at(0) == '{');
CHECK(json::from_bjdata(out_num) == j_num);
}
SECTION("ndarray with out-of-range _ArrayData_ elements stays as object")
{
// each element is cast to the (possibly narrower) C++ type
// named by _ArrayType_ before being written; a value that
// does not fit that type would silently wrap instead of
// being reported, so such an object falls back to a plain
// object encoding that still round-trips (see GitHub issue #5403)
// an unsigned element that does not fit uint8
json const j_uint8 = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 256}}});
const auto out_uint8 = json::to_bjdata(j_uint8);
CHECK(out_uint8.at(0) == '{');
CHECK(json::from_bjdata(out_uint8) == j_uint8);
// a signed element that does not fit int8
json const j_int8 = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 200}}});
const auto out_int8 = json::to_bjdata(j_int8);
CHECK(out_int8.at(0) == '{');
CHECK(json::from_bjdata(out_int8) == j_int8);
// a negative element is likewise out of range for an
// unsigned _ArrayType_
json const j_uint16_neg = json({{"_ArrayType_", "uint16"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, -1}}});
const auto out_uint16_neg = json::to_bjdata(j_uint16_neg);
CHECK(out_uint16_neg.at(0) == '{');
CHECK(json::from_bjdata(out_uint16_neg) == j_uint16_neg);
// a double element that overflows to infinity when narrowed
// to the "single" (float) precision named by _ArrayType_
json const j_single = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1.5, 1e40}}});
const auto out_single = json::to_bjdata(j_single);
CHECK(out_single.at(0) == '{');
CHECK(json::from_bjdata(out_single) == j_single);
// in-range boundary values still use the compact ndarray encoding
json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {0, 255}}});
CHECK(json::to_bjdata(j_uint8_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, ']', 0, 255}));
json const j_int8_ok = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {-128, 127}}});
CHECK(json::to_bjdata(j_int8_ok) == std::vector<uint8_t>({'[', '$', 'i', '#', '[', 'i', 2, ']', 0x80, 0x7F}));
json const j_single_ok = json({{"_ArrayType_", "single"}, {"_ArraySize_", {1}}, {"_ArrayData_", {1.5}}});
const auto out_single_ok = json::to_bjdata(j_single_ok);
CHECK(out_single_ok.at(0) == '[');
CHECK(json::from_bjdata(out_single_ok) == json({1.5f}));
}
}
}
-31
View File
@@ -273,36 +273,5 @@ TEST_CASE("Regression tests for extended diagnostics")
CHECK(j1["numbers"]["two"] == 2);
CHECK(j1["string"] == "t");
}
SECTION("Regression test - swap(array_t&)/swap(object_t&) must update JSON_DIAGNOSTICS parent pointers")
{
// swap(array_t&)
{
json j = json::array();
json::array_t arr = {json::array({1})};
j.swap(arr);
// parent pointers of the moved-in elements must point into j, not
// into the now-defunct free-standing array_t
CHECK_THROWS_WITH_AS(j[0][0].get<std::string>(), "[json.exception.type_error.302] (/0/0) type must be string, but is number", json::type_error);
// must not trigger assert_invariant() in a debug/assert-enabled build
json const k = j;
CHECK(k == j);
}
// swap(object_t&)
{
json o = json::object();
json::object_t obj = {{"a", json::array({1})}};
o.swap(obj);
CHECK_THROWS_WITH_AS(o["a"][0].get<std::string>(), "[json.exception.type_error.302] (/a/0) type must be string, but is number", json::type_error);
// must not trigger assert_invariant() in a debug/assert-enabled build
json const p = o;
CHECK(p == o);
}
}
}
-81
View File
@@ -81,84 +81,3 @@ TEST_CASE("regression test for issue #3732 - iteration_proxy_value<iter_impl<ord
};
static_cast<void>(fn);
}
TEST_CASE("regression test - diff() must account for ordered_json member order")
{
SECTION("pure reorder, no value changes")
{
ordered_json a = {{"a", 1}, {"b", 2}};
ordered_json b = {{"b", 2}, {"a", 1}};
CHECK(a != b); // order-sensitive equality
CHECK(a.patch(ordered_json::diff(a, b)) == b);
}
SECTION("new key must land at the front")
{
ordered_json c = {{"b", 2}};
ordered_json e = {{"a", 1}, {"b", 2}};
CHECK(c.patch(ordered_json::diff(c, e)) == e);
}
SECTION("reorder plus a value change on one of the reordered keys")
{
ordered_json a = {{"a", 1}, {"b", 2}};
ordered_json b = {{"b", 20}, {"a", 1}};
CHECK(a != b);
CHECK(a.patch(ordered_json::diff(a, b)) == b);
}
SECTION("reorder plus a deleted key")
{
ordered_json a = {{"a", 1}, {"b", 2}, {"c", 3}};
ordered_json b = {{"b", 2}, {"a", 1}};
CHECK(a != b);
CHECK(a.patch(ordered_json::diff(a, b)) == b);
}
SECTION("reorder plus a nested value that itself needs a recursive diff")
{
ordered_json a = {{"a", {{"x", 1}, {"y", 2}}}, {"b", 2}};
ordered_json b = {{"b", 2}, {"a", {{"x", 1}, {"y", 99}}}};
CHECK(a != b);
CHECK(a.patch(ordered_json::diff(a, b)) == b);
}
SECTION("three or more keys shuffled into a different order")
{
ordered_json a = {{"a", 1}, {"b", 2}, {"c", 3}, {"d", 4}};
ordered_json b = {{"d", 4}, {"b", 2}, {"a", 1}, {"c", 3}};
CHECK(a != b);
CHECK(a.patch(ordered_json::diff(a, b)) == b);
}
SECTION("matching order still produces a minimal patch (fast path unaffected)")
{
ordered_json a = {{"a", 1}, {"b", 2}, {"c", 3}};
ordered_json b = {{"a", 1}, {"b", 20}, {"c", 3}};
auto p = ordered_json::diff(a, b);
// only the changed value should be touched, not a wholesale remove+add
CHECK(p.size() == 1);
CHECK(p[0]["op"] == "replace");
CHECK(p[0]["path"] == "/b");
CHECK(a.patch(p) == b);
}
SECTION("plain json (std::map-backed) is unaffected by same-key-different-insertion-order")
{
json a;
a["b"] = 2;
a["a"] = 1;
json b;
b["a"] = 1;
b["b"] = 2;
// std::map iteration is always sorted by key, so a == b regardless of
// insertion order, and diff() must still produce the same minimal
// (empty) result as before this fix
CHECK(a == b);
auto p = json::diff(a, b);
CHECK(p.empty());
CHECK(a.patch(p) == b);
}
}