Compare commits

..
Author SHA1 Message Date
Niels Lohmann 32349b379b Make contains(json_pointer) return false instead of throwing on unrepresentable array-index tokens
contains(const json_pointer&) is documented to never throw, but a purely
numeric reference token that is syntactically a valid array index yet
numerically too large to be represented (exceeding size_type's max, or
exceeding ULLONG_MAX and causing strtoull() to set errno to ERANGE) made
it fall through to array_index(), which throws out_of_range.410/404.

Pre-check the token's magnitude the same way array_index() does, but
return false instead of throwing, mirroring how the surrounding code
already rejects other malformed tokens (leading zero, non-digit
characters, "-") without throwing. operator[]/at() are untouched and
keep throwing for these inputs.

Fixes #5395

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-05 21:57:41 +02:00
5 changed files with 72 additions and 250 deletions
+14
View File
@@ -748,6 +748,20 @@ class json_pointer
} }
} }
// the reference token consists only of digits at this point (cf. checks
// above); however, its numeric value might not be representable, in which
// case array_index() would throw out_of_range.404/410 -- contains() must
// not throw (see #5395), so such a reference token is treated as "not found"
errno = 0; // strtoull() does not reset errno on success
char* p_end = nullptr; // NOLINT(misc-const-correctness)
const unsigned long long magnitude = std::strtoull(reference_token.c_str(), &p_end, 10); // NOLINT(runtime/int)
if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX
|| magnitude >= static_cast<unsigned long long>((std::numeric_limits<typename BasicJsonType::size_type>::max)()))) // NOLINT(runtime/int)
{
// the array index cannot be represented as size_type
return false;
}
const auto idx = array_index<BasicJsonType>(reference_token); const auto idx = array_index<BasicJsonType>(reference_token);
if (idx >= ptr->size()) if (idx >= ptr->size())
{ {
@@ -1647,20 +1647,6 @@ class binary_writer
return 'D'; // float 64 return 'D'; // float 64
} }
/*!
@brief checks whether a JSON number fits into @a TargetType
@param[in] el a JSON number of either the signed or unsigned integer kind
@return whether @a el's value can be represented by @a TargetType without
wrapping, regardless of which of the two kinds it is stored as
*/
template<typename TargetType>
static bool bjdata_ndarray_value_in_range(const BasicJsonType& el)
{
return el.is_number_unsigned()
? value_in_range_of<TargetType>(el.template get<std::uint64_t>())
: value_in_range_of<TargetType>(el.template get<std::int64_t>());
}
/*! /*!
@return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid @return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid
*/ */
@@ -1681,16 +1667,6 @@ class binary_writer
} }
CharType dtype = it->second; CharType dtype = it->second;
// the 'B' (byte) marker is only defined by BJData Draft 3; emitting it
// under the default Draft 2 mode would produce a stream that Draft 2
// readers reject, so such an object falls back to a plain object
// encoding instead (see the "Binary values" section of the BJData
// documentation)
if (dtype == 'B' && bjdata_version != bjdata_version_t::draft3)
{
return true;
}
key = "_ArraySize_"; key = "_ArraySize_";
// the dimensions are written verbatim as the header length below, so a // the dimensions are written verbatim as the header length below, so a
// value that is not an array cannot produce a valid one: null emits 'Z' // value that is not an array cannot produce a valid one: null emits 'Z'
@@ -1755,60 +1731,6 @@ class binary_writer
} }
} }
// every element is cast to the (possibly narrower) C++ type matching
// dtype below; a value that does not fit that type would silently
// wrap (integers) or overflow to infinity (the "single" precision
// float) instead of being reported, so such an object falls back to
// a plain object encoding as well
for (const auto& el : value.at(key))
{
bool in_range = true;
switch (dtype)
{
case 'U':
case 'C':
case 'B':
in_range = bjdata_ndarray_value_in_range<std::uint8_t>(el);
break;
case 'i':
in_range = bjdata_ndarray_value_in_range<std::int8_t>(el);
break;
case 'u':
in_range = bjdata_ndarray_value_in_range<std::uint16_t>(el);
break;
case 'I':
in_range = bjdata_ndarray_value_in_range<std::int16_t>(el);
break;
case 'm':
in_range = bjdata_ndarray_value_in_range<std::uint32_t>(el);
break;
case 'l':
in_range = bjdata_ndarray_value_in_range<std::int32_t>(el);
break;
case 'M':
in_range = bjdata_ndarray_value_in_range<std::uint64_t>(el);
break;
case 'L':
in_range = bjdata_ndarray_value_in_range<std::int64_t>(el);
break;
case 'd':
{
const auto dval = el.template get<double>();
in_range = !std::isfinite(dval) ||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
break;
}
default:
// 'D' (double) already spans the full range of number_float_t
break;
}
if (!in_range)
{
return true;
}
}
oa->write_character('['); oa->write_character('[');
oa->write_character('$'); oa->write_character('$');
oa->write_character(dtype); oa->write_character(dtype);
+14 -78
View File
@@ -16394,6 +16394,20 @@ class json_pointer
} }
} }
// the reference token consists only of digits at this point (cf. checks
// above); however, its numeric value might not be representable, in which
// case array_index() would throw out_of_range.404/410 -- contains() must
// not throw (see #5395), so such a reference token is treated as "not found"
errno = 0; // strtoull() does not reset errno on success
char* p_end = nullptr; // NOLINT(misc-const-correctness)
const unsigned long long magnitude = std::strtoull(reference_token.c_str(), &p_end, 10); // NOLINT(runtime/int)
if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX
|| magnitude >= static_cast<unsigned long long>((std::numeric_limits<typename BasicJsonType::size_type>::max)()))) // NOLINT(runtime/int)
{
// the array index cannot be represented as size_type
return false;
}
const auto idx = array_index<BasicJsonType>(reference_token); const auto idx = array_index<BasicJsonType>(reference_token);
if (idx >= ptr->size()) if (idx >= ptr->size())
{ {
@@ -18655,20 +18669,6 @@ class binary_writer
return 'D'; // float 64 return 'D'; // float 64
} }
/*!
@brief checks whether a JSON number fits into @a TargetType
@param[in] el a JSON number of either the signed or unsigned integer kind
@return whether @a el's value can be represented by @a TargetType without
wrapping, regardless of which of the two kinds it is stored as
*/
template<typename TargetType>
static bool bjdata_ndarray_value_in_range(const BasicJsonType& el)
{
return el.is_number_unsigned()
? value_in_range_of<TargetType>(el.template get<std::uint64_t>())
: value_in_range_of<TargetType>(el.template get<std::int64_t>());
}
/*! /*!
@return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid @return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid
*/ */
@@ -18689,16 +18689,6 @@ class binary_writer
} }
CharType dtype = it->second; CharType dtype = it->second;
// the 'B' (byte) marker is only defined by BJData Draft 3; emitting it
// under the default Draft 2 mode would produce a stream that Draft 2
// readers reject, so such an object falls back to a plain object
// encoding instead (see the "Binary values" section of the BJData
// documentation)
if (dtype == 'B' && bjdata_version != bjdata_version_t::draft3)
{
return true;
}
key = "_ArraySize_"; key = "_ArraySize_";
// the dimensions are written verbatim as the header length below, so a // the dimensions are written verbatim as the header length below, so a
// value that is not an array cannot produce a valid one: null emits 'Z' // value that is not an array cannot produce a valid one: null emits 'Z'
@@ -18763,60 +18753,6 @@ class binary_writer
} }
} }
// every element is cast to the (possibly narrower) C++ type matching
// dtype below; a value that does not fit that type would silently
// wrap (integers) or overflow to infinity (the "single" precision
// float) instead of being reported, so such an object falls back to
// a plain object encoding as well
for (const auto& el : value.at(key))
{
bool in_range = true;
switch (dtype)
{
case 'U':
case 'C':
case 'B':
in_range = bjdata_ndarray_value_in_range<std::uint8_t>(el);
break;
case 'i':
in_range = bjdata_ndarray_value_in_range<std::int8_t>(el);
break;
case 'u':
in_range = bjdata_ndarray_value_in_range<std::uint16_t>(el);
break;
case 'I':
in_range = bjdata_ndarray_value_in_range<std::int16_t>(el);
break;
case 'm':
in_range = bjdata_ndarray_value_in_range<std::uint32_t>(el);
break;
case 'l':
in_range = bjdata_ndarray_value_in_range<std::int32_t>(el);
break;
case 'M':
in_range = bjdata_ndarray_value_in_range<std::uint64_t>(el);
break;
case 'L':
in_range = bjdata_ndarray_value_in_range<std::int64_t>(el);
break;
case 'd':
{
const auto dval = el.template get<double>();
in_range = !std::isfinite(dval) ||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
break;
}
default:
// 'D' (double) already spans the full range of number_float_t
break;
}
if (!in_range)
{
return true;
}
}
oa->write_character('['); oa->write_character('[');
oa->write_character('$'); oa->write_character('$');
oa->write_character(dtype); oa->write_character(dtype);
+2 -94
View File
@@ -2586,12 +2586,7 @@ TEST_CASE("BJData")
CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d); CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d);
CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D); CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D);
CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C); CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C);
// v_B uses the Draft-3-only 'B' marker, so it round-trips only when CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true) == v_B);
// Draft 3 is explicitly selected (see GitHub issue #5404); the
// default Draft 2 falls back to a plain object instead, covered by
// the "ndarray with _ArrayType_ "byte" is gated by the BJData draft
// version" section below
CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true, json::bjdata_version_t::draft3) == v_B);
} }
SECTION("ndarray with data not matching _ArrayType_ is written as an object") SECTION("ndarray with data not matching _ArrayType_ is written as an object")
@@ -2634,10 +2629,8 @@ TEST_CASE("BJData")
// the C++ API stores an int literal as number_integer, so _ArrayType_ // the C++ API stores an int literal as number_integer, so _ArrayType_
// names the wire type rather than the storage. Both storages have to // names the wire type rather than the storage. Both storages have to
// produce the same typed array for every type. // produce the same typed array for every type.
// "byte" is checked separately below since it additionally requires
// BJData Draft 3 to be selected explicitly (see GitHub issue #5404).
for (const char* type : for (const char* type :
{"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char" {"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char", "byte"
}) })
{ {
CAPTURE(type); CAPTURE(type);
@@ -2648,14 +2641,6 @@ TEST_CASE("BJData")
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}))); CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}})));
} }
{
const std::string text = R"({"_ArrayType_":"byte","_ArraySize_":[2,3],"_ArrayData_":[1,2,3,4,5,6]})";
const auto from_text = json::to_bjdata(json::parse(text), true, true, json::bjdata_version_t::draft3);
CHECK(from_text.at(0) == '[');
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}),
true, true, json::bjdata_version_t::draft3));
}
// negative values under a signed type behave the same way // negative values under a signed type behave the same way
const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})")); const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})"));
CHECK(from_neg.at(0) == '['); CHECK(from_neg.at(0) == '[');
@@ -2791,83 +2776,6 @@ TEST_CASE("BJData")
CHECK(out_num.at(0) == '{'); CHECK(out_num.at(0) == '{');
CHECK(json::from_bjdata(out_num) == j_num); CHECK(json::from_bjdata(out_num) == j_num);
} }
SECTION("ndarray with out-of-range _ArrayData_ elements stays as object")
{
// each element is cast to the (possibly narrower) C++ type
// named by _ArrayType_ before being written; a value that
// does not fit that type would silently wrap instead of
// being reported, so such an object falls back to a plain
// object encoding that still round-trips (see GitHub issue #5403)
// an unsigned element that does not fit uint8
json const j_uint8 = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 256}}});
const auto out_uint8 = json::to_bjdata(j_uint8);
CHECK(out_uint8.at(0) == '{');
CHECK(json::from_bjdata(out_uint8) == j_uint8);
// a signed element that does not fit int8
json const j_int8 = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 200}}});
const auto out_int8 = json::to_bjdata(j_int8);
CHECK(out_int8.at(0) == '{');
CHECK(json::from_bjdata(out_int8) == j_int8);
// a negative element is likewise out of range for an
// unsigned _ArrayType_
json const j_uint16_neg = json({{"_ArrayType_", "uint16"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, -1}}});
const auto out_uint16_neg = json::to_bjdata(j_uint16_neg);
CHECK(out_uint16_neg.at(0) == '{');
CHECK(json::from_bjdata(out_uint16_neg) == j_uint16_neg);
// a double element that overflows to infinity when narrowed
// to the "single" (float) precision named by _ArrayType_
json const j_single = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1.5, 1e40}}});
const auto out_single = json::to_bjdata(j_single);
CHECK(out_single.at(0) == '{');
CHECK(json::from_bjdata(out_single) == j_single);
// in-range boundary values still use the compact ndarray encoding
json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {0, 255}}});
CHECK(json::to_bjdata(j_uint8_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, ']', 0, 255}));
json const j_int8_ok = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {-128, 127}}});
CHECK(json::to_bjdata(j_int8_ok) == std::vector<uint8_t>({'[', '$', 'i', '#', '[', 'i', 2, ']', 0x80, 0x7F}));
json const j_single_ok = json({{"_ArrayType_", "single"}, {"_ArraySize_", {1}}, {"_ArrayData_", {1.5}}});
const auto out_single_ok = json::to_bjdata(j_single_ok);
CHECK(out_single_ok.at(0) == '[');
CHECK(json::from_bjdata(out_single_ok) == json({1.5f}));
}
SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version")
{
// the 'B' (byte) marker used by _ArrayType_ "byte" is only defined
// by BJData Draft 3; Draft 2 (the default) has no such marker, so
// emitting it unconditionally produced a stream that a Draft 2
// reader could not parse as intended (see GitHub issue #5404).
// Two dimensions are used so that a successfully written ndarray
// round-trips back into the annotated object (a single dimension
// is, by the BJData ndarray convention, read back as a plain
// binary value rather than the annotated object, same as every
// other single-dimension ndarray of a non-"byte" type is read
// back as a plain array instead of the annotated object).
json const j_byte = json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}});
// default (Draft 2): falls back to a plain object and round-trips
const auto out_draft2 = json::to_bjdata(j_byte);
CHECK(out_draft2.at(0) == '{');
CHECK(json::from_bjdata(out_draft2) == j_byte);
// explicit Draft 2: same as the default
const auto out_draft2_explicit = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft2);
CHECK(out_draft2_explicit.at(0) == '{');
CHECK(json::from_bjdata(out_draft2_explicit) == j_byte);
// Draft 3 explicitly selected: still uses the compact 'B' ndarray encoding
const auto out_draft3 = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft3);
CHECK(out_draft3 == std::vector<uint8_t>({'[', '$', 'B', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6}));
CHECK(json::from_bjdata(out_draft3) == j_byte);
}
} }
} }
+42
View File
@@ -319,6 +319,44 @@ TEST_CASE("JSON pointers")
CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&); CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&); CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&);
// #5395: contains() must not throw for a reference token that is a
// syntactically valid array index but numerically exceeds ULLONG_MAX
// (causing strtoull() to set errno to ERANGE) -- it should just report
// that the pointer does not resolve to an element
CHECK(!j.contains(jp));
CHECK(!j_const.contains(jp));
}
{
// #5395: same as above, but using the exact reproduction from the issue
json::json_pointer const jp("/99999999999999999999");
std::string const throw_msg = "[json.exception.out_of_range.404] unresolved reference token '99999999999999999999'";
CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j.at(jp) = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const.at(jp) == 1, throw_msg.c_str(), json::out_of_range&);
CHECK(!j.contains(jp));
CHECK(!j_const.contains(jp));
}
{
// #5395: a reference token that is numerically representable in
// unsigned long long but exceeds size_type's max (e.g. ULLONG_MAX
// itself on typical 64-bit platforms, where size_type's max equals
// ULLONG_MAX) must not make contains() throw either
json::json_pointer const jp("/18446744073709551615");
std::string const throw_msg = "[json.exception.out_of_range.410] array index 18446744073709551615 exceeds size_type";
CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j.at(jp) = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const.at(jp) == 1, throw_msg.c_str(), json::out_of_range&);
CHECK(!j.contains(jp));
CHECK(!j_const.contains(jp));
} }
// on some machines, the check below is not constant // on some machines, the check below is not constant
@@ -334,6 +372,10 @@ TEST_CASE("JSON pointers")
CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&); CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&); CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&);
// #5395: contains() must not throw for a reference token exceeding size_type's max
CHECK(!j.contains(jp));
CHECK(!j_const.contains(jp));
} }
DOCTEST_MSVC_SUPPRESS_WARNING_POP DOCTEST_MSVC_SUPPRESS_WARNING_POP