Compare commits

..
Author SHA1 Message Date
Niels Lohmann 1c9475d68f Fix remaining CI failures in the huge-claimed-length DoS regression tests
- Apply the same google-readability-casting fix (std::size_t{N} instead
  of std::size_t(N)) to the "arrays of various sizes" section in
  unit-msgpack.cpp, unit-ubjson.cpp, and unit-bjdata.cpp; only
  unit-cbor.cpp had been fixed previously, since clang-tidy's build
  didn't get far enough to report the other three in the same pass.

- json_sax_dom_parser::start_array()'s max_size() check calls JSON_THROW
  directly rather than going through sax->parse_error(), so unlike the
  scanner's own "not enough data" parse_error it is not gated by
  allow_exceptions=false. On a platform where a header's claimed count
  exceeds max_size() (e.g. 32-bit, for UBJSON/BJData's 0x7FFFFFFF test
  value), from_ubjson/from_bjdata(input, true, false) can therefore still
  throw instead of returning a discarded value. Make that assertion
  tolerant of either outcome, same as the main exception-catching check
  above it.

- Guard all four "a huge claimed length..." SECTIONs with
  #if !defined(JSON_NOEXCEPTION), matching this test suite's existing
  convention for exception-dependent tests: under JSON_NOEXCEPTION,
  JSON_THROW never produces a catchable C++ exception at all (it aborts
  the process), so a section that relies on try/catch to distinguish
  between two acceptable outcomes cannot be expressed under that build
  configuration regardless of platform.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-06 20:08:27 +02:00
Niels Lohmann 98ab9f31c3 Make the huge-claimed-length DoS regression tests portable across size_t widths
On a platform where size_t is narrower than 64 bits (e.g. 32-bit mingw/msvc
x86), the previously-hardcoded huge test lengths either collide with that
platform's unknown_size() sentinel (CBOR/MessagePack, both using exactly
SIZE_MAX) or exceed the platform's smaller vector<json>::max_size()
(UBJSON/BJData's 0x7FFFFFFF), so the header is now rejected outright
(out_of_range.408) instead of being accepted and only found short of data
(parse_error.110). Both are safe, bounded rejections of the hostile input;
the property under test -- no attempt to allocate space for billions of
elements -- holds either way. Accept both outcomes instead of pinning the
64-bit-only exact result.

Also fixed an unrelated clang-tidy finding (google-readability-casting) on
the functional-style std::size_t(...) casts in the neighboring "arrays of
various sizes" section.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-06 11:19:06 +02:00
Niels Lohmann aee9421883 Reserve capped array capacity for definite-length binary arrays
CBOR, MessagePack, and the optimized [$type#count UBJSON/BJData form all
pass an exact element count to sax->start_array(len), but
json_sax_dom_parser::start_array() (and the callback variant) only used
len for an overflow check against max_size() and never reserved the
underlying vector, so each element triggered a reallocation cascade via
emplace_back().

Reserve upfront, but cap the reservation at 16384 elements: max_size()
for a std::vector is far larger than any realistic input, so an
unbounded reserve(len) would let a crafted/truncated header (e.g. CBOR
0x9A + a huge uint32 count with no data) trigger a multi-gigabyte
allocation attempt instead of the normal graceful parse_error. With the
cap, a hostile length still fails fast with the existing parse_error,
while realistic arrays get a single up-front allocation.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-05 20:51:47 +02:00
7 changed files with 423 additions and 250 deletions
@@ -301,6 +301,16 @@ class json_sax_dom_parser
JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
} }
if (len != detail::unknown_size())
{
// reserve upfront to avoid repeated reallocations while adding elements,
// but cap the reservation so a bogus/hostile length (which is not bounded
// by max_size(), unlike e.g. std::vector) cannot trigger an oversized
// allocation for a small or truncated input
constexpr std::size_t reserve_cap = 16384;
ref_stack.back()->m_data.m_value.array->reserve(len < reserve_cap ? len : reserve_cap);
}
return true; return true;
} }
@@ -661,6 +671,16 @@ class json_sax_dom_callback_parser
{ {
JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
} }
if (len != detail::unknown_size())
{
// reserve upfront to avoid repeated reallocations while adding elements,
// but cap the reservation so a bogus/hostile length (which is not bounded
// by max_size(), unlike e.g. std::vector) cannot trigger an oversized
// allocation for a small or truncated input
constexpr std::size_t reserve_cap = 16384;
ref_stack.back()->m_data.m_value.array->reserve(len < reserve_cap ? len : reserve_cap);
}
} }
return true; return true;
@@ -1647,20 +1647,6 @@ class binary_writer
return 'D'; // float 64 return 'D'; // float 64
} }
/*!
@brief checks whether a JSON number fits into @a TargetType
@param[in] el a JSON number of either the signed or unsigned integer kind
@return whether @a el's value can be represented by @a TargetType without
wrapping, regardless of which of the two kinds it is stored as
*/
template<typename TargetType>
static bool bjdata_ndarray_value_in_range(const BasicJsonType& el)
{
return el.is_number_unsigned()
? value_in_range_of<TargetType>(el.template get<std::uint64_t>())
: value_in_range_of<TargetType>(el.template get<std::int64_t>());
}
/*! /*!
@return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid @return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid
*/ */
@@ -1681,16 +1667,6 @@ class binary_writer
} }
CharType dtype = it->second; CharType dtype = it->second;
// the 'B' (byte) marker is only defined by BJData Draft 3; emitting it
// under the default Draft 2 mode would produce a stream that Draft 2
// readers reject, so such an object falls back to a plain object
// encoding instead (see the "Binary values" section of the BJData
// documentation)
if (dtype == 'B' && bjdata_version != bjdata_version_t::draft3)
{
return true;
}
key = "_ArraySize_"; key = "_ArraySize_";
// the dimensions are written verbatim as the header length below, so a // the dimensions are written verbatim as the header length below, so a
// value that is not an array cannot produce a valid one: null emits 'Z' // value that is not an array cannot produce a valid one: null emits 'Z'
@@ -1755,60 +1731,6 @@ class binary_writer
} }
} }
// every element is cast to the (possibly narrower) C++ type matching
// dtype below; a value that does not fit that type would silently
// wrap (integers) or overflow to infinity (the "single" precision
// float) instead of being reported, so such an object falls back to
// a plain object encoding as well
for (const auto& el : value.at(key))
{
bool in_range = true;
switch (dtype)
{
case 'U':
case 'C':
case 'B':
in_range = bjdata_ndarray_value_in_range<std::uint8_t>(el);
break;
case 'i':
in_range = bjdata_ndarray_value_in_range<std::int8_t>(el);
break;
case 'u':
in_range = bjdata_ndarray_value_in_range<std::uint16_t>(el);
break;
case 'I':
in_range = bjdata_ndarray_value_in_range<std::int16_t>(el);
break;
case 'm':
in_range = bjdata_ndarray_value_in_range<std::uint32_t>(el);
break;
case 'l':
in_range = bjdata_ndarray_value_in_range<std::int32_t>(el);
break;
case 'M':
in_range = bjdata_ndarray_value_in_range<std::uint64_t>(el);
break;
case 'L':
in_range = bjdata_ndarray_value_in_range<std::int64_t>(el);
break;
case 'd':
{
const auto dval = el.template get<double>();
in_range = !std::isfinite(dval) ||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
break;
}
default:
// 'D' (double) already spans the full range of number_float_t
break;
}
if (!in_range)
{
return true;
}
}
oa->write_character('['); oa->write_character('[');
oa->write_character('$'); oa->write_character('$');
oa->write_character(dtype); oa->write_character(dtype);
+20 -78
View File
@@ -9829,6 +9829,16 @@ class json_sax_dom_parser
JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
} }
if (len != detail::unknown_size())
{
// reserve upfront to avoid repeated reallocations while adding elements,
// but cap the reservation so a bogus/hostile length (which is not bounded
// by max_size(), unlike e.g. std::vector) cannot trigger an oversized
// allocation for a small or truncated input
constexpr std::size_t reserve_cap = 16384;
ref_stack.back()->m_data.m_value.array->reserve(len < reserve_cap ? len : reserve_cap);
}
return true; return true;
} }
@@ -10189,6 +10199,16 @@ class json_sax_dom_callback_parser
{ {
JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
} }
if (len != detail::unknown_size())
{
// reserve upfront to avoid repeated reallocations while adding elements,
// but cap the reservation so a bogus/hostile length (which is not bounded
// by max_size(), unlike e.g. std::vector) cannot trigger an oversized
// allocation for a small or truncated input
constexpr std::size_t reserve_cap = 16384;
ref_stack.back()->m_data.m_value.array->reserve(len < reserve_cap ? len : reserve_cap);
}
} }
return true; return true;
@@ -18655,20 +18675,6 @@ class binary_writer
return 'D'; // float 64 return 'D'; // float 64
} }
/*!
@brief checks whether a JSON number fits into @a TargetType
@param[in] el a JSON number of either the signed or unsigned integer kind
@return whether @a el's value can be represented by @a TargetType without
wrapping, regardless of which of the two kinds it is stored as
*/
template<typename TargetType>
static bool bjdata_ndarray_value_in_range(const BasicJsonType& el)
{
return el.is_number_unsigned()
? value_in_range_of<TargetType>(el.template get<std::uint64_t>())
: value_in_range_of<TargetType>(el.template get<std::int64_t>());
}
/*! /*!
@return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid @return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid
*/ */
@@ -18689,16 +18695,6 @@ class binary_writer
} }
CharType dtype = it->second; CharType dtype = it->second;
// the 'B' (byte) marker is only defined by BJData Draft 3; emitting it
// under the default Draft 2 mode would produce a stream that Draft 2
// readers reject, so such an object falls back to a plain object
// encoding instead (see the "Binary values" section of the BJData
// documentation)
if (dtype == 'B' && bjdata_version != bjdata_version_t::draft3)
{
return true;
}
key = "_ArraySize_"; key = "_ArraySize_";
// the dimensions are written verbatim as the header length below, so a // the dimensions are written verbatim as the header length below, so a
// value that is not an array cannot produce a valid one: null emits 'Z' // value that is not an array cannot produce a valid one: null emits 'Z'
@@ -18763,60 +18759,6 @@ class binary_writer
} }
} }
// every element is cast to the (possibly narrower) C++ type matching
// dtype below; a value that does not fit that type would silently
// wrap (integers) or overflow to infinity (the "single" precision
// float) instead of being reported, so such an object falls back to
// a plain object encoding as well
for (const auto& el : value.at(key))
{
bool in_range = true;
switch (dtype)
{
case 'U':
case 'C':
case 'B':
in_range = bjdata_ndarray_value_in_range<std::uint8_t>(el);
break;
case 'i':
in_range = bjdata_ndarray_value_in_range<std::int8_t>(el);
break;
case 'u':
in_range = bjdata_ndarray_value_in_range<std::uint16_t>(el);
break;
case 'I':
in_range = bjdata_ndarray_value_in_range<std::int16_t>(el);
break;
case 'm':
in_range = bjdata_ndarray_value_in_range<std::uint32_t>(el);
break;
case 'l':
in_range = bjdata_ndarray_value_in_range<std::int32_t>(el);
break;
case 'M':
in_range = bjdata_ndarray_value_in_range<std::uint64_t>(el);
break;
case 'L':
in_range = bjdata_ndarray_value_in_range<std::int64_t>(el);
break;
case 'd':
{
const auto dval = el.template get<double>();
in_range = !std::isfinite(dval) ||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
break;
}
default:
// 'D' (double) already spans the full range of number_float_t
break;
}
if (!in_range)
{
return true;
}
}
oa->write_character('['); oa->write_character('[');
oa->write_character('$'); oa->write_character('$');
oa->write_character(dtype); oa->write_character(dtype);
+107 -94
View File
@@ -2586,12 +2586,7 @@ TEST_CASE("BJData")
CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d); CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d);
CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D); CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D);
CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C); CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C);
// v_B uses the Draft-3-only 'B' marker, so it round-trips only when CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true) == v_B);
// Draft 3 is explicitly selected (see GitHub issue #5404); the
// default Draft 2 falls back to a plain object instead, covered by
// the "ndarray with _ArrayType_ "byte" is gated by the BJData draft
// version" section below
CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true, json::bjdata_version_t::draft3) == v_B);
} }
SECTION("ndarray with data not matching _ArrayType_ is written as an object") SECTION("ndarray with data not matching _ArrayType_ is written as an object")
@@ -2634,10 +2629,8 @@ TEST_CASE("BJData")
// the C++ API stores an int literal as number_integer, so _ArrayType_ // the C++ API stores an int literal as number_integer, so _ArrayType_
// names the wire type rather than the storage. Both storages have to // names the wire type rather than the storage. Both storages have to
// produce the same typed array for every type. // produce the same typed array for every type.
// "byte" is checked separately below since it additionally requires
// BJData Draft 3 to be selected explicitly (see GitHub issue #5404).
for (const char* type : for (const char* type :
{"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char" {"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char", "byte"
}) })
{ {
CAPTURE(type); CAPTURE(type);
@@ -2648,14 +2641,6 @@ TEST_CASE("BJData")
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}))); CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}})));
} }
{
const std::string text = R"({"_ArrayType_":"byte","_ArraySize_":[2,3],"_ArrayData_":[1,2,3,4,5,6]})";
const auto from_text = json::to_bjdata(json::parse(text), true, true, json::bjdata_version_t::draft3);
CHECK(from_text.at(0) == '[');
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}),
true, true, json::bjdata_version_t::draft3));
}
// negative values under a signed type behave the same way // negative values under a signed type behave the same way
const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})")); const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})"));
CHECK(from_neg.at(0) == '['); CHECK(from_neg.at(0) == '[');
@@ -2791,83 +2776,6 @@ TEST_CASE("BJData")
CHECK(out_num.at(0) == '{'); CHECK(out_num.at(0) == '{');
CHECK(json::from_bjdata(out_num) == j_num); CHECK(json::from_bjdata(out_num) == j_num);
} }
SECTION("ndarray with out-of-range _ArrayData_ elements stays as object")
{
// each element is cast to the (possibly narrower) C++ type
// named by _ArrayType_ before being written; a value that
// does not fit that type would silently wrap instead of
// being reported, so such an object falls back to a plain
// object encoding that still round-trips (see GitHub issue #5403)
// an unsigned element that does not fit uint8
json const j_uint8 = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 256}}});
const auto out_uint8 = json::to_bjdata(j_uint8);
CHECK(out_uint8.at(0) == '{');
CHECK(json::from_bjdata(out_uint8) == j_uint8);
// a signed element that does not fit int8
json const j_int8 = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 200}}});
const auto out_int8 = json::to_bjdata(j_int8);
CHECK(out_int8.at(0) == '{');
CHECK(json::from_bjdata(out_int8) == j_int8);
// a negative element is likewise out of range for an
// unsigned _ArrayType_
json const j_uint16_neg = json({{"_ArrayType_", "uint16"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, -1}}});
const auto out_uint16_neg = json::to_bjdata(j_uint16_neg);
CHECK(out_uint16_neg.at(0) == '{');
CHECK(json::from_bjdata(out_uint16_neg) == j_uint16_neg);
// a double element that overflows to infinity when narrowed
// to the "single" (float) precision named by _ArrayType_
json const j_single = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1.5, 1e40}}});
const auto out_single = json::to_bjdata(j_single);
CHECK(out_single.at(0) == '{');
CHECK(json::from_bjdata(out_single) == j_single);
// in-range boundary values still use the compact ndarray encoding
json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {0, 255}}});
CHECK(json::to_bjdata(j_uint8_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, ']', 0, 255}));
json const j_int8_ok = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {-128, 127}}});
CHECK(json::to_bjdata(j_int8_ok) == std::vector<uint8_t>({'[', '$', 'i', '#', '[', 'i', 2, ']', 0x80, 0x7F}));
json const j_single_ok = json({{"_ArrayType_", "single"}, {"_ArraySize_", {1}}, {"_ArrayData_", {1.5}}});
const auto out_single_ok = json::to_bjdata(j_single_ok);
CHECK(out_single_ok.at(0) == '[');
CHECK(json::from_bjdata(out_single_ok) == json({1.5f}));
}
SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version")
{
// the 'B' (byte) marker used by _ArrayType_ "byte" is only defined
// by BJData Draft 3; Draft 2 (the default) has no such marker, so
// emitting it unconditionally produced a stream that a Draft 2
// reader could not parse as intended (see GitHub issue #5404).
// Two dimensions are used so that a successfully written ndarray
// round-trips back into the annotated object (a single dimension
// is, by the BJData ndarray convention, read back as a plain
// binary value rather than the annotated object, same as every
// other single-dimension ndarray of a non-"byte" type is read
// back as a plain array instead of the annotated object).
json const j_byte = json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}});
// default (Draft 2): falls back to a plain object and round-trips
const auto out_draft2 = json::to_bjdata(j_byte);
CHECK(out_draft2.at(0) == '{');
CHECK(json::from_bjdata(out_draft2) == j_byte);
// explicit Draft 2: same as the default
const auto out_draft2_explicit = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft2);
CHECK(out_draft2_explicit.at(0) == '{');
CHECK(json::from_bjdata(out_draft2_explicit) == j_byte);
// Draft 3 explicitly selected: still uses the compact 'B' ndarray encoding
const auto out_draft3 = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft3);
CHECK(out_draft3 == std::vector<uint8_t>({'[', '$', 'B', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6}));
CHECK(json::from_bjdata(out_draft3) == j_byte);
}
} }
} }
@@ -3581,6 +3489,111 @@ TEST_CASE("BJData")
} }
} }
TEST_CASE("issue #5405 - array reserve for definite-length BJData arrays")
{
#if !defined(JSON_NOEXCEPTION)
// this SECTION relies on catching a thrown exception to distinguish
// which of two acceptable, bounded rejections a hostile header took;
// under JSON_NOEXCEPTION, JSON_THROW never produces a catchable C++
// exception (it aborts instead), so this cannot be tested that way here
SECTION("a huge claimed length with no element data must not over-allocate")
{
// optimized form [$type#count: type 'i' (int8), count as a four-byte
// little-endian 'l' (int32) of 0x7FFFFFFF (2147483647), but no
// element data at all. max_size() for a std::vector is far larger
// than this count, so it does not reject the header outright; the
// (capped) reservation must not attempt to allocate space for
// billions of elements before the missing data is detected.
json _;
const std::vector<uint8_t> input = {'[', '$', 'i', '#', 'l', 0xFF, 0xFF, 0xFF, 0x7F};
// On a platform where std::vector<json>::max_size() is smaller than
// the claimed count (e.g. 32-bit, where max_size() is bounded by a
// 32-bit SIZE_MAX divided by sizeof(json)), the SAX consumer's own
// check rejects the header outright (out_of_range.408, with the
// claimed count in the message) instead of accepting it and only
// finding it short of data once the (capped) reservation looks for
// element bytes that were never provided (parse_error.110). Either
// is an acceptable, bounded rejection of the hostile header -- the
// property under test is that no path attempts to allocate space
// for billions of elements.
bool threw = false;
try
{
_ = json::from_bjdata(input);
}
catch (const json::parse_error& e)
{
threw = true;
CHECK(e.id == 110);
CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 10: syntax error while parsing BJData number: unexpected end of input");
}
catch (const json::out_of_range& e)
{
threw = true;
CHECK(e.id == 408);
CHECK(std::string(e.what()).find("excessive array size") != std::string::npos);
}
CHECK(threw);
// json_sax_dom_parser::start_array()'s max_size() check (unlike the
// scanner's own parse_error path) throws unconditionally via
// JSON_THROW rather than going through sax->parse_error(), so it is
// not gated by allow_exceptions=false on a platform where this
// header hits that check (e.g. 32-bit, see above) -- allow either
// a discarded result or the same out_of_range it throws with
// exceptions enabled.
try
{
CHECK(json::from_bjdata(input, true, false).is_discarded());
}
catch (const json::out_of_range& e)
{
CHECK(e.id == 408);
}
}
#endif
SECTION("arrays of various sizes decode to the same value as before the reserve optimization")
{
for (const auto size :
{
std::size_t{0}, std::size_t{1}, std::size_t{5}, // small
std::size_t{16384}, // exactly at the reserve cap
std::size_t{20000} // above the reserve cap
})
{
CAPTURE(size)
json j = json::array();
for (std::size_t i = 0; i < size; ++i)
{
j.push_back(static_cast<int>(i % 1000));
}
// exercise both the plain and the optimized [$type#count encoding
const auto packed_plain = json::to_bjdata(j);
CHECK(json::from_bjdata(packed_plain) == j);
const auto packed_optimized = json::to_bjdata(j, true, true);
CHECK(json::from_bjdata(packed_optimized) == j);
}
}
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
{
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
// a custom SAX consumer that does not touch a DOM array sees identical events
json j = json::array();
for (int i = 0; i < 100; ++i)
{
j.push_back(i);
}
const auto packed = json::to_bjdata(j, true, true);
SaxCountdown scp(1000000); // large enough to never trigger an abort
CHECK(json::sax_parse(packed, &scp, json::input_format_t::bjdata));
}
}
TEST_CASE("Universal Binary JSON Specification Examples 1") TEST_CASE("Universal Binary JSON Specification Examples 1")
{ {
SECTION("Null Value") SECTION("Null Value")
+86
View File
@@ -2035,6 +2035,92 @@ TEST_CASE("CBOR definite length equal to the indefinite-length sentinel")
} }
} }
TEST_CASE("issue #5405 - array reserve for definite-length CBOR arrays")
{
#if !defined(JSON_NOEXCEPTION)
// this SECTION relies on catching a thrown exception to distinguish
// which of two acceptable, bounded rejections a hostile header took;
// under JSON_NOEXCEPTION, JSON_THROW never produces a catchable C++
// exception (it aborts instead), so this cannot be tested that way here
SECTION("a huge claimed length with no element data must not over-allocate")
{
// 0x9A: array with a four-byte length; claims 0xFFFFFFFF (4294967295)
// elements but provides none. max_size() for a std::vector is far
// larger than this count, so it does not reject the header outright;
// the (capped) reservation must not attempt to allocate space for
// billions of elements before the missing data is detected.
json _;
const std::vector<uint8_t> input = {0x9A, 0xFF, 0xFF, 0xFF, 0xFF};
// On a platform where std::size_t is narrower than 64 bits (e.g.
// 32-bit), the claimed count 0xFFFFFFFF coincides with that
// platform's detail::unknown_size() sentinel (SIZE_MAX), so the
// format-level size check rejects it outright (out_of_range.408,
// "excessive ... size") before the SAX consumer's own max_size()
// check would even run; on a 64-bit platform it passes both of
// those checks and is only found short of data once the (capped)
// reservation looks for element bytes that were never provided
// (parse_error.110). Either is an acceptable, bounded rejection of
// the hostile header -- the property under test is that no path
// attempts to allocate space for billions of elements.
bool threw = false;
try
{
_ = json::from_cbor(input);
}
catch (const json::parse_error& e)
{
threw = true;
CHECK(e.id == 110);
CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 6: syntax error while parsing CBOR value: unexpected end of input");
}
catch (const json::out_of_range& e)
{
threw = true;
CHECK(e.id == 408);
CHECK(std::string(e.what()).find("excessive") != std::string::npos);
}
CHECK(threw);
CHECK(json::from_cbor(input, true, false).is_discarded());
}
#endif
SECTION("arrays of various sizes decode to the same value as before the reserve optimization")
{
for (const auto size :
{
std::size_t{0}, std::size_t{1}, std::size_t{5}, // small
std::size_t{16384}, // exactly at the reserve cap
std::size_t{20000} // above the reserve cap
})
{
CAPTURE(size)
json j = json::array();
for (std::size_t i = 0; i < size; ++i)
{
j.push_back(static_cast<int>(i % 1000));
}
const auto packed = json::to_cbor(j);
CHECK(json::from_cbor(packed) == j);
}
}
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
{
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
// a custom SAX consumer that does not touch a DOM array sees identical events
json j = json::array();
for (int i = 0; i < 100; ++i)
{
j.push_back(i);
}
const auto packed = json::to_cbor(j);
SaxCountdown scp(1000000); // large enough to never trigger an abort
CHECK(json::sax_parse(packed, &scp, json::input_format_t::cbor));
}
}
TEST_CASE("CBOR roundtrips" * doctest::skip()) TEST_CASE("CBOR roundtrips" * doctest::skip())
{ {
SECTION("input from flynn") SECTION("input from flynn")
+85
View File
@@ -1597,6 +1597,91 @@ TEST_CASE("MessagePack")
} }
} }
TEST_CASE("issue #5405 - array reserve for definite-length MessagePack arrays")
{
#if !defined(JSON_NOEXCEPTION)
// this SECTION relies on catching a thrown exception to distinguish
// which of two acceptable, bounded rejections a hostile header took;
// under JSON_NOEXCEPTION, JSON_THROW never produces a catchable C++
// exception (it aborts instead), so this cannot be tested that way here
SECTION("a huge claimed length with no element data must not over-allocate")
{
// 0xdd: array 32 (four-byte length); claims 0xFFFFFFFF (4294967295)
// elements but provides none. max_size() for a std::vector is far
// larger than this count, so it does not reject the header outright;
// the (capped) reservation must not attempt to allocate space for
// billions of elements before the missing data is detected.
json _;
const std::vector<uint8_t> input = {0xdd, 0xFF, 0xFF, 0xFF, 0xFF};
// On a platform where std::size_t is narrower than 64 bits (e.g.
// 32-bit), the claimed count 0xFFFFFFFF coincides with that
// platform's SIZE_MAX, which some size-narrowing checks treat the
// same as detail::unknown_size(); it may then be rejected before
// the SAX consumer's own max_size() check (out_of_range.408) rather
// than being accepted and only found short of data once the
// (capped) reservation looks for element bytes that were never
// provided (parse_error.110). Either is an acceptable, bounded
// rejection of the hostile header -- the property under test is
// that no path attempts to allocate space for billions of elements.
bool threw = false;
try
{
_ = json::from_msgpack(input);
}
catch (const json::parse_error& e)
{
threw = true;
CHECK(e.id == 110);
CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 6: syntax error while parsing MessagePack value: unexpected end of input");
}
catch (const json::out_of_range& e)
{
threw = true;
CHECK(e.id == 408);
CHECK(std::string(e.what()).find("excessive") != std::string::npos);
}
CHECK(threw);
CHECK(json::from_msgpack(input, true, false).is_discarded());
}
#endif
SECTION("arrays of various sizes decode to the same value as before the reserve optimization")
{
for (const auto size :
{
std::size_t{0}, std::size_t{1}, std::size_t{5}, // small
std::size_t{16384}, // exactly at the reserve cap
std::size_t{20000} // above the reserve cap
})
{
CAPTURE(size)
json j = json::array();
for (std::size_t i = 0; i < size; ++i)
{
j.push_back(static_cast<int>(i % 1000));
}
const auto packed = json::to_msgpack(j);
CHECK(json::from_msgpack(packed) == j);
}
}
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
{
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
// a custom SAX consumer that does not touch a DOM array sees identical events
json j = json::array();
for (int i = 0; i < 100; ++i)
{
j.push_back(i);
}
const auto packed = json::to_msgpack(j);
SaxCountdown scp(1000000); // large enough to never trigger an abort
CHECK(json::sax_parse(packed, &scp, json::input_format_t::msgpack));
}
}
// use this testcase outside [hide] to run it with Valgrind // use this testcase outside [hide] to run it with Valgrind
TEST_CASE("single MessagePack roundtrip") TEST_CASE("single MessagePack roundtrip")
{ {
+105
View File
@@ -2149,6 +2149,111 @@ TEST_CASE("UBJSON")
} }
} }
TEST_CASE("issue #5405 - array reserve for definite-length UBJSON arrays")
{
#if !defined(JSON_NOEXCEPTION)
// this SECTION relies on catching a thrown exception to distinguish
// which of two acceptable, bounded rejections a hostile header took;
// under JSON_NOEXCEPTION, JSON_THROW never produces a catchable C++
// exception (it aborts instead), so this cannot be tested that way here
SECTION("a huge claimed length with no element data must not over-allocate")
{
// optimized form [$type#count: type 'i' (int8), count as a four-byte
// 'l' (int32) of 0x7FFFFFFF (2147483647), but no element data at all.
// max_size() for a std::vector is far larger than this count, so it
// does not reject the header outright; the (capped) reservation must
// not attempt to allocate space for billions of elements before the
// missing data is detected.
json _;
const std::vector<uint8_t> input = {'[', '$', 'i', '#', 'l', 0x7F, 0xFF, 0xFF, 0xFF};
// On a platform where std::vector<json>::max_size() is smaller than
// the claimed count (e.g. 32-bit, where max_size() is bounded by a
// 32-bit SIZE_MAX divided by sizeof(json)), the SAX consumer's own
// check rejects the header outright (out_of_range.408, with the
// claimed count in the message) instead of accepting it and only
// finding it short of data once the (capped) reservation looks for
// element bytes that were never provided (parse_error.110). Either
// is an acceptable, bounded rejection of the hostile header -- the
// property under test is that no path attempts to allocate space
// for billions of elements.
bool threw = false;
try
{
_ = json::from_ubjson(input);
}
catch (const json::parse_error& e)
{
threw = true;
CHECK(e.id == 110);
CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 10: syntax error while parsing UBJSON number: unexpected end of input");
}
catch (const json::out_of_range& e)
{
threw = true;
CHECK(e.id == 408);
CHECK(std::string(e.what()).find("excessive array size") != std::string::npos);
}
CHECK(threw);
// json_sax_dom_parser::start_array()'s max_size() check (unlike the
// scanner's own parse_error path) throws unconditionally via
// JSON_THROW rather than going through sax->parse_error(), so it is
// not gated by allow_exceptions=false on a platform where this
// header hits that check (e.g. 32-bit, see above) -- allow either
// a discarded result or the same out_of_range it throws with
// exceptions enabled.
try
{
CHECK(json::from_ubjson(input, true, false).is_discarded());
}
catch (const json::out_of_range& e)
{
CHECK(e.id == 408);
}
}
#endif
SECTION("arrays of various sizes decode to the same value as before the reserve optimization")
{
for (const auto size :
{
std::size_t{0}, std::size_t{1}, std::size_t{5}, // small
std::size_t{16384}, // exactly at the reserve cap
std::size_t{20000} // above the reserve cap
})
{
CAPTURE(size)
json j = json::array();
for (std::size_t i = 0; i < size; ++i)
{
j.push_back(static_cast<int>(i % 1000));
}
// exercise both the plain and the optimized [$type#count encoding
const auto packed_plain = json::to_ubjson(j);
CHECK(json::from_ubjson(packed_plain) == j);
const auto packed_optimized = json::to_ubjson(j, true, true);
CHECK(json::from_ubjson(packed_optimized) == j);
}
}
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
{
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
// a custom SAX consumer that does not touch a DOM array sees identical events
json j = json::array();
for (int i = 0; i < 100; ++i)
{
j.push_back(i);
}
const auto packed = json::to_ubjson(j, true, true);
SaxCountdown scp(1000000); // large enough to never trigger an abort
CHECK(json::sax_parse(packed, &scp, json::input_format_t::ubjson));
}
}
TEST_CASE("Universal Binary JSON Specification Examples 1") TEST_CASE("Universal Binary JSON Specification Examples 1")
{ {
SECTION("Null Value") SECTION("Null Value")