Merge remote-tracking branch 'origin/develop' into claude/fix-issue-3989-db7e45

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann committed 2026-10-07 08:54:17 +02:00
commit 1d2cb49ff3
17 files changed
+694 -33

No files matched your search

+44 -4
View File
@@ -10,7 +10,9 @@
This file implements a parser test suitable for fuzz testing. Given a byte
array data, it performs the following steps:
- j0 = from_bjdata(data, allow_exceptions = false)
- j1 = from_bjdata(data)
- assert(j0 is discarded if parsing j1 fails, and j0 == j1 otherwise)
- vec2 = to_bjdata(j1, use_size = false, use_type = false)
- vec3 = to_bjdata(j1, use_size = true, use_type = false)
- vec4 = to_bjdata(j1, use_size = true, use_type = true)
@@ -65,6 +67,13 @@ drivers.
using json = nlohmann::json;
// compares dumps rather than values, because NaN != NaN; keep writes strings
// byte for byte, so ill-formed UTF-8 that a binary reader accepts cannot throw
static bool same_value(const json& lhs, const json& rhs)
{
return lhs.dump(-1, ' ', false, json::error_handler_t::keep) == rhs.dump(-1, ' ', false, json::error_handler_t::keep);
}
// value-stable comparison for the round-trip checks below; see the note
// above on why this compares dump()s rather than the json values directly
static bool is_value_stable(const json& lhs, const json& rhs)
@@ -75,14 +84,42 @@ static bool is_value_stable(const json& lhs, const json& rhs)
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
{
// step 0: recover from all errors, reading from memory and from a stream
// recover from all errors, reading from memory and from a stream
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bjdata).errors == 0;
std::vector<uint8_t> const vec1(data, data + size);
// step 0: parse input without exceptions; a parse error must then be
// reported as a discarded value, never thrown
json j_noexcept;
bool noexcept_threw = false;
try
{
j_noexcept = json::from_bjdata(vec1, true, false);
}
catch (const json::parse_error&)
{
assert(false);
}
catch (const json::exception&)
{
// type and out-of-range errors are not parse errors and still throw
noexcept_threw = true;
}
// whether step 1 succeeded; if not, the catch blocks below check that
// step 0 failed, too
bool parsed = false;
try
{
// step 1: parse input
std::vector<uint8_t> const vec1(data, data + size);
json const j1 = json::from_bjdata(vec1);
parsed = true;
// without exceptions, the same input must give the same value
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
// the recovering parser must not have reported an error either
assert(recovered_without_errors);
try
@@ -117,16 +154,19 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
catch (const json::parse_error&)
{
// parse errors are ok, because input may be random bytes
assert(!recovered_without_errors);
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
assert(parsed || !recovered_without_errors);
}
catch (const json::type_error&)
{
// type errors can occur during parsing, too
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
}
catch (const json::out_of_range&)
{
// out of range errors may happen if provided sizes are excessive
assert(!recovered_without_errors);
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
assert(parsed || !recovered_without_errors);
}
// return 0 - non-zero return values are reserved for future use
+44 -4
View File
@@ -10,7 +10,9 @@
This file implements a parser test suitable for fuzz testing. Given a byte
array data, it performs the following steps:
- j0 = from_bon8(data, allow_exceptions = false)
- j1 = from_bon8(data)
- assert(j0 is discarded if parsing j1 fails, and j0 == j1 otherwise)
- vec = to_bon8(j1)
- j2 = from_bon8(vec)
- assert(to_bon8(j2) == vec)
@@ -40,6 +42,13 @@ drivers.
using json = nlohmann::json;
// compares dumps rather than values, because NaN != NaN; keep writes strings
// byte for byte, so ill-formed UTF-8 that a binary reader accepts cannot throw
static bool same_value(const json& lhs, const json& rhs)
{
return lhs.dump(-1, ' ', false, json::error_handler_t::keep) == rhs.dump(-1, ' ', false, json::error_handler_t::keep);
}
namespace
{
// the serialization of the value read from @a input, or the error message
@@ -61,7 +70,7 @@ std::string read_bon8(InputType&& input)
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
{
// step 0: recover from all errors, reading from memory and from a stream
// recover from all errors, reading from memory and from a stream
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bon8).errors == 0;
// contiguous and stream input must be read alike
@@ -70,11 +79,39 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
assert(read_bon8(std::vector<uint8_t>(data, data + size)) == read_bon8(stream));
}
std::vector<uint8_t> const vec1(data, data + size);
// step 0: parse input without exceptions; a parse error must then be
// reported as a discarded value, never thrown
json j_noexcept;
bool noexcept_threw = false;
try
{
j_noexcept = json::from_bon8(vec1, true, false);
}
catch (const json::parse_error&)
{
assert(false);
}
catch (const json::exception&)
{
// type and out-of-range errors are not parse errors and still throw
noexcept_threw = true;
}
// whether step 1 succeeded; if not, the catch blocks below check that
// step 0 failed, too
bool parsed = false;
try
{
// step 1: parse input
std::vector<uint8_t> const vec1(data, data + size);
json const j1 = json::from_bon8(vec1);
parsed = true;
// without exceptions, the same input must give the same value
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
// the recovering parser must not have reported an error either
assert(recovered_without_errors);
try
@@ -97,16 +134,19 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
catch (const json::parse_error&)
{
// parse errors are ok, because input may be random bytes
assert(!recovered_without_errors);
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
assert(parsed || !recovered_without_errors);
}
catch (const json::type_error&)
{
// type errors can occur during parsing, too
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
}
catch (const json::out_of_range&)
{
// out of range errors may happen if provided sizes are excessive
assert(!recovered_without_errors);
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
assert(parsed || !recovered_without_errors);
}
// return 0 - non-zero return values are reserved for future use
+44 -4
View File
@@ -10,7 +10,9 @@
This file implements a parser test suitable for fuzz testing. Given a byte
array data, it performs the following steps:
- j0 = from_bson(data, allow_exceptions = false)
- j1 = from_bson(data)
- assert(j0 is discarded if parsing j1 fails, and j0 == j1 otherwise)
- vec = to_bson(j1)
- j2 = from_bson(vec)
- assert(to_bson(j2) == vec)
@@ -35,17 +37,52 @@ drivers.
using json = nlohmann::json;
// compares dumps rather than values, because NaN != NaN; keep writes strings
// byte for byte, so ill-formed UTF-8 that a binary reader accepts cannot throw
static bool same_value(const json& lhs, const json& rhs)
{
return lhs.dump(-1, ' ', false, json::error_handler_t::keep) == rhs.dump(-1, ' ', false, json::error_handler_t::keep);
}
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
{
// step 0: recover from all errors, reading from memory and from a stream
// recover from all errors, reading from memory and from a stream
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bson).errors == 0;
std::vector<uint8_t> const vec1(data, data + size);
// step 0: parse input without exceptions; a parse error must then be
// reported as a discarded value, never thrown
json j_noexcept;
bool noexcept_threw = false;
try
{
j_noexcept = json::from_bson(vec1, true, false);
}
catch (const json::parse_error&)
{
assert(false);
}
catch (const json::exception&)
{
// type and out-of-range errors are not parse errors and still throw
noexcept_threw = true;
}
// whether step 1 succeeded; if not, the catch blocks below check that
// step 0 failed, too
bool parsed = false;
try
{
// step 1: parse input
std::vector<uint8_t> const vec1(data, data + size);
json const j1 = json::from_bson(vec1);
parsed = true;
// without exceptions, the same input must give the same value
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
// the recovering parser must not have reported an error either
assert(recovered_without_errors);
try
@@ -68,16 +105,19 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
catch (const json::parse_error&)
{
// parse errors are ok, because input may be random bytes
assert(!recovered_without_errors);
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
assert(parsed || !recovered_without_errors);
}
catch (const json::type_error&)
{
// type errors can occur during parsing, too
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
}
catch (const json::out_of_range&)
{
// out of range errors can occur during parsing, too
assert(!recovered_without_errors);
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
assert(parsed || !recovered_without_errors);
}
// return 0 - non-zero return values are reserved for future use
+44 -4
View File
@@ -10,7 +10,9 @@
This file implements a parser test suitable for fuzz testing. Given a byte
array data, it performs the following steps:
- j0 = from_cbor(data, allow_exceptions = false)
- j1 = from_cbor(data)
- assert(j0 is discarded if parsing j1 fails, and j0 == j1 otherwise)
- vec = to_cbor(j1)
- j2 = from_cbor(vec)
- assert(to_cbor(j2) == vec)
@@ -35,17 +37,52 @@ drivers.
using json = nlohmann::json;
// compares dumps rather than values, because NaN != NaN; keep writes strings
// byte for byte, so ill-formed UTF-8 that a binary reader accepts cannot throw
static bool same_value(const json& lhs, const json& rhs)
{
return lhs.dump(-1, ' ', false, json::error_handler_t::keep) == rhs.dump(-1, ' ', false, json::error_handler_t::keep);
}
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
{
// step 0: recover from all errors, reading from memory and from a stream
// recover from all errors, reading from memory and from a stream
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::cbor).errors == 0;
std::vector<uint8_t> const vec1(data, data + size);
// step 0: parse input without exceptions; a parse error must then be
// reported as a discarded value, never thrown
json j_noexcept;
bool noexcept_threw = false;
try
{
j_noexcept = json::from_cbor(vec1, true, false);
}
catch (const json::parse_error&)
{
assert(false);
}
catch (const json::exception&)
{
// type and out-of-range errors are not parse errors and still throw
noexcept_threw = true;
}
// whether step 1 succeeded; if not, the catch blocks below check that
// step 0 failed, too
bool parsed = false;
try
{
// step 1: parse input
std::vector<uint8_t> const vec1(data, data + size);
json const j1 = json::from_cbor(vec1);
parsed = true;
// without exceptions, the same input must give the same value
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
// the recovering parser must not have reported an error either
assert(recovered_without_errors);
try
@@ -68,16 +105,19 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
catch (const json::parse_error&)
{
// parse errors are ok, because input may be random bytes
assert(!recovered_without_errors);
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
assert(parsed || !recovered_without_errors);
}
catch (const json::type_error&)
{
// type errors can occur during parsing, too
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
}
catch (const json::out_of_range&)
{
// out of range errors can occur during parsing, too
assert(!recovered_without_errors);
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
assert(parsed || !recovered_without_errors);
}
// return 0 - non-zero return values are reserved for future use
+37 -1
View File
@@ -10,7 +10,9 @@
This file implements a parser test suitable for fuzz testing. Given a byte
array data, it performs the following steps:
- j0 = parse(data, allow_exceptions = false)
- j1 = parse(data)
- assert(j0 is discarded if parsing j1 fails, and j0 == j1 otherwise)
- s1 = serialize(j1)
- j2 = parse(s1)
- s2 = serialize(j2)
@@ -36,20 +38,52 @@ drivers.
using json = nlohmann::json;
// compares dumps rather than values, because NaN != NaN; keep writes strings
// byte for byte, so ill-formed UTF-8 that a binary reader accepts cannot throw
static bool same_value(const json& lhs, const json& rhs)
{
return lhs.dump(-1, ' ', false, json::error_handler_t::keep) == rhs.dump(-1, ' ', false, json::error_handler_t::keep);
}
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
{
// step 0: recover from all errors, reading from memory and from a stream
// recover from all errors, reading from memory and from a stream
{
const auto checker = check_recovering_parse(data, size, json::input_format_t::json);
assert(checker.events <= (4 * size) + 4);
assert((checker.errors == 0) == json::accept(data, data + size));
}
// step 0: parse input without exceptions; a parse error must then be
// reported as a discarded value, never thrown
json j_noexcept;
bool noexcept_threw = false;
try
{
j_noexcept = json::parse(data, data + size, nullptr, false);
}
catch (const json::parse_error&)
{
assert(false);
}
catch (const json::exception&)
{
// type and out-of-range errors are not parse errors and still throw
noexcept_threw = true;
}
// whether step 1 succeeded; if not, the catch blocks below check that
// step 0 failed, too
bool parsed = false;
try
{
// step 1: parse input
json const j1 = json::parse(data, data + size);
parsed = true;
// without exceptions, the same input must give the same value
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
try
{
@@ -76,10 +110,12 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
catch (const json::parse_error&)
{
// parse errors are ok, because input may be random bytes
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
}
catch (const json::out_of_range&)
{
// out of range errors may happen if provided sizes are excessive
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
}
// return 0 - non-zero return values are reserved for future use
+44 -4
View File
@@ -10,7 +10,9 @@
This file implements a parser test suitable for fuzz testing. Given a byte
array data, it performs the following steps:
- j0 = from_msgpack(data, allow_exceptions = false)
- j1 = from_msgpack(data)
- assert(j0 is discarded if parsing j1 fails, and j0 == j1 otherwise)
- vec = to_msgpack(j1)
- j2 = from_msgpack(vec)
- assert(to_msgpack(j2) == vec)
@@ -35,17 +37,52 @@ drivers.
using json = nlohmann::json;
// compares dumps rather than values, because NaN != NaN; keep writes strings
// byte for byte, so ill-formed UTF-8 that a binary reader accepts cannot throw
static bool same_value(const json& lhs, const json& rhs)
{
return lhs.dump(-1, ' ', false, json::error_handler_t::keep) == rhs.dump(-1, ' ', false, json::error_handler_t::keep);
}
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
{
// step 0: recover from all errors, reading from memory and from a stream
// recover from all errors, reading from memory and from a stream
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::msgpack).errors == 0;
std::vector<uint8_t> const vec1(data, data + size);
// step 0: parse input without exceptions; a parse error must then be
// reported as a discarded value, never thrown
json j_noexcept;
bool noexcept_threw = false;
try
{
j_noexcept = json::from_msgpack(vec1, true, false);
}
catch (const json::parse_error&)
{
assert(false);
}
catch (const json::exception&)
{
// type and out-of-range errors are not parse errors and still throw
noexcept_threw = true;
}
// whether step 1 succeeded; if not, the catch blocks below check that
// step 0 failed, too
bool parsed = false;
try
{
// step 1: parse input
std::vector<uint8_t> const vec1(data, data + size);
json const j1 = json::from_msgpack(vec1);
parsed = true;
// without exceptions, the same input must give the same value
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
// the recovering parser must not have reported an error either
assert(recovered_without_errors);
try
@@ -68,16 +105,19 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
catch (const json::parse_error&)
{
// parse errors are ok, because input may be random bytes
assert(!recovered_without_errors);
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
assert(parsed || !recovered_without_errors);
}
catch (const json::type_error&)
{
// type errors can occur during parsing, too
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
}
catch (const json::out_of_range&)
{
// out of range errors may happen if provided sizes are excessive
assert(!recovered_without_errors);
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
assert(parsed || !recovered_without_errors);
}
// return 0 - non-zero return values are reserved for future use
+44 -4
View File
@@ -10,7 +10,9 @@
This file implements a parser test suitable for fuzz testing. Given a byte
array data, it performs the following steps:
- j0 = from_ubjson(data, allow_exceptions = false)
- j1 = from_ubjson(data)
- assert(j0 is discarded if parsing j1 fails, and j0 == j1 otherwise)
- vec2 = to_ubjson(j1, use_size = false, use_type = false)
- vec3 = to_ubjson(j1, use_size = true, use_type = false)
- vec4 = to_ubjson(j1, use_size = true, use_type = true)
@@ -44,17 +46,52 @@ drivers.
using json = nlohmann::json;
// compares dumps rather than values, because NaN != NaN; keep writes strings
// byte for byte, so ill-formed UTF-8 that a binary reader accepts cannot throw
static bool same_value(const json& lhs, const json& rhs)
{
return lhs.dump(-1, ' ', false, json::error_handler_t::keep) == rhs.dump(-1, ' ', false, json::error_handler_t::keep);
}
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
{
// step 0: recover from all errors, reading from memory and from a stream
// recover from all errors, reading from memory and from a stream
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::ubjson).errors == 0;
std::vector<uint8_t> const vec1(data, data + size);
// step 0: parse input without exceptions; a parse error must then be
// reported as a discarded value, never thrown
json j_noexcept;
bool noexcept_threw = false;
try
{
j_noexcept = json::from_ubjson(vec1, true, false);
}
catch (const json::parse_error&)
{
assert(false);
}
catch (const json::exception&)
{
// type and out-of-range errors are not parse errors and still throw
noexcept_threw = true;
}
// whether step 1 succeeded; if not, the catch blocks below check that
// step 0 failed, too
bool parsed = false;
try
{
// step 1: parse input
std::vector<uint8_t> const vec1(data, data + size);
json const j1 = json::from_ubjson(vec1);
parsed = true;
// without exceptions, the same input must give the same value
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
// the recovering parser must not have reported an error either
assert(recovered_without_errors);
try
@@ -87,16 +124,19 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
catch (const json::parse_error&)
{
// parse errors are ok, because input may be random bytes
assert(!recovered_without_errors);
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
assert(parsed || !recovered_without_errors);
}
catch (const json::type_error&)
{
// type errors can occur during parsing, too
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
}
catch (const json::out_of_range&)
{
// out of range errors may happen if provided sizes are excessive
assert(!recovered_without_errors);
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
assert(parsed || !recovered_without_errors);
}
// return 0 - non-zero return values are reserved for future use
+46
View File
@@ -4603,3 +4603,49 @@ TEST_CASE("issue #5648 - from_bjdata(ptr, len) must read len bytes, not treat pt
CHECK(json::from_bjdata(packed.data(), packed.size(), false) == j);
#endif
}
TEST_CASE("BJData large strings and binaries (chunked reader)")
{
// Strings share get_ubjson_string() -> get_string() -> get_bytes() with
// plain UBJSON. Binary values are different: only a Draft 3 optimized
// array (type marker 'B') is read back as a binary value, through
// get_binary() -> get_bytes() (see parse_ubjson_internal()'s "If BJData
// type marker is 'B'" branch); Draft 2 (the default) writes a binary
// value as a plain array of uint8_t numbers instead (see the "round trip
// of a binary value is value-stable, not byte-stable" test above), which
// never reaches get_bytes(). Both reads happen in bounded chunks
// (binary_reader.hpp, chunk_size == 4096); check lengths around and
// beyond that size, for both vector (iterator) and pointer inputs.
for (const std::size_t len :
{
std::size_t{0}, std::size_t{1}, std::size_t{4095}, std::size_t{4096},
std::size_t{4097}, std::size_t{8192}, std::size_t{100000}
})
{
CAPTURE(len)
// string
const json j_string = std::string(len, 'x');
const std::vector<std::uint8_t> v_string = json::to_bjdata(j_string);
CHECK(json::from_bjdata(v_string) == j_string);
// pointer input exercises the std::memcpy fast path
CHECK(json::from_bjdata(reinterpret_cast<const char*>(v_string.data()),
reinterpret_cast<const char*>(v_string.data()) + v_string.size()) == j_string);
// binary, forced into the Draft 3 optimized ('B' marker) encoding
const json j_binary = json::binary(std::vector<std::uint8_t>(len, 0xCD));
const std::vector<std::uint8_t> v_binary = json::to_bjdata(j_binary, true, true, json::bjdata_version_t::draft3);
CHECK(json::from_bjdata(v_binary) == j_binary);
CHECK(json::from_bjdata(reinterpret_cast<const char*>(v_binary.data()),
reinterpret_cast<const char*>(v_binary.data()) + v_binary.size()) == j_binary);
// a truncated payload must still be reported as an error
if (len > 16)
{
std::vector<std::uint8_t> truncated = v_string;
truncated.resize(truncated.size() - 8);
json _;
CHECK_THROWS_AS(_ = json::from_bjdata(truncated), json::parse_error);
}
}
}
+44
View File
@@ -1993,3 +1993,47 @@ TEST_CASE("Invalid document size handling")
CHECK(json::from_bson(v, true, false).is_discarded());
}
}
TEST_CASE("BSON large strings and binaries (chunked reader)")
{
// get_bson_string()/get_bson_binary() both read through get_string()/
// get_binary(), which read in bounded chunks (binary_reader.hpp,
// chunk_size == 4096); make sure roundtripping is correct for lengths
// around and beyond that chunk size, for both vector (iterator) and
// pointer inputs. BSON only accepts an object at the top level, so the
// string/binary value is wrapped in one.
for (const std::size_t len :
{
std::size_t{0}, std::size_t{1}, std::size_t{4095}, std::size_t{4096},
std::size_t{4097}, std::size_t{8192}, std::size_t{100000}
})
{
CAPTURE(len)
// string
const json j_string = {{"k", std::string(len, 'x')}};
const std::vector<std::uint8_t> v_string = json::to_bson(j_string);
CHECK(json::from_bson(v_string) == j_string);
// pointer input exercises the std::memcpy fast path
CHECK(json::from_bson(reinterpret_cast<const char*>(v_string.data()),
reinterpret_cast<const char*>(v_string.data()) + v_string.size()) == j_string);
// binary (BSON binary values always carry a subtype, so give one
// explicitly; otherwise from_bson() would round-trip to subtype 0
// rather than back to the original "no subtype" value)
const json j_binary = {{"k", json::binary(std::vector<std::uint8_t>(len, 0xCD), std::uint8_t{0})}};
const std::vector<std::uint8_t> v_binary = json::to_bson(j_binary);
CHECK(json::from_bson(v_binary) == j_binary);
CHECK(json::from_bson(reinterpret_cast<const char*>(v_binary.data()),
reinterpret_cast<const char*>(v_binary.data()) + v_binary.size()) == j_binary);
// a truncated payload must still be reported as an error
if (len > 16)
{
std::vector<std::uint8_t> truncated = v_string;
truncated.resize(truncated.size() - 8);
json _;
CHECK_THROWS_AS(_ = json::from_bson(truncated), json::parse_error);
}
}
}
+40
View File
@@ -2482,3 +2482,43 @@ TEST_CASE("MessagePack numbers use the active union member (see #5644)")
CHECK(json::from_msgpack(result) == j);
}
}
TEST_CASE("MessagePack large strings and binaries (chunked reader)")
{
// get_msgpack_string()/get_msgpack_binary() both read through get_binary(),
// which reads in bounded chunks (binary_reader.hpp, chunk_size == 4096);
// make sure roundtripping is correct for lengths around and beyond that
// chunk size, for both vector (iterator) and pointer inputs.
for (const std::size_t len :
{
std::size_t{0}, std::size_t{1}, std::size_t{4095}, std::size_t{4096},
std::size_t{4097}, std::size_t{8192}, std::size_t{100000}
})
{
CAPTURE(len)
// string
const json j_string = std::string(len, 'x');
const std::vector<std::uint8_t> v_string = json::to_msgpack(j_string);
CHECK(json::from_msgpack(v_string) == j_string);
// pointer input exercises the std::memcpy fast path
CHECK(json::from_msgpack(reinterpret_cast<const char*>(v_string.data()),
reinterpret_cast<const char*>(v_string.data()) + v_string.size()) == j_string);
// binary
const json j_binary = json::binary(std::vector<std::uint8_t>(len, 0xCD));
const std::vector<std::uint8_t> v_binary = json::to_msgpack(j_binary);
CHECK(json::from_msgpack(v_binary) == j_binary);
CHECK(json::from_msgpack(reinterpret_cast<const char*>(v_binary.data()),
reinterpret_cast<const char*>(v_binary.data()) + v_binary.size()) == j_binary);
// a truncated payload must still be reported as an error
if (len > 16)
{
std::vector<std::uint8_t> truncated = v_string;
truncated.resize(truncated.size() - 8);
json _;
CHECK_THROWS_AS(_ = json::from_msgpack(truncated), json::parse_error);
}
}
}
+3 -3
View File
@@ -533,7 +533,7 @@ TEST_CASE("regression tests 3")
}
#endif
#if JSON_HAS_RANGES && !defined(__MINGW32__)
#if JSON_HAS_RANGE_VIEW_CONVERSION
SECTION("issue #4916 - constructing array from C++20 ranges view does not work")
{
std::vector<int> nums{1, 2, 37, 42, 21};
@@ -548,7 +548,7 @@ TEST_CASE("regression tests 3")
#endif
// owning_view is not available in libstdc++ < 12
#if JSON_HAS_RANGES && !defined(__MINGW32__) && !(defined(__GLIBCXX__) && _GLIBCXX_RELEASE < 12)
#if JSON_HAS_RANGE_VIEW_CONVERSION && !(defined(__GLIBCXX__) && _GLIBCXX_RELEASE < 12)
SECTION("issue #4916 - constructing array from prvalue C++20 ranges view (owning_view)")
{
json const j(std::vector<int> {1, 2, 37, 42, 21} | std::views::filter([](int i)
@@ -560,7 +560,7 @@ TEST_CASE("regression tests 3")
}
#endif
#if JSON_HAS_RANGES && !defined(__MINGW32__)
#if JSON_HAS_RANGE_VIEW_CONVERSION
SECTION("issue #4916 - constructing array from C++20 transform view (prvalue elements)")
{
std::vector<int> nums{1, 2, 3};
+175
View File
@@ -809,3 +809,178 @@ TEST_CASE("serializer buffers are flushed mid-string and mid-binary")
CHECK(j.dump(2) == "{\n \"bytes\": [" + expected_pretty_bytes + "],\n \"subtype\": null\n}");
}
}
TEST_CASE("serialization boundary values for the write buffer")
{
// write_buffer is a std::array<char, 1024> (write_buffer_size). put_string()
// guards it with two checks, and each must be exercised exactly on and one
// past its own boundary: a heap overflow in a different manual buffer path
// (the dump(1100) indent buffer) once survived 100% line coverage because
// every test that touched it only ever grew the buffer by a single step,
// never landing on the exact edge of the comparison that protects it.
//
// - straight-through: put_string() bypasses write_buffer entirely and
// writes directly to the output adapter once `length >= write_buffer.size()`.
// - flush-then-copy: otherwise, if `write_buffer_pos + length > write_buffer.size()`,
// put_string() flushes what is pending and then memcpy's the new run into
// the freshly emptied buffer.
SECTION("top-level string exercises the straight-through guard (length >= 1024)")
{
// dump() of a bare string writes the opening quote with put_char()
// (write_buffer_pos: 0 -> 1), then the body with put_string(). With
// write_buffer_pos == 1, `1 + length > 1024` and `length >= 1024` flip
// together at length 1024, so 1023/1024/1025 cover "just under",
// "exactly at" and "just over" the guard in one move: 1023 is copied
// into the buffer (filling it exactly), 1024 and 1025 bypass it.
for (const std::size_t len :
{
std::size_t{1023}, std::size_t{1024}, std::size_t{1025}
})
{
CAPTURE(len)
const std::string body(len, 'a');
const json j = body;
const std::string expected = '"' + body + '"';
CHECK(j.dump() == expected);
std::ostringstream o;
o << j;
CHECK(o.str() == expected);
}
}
SECTION("string nested in an array exercises the flush-then-copy guard")
{
// json::array({body}) writes '[' then '"' before the body, so
// write_buffer_pos == 2 when put_string() is entered for it. The
// body's last byte then lands at logical offset 2 + len: len == 1022
// lands exactly on offset 1024 (2 + 1022 == write_buffer.size(), so the
// strict "> " guard does not fire and the body fits snugly), while
// len == 1023 lands one past it at offset 1025 (2 + 1023 > 1024),
// which must flush what's pending before copying the body in.
for (const std::size_t len :
{
std::size_t{1022}, std::size_t{1023}
})
{
CAPTURE(len)
const std::string body(len, 'a');
const json j = json::array({body});
const std::string expected = "[\"" + body + "\"]";
CHECK(j.dump() == expected);
std::ostringstream o;
o << j;
CHECK(o.str() == expected);
CHECK(json::parse(j.dump()) == j);
}
}
}
TEST_CASE("serialization boundary values for the string buffer")
{
// string_buffer is a std::array<char, 512>. dump_escaped_impl() flushes it
// mid-string once fewer than 13 bytes remain (`string_buffer.size() - bytes
// < 13`), 13 being one more than the most a single code point can ever
// write at once (a surrogate pair: two back-to-back "\uXXXX" escapes, 12
// bytes). Every write into string_buffer that this check protects happens
// in steps of 2 (a simple "\\x" escape) or 6 (one "\uXXXX" unit), so
// `bytes` only ever takes even values at the point the check runs - the
// tightest values actually reachable are therefore 498 (512 - 498 == 14,
// one simple escape away from the threshold) and 500 (512 - 500 == 12,
// where the flush fires immediately and resets bytes to 0).
SECTION("a run of 2-byte escapes lands bytes on, and one step past, the flush threshold")
{
for (const int count :
{
249, 250, 251
})
{
CAPTURE(count)
const json j = std::string(static_cast<std::size_t>(count), '\n');
std::string expected = "\"";
for (int i = 0; i < count; ++i)
{
expected += "\\n";
}
expected += '"';
CHECK(j.dump() == expected);
}
}
SECTION("an ASCII prefix leaves the tightest reachable margin before a 12-byte surrogate pair")
{
// U+1F600 (the "\xF0\x9F\x98\x80" UTF-8 bytes) is dumped under
// ensure_ascii as the 12-byte surrogate pair "\ud83d\ude00"; that
// write happens in a single step with no intermediate flush check, so
// it is the write most exposed by an off-by-one in the "< 13" guard.
// A prefix of 249 newlines leaves exactly 14 bytes of headroom
// (512 - 498), the smallest margin the guard ever actually allows
// into a new code point; 250 newlines instead trigger the guard's own
// flush first, so the emoji starts from a freshly emptied (512-byte)
// buffer, and 251 repeats that with one more escape already past the
// reset. Together they cover the margin the guard allows landing on,
// one step before, and one step after - all must still produce the
// identical, correct escapes.
for (const int prefix_count :
{
249, 250, 251
})
{
CAPTURE(prefix_count)
const std::string prefix(static_cast<std::size_t>(prefix_count), '\n');
const std::string emoji = "\xF0\x9F\x98\x80";
const json j = prefix + emoji;
std::string expected_prefix;
for (int i = 0; i < prefix_count; ++i)
{
expected_prefix += "\\n";
}
// newline escaping does not depend on ensure_ascii: only the
// emoji differs (raw UTF-8 bytes vs. a \u-escaped surrogate pair)
CHECK(j.dump(-1, ' ', false) == '"' + expected_prefix + emoji + '"');
CHECK(j.dump(-1, ' ', true) == '"' + expected_prefix + "\\ud83d\\ude00\"");
CHECK(json::parse(j.dump(-1, ' ', true)) == j);
CHECK(json::parse(j.dump(-1, ' ', false)) == j);
}
}
SECTION("SWAR bulk-copy stride: k plain bytes followed by a byte handled individually")
{
// string_bulk_run()/find_ascii_copyable_run() (string_scan.hpp) scan 8
// bytes at a time and fall back to a byte-at-a-time tail scan for
// what is left over. k from 0 to 17 spans zero, one and two full
// 8-byte strides plus a 1-byte tail, so every possible stopping point
// within and right after the SIMD stride is covered.
for (std::size_t k = 0; k <= 17; ++k)
{
CAPTURE(k)
const std::string prefix(k, 'a');
// (a) the run is stopped by a quote that must itself be escaped
{
const json j = prefix + "\"";
CHECK(j.dump() == '"' + prefix + "\\\"" + '"');
}
// (b) the run is stopped by a control character
{
const json j = prefix + "\x01";
CHECK(j.dump() == '"' + prefix + "\\u0001" + '"');
}
// (c) the run is stopped by a non-ASCII byte under ensure_ascii
{
const json j = prefix + "\xC3\xA9"; // prefix + 'é'
CHECK(j.dump(-1, ' ', true) == '"' + prefix + "\\u00e9" + '"');
}
}
}
}
+40
View File
@@ -3308,3 +3308,43 @@ TEST_CASE("UBJSON and BJData integer markers at every range edge")
}
}
}
TEST_CASE("UBJSON large strings (chunked reader)")
{
// get_ubjson_string() reads through get_string(), which reads in bounded
// chunks (binary_reader.hpp, chunk_size == 4096); make sure roundtripping
// is correct for lengths around and beyond that chunk size, for both
// vector (iterator) and pointer inputs.
//
// A binary value is not included here: plain UBJSON (unlike BJData, see
// the "BJData large strings and binaries" test) has no reader-side binary
// type, so even the optimized uint8_t-array encoding of a binary value is
// read back element-by-element as a JSON array of numbers rather than
// through get_binary() - it never reaches the chunked path this test is
// about (see the "roundtrip only works to an array of numbers" case
// above).
for (const std::size_t len :
{
std::size_t{0}, std::size_t{1}, std::size_t{4095}, std::size_t{4096},
std::size_t{4097}, std::size_t{8192}, std::size_t{100000}
})
{
CAPTURE(len)
const json j_string = std::string(len, 'x');
const std::vector<std::uint8_t> v_string = json::to_ubjson(j_string);
CHECK(json::from_ubjson(v_string) == j_string);
// pointer input exercises the std::memcpy fast path
CHECK(json::from_ubjson(reinterpret_cast<const char*>(v_string.data()),
reinterpret_cast<const char*>(v_string.data()) + v_string.size()) == j_string);
// a truncated payload must still be reported as an error
if (len > 16)
{
std::vector<std::uint8_t> truncated = v_string;
truncated.resize(truncated.size() - 8);
json _;
CHECK_THROWS_AS(_ = json::from_ubjson(truncated), json::parse_error);
}
}
}