mirror of
https://github.com/nlohmann/json.git
synced 2026-09-12 19:27:59 +00:00
Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
93295e08b7 | ||
|
|
aa391dc0a5 | ||
|
|
d2514a46f7 |
@@ -432,21 +432,7 @@ class binary_reader
|
|||||||
exception_message(input_format_t::bson, concat("string length must be at least 1, is ", std::to_string(len)), "string"), nullptr));
|
exception_message(input_format_t::bson, concat("string length must be at least 1, is ", std::to_string(len)), "string"), nullptr));
|
||||||
}
|
}
|
||||||
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(!get_string(input_format_t::bson, len - static_cast<NumberType>(1), result)))
|
return get_string(input_format_t::bson, len - static_cast<NumberType>(1), result) && get() != char_traits<char_type>::eof();
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(get() != 0x00))
|
|
||||||
{
|
|
||||||
auto last_token = get_token_string();
|
|
||||||
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
|
|
||||||
exception_message(input_format_t::bson,
|
|
||||||
"BSON string is not null-terminated",
|
|
||||||
"string"), nullptr));
|
|
||||||
}
|
|
||||||
|
|
||||||
return true;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@@ -564,6 +550,8 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
//////////
|
//////////
|
||||||
// CBOR //
|
// CBOR //
|
||||||
//////////
|
//////////
|
||||||
@@ -1446,7 +1434,7 @@ class binary_reader
|
|||||||
// a copy, not a reference: it must stay valid across the
|
// a copy, not a reference: it must stay valid across the
|
||||||
// pop_back() below, which destroys the container_stack element
|
// pop_back() below, which destroys the container_stack element
|
||||||
// it would otherwise alias
|
// it would otherwise alias
|
||||||
container_frame top = container_stack.back();
|
const container_frame top = container_stack.back();
|
||||||
bool at_end = false;
|
bool at_end = false;
|
||||||
|
|
||||||
if (top.remaining != npos)
|
if (top.remaining != npos)
|
||||||
@@ -2223,7 +2211,7 @@ class binary_reader
|
|||||||
// would otherwise alias.
|
// would otherwise alias.
|
||||||
for (;;)
|
for (;;)
|
||||||
{
|
{
|
||||||
container_frame top = container_stack.back();
|
const container_frame top = container_stack.back();
|
||||||
|
|
||||||
if (top.remaining != npos)
|
if (top.remaining != npos)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -1691,6 +1691,16 @@ class binary_writer
|
|||||||
}
|
}
|
||||||
CharType dtype = it->second;
|
CharType dtype = it->second;
|
||||||
|
|
||||||
|
// the 'B' (byte) marker is only defined by BJData Draft 3; emitting it
|
||||||
|
// under the default Draft 2 mode would produce a stream that Draft 2
|
||||||
|
// readers reject, so such an object falls back to a plain object
|
||||||
|
// encoding instead (see the "Binary values" section of the BJData
|
||||||
|
// documentation)
|
||||||
|
if (dtype == 'B' && bjdata_version != bjdata_version_t::draft3)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
key = "_ArraySize_";
|
key = "_ArraySize_";
|
||||||
// the dimensions are written verbatim as the header length below, so a
|
// the dimensions are written verbatim as the header length below, so a
|
||||||
// value that is not an array cannot produce a valid one: null emits 'Z'
|
// value that is not an array cannot produce a valid one: null emits 'Z'
|
||||||
|
|||||||
@@ -12381,21 +12381,7 @@ class binary_reader
|
|||||||
exception_message(input_format_t::bson, concat("string length must be at least 1, is ", std::to_string(len)), "string"), nullptr));
|
exception_message(input_format_t::bson, concat("string length must be at least 1, is ", std::to_string(len)), "string"), nullptr));
|
||||||
}
|
}
|
||||||
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(!get_string(input_format_t::bson, len - static_cast<NumberType>(1), result)))
|
return get_string(input_format_t::bson, len - static_cast<NumberType>(1), result) && get() != char_traits<char_type>::eof();
|
||||||
{
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(get() != 0x00))
|
|
||||||
{
|
|
||||||
auto last_token = get_token_string();
|
|
||||||
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
|
|
||||||
exception_message(input_format_t::bson,
|
|
||||||
"BSON string is not null-terminated",
|
|
||||||
"string"), nullptr));
|
|
||||||
}
|
|
||||||
|
|
||||||
return true;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@@ -12513,6 +12499,8 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
//////////
|
//////////
|
||||||
// CBOR //
|
// CBOR //
|
||||||
//////////
|
//////////
|
||||||
@@ -13395,7 +13383,7 @@ class binary_reader
|
|||||||
// a copy, not a reference: it must stay valid across the
|
// a copy, not a reference: it must stay valid across the
|
||||||
// pop_back() below, which destroys the container_stack element
|
// pop_back() below, which destroys the container_stack element
|
||||||
// it would otherwise alias
|
// it would otherwise alias
|
||||||
container_frame top = container_stack.back();
|
const container_frame top = container_stack.back();
|
||||||
bool at_end = false;
|
bool at_end = false;
|
||||||
|
|
||||||
if (top.remaining != npos)
|
if (top.remaining != npos)
|
||||||
@@ -14172,7 +14160,7 @@ class binary_reader
|
|||||||
// would otherwise alias.
|
// would otherwise alias.
|
||||||
for (;;)
|
for (;;)
|
||||||
{
|
{
|
||||||
container_frame top = container_stack.back();
|
const container_frame top = container_stack.back();
|
||||||
|
|
||||||
if (top.remaining != npos)
|
if (top.remaining != npos)
|
||||||
{
|
{
|
||||||
@@ -20238,6 +20226,16 @@ class binary_writer
|
|||||||
}
|
}
|
||||||
CharType dtype = it->second;
|
CharType dtype = it->second;
|
||||||
|
|
||||||
|
// the 'B' (byte) marker is only defined by BJData Draft 3; emitting it
|
||||||
|
// under the default Draft 2 mode would produce a stream that Draft 2
|
||||||
|
// readers reject, so such an object falls back to a plain object
|
||||||
|
// encoding instead (see the "Binary values" section of the BJData
|
||||||
|
// documentation)
|
||||||
|
if (dtype == 'B' && bjdata_version != bjdata_version_t::draft3)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
key = "_ArraySize_";
|
key = "_ArraySize_";
|
||||||
// the dimensions are written verbatim as the header length below, so a
|
// the dimensions are written verbatim as the header length below, so a
|
||||||
// value that is not an array cannot produce a valid one: null emits 'Z'
|
// value that is not an array cannot produce a valid one: null emits 'Z'
|
||||||
|
|||||||
@@ -252,4 +252,323 @@ static void BinaryToCbor(benchmark::State& state)
|
|||||||
}
|
}
|
||||||
BENCHMARK(BinaryToCbor)->RangeMultiplier(2)->Range(8, 8 << 12);
|
BENCHMARK(BinaryToCbor)->RangeMultiplier(2)->Range(8, 8 << 12);
|
||||||
|
|
||||||
|
//////////////////////////////////////////////////////////////////////////////
|
||||||
|
// parse binary formats
|
||||||
|
//////////////////////////////////////////////////////////////////////////////
|
||||||
|
|
||||||
|
// Only MessagePack had a read benchmark (FromMsgpack above, left untouched so
|
||||||
|
// its numbers stay comparable across releases). The benchmarks below cover the
|
||||||
|
// other formats, and read from a contiguous buffer as well as from a FILE*:
|
||||||
|
// most callers pass a container, and the two adapters compile to different
|
||||||
|
// code. The test data repository ships JSON only, so the input for each is
|
||||||
|
// derived at setup time by serializing a parsed test file.
|
||||||
|
|
||||||
|
/// binary format to benchmark; the _optimized variants add UBJSON/BJData size
|
||||||
|
/// and type annotations, which the readers handle in a separate code path
|
||||||
|
enum class binary_format
|
||||||
|
{
|
||||||
|
cbor,
|
||||||
|
msgpack,
|
||||||
|
ubjson,
|
||||||
|
ubjson_optimized,
|
||||||
|
bjdata,
|
||||||
|
bjdata_optimized,
|
||||||
|
bson
|
||||||
|
};
|
||||||
|
|
||||||
|
static std::vector<std::uint8_t> to_binary(const json& j, const binary_format format)
|
||||||
|
{
|
||||||
|
switch (format)
|
||||||
|
{
|
||||||
|
case binary_format::cbor:
|
||||||
|
return json::to_cbor(j);
|
||||||
|
case binary_format::msgpack:
|
||||||
|
return json::to_msgpack(j);
|
||||||
|
case binary_format::ubjson:
|
||||||
|
return json::to_ubjson(j);
|
||||||
|
case binary_format::ubjson_optimized:
|
||||||
|
return json::to_ubjson(j, true, true);
|
||||||
|
case binary_format::bjdata:
|
||||||
|
return json::to_bjdata(j);
|
||||||
|
case binary_format::bjdata_optimized:
|
||||||
|
return json::to_bjdata(j, true, true);
|
||||||
|
case binary_format::bson:
|
||||||
|
default:
|
||||||
|
return json::to_bson(j);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static json from_binary(const std::vector<std::uint8_t>& bytes, const binary_format format)
|
||||||
|
{
|
||||||
|
switch (format)
|
||||||
|
{
|
||||||
|
case binary_format::cbor:
|
||||||
|
return json::from_cbor(bytes);
|
||||||
|
case binary_format::msgpack:
|
||||||
|
return json::from_msgpack(bytes);
|
||||||
|
case binary_format::ubjson:
|
||||||
|
case binary_format::ubjson_optimized:
|
||||||
|
return json::from_ubjson(bytes);
|
||||||
|
case binary_format::bjdata:
|
||||||
|
case binary_format::bjdata_optimized:
|
||||||
|
return json::from_bjdata(bytes);
|
||||||
|
case binary_format::bson:
|
||||||
|
default:
|
||||||
|
return json::from_bson(bytes);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static json from_binary(std::FILE* file, const binary_format format)
|
||||||
|
{
|
||||||
|
switch (format)
|
||||||
|
{
|
||||||
|
case binary_format::cbor:
|
||||||
|
return json::from_cbor(file);
|
||||||
|
case binary_format::msgpack:
|
||||||
|
return json::from_msgpack(file);
|
||||||
|
case binary_format::ubjson:
|
||||||
|
case binary_format::ubjson_optimized:
|
||||||
|
return json::from_ubjson(file);
|
||||||
|
case binary_format::bjdata:
|
||||||
|
case binary_format::bjdata_optimized:
|
||||||
|
return json::from_bjdata(file);
|
||||||
|
case binary_format::bson:
|
||||||
|
default:
|
||||||
|
return json::from_bson(file);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief serialize a parsed test file to @a format
|
||||||
|
|
||||||
|
Returns an empty vector and marks the benchmark as skipped if the file cannot
|
||||||
|
be represented in the format, rather than letting the exception escape: BSON
|
||||||
|
requires an object at the top level, and several test files are arrays.
|
||||||
|
*/
|
||||||
|
static std::vector<std::uint8_t> binary_input(benchmark::State& state, const char* filename, const binary_format format)
|
||||||
|
{
|
||||||
|
std::ifstream f(filename);
|
||||||
|
std::string const str((std::istreambuf_iterator<char>(f)), std::istreambuf_iterator<char>());
|
||||||
|
const json j = json::parse(str);
|
||||||
|
|
||||||
|
if (format == binary_format::bson && !j.is_object())
|
||||||
|
{
|
||||||
|
state.SkipWithError("BSON requires an object at the top level");
|
||||||
|
return {};
|
||||||
|
}
|
||||||
|
|
||||||
|
return to_binary(j, format);
|
||||||
|
}
|
||||||
|
|
||||||
|
static void FromBinaryBuffer(benchmark::State& state, const char* filename, const binary_format format)
|
||||||
|
{
|
||||||
|
const std::vector<std::uint8_t> bytes = binary_input(state, filename, format);
|
||||||
|
if (bytes.empty())
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (auto _ : state)
|
||||||
|
{
|
||||||
|
// the value is destroyed outside the timed section, because destroying
|
||||||
|
// a large DOM is not what this benchmark measures
|
||||||
|
state.PauseTiming();
|
||||||
|
auto* j = new json();
|
||||||
|
state.ResumeTiming();
|
||||||
|
|
||||||
|
*j = from_binary(bytes, format);
|
||||||
|
|
||||||
|
state.PauseTiming();
|
||||||
|
delete j;
|
||||||
|
state.ResumeTiming();
|
||||||
|
}
|
||||||
|
|
||||||
|
state.SetBytesProcessed(state.iterations() * bytes.size());
|
||||||
|
}
|
||||||
|
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / floats, TEST_DATA_DIRECTORY "/regression/floats.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / signed_ints, TEST_DATA_DIRECTORY "/regression/signed_ints.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson_optimized / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson_optimized);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson_optimized / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson_optimized);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bjdata);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata_optimized / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bjdata_optimized);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata_optimized / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata_optimized);
|
||||||
|
// BSON requires an object at the top level, so the array-rooted test files
|
||||||
|
// (jeopardy and the regression files) cannot be captured here
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bson / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::bson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bson);
|
||||||
|
|
||||||
|
static void FromBinaryFile(benchmark::State& state, const char* filename, const binary_format format)
|
||||||
|
{
|
||||||
|
const std::vector<std::uint8_t> bytes = binary_input(state, filename, format);
|
||||||
|
if (bytes.empty())
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
const char* tmp = "benchmark_input.bin";
|
||||||
|
std::ofstream o(tmp, std::ios::binary);
|
||||||
|
o.write(reinterpret_cast<const char*>(bytes.data()), static_cast<std::streamsize>(bytes.size()));
|
||||||
|
o.flush();
|
||||||
|
o.close();
|
||||||
|
|
||||||
|
for (auto _ : state)
|
||||||
|
{
|
||||||
|
state.PauseTiming();
|
||||||
|
auto* j = new json();
|
||||||
|
auto* file = std::fopen(tmp, "rb");
|
||||||
|
state.ResumeTiming();
|
||||||
|
|
||||||
|
*j = from_binary(file, format);
|
||||||
|
|
||||||
|
state.PauseTiming();
|
||||||
|
std::fclose(file);
|
||||||
|
delete j;
|
||||||
|
state.ResumeTiming();
|
||||||
|
}
|
||||||
|
|
||||||
|
state.SetBytesProcessed(state.iterations() * bytes.size());
|
||||||
|
}
|
||||||
|
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryFile, cbor / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryFile, cbor / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryFile, ubjson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryFile, ubjson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryFile, bjdata / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryFile, bson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bson);
|
||||||
|
|
||||||
|
//////////////////////////////////////////////////////////////////////////////
|
||||||
|
// parse binary formats: value shapes
|
||||||
|
//////////////////////////////////////////////////////////////////////////////
|
||||||
|
|
||||||
|
// The test files above are wide and shallow, but the readers' cost is per
|
||||||
|
// container, so these cover the shapes that stress the container handling
|
||||||
|
// itself. Every shape is wrapped in an object so that BSON, which requires an
|
||||||
|
// object at the top level, measures the same value as the other formats.
|
||||||
|
|
||||||
|
/// deeply nested arrays: one container per level, no other work
|
||||||
|
static json make_nested()
|
||||||
|
{
|
||||||
|
json nested = json::array();
|
||||||
|
json* p = &nested;
|
||||||
|
for (std::size_t i = 1; i < 1000; ++i)
|
||||||
|
{
|
||||||
|
p->push_back(json::array());
|
||||||
|
p = &p->operator[](0);
|
||||||
|
}
|
||||||
|
|
||||||
|
json j = json::object();
|
||||||
|
j["data"] = std::move(nested);
|
||||||
|
return j;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// many sibling containers: maximum container churn, minimum nesting
|
||||||
|
static json make_containers()
|
||||||
|
{
|
||||||
|
json data = json::array();
|
||||||
|
for (std::size_t i = 0; i < 100000; ++i)
|
||||||
|
{
|
||||||
|
data.push_back(json::array({1, 2}));
|
||||||
|
}
|
||||||
|
|
||||||
|
json j = json::object();
|
||||||
|
j["data"] = std::move(data);
|
||||||
|
return j;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// one flat array of numbers: the scalar decoding path, which must not move
|
||||||
|
static json make_scalars()
|
||||||
|
{
|
||||||
|
json data = json::array();
|
||||||
|
for (std::size_t i = 0; i < 1000000; ++i)
|
||||||
|
{
|
||||||
|
data.push_back(i);
|
||||||
|
}
|
||||||
|
|
||||||
|
json j = json::object();
|
||||||
|
j["data"] = std::move(data);
|
||||||
|
return j;
|
||||||
|
}
|
||||||
|
|
||||||
|
static void FromBinaryShape(benchmark::State& state, json (*build)(), const binary_format format)
|
||||||
|
{
|
||||||
|
const std::vector<std::uint8_t> bytes = to_binary(build(), format);
|
||||||
|
|
||||||
|
for (auto _ : state)
|
||||||
|
{
|
||||||
|
state.PauseTiming();
|
||||||
|
auto* j = new json();
|
||||||
|
state.ResumeTiming();
|
||||||
|
|
||||||
|
*j = from_binary(bytes, format);
|
||||||
|
|
||||||
|
state.PauseTiming();
|
||||||
|
delete j;
|
||||||
|
state.ResumeTiming();
|
||||||
|
}
|
||||||
|
|
||||||
|
state.SetBytesProcessed(state.iterations() * bytes.size());
|
||||||
|
}
|
||||||
|
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, nested / cbor, make_nested, binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, nested / msgpack, make_nested, binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, nested / ubjson, make_nested, binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, nested / bjdata, make_nested, binary_format::bjdata);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, nested / bson, make_nested, binary_format::bson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, containers / cbor, make_containers, binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, containers / msgpack, make_containers, binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, containers / ubjson, make_containers, binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, containers / ubjson_optimized, make_containers, binary_format::ubjson_optimized);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, containers / bjdata, make_containers, binary_format::bjdata);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, containers / bson, make_containers, binary_format::bson);
|
||||||
|
// BSON names every array element, so a large array measures key generation
|
||||||
|
// rather than scalar decoding and is left out here
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, scalars / cbor, make_scalars, binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, scalars / msgpack, make_scalars, binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, scalars / ubjson, make_scalars, binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, scalars / bjdata, make_scalars, binary_format::bjdata);
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief parse an indefinite-length CBOR string
|
||||||
|
|
||||||
|
The writer never emits this form, so the input is assembled by hand: 0x7F
|
||||||
|
opens the string, each chunk is a one-character string, and 0xFF closes it.
|
||||||
|
*/
|
||||||
|
static void FromCborChunkedString(benchmark::State& state, const std::size_t chunks)
|
||||||
|
{
|
||||||
|
std::vector<std::uint8_t> bytes;
|
||||||
|
bytes.reserve(2 * chunks + 2);
|
||||||
|
bytes.push_back(0x7F);
|
||||||
|
for (std::size_t i = 0; i < chunks; ++i)
|
||||||
|
{
|
||||||
|
bytes.push_back(0x61); // string of length 1
|
||||||
|
bytes.push_back(0x61); // 'a'
|
||||||
|
}
|
||||||
|
bytes.push_back(0xFF);
|
||||||
|
|
||||||
|
for (auto _ : state)
|
||||||
|
{
|
||||||
|
json j = json::from_cbor(bytes);
|
||||||
|
benchmark::DoNotOptimize(j);
|
||||||
|
}
|
||||||
|
|
||||||
|
state.SetBytesProcessed(state.iterations() * bytes.size());
|
||||||
|
}
|
||||||
|
|
||||||
|
BENCHMARK_CAPTURE(FromCborChunkedString, 10000 chunks, 10000);
|
||||||
|
|
||||||
BENCHMARK_MAIN();
|
BENCHMARK_MAIN();
|
||||||
|
|||||||
@@ -2586,7 +2586,12 @@ TEST_CASE("BJData")
|
|||||||
CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d);
|
CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d);
|
||||||
CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D);
|
CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D);
|
||||||
CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C);
|
CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C);
|
||||||
CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true) == v_B);
|
// v_B uses the Draft-3-only 'B' marker, so it round-trips only when
|
||||||
|
// Draft 3 is explicitly selected (see GitHub issue #5404); the
|
||||||
|
// default Draft 2 falls back to a plain object instead, covered by
|
||||||
|
// the "ndarray with _ArrayType_ "byte" is gated by the BJData draft
|
||||||
|
// version" section below
|
||||||
|
CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true, json::bjdata_version_t::draft3) == v_B);
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("ndarray with data not matching _ArrayType_ is written as an object")
|
SECTION("ndarray with data not matching _ArrayType_ is written as an object")
|
||||||
@@ -2629,8 +2634,10 @@ TEST_CASE("BJData")
|
|||||||
// the C++ API stores an int literal as number_integer, so _ArrayType_
|
// the C++ API stores an int literal as number_integer, so _ArrayType_
|
||||||
// names the wire type rather than the storage. Both storages have to
|
// names the wire type rather than the storage. Both storages have to
|
||||||
// produce the same typed array for every type.
|
// produce the same typed array for every type.
|
||||||
|
// "byte" is checked separately below since it additionally requires
|
||||||
|
// BJData Draft 3 to be selected explicitly (see GitHub issue #5404).
|
||||||
for (const char* type :
|
for (const char* type :
|
||||||
{"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char", "byte"
|
{"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char"
|
||||||
})
|
})
|
||||||
{
|
{
|
||||||
CAPTURE(type);
|
CAPTURE(type);
|
||||||
@@ -2641,6 +2648,14 @@ TEST_CASE("BJData")
|
|||||||
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}})));
|
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}})));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
{
|
||||||
|
const std::string text = R"({"_ArrayType_":"byte","_ArraySize_":[2,3],"_ArrayData_":[1,2,3,4,5,6]})";
|
||||||
|
const auto from_text = json::to_bjdata(json::parse(text), true, true, json::bjdata_version_t::draft3);
|
||||||
|
CHECK(from_text.at(0) == '[');
|
||||||
|
CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}),
|
||||||
|
true, true, json::bjdata_version_t::draft3));
|
||||||
|
}
|
||||||
|
|
||||||
// negative values under a signed type behave the same way
|
// negative values under a signed type behave the same way
|
||||||
const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})"));
|
const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})"));
|
||||||
CHECK(from_neg.at(0) == '[');
|
CHECK(from_neg.at(0) == '[');
|
||||||
@@ -2823,6 +2838,36 @@ TEST_CASE("BJData")
|
|||||||
CHECK(out_single_ok.at(0) == '[');
|
CHECK(out_single_ok.at(0) == '[');
|
||||||
CHECK(json::from_bjdata(out_single_ok) == json({1.5f}));
|
CHECK(json::from_bjdata(out_single_ok) == json({1.5f}));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version")
|
||||||
|
{
|
||||||
|
// the 'B' (byte) marker used by _ArrayType_ "byte" is only defined
|
||||||
|
// by BJData Draft 3; Draft 2 (the default) has no such marker, so
|
||||||
|
// emitting it unconditionally produced a stream that a Draft 2
|
||||||
|
// reader could not parse as intended (see GitHub issue #5404).
|
||||||
|
// Two dimensions are used so that a successfully written ndarray
|
||||||
|
// round-trips back into the annotated object (a single dimension
|
||||||
|
// is, by the BJData ndarray convention, read back as a plain
|
||||||
|
// binary value rather than the annotated object, same as every
|
||||||
|
// other single-dimension ndarray of a non-"byte" type is read
|
||||||
|
// back as a plain array instead of the annotated object).
|
||||||
|
json const j_byte = json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}});
|
||||||
|
|
||||||
|
// default (Draft 2): falls back to a plain object and round-trips
|
||||||
|
const auto out_draft2 = json::to_bjdata(j_byte);
|
||||||
|
CHECK(out_draft2.at(0) == '{');
|
||||||
|
CHECK(json::from_bjdata(out_draft2) == j_byte);
|
||||||
|
|
||||||
|
// explicit Draft 2: same as the default
|
||||||
|
const auto out_draft2_explicit = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft2);
|
||||||
|
CHECK(out_draft2_explicit.at(0) == '{');
|
||||||
|
CHECK(json::from_bjdata(out_draft2_explicit) == j_byte);
|
||||||
|
|
||||||
|
// Draft 3 explicitly selected: still uses the compact 'B' ndarray encoding
|
||||||
|
const auto out_draft3 = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft3);
|
||||||
|
CHECK(out_draft3 == std::vector<uint8_t>({'[', '$', 'B', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6}));
|
||||||
|
CHECK(json::from_bjdata(out_draft3) == j_byte);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1688,76 +1688,3 @@ TEST_CASE("BSON roundtrips" * doctest::skip())
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST_CASE("Invalid document size handling")
|
|
||||||
{
|
|
||||||
SECTION("document size must be at least 5")
|
|
||||||
{
|
|
||||||
std::vector<std::uint8_t> const v = {0x04, 0x00, 0x00, 0x00, 0x00};
|
|
||||||
json _;
|
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.112] parse error at byte 5: syntax error while parsing BSON document: document size 4 does not match the number of bytes read (5)", json::parse_error&);
|
|
||||||
CHECK(json::from_bson(v, true, false).is_discarded());
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("declared document size must match consumed bytes (extra trailing element)")
|
|
||||||
{
|
|
||||||
// Declares 5-byte empty document but appends an int32 element after the declared end.
|
|
||||||
std::vector<std::uint8_t> const v =
|
|
||||||
{
|
|
||||||
0x05, 0x00, 0x00, 0x00,
|
|
||||||
0x10, 'a', 'd', 'm', 'i', 'n', 0x00,
|
|
||||||
0x01, 0x00, 0x00, 0x00,
|
|
||||||
0x00
|
|
||||||
};
|
|
||||||
json _;
|
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.112] parse error at byte 16: syntax error while parsing BSON document: document size 5 does not match the number of bytes read (16)", json::parse_error&);
|
|
||||||
CHECK(json::from_bson(v, true, false).is_discarded());
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("declared document size must match consumed bytes (premature terminator)")
|
|
||||||
{
|
|
||||||
// Declares 32-byte document but only contains the size field followed by an immediate terminator.
|
|
||||||
std::vector<std::uint8_t> const v =
|
|
||||||
{
|
|
||||||
0x20, 0x00, 0x00, 0x00,
|
|
||||||
0x00
|
|
||||||
};
|
|
||||||
json _;
|
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.112] parse error at byte 5: syntax error while parsing BSON document: document size 32 does not match the number of bytes read (5)", json::parse_error&);
|
|
||||||
CHECK(json::from_bson(v, true, false).is_discarded());
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("array declared size must match consumed bytes")
|
|
||||||
{
|
|
||||||
// Outer object contains an array "a" that declares 5 bytes (empty) but
|
|
||||||
// actually contains an int32 element before its terminator.
|
|
||||||
std::vector<std::uint8_t> const v =
|
|
||||||
{
|
|
||||||
0x14, 0x00, 0x00, 0x00, // object size = 20
|
|
||||||
0x04, 'a', 0x00, // key "a", array type
|
|
||||||
0x05, 0x00, 0x00, 0x00, // array declared size = 5 (empty)
|
|
||||||
0x10, '0', 0x00, 0x01, 0x00, 0x00, 0x00, // extra int32 element "0" = 1
|
|
||||||
0x00, // array terminator
|
|
||||||
0x00 // object terminator
|
|
||||||
};
|
|
||||||
json _;
|
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.112] parse error at byte 19: syntax error while parsing BSON document: document size 5 does not match the number of bytes read (12)", json::parse_error&);
|
|
||||||
CHECK(json::from_bson(v, true, false).is_discarded());
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("BSON string must end with 0x00")
|
|
||||||
{
|
|
||||||
// Length-prefixed string whose terminator byte is 'X' (0x58), not 0x00.
|
|
||||||
std::vector<std::uint8_t> const v =
|
|
||||||
{
|
|
||||||
0x0F, 0x00, 0x00, 0x00,
|
|
||||||
0x02, 's', 0x00,
|
|
||||||
0x02, 0x00, 0x00, 0x00,
|
|
||||||
'A', 'X',
|
|
||||||
0x00
|
|
||||||
};
|
|
||||||
json _;
|
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.112] parse error at byte 13: syntax error while parsing BSON string: BSON string is not null-terminated", json::parse_error&);
|
|
||||||
CHECK(json::from_bson(v, true, false).is_discarded());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -469,7 +469,7 @@ TEST_CASE("serialization of strings (bulk fast path)")
|
|||||||
SECTION("invalid UTF-8 handling is unaffected by the fast path")
|
SECTION("invalid UTF-8 handling is unaffected by the fast path")
|
||||||
{
|
{
|
||||||
const json j = std::string("valid\xff" "more");
|
const json j = std::string("valid\xff" "more");
|
||||||
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0xFF", json::type_error&);
|
CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0xFF", json::type_error&);
|
||||||
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"valid\xef\xbf\xbd" "more\"");
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"valid\xef\xbf\xbd" "more\"");
|
||||||
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"valid\\ufffdmore\"");
|
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"valid\\ufffdmore\"");
|
||||||
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"validmore\"");
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"validmore\"");
|
||||||
|
|||||||
Reference in New Issue
Block a user