mirror of
https://github.com/nlohmann/json.git
synced 2026-09-08 09:18:00 +00:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
81bc423ead |
@@ -149,11 +149,10 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
public:
|
public:
|
||||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||||
|
|
||||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept
|
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept
|
||||||
: ia(std::move(adapter))
|
: ia(std::move(adapter))
|
||||||
, ignore_comments(ignore_comments_)
|
, ignore_comments(ignore_comments_)
|
||||||
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
||||||
, discard_number_values(discard_number_values_)
|
|
||||||
{}
|
{}
|
||||||
|
|
||||||
// deleted because of pointer members
|
// deleted because of pointer members
|
||||||
@@ -1280,58 +1279,6 @@ scan_number_done:
|
|||||||
// we are done scanning a number)
|
// we are done scanning a number)
|
||||||
unget();
|
unget();
|
||||||
|
|
||||||
// If the caller does not need the converted value (only whether the
|
|
||||||
// input is syntactically valid; see json_sax_acceptor/accept()), an
|
|
||||||
// unsigned/integer token can be reported without calling
|
|
||||||
// strtoull()/strtoll() at all, *provided* we can already tell from
|
|
||||||
// the digit count alone that the conversion cannot overflow 64 bits.
|
|
||||||
// Such tokens are always finite and are accepted unconditionally by
|
|
||||||
// the parser regardless of their actual value (parser::sax_parse_internal()
|
|
||||||
// never checks finiteness for value_unsigned/value_integer), so the
|
|
||||||
// classification below is all that is needed.
|
|
||||||
//
|
|
||||||
// A decimal number with up to 18 digits is always representable in
|
|
||||||
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
|
|
||||||
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
|
|
||||||
// could not have set errno to ERANGE for it. Numbers with more digits
|
|
||||||
// (rare in practice) fall through to the exact code below, unchanged,
|
|
||||||
// so their handling -- including reclassification to value_float when
|
|
||||||
// the value overflows 64 bits, and rejection when it is not even
|
|
||||||
// finite as a double -- is bit-for-bit identical to before this
|
|
||||||
// optimization.
|
|
||||||
//
|
|
||||||
// Note this reasons about std::uint64_t/std::int64_t, not about
|
|
||||||
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
|
|
||||||
// narrower, template parameters -- e.g. std::uint32_t). That is fine
|
|
||||||
// *only* because discard_number_values is exclusively set by
|
|
||||||
// accept() (see json.hpp), and accept() always parses through the
|
|
||||||
// library's own json_sax_acceptor -- never a user-supplied SAX
|
|
||||||
// consumer -- whose number_unsigned()/number_integer()/number_float()
|
|
||||||
// callbacks unconditionally discard their argument and return true.
|
|
||||||
// So for every caller that can reach this branch, neither the token
|
|
||||||
// classification below nor the eventual (possibly narrowed, and on
|
|
||||||
// this fast path left stale/unset) value_unsigned/value_integer is
|
|
||||||
// ever consulted -- an unsigned/integer token is accepted outright,
|
|
||||||
// and even a >18-digit token that this fast path deliberately falls
|
|
||||||
// through for is, once reclassified to value_float, still finite
|
|
||||||
// (and thus accepted) for any digit count that fits in number_unsigned_t
|
|
||||||
// or number_integer_t regardless of that type's width. If this
|
|
||||||
// function is ever taught to run with discard_number_values true for
|
|
||||||
// a caller that *does* read the converted value, this reasoning (and
|
|
||||||
// the fast path below) would need to be revisited.
|
|
||||||
if (discard_number_values)
|
|
||||||
{
|
|
||||||
constexpr std::size_t safe_digit_count = 18;
|
|
||||||
if (number_type == token_type::value_unsigned && token_buffer.size() <= safe_digit_count)
|
|
||||||
{
|
|
||||||
return token_type::value_unsigned;
|
|
||||||
}
|
|
||||||
if (number_type == token_type::value_integer && token_buffer.size() - 1 <= safe_digit_count)
|
|
||||||
{
|
|
||||||
return token_type::value_integer;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||||
errno = 0;
|
errno = 0;
|
||||||
|
|
||||||
@@ -1807,13 +1754,6 @@ scan_number_done:
|
|||||||
const char_int_type decimal_point_char = '.';
|
const char_int_type decimal_point_char = '.';
|
||||||
/// the position of the decimal point in the input
|
/// the position of the decimal point in the input
|
||||||
std::size_t decimal_point_position = std::string::npos;
|
std::size_t decimal_point_position = std::string::npos;
|
||||||
|
|
||||||
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
|
|
||||||
/// token classification and never looks at the converted numeric value;
|
|
||||||
/// when set, scan_number() may skip strtoull()/strtoll() for
|
|
||||||
/// value_unsigned/value_integer tokens whose digit count guarantees they
|
|
||||||
/// fit into 64 bits (see scan_number())
|
|
||||||
const bool discard_number_values = false;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
} // namespace detail
|
} // namespace detail
|
||||||
|
|||||||
@@ -72,10 +72,9 @@ class parser
|
|||||||
parser_callback_t<BasicJsonType> cb = nullptr,
|
parser_callback_t<BasicJsonType> cb = nullptr,
|
||||||
const bool allow_exceptions_ = true,
|
const bool allow_exceptions_ = true,
|
||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas_ = false,
|
const bool ignore_trailing_commas_ = false)
|
||||||
const bool discard_number_values_ = false)
|
|
||||||
: callback(std::move(cb))
|
: callback(std::move(cb))
|
||||||
, m_lexer(std::move(adapter), ignore_comments, discard_number_values_)
|
, m_lexer(std::move(adapter), ignore_comments)
|
||||||
, allow_exceptions(allow_exceptions_)
|
, allow_exceptions(allow_exceptions_)
|
||||||
, ignore_trailing_commas(ignore_trailing_commas_)
|
, ignore_trailing_commas(ignore_trailing_commas_)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -164,12 +164,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
detail::parser_callback_t<basic_json>cb = nullptr,
|
detail::parser_callback_t<basic_json>cb = nullptr,
|
||||||
const bool allow_exceptions = true,
|
const bool allow_exceptions = true,
|
||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false,
|
const bool ignore_trailing_commas = false
|
||||||
const bool discard_number_values = false
|
|
||||||
)
|
)
|
||||||
{
|
{
|
||||||
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
||||||
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas, discard_number_values);
|
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas);
|
||||||
}
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
@@ -4134,7 +4133,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||||
@@ -4145,7 +4144,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||||
@@ -4154,7 +4153,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief generate SAX events
|
/// @brief generate SAX events
|
||||||
|
|||||||
@@ -7932,11 +7932,10 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
public:
|
public:
|
||||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||||
|
|
||||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept
|
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept
|
||||||
: ia(std::move(adapter))
|
: ia(std::move(adapter))
|
||||||
, ignore_comments(ignore_comments_)
|
, ignore_comments(ignore_comments_)
|
||||||
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
||||||
, discard_number_values(discard_number_values_)
|
|
||||||
{}
|
{}
|
||||||
|
|
||||||
// deleted because of pointer members
|
// deleted because of pointer members
|
||||||
@@ -9063,58 +9062,6 @@ scan_number_done:
|
|||||||
// we are done scanning a number)
|
// we are done scanning a number)
|
||||||
unget();
|
unget();
|
||||||
|
|
||||||
// If the caller does not need the converted value (only whether the
|
|
||||||
// input is syntactically valid; see json_sax_acceptor/accept()), an
|
|
||||||
// unsigned/integer token can be reported without calling
|
|
||||||
// strtoull()/strtoll() at all, *provided* we can already tell from
|
|
||||||
// the digit count alone that the conversion cannot overflow 64 bits.
|
|
||||||
// Such tokens are always finite and are accepted unconditionally by
|
|
||||||
// the parser regardless of their actual value (parser::sax_parse_internal()
|
|
||||||
// never checks finiteness for value_unsigned/value_integer), so the
|
|
||||||
// classification below is all that is needed.
|
|
||||||
//
|
|
||||||
// A decimal number with up to 18 digits is always representable in
|
|
||||||
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
|
|
||||||
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
|
|
||||||
// could not have set errno to ERANGE for it. Numbers with more digits
|
|
||||||
// (rare in practice) fall through to the exact code below, unchanged,
|
|
||||||
// so their handling -- including reclassification to value_float when
|
|
||||||
// the value overflows 64 bits, and rejection when it is not even
|
|
||||||
// finite as a double -- is bit-for-bit identical to before this
|
|
||||||
// optimization.
|
|
||||||
//
|
|
||||||
// Note this reasons about std::uint64_t/std::int64_t, not about
|
|
||||||
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
|
|
||||||
// narrower, template parameters -- e.g. std::uint32_t). That is fine
|
|
||||||
// *only* because discard_number_values is exclusively set by
|
|
||||||
// accept() (see json.hpp), and accept() always parses through the
|
|
||||||
// library's own json_sax_acceptor -- never a user-supplied SAX
|
|
||||||
// consumer -- whose number_unsigned()/number_integer()/number_float()
|
|
||||||
// callbacks unconditionally discard their argument and return true.
|
|
||||||
// So for every caller that can reach this branch, neither the token
|
|
||||||
// classification below nor the eventual (possibly narrowed, and on
|
|
||||||
// this fast path left stale/unset) value_unsigned/value_integer is
|
|
||||||
// ever consulted -- an unsigned/integer token is accepted outright,
|
|
||||||
// and even a >18-digit token that this fast path deliberately falls
|
|
||||||
// through for is, once reclassified to value_float, still finite
|
|
||||||
// (and thus accepted) for any digit count that fits in number_unsigned_t
|
|
||||||
// or number_integer_t regardless of that type's width. If this
|
|
||||||
// function is ever taught to run with discard_number_values true for
|
|
||||||
// a caller that *does* read the converted value, this reasoning (and
|
|
||||||
// the fast path below) would need to be revisited.
|
|
||||||
if (discard_number_values)
|
|
||||||
{
|
|
||||||
constexpr std::size_t safe_digit_count = 18;
|
|
||||||
if (number_type == token_type::value_unsigned && token_buffer.size() <= safe_digit_count)
|
|
||||||
{
|
|
||||||
return token_type::value_unsigned;
|
|
||||||
}
|
|
||||||
if (number_type == token_type::value_integer && token_buffer.size() - 1 <= safe_digit_count)
|
|
||||||
{
|
|
||||||
return token_type::value_integer;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||||
errno = 0;
|
errno = 0;
|
||||||
|
|
||||||
@@ -9590,13 +9537,6 @@ scan_number_done:
|
|||||||
const char_int_type decimal_point_char = '.';
|
const char_int_type decimal_point_char = '.';
|
||||||
/// the position of the decimal point in the input
|
/// the position of the decimal point in the input
|
||||||
std::size_t decimal_point_position = std::string::npos;
|
std::size_t decimal_point_position = std::string::npos;
|
||||||
|
|
||||||
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
|
|
||||||
/// token classification and never looks at the converted numeric value;
|
|
||||||
/// when set, scan_number() may skip strtoull()/strtoll() for
|
|
||||||
/// value_unsigned/value_integer tokens whose digit count guarantees they
|
|
||||||
/// fit into 64 bits (see scan_number())
|
|
||||||
const bool discard_number_values = false;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
} // namespace detail
|
} // namespace detail
|
||||||
@@ -14103,10 +14043,9 @@ class parser
|
|||||||
parser_callback_t<BasicJsonType> cb = nullptr,
|
parser_callback_t<BasicJsonType> cb = nullptr,
|
||||||
const bool allow_exceptions_ = true,
|
const bool allow_exceptions_ = true,
|
||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas_ = false,
|
const bool ignore_trailing_commas_ = false)
|
||||||
const bool discard_number_values_ = false)
|
|
||||||
: callback(std::move(cb))
|
: callback(std::move(cb))
|
||||||
, m_lexer(std::move(adapter), ignore_comments, discard_number_values_)
|
, m_lexer(std::move(adapter), ignore_comments)
|
||||||
, allow_exceptions(allow_exceptions_)
|
, allow_exceptions(allow_exceptions_)
|
||||||
, ignore_trailing_commas(ignore_trailing_commas_)
|
, ignore_trailing_commas(ignore_trailing_commas_)
|
||||||
{
|
{
|
||||||
@@ -21653,12 +21592,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
detail::parser_callback_t<basic_json>cb = nullptr,
|
detail::parser_callback_t<basic_json>cb = nullptr,
|
||||||
const bool allow_exceptions = true,
|
const bool allow_exceptions = true,
|
||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false,
|
const bool ignore_trailing_commas = false
|
||||||
const bool discard_number_values = false
|
|
||||||
)
|
)
|
||||||
{
|
{
|
||||||
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
||||||
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas, discard_number_values);
|
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas);
|
||||||
}
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
@@ -25623,7 +25561,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||||
@@ -25634,7 +25572,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||||
@@ -25643,7 +25581,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief generate SAX events
|
/// @brief generate SAX events
|
||||||
|
|||||||
@@ -214,4 +214,323 @@ static void BinaryToCbor(benchmark::State& state)
|
|||||||
}
|
}
|
||||||
BENCHMARK(BinaryToCbor)->RangeMultiplier(2)->Range(8, 8 << 12);
|
BENCHMARK(BinaryToCbor)->RangeMultiplier(2)->Range(8, 8 << 12);
|
||||||
|
|
||||||
|
//////////////////////////////////////////////////////////////////////////////
|
||||||
|
// parse binary formats
|
||||||
|
//////////////////////////////////////////////////////////////////////////////
|
||||||
|
|
||||||
|
// Only MessagePack had a read benchmark (FromMsgpack above, left untouched so
|
||||||
|
// its numbers stay comparable across releases). The benchmarks below cover the
|
||||||
|
// other formats, and read from a contiguous buffer as well as from a FILE*:
|
||||||
|
// most callers pass a container, and the two adapters compile to different
|
||||||
|
// code. The test data repository ships JSON only, so the input for each is
|
||||||
|
// derived at setup time by serializing a parsed test file.
|
||||||
|
|
||||||
|
/// binary format to benchmark; the _optimized variants add UBJSON/BJData size
|
||||||
|
/// and type annotations, which the readers handle in a separate code path
|
||||||
|
enum class binary_format
|
||||||
|
{
|
||||||
|
cbor,
|
||||||
|
msgpack,
|
||||||
|
ubjson,
|
||||||
|
ubjson_optimized,
|
||||||
|
bjdata,
|
||||||
|
bjdata_optimized,
|
||||||
|
bson
|
||||||
|
};
|
||||||
|
|
||||||
|
static std::vector<std::uint8_t> to_binary(const json& j, const binary_format format)
|
||||||
|
{
|
||||||
|
switch (format)
|
||||||
|
{
|
||||||
|
case binary_format::cbor:
|
||||||
|
return json::to_cbor(j);
|
||||||
|
case binary_format::msgpack:
|
||||||
|
return json::to_msgpack(j);
|
||||||
|
case binary_format::ubjson:
|
||||||
|
return json::to_ubjson(j);
|
||||||
|
case binary_format::ubjson_optimized:
|
||||||
|
return json::to_ubjson(j, true, true);
|
||||||
|
case binary_format::bjdata:
|
||||||
|
return json::to_bjdata(j);
|
||||||
|
case binary_format::bjdata_optimized:
|
||||||
|
return json::to_bjdata(j, true, true);
|
||||||
|
case binary_format::bson:
|
||||||
|
default:
|
||||||
|
return json::to_bson(j);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static json from_binary(const std::vector<std::uint8_t>& bytes, const binary_format format)
|
||||||
|
{
|
||||||
|
switch (format)
|
||||||
|
{
|
||||||
|
case binary_format::cbor:
|
||||||
|
return json::from_cbor(bytes);
|
||||||
|
case binary_format::msgpack:
|
||||||
|
return json::from_msgpack(bytes);
|
||||||
|
case binary_format::ubjson:
|
||||||
|
case binary_format::ubjson_optimized:
|
||||||
|
return json::from_ubjson(bytes);
|
||||||
|
case binary_format::bjdata:
|
||||||
|
case binary_format::bjdata_optimized:
|
||||||
|
return json::from_bjdata(bytes);
|
||||||
|
case binary_format::bson:
|
||||||
|
default:
|
||||||
|
return json::from_bson(bytes);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static json from_binary(std::FILE* file, const binary_format format)
|
||||||
|
{
|
||||||
|
switch (format)
|
||||||
|
{
|
||||||
|
case binary_format::cbor:
|
||||||
|
return json::from_cbor(file);
|
||||||
|
case binary_format::msgpack:
|
||||||
|
return json::from_msgpack(file);
|
||||||
|
case binary_format::ubjson:
|
||||||
|
case binary_format::ubjson_optimized:
|
||||||
|
return json::from_ubjson(file);
|
||||||
|
case binary_format::bjdata:
|
||||||
|
case binary_format::bjdata_optimized:
|
||||||
|
return json::from_bjdata(file);
|
||||||
|
case binary_format::bson:
|
||||||
|
default:
|
||||||
|
return json::from_bson(file);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief serialize a parsed test file to @a format
|
||||||
|
|
||||||
|
Returns an empty vector and marks the benchmark as skipped if the file cannot
|
||||||
|
be represented in the format, rather than letting the exception escape: BSON
|
||||||
|
requires an object at the top level, and several test files are arrays.
|
||||||
|
*/
|
||||||
|
static std::vector<std::uint8_t> binary_input(benchmark::State& state, const char* filename, const binary_format format)
|
||||||
|
{
|
||||||
|
std::ifstream f(filename);
|
||||||
|
std::string const str((std::istreambuf_iterator<char>(f)), std::istreambuf_iterator<char>());
|
||||||
|
const json j = json::parse(str);
|
||||||
|
|
||||||
|
if (format == binary_format::bson && !j.is_object())
|
||||||
|
{
|
||||||
|
state.SkipWithError("BSON requires an object at the top level");
|
||||||
|
return {};
|
||||||
|
}
|
||||||
|
|
||||||
|
return to_binary(j, format);
|
||||||
|
}
|
||||||
|
|
||||||
|
static void FromBinaryBuffer(benchmark::State& state, const char* filename, const binary_format format)
|
||||||
|
{
|
||||||
|
const std::vector<std::uint8_t> bytes = binary_input(state, filename, format);
|
||||||
|
if (bytes.empty())
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (auto _ : state)
|
||||||
|
{
|
||||||
|
// the value is destroyed outside the timed section, because destroying
|
||||||
|
// a large DOM is not what this benchmark measures
|
||||||
|
state.PauseTiming();
|
||||||
|
auto* j = new json();
|
||||||
|
state.ResumeTiming();
|
||||||
|
|
||||||
|
*j = from_binary(bytes, format);
|
||||||
|
|
||||||
|
state.PauseTiming();
|
||||||
|
delete j;
|
||||||
|
state.ResumeTiming();
|
||||||
|
}
|
||||||
|
|
||||||
|
state.SetBytesProcessed(state.iterations() * bytes.size());
|
||||||
|
}
|
||||||
|
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / floats, TEST_DATA_DIRECTORY "/regression/floats.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, cbor / signed_ints, TEST_DATA_DIRECTORY "/regression/signed_ints.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, msgpack / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson_optimized / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson_optimized);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, ubjson_optimized / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson_optimized);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bjdata);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata_optimized / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bjdata_optimized);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata_optimized / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata_optimized);
|
||||||
|
// BSON requires an object at the top level, so the array-rooted test files
|
||||||
|
// (jeopardy and the regression files) cannot be captured here
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bson / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::bson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryBuffer, bson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bson);
|
||||||
|
|
||||||
|
static void FromBinaryFile(benchmark::State& state, const char* filename, const binary_format format)
|
||||||
|
{
|
||||||
|
const std::vector<std::uint8_t> bytes = binary_input(state, filename, format);
|
||||||
|
if (bytes.empty())
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
const char* tmp = "benchmark_input.bin";
|
||||||
|
std::ofstream o(tmp, std::ios::binary);
|
||||||
|
o.write(reinterpret_cast<const char*>(bytes.data()), static_cast<std::streamsize>(bytes.size()));
|
||||||
|
o.flush();
|
||||||
|
o.close();
|
||||||
|
|
||||||
|
for (auto _ : state)
|
||||||
|
{
|
||||||
|
state.PauseTiming();
|
||||||
|
auto* j = new json();
|
||||||
|
auto* file = std::fopen(tmp, "rb");
|
||||||
|
state.ResumeTiming();
|
||||||
|
|
||||||
|
*j = from_binary(file, format);
|
||||||
|
|
||||||
|
state.PauseTiming();
|
||||||
|
std::fclose(file);
|
||||||
|
delete j;
|
||||||
|
state.ResumeTiming();
|
||||||
|
}
|
||||||
|
|
||||||
|
state.SetBytesProcessed(state.iterations() * bytes.size());
|
||||||
|
}
|
||||||
|
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryFile, cbor / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryFile, cbor / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryFile, ubjson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryFile, ubjson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryFile, bjdata / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryFile, bson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bson);
|
||||||
|
|
||||||
|
//////////////////////////////////////////////////////////////////////////////
|
||||||
|
// parse binary formats: value shapes
|
||||||
|
//////////////////////////////////////////////////////////////////////////////
|
||||||
|
|
||||||
|
// The test files above are wide and shallow, but the readers' cost is per
|
||||||
|
// container, so these cover the shapes that stress the container handling
|
||||||
|
// itself. Every shape is wrapped in an object so that BSON, which requires an
|
||||||
|
// object at the top level, measures the same value as the other formats.
|
||||||
|
|
||||||
|
/// deeply nested arrays: one container per level, no other work
|
||||||
|
static json make_nested()
|
||||||
|
{
|
||||||
|
json nested = json::array();
|
||||||
|
json* p = &nested;
|
||||||
|
for (std::size_t i = 1; i < 1000; ++i)
|
||||||
|
{
|
||||||
|
p->push_back(json::array());
|
||||||
|
p = &p->operator[](0);
|
||||||
|
}
|
||||||
|
|
||||||
|
json j = json::object();
|
||||||
|
j["data"] = std::move(nested);
|
||||||
|
return j;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// many sibling containers: maximum container churn, minimum nesting
|
||||||
|
static json make_containers()
|
||||||
|
{
|
||||||
|
json data = json::array();
|
||||||
|
for (std::size_t i = 0; i < 100000; ++i)
|
||||||
|
{
|
||||||
|
data.push_back(json::array({1, 2}));
|
||||||
|
}
|
||||||
|
|
||||||
|
json j = json::object();
|
||||||
|
j["data"] = std::move(data);
|
||||||
|
return j;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// one flat array of numbers: the scalar decoding path, which must not move
|
||||||
|
static json make_scalars()
|
||||||
|
{
|
||||||
|
json data = json::array();
|
||||||
|
for (std::size_t i = 0; i < 1000000; ++i)
|
||||||
|
{
|
||||||
|
data.push_back(i);
|
||||||
|
}
|
||||||
|
|
||||||
|
json j = json::object();
|
||||||
|
j["data"] = std::move(data);
|
||||||
|
return j;
|
||||||
|
}
|
||||||
|
|
||||||
|
static void FromBinaryShape(benchmark::State& state, json (*build)(), const binary_format format)
|
||||||
|
{
|
||||||
|
const std::vector<std::uint8_t> bytes = to_binary(build(), format);
|
||||||
|
|
||||||
|
for (auto _ : state)
|
||||||
|
{
|
||||||
|
state.PauseTiming();
|
||||||
|
auto* j = new json();
|
||||||
|
state.ResumeTiming();
|
||||||
|
|
||||||
|
*j = from_binary(bytes, format);
|
||||||
|
|
||||||
|
state.PauseTiming();
|
||||||
|
delete j;
|
||||||
|
state.ResumeTiming();
|
||||||
|
}
|
||||||
|
|
||||||
|
state.SetBytesProcessed(state.iterations() * bytes.size());
|
||||||
|
}
|
||||||
|
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, nested / cbor, make_nested, binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, nested / msgpack, make_nested, binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, nested / ubjson, make_nested, binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, nested / bjdata, make_nested, binary_format::bjdata);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, nested / bson, make_nested, binary_format::bson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, containers / cbor, make_containers, binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, containers / msgpack, make_containers, binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, containers / ubjson, make_containers, binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, containers / ubjson_optimized, make_containers, binary_format::ubjson_optimized);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, containers / bjdata, make_containers, binary_format::bjdata);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, containers / bson, make_containers, binary_format::bson);
|
||||||
|
// BSON names every array element, so a large array measures key generation
|
||||||
|
// rather than scalar decoding and is left out here
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, scalars / cbor, make_scalars, binary_format::cbor);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, scalars / msgpack, make_scalars, binary_format::msgpack);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, scalars / ubjson, make_scalars, binary_format::ubjson);
|
||||||
|
BENCHMARK_CAPTURE(FromBinaryShape, scalars / bjdata, make_scalars, binary_format::bjdata);
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief parse an indefinite-length CBOR string
|
||||||
|
|
||||||
|
The writer never emits this form, so the input is assembled by hand: 0x7F
|
||||||
|
opens the string, each chunk is a one-character string, and 0xFF closes it.
|
||||||
|
*/
|
||||||
|
static void FromCborChunkedString(benchmark::State& state, const std::size_t chunks)
|
||||||
|
{
|
||||||
|
std::vector<std::uint8_t> bytes;
|
||||||
|
bytes.reserve(2 * chunks + 2);
|
||||||
|
bytes.push_back(0x7F);
|
||||||
|
for (std::size_t i = 0; i < chunks; ++i)
|
||||||
|
{
|
||||||
|
bytes.push_back(0x61); // string of length 1
|
||||||
|
bytes.push_back(0x61); // 'a'
|
||||||
|
}
|
||||||
|
bytes.push_back(0xFF);
|
||||||
|
|
||||||
|
for (auto _ : state)
|
||||||
|
{
|
||||||
|
json j = json::from_cbor(bytes);
|
||||||
|
benchmark::DoNotOptimize(j);
|
||||||
|
}
|
||||||
|
|
||||||
|
state.SetBytesProcessed(state.iterations() * bytes.size());
|
||||||
|
}
|
||||||
|
|
||||||
|
BENCHMARK_CAPTURE(FromCborChunkedString, 10000 chunks, 10000);
|
||||||
|
|
||||||
BENCHMARK_MAIN();
|
BENCHMARK_MAIN();
|
||||||
|
|||||||
@@ -930,98 +930,6 @@ TEST_CASE("parser class")
|
|||||||
CHECK(accept_helper("+1") == false);
|
CHECK(accept_helper("+1") == false);
|
||||||
CHECK(accept_helper("+0") == false);
|
CHECK(accept_helper("+0") == false);
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("issue #5411 - skip conversion when accept() does not need the numeric value")
|
|
||||||
{
|
|
||||||
// lexer::scan_number() may skip strtoull()/strtoll() for
|
|
||||||
// value_unsigned/value_integer tokens when the caller (e.g.
|
|
||||||
// json::accept()) does not need the converted value, as long
|
|
||||||
// as the digit count alone guarantees no 64-bit overflow (see
|
|
||||||
// the "safe_digit_count" fast path in scan_number()). This
|
|
||||||
// differential test checks that json::accept() (which enables
|
|
||||||
// the fast path) and json::parse() (which never does) always
|
|
||||||
// agree, over a corpus that exercises both the fast path
|
|
||||||
// (<=18 digits) and the untouched, exact fallback path (>=19
|
|
||||||
// digits) -- including reclassification of huge digit-only
|
|
||||||
// integers to a (possibly non-finite) floating-point value.
|
|
||||||
const std::vector<std::pair<std::string, bool>> cases =
|
|
||||||
{
|
|
||||||
// normal small/large integers, both signs
|
|
||||||
{"0", true}, {"1", true}, {"-1", true}, {"42", true}, {"-42", true},
|
|
||||||
{"123456789", true}, {"-123456789", true},
|
|
||||||
|
|
||||||
// digit-count boundary around the 18-digit safe cutoff (both signs)
|
|
||||||
{std::string(17, '9'), true},
|
|
||||||
{std::string(18, '9'), true},
|
|
||||||
{std::string(19, '9'), true},
|
|
||||||
{std::string(20, '9'), true},
|
|
||||||
{"-" + std::string(17, '9'), true},
|
|
||||||
{"-" + std::string(18, '9'), true},
|
|
||||||
{"-" + std::string(19, '9'), true},
|
|
||||||
{"-" + std::string(20, '9'), true},
|
|
||||||
|
|
||||||
// 64-bit boundaries
|
|
||||||
{"9223372036854775807", true}, // INT64_MAX
|
|
||||||
{"-9223372036854775808", true}, // INT64_MIN
|
|
||||||
{"18446744073709551615", true}, // UINT64_MAX
|
|
||||||
{"18446744073709551616", true}, // UINT64_MAX + 1 (overflows uint64_t, finite double)
|
|
||||||
|
|
||||||
// the 28-digit example from the issue: overflows uint64_t
|
|
||||||
// but is finite as a double, so the scanner reclassifies
|
|
||||||
// it to value_float and it is accepted
|
|
||||||
{"9999999999999999999999999999", true},
|
|
||||||
|
|
||||||
// huge digit-only integers that overflow even a double -> rejected
|
|
||||||
{std::string(309, '9'), false},
|
|
||||||
{std::string(400, '9'), false},
|
|
||||||
{"1" + std::string(400, '0'), false},
|
|
||||||
|
|
||||||
// 1e999 / 1e400 style overflow -> rejected
|
|
||||||
{"1e999", false},
|
|
||||||
{"1e400", false},
|
|
||||||
{"-1e999", false},
|
|
||||||
{"1E999", false},
|
|
||||||
|
|
||||||
// values straddling DBL_MAX
|
|
||||||
{"1.7976931348623157e308", true}, // <= DBL_MAX, finite
|
|
||||||
{"1.7976931348623159e308", false}, // > DBL_MAX, overflows to inf
|
|
||||||
|
|
||||||
// a mix of other valid/invalid numeric syntax
|
|
||||||
{"3.14159", true},
|
|
||||||
{"-0.0", true},
|
|
||||||
{"1.0e10", true},
|
|
||||||
{"01", false},
|
|
||||||
{"-", false},
|
|
||||||
{"1.", false},
|
|
||||||
{"1e", false},
|
|
||||||
{"+1", false},
|
|
||||||
};
|
|
||||||
|
|
||||||
for (const auto& c : cases)
|
|
||||||
{
|
|
||||||
const std::string& number = c.first;
|
|
||||||
const bool expected = c.second;
|
|
||||||
CAPTURE(number)
|
|
||||||
CAPTURE(expected)
|
|
||||||
|
|
||||||
// accept() takes the fast path (skips conversion when possible)
|
|
||||||
CHECK(json::accept(number) == expected);
|
|
||||||
|
|
||||||
// parse() always performs the full conversion; it must agree
|
|
||||||
json j;
|
|
||||||
CHECK_NOTHROW(json::parser(nlohmann::detail::input_adapter(number), nullptr, false).parse(true, j));
|
|
||||||
CHECK(!j.is_discarded() == expected);
|
|
||||||
|
|
||||||
// wrap in an array so get_token() is exercised beyond the
|
|
||||||
// very first (constructor-time) scan as well
|
|
||||||
std::string wrapped = "[";
|
|
||||||
wrapped += number;
|
|
||||||
wrapped += ",";
|
|
||||||
wrapped += number;
|
|
||||||
wrapped += "]";
|
|
||||||
CHECK(json::accept(wrapped) == expected);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user