mirror of
https://github.com/nlohmann/json.git
synced 2026-09-06 16:27:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
98ab9f31c3 | ||
|
|
aee9421883 |
@@ -301,6 +301,16 @@ class json_sax_dom_parser
|
|||||||
JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
|
JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (len != detail::unknown_size())
|
||||||
|
{
|
||||||
|
// reserve upfront to avoid repeated reallocations while adding elements,
|
||||||
|
// but cap the reservation so a bogus/hostile length (which is not bounded
|
||||||
|
// by max_size(), unlike e.g. std::vector) cannot trigger an oversized
|
||||||
|
// allocation for a small or truncated input
|
||||||
|
constexpr std::size_t reserve_cap = 16384;
|
||||||
|
ref_stack.back()->m_data.m_value.array->reserve(len < reserve_cap ? len : reserve_cap);
|
||||||
|
}
|
||||||
|
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -661,6 +671,16 @@ class json_sax_dom_callback_parser
|
|||||||
{
|
{
|
||||||
JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
|
JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (len != detail::unknown_size())
|
||||||
|
{
|
||||||
|
// reserve upfront to avoid repeated reallocations while adding elements,
|
||||||
|
// but cap the reservation so a bogus/hostile length (which is not bounded
|
||||||
|
// by max_size(), unlike e.g. std::vector) cannot trigger an oversized
|
||||||
|
// allocation for a small or truncated input
|
||||||
|
constexpr std::size_t reserve_cap = 16384;
|
||||||
|
ref_stack.back()->m_data.m_value.array->reserve(len < reserve_cap ? len : reserve_cap);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return true;
|
return true;
|
||||||
|
|||||||
@@ -149,11 +149,10 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
public:
|
public:
|
||||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||||
|
|
||||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept
|
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept
|
||||||
: ia(std::move(adapter))
|
: ia(std::move(adapter))
|
||||||
, ignore_comments(ignore_comments_)
|
, ignore_comments(ignore_comments_)
|
||||||
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
||||||
, discard_number_values(discard_number_values_)
|
|
||||||
{}
|
{}
|
||||||
|
|
||||||
// deleted because of pointer members
|
// deleted because of pointer members
|
||||||
@@ -1280,58 +1279,6 @@ scan_number_done:
|
|||||||
// we are done scanning a number)
|
// we are done scanning a number)
|
||||||
unget();
|
unget();
|
||||||
|
|
||||||
// If the caller does not need the converted value (only whether the
|
|
||||||
// input is syntactically valid; see json_sax_acceptor/accept()), an
|
|
||||||
// unsigned/integer token can be reported without calling
|
|
||||||
// strtoull()/strtoll() at all, *provided* we can already tell from
|
|
||||||
// the digit count alone that the conversion cannot overflow 64 bits.
|
|
||||||
// Such tokens are always finite and are accepted unconditionally by
|
|
||||||
// the parser regardless of their actual value (parser::sax_parse_internal()
|
|
||||||
// never checks finiteness for value_unsigned/value_integer), so the
|
|
||||||
// classification below is all that is needed.
|
|
||||||
//
|
|
||||||
// A decimal number with up to 18 digits is always representable in
|
|
||||||
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
|
|
||||||
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
|
|
||||||
// could not have set errno to ERANGE for it. Numbers with more digits
|
|
||||||
// (rare in practice) fall through to the exact code below, unchanged,
|
|
||||||
// so their handling -- including reclassification to value_float when
|
|
||||||
// the value overflows 64 bits, and rejection when it is not even
|
|
||||||
// finite as a double -- is bit-for-bit identical to before this
|
|
||||||
// optimization.
|
|
||||||
//
|
|
||||||
// Note this reasons about std::uint64_t/std::int64_t, not about
|
|
||||||
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
|
|
||||||
// narrower, template parameters -- e.g. std::uint32_t). That is fine
|
|
||||||
// *only* because discard_number_values is exclusively set by
|
|
||||||
// accept() (see json.hpp), and accept() always parses through the
|
|
||||||
// library's own json_sax_acceptor -- never a user-supplied SAX
|
|
||||||
// consumer -- whose number_unsigned()/number_integer()/number_float()
|
|
||||||
// callbacks unconditionally discard their argument and return true.
|
|
||||||
// So for every caller that can reach this branch, neither the token
|
|
||||||
// classification below nor the eventual (possibly narrowed, and on
|
|
||||||
// this fast path left stale/unset) value_unsigned/value_integer is
|
|
||||||
// ever consulted -- an unsigned/integer token is accepted outright,
|
|
||||||
// and even a >18-digit token that this fast path deliberately falls
|
|
||||||
// through for is, once reclassified to value_float, still finite
|
|
||||||
// (and thus accepted) for any digit count that fits in number_unsigned_t
|
|
||||||
// or number_integer_t regardless of that type's width. If this
|
|
||||||
// function is ever taught to run with discard_number_values true for
|
|
||||||
// a caller that *does* read the converted value, this reasoning (and
|
|
||||||
// the fast path below) would need to be revisited.
|
|
||||||
if (discard_number_values)
|
|
||||||
{
|
|
||||||
constexpr std::size_t safe_digit_count = 18;
|
|
||||||
if (number_type == token_type::value_unsigned && token_buffer.size() <= safe_digit_count)
|
|
||||||
{
|
|
||||||
return token_type::value_unsigned;
|
|
||||||
}
|
|
||||||
if (number_type == token_type::value_integer && token_buffer.size() - 1 <= safe_digit_count)
|
|
||||||
{
|
|
||||||
return token_type::value_integer;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||||
errno = 0;
|
errno = 0;
|
||||||
|
|
||||||
@@ -1446,7 +1393,8 @@ scan_number_done:
|
|||||||
*/
|
*/
|
||||||
char_int_type get()
|
char_int_type get()
|
||||||
{
|
{
|
||||||
advance_position();
|
++position.chars_read_total;
|
||||||
|
++position.chars_read_current_line;
|
||||||
|
|
||||||
if (next_unget)
|
if (next_unget)
|
||||||
{
|
{
|
||||||
@@ -1458,23 +1406,6 @@ scan_number_done:
|
|||||||
current = ia.get_character();
|
current = ia.get_character();
|
||||||
}
|
}
|
||||||
|
|
||||||
return track_after_read();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// shared head of get() / get_ignoring_pending_unget(): bump the
|
|
||||||
/// per-character position counters (line-count-on-'\n' bookkeeping is
|
|
||||||
/// handled afterwards, in track_after_read(), once `current` is known)
|
|
||||||
void advance_position() noexcept
|
|
||||||
{
|
|
||||||
++position.chars_read_total;
|
|
||||||
++position.chars_read_current_line;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// shared tail of get() / get_ignoring_pending_unget(): capture the
|
|
||||||
/// character for error messages (if needed) and update line/column
|
|
||||||
/// bookkeeping for the character now in `current`
|
|
||||||
char_int_type track_after_read()
|
|
||||||
{
|
|
||||||
// seekable adapters reconstruct the token lazily on error (see
|
// seekable adapters reconstruct the token lazily on error (see
|
||||||
// get_token_string), so the eager per-character copy is skipped
|
// get_token_string), so the eager per-character copy is skipped
|
||||||
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
||||||
@@ -1488,29 +1419,6 @@ scan_number_done:
|
|||||||
return current;
|
return current;
|
||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
|
||||||
@brief like get(), but for call sites that can prove no unget() is pending
|
|
||||||
|
|
||||||
get() has to check the `next_unget` flag on every call, because a
|
|
||||||
previous token may have ended with unget() (e.g. scan_number() always
|
|
||||||
ungets the character that terminated the number, so the next call to
|
|
||||||
scan() can see it again). skip_whitespace() reads that first,
|
|
||||||
possibly-ungotten character via a plain get(), but every further
|
|
||||||
character it reads is guaranteed to be a fresh read: nothing between
|
|
||||||
those calls invokes unget(). This variant skips the (otherwise always
|
|
||||||
false) next_unget branch for those calls; it is not a general
|
|
||||||
replacement for get().
|
|
||||||
*/
|
|
||||||
char_int_type get_ignoring_pending_unget()
|
|
||||||
{
|
|
||||||
JSON_ASSERT(!next_unget);
|
|
||||||
|
|
||||||
advance_position();
|
|
||||||
current = ia.get_character();
|
|
||||||
|
|
||||||
return track_after_read();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
||||||
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
||||||
|
|
||||||
@@ -1704,37 +1612,13 @@ scan_number_done:
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// whether `current` is one of the four JSON whitespace characters
|
|
||||||
bool current_is_whitespace() const noexcept
|
|
||||||
{
|
|
||||||
return current == ' ' || current == '\t' || current == '\n' || current == '\r';
|
|
||||||
}
|
|
||||||
|
|
||||||
void skip_whitespace()
|
void skip_whitespace()
|
||||||
{
|
{
|
||||||
// the first character may be a pending unget() left over from the
|
|
||||||
// previous token (see get_ignoring_pending_unget()); every
|
|
||||||
// subsequent character read by this loop is guaranteed fresh, since
|
|
||||||
// nothing below calls unget()
|
|
||||||
get();
|
|
||||||
|
|
||||||
if (!current_is_whitespace())
|
|
||||||
{
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
// this is written as an if-guarded do-while (rather than a plain
|
|
||||||
// while loop) because that shape is what lets both GCC and Clang
|
|
||||||
// keep the input adapter's read pointer in a register across
|
|
||||||
// iterations; the equivalent while-loop measurably defeated that
|
|
||||||
// optimization in testing, turning long whitespace runs (e.g. the
|
|
||||||
// indentation of pretty-printed JSON) from a register-only loop
|
|
||||||
// into one that reloads the pointer from memory every character
|
|
||||||
do
|
do
|
||||||
{
|
{
|
||||||
get_ignoring_pending_unget();
|
get();
|
||||||
}
|
}
|
||||||
while (current_is_whitespace());
|
while (current == ' ' || current == '\t' || current == '\n' || current == '\r');
|
||||||
}
|
}
|
||||||
|
|
||||||
token_type scan()
|
token_type scan()
|
||||||
@@ -1870,13 +1754,6 @@ scan_number_done:
|
|||||||
const char_int_type decimal_point_char = '.';
|
const char_int_type decimal_point_char = '.';
|
||||||
/// the position of the decimal point in the input
|
/// the position of the decimal point in the input
|
||||||
std::size_t decimal_point_position = std::string::npos;
|
std::size_t decimal_point_position = std::string::npos;
|
||||||
|
|
||||||
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
|
|
||||||
/// token classification and never looks at the converted numeric value;
|
|
||||||
/// when set, scan_number() may skip strtoull()/strtoll() for
|
|
||||||
/// value_unsigned/value_integer tokens whose digit count guarantees they
|
|
||||||
/// fit into 64 bits (see scan_number())
|
|
||||||
const bool discard_number_values = false;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
} // namespace detail
|
} // namespace detail
|
||||||
|
|||||||
@@ -72,10 +72,9 @@ class parser
|
|||||||
parser_callback_t<BasicJsonType> cb = nullptr,
|
parser_callback_t<BasicJsonType> cb = nullptr,
|
||||||
const bool allow_exceptions_ = true,
|
const bool allow_exceptions_ = true,
|
||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas_ = false,
|
const bool ignore_trailing_commas_ = false)
|
||||||
const bool discard_number_values_ = false)
|
|
||||||
: callback(std::move(cb))
|
: callback(std::move(cb))
|
||||||
, m_lexer(std::move(adapter), ignore_comments, discard_number_values_)
|
, m_lexer(std::move(adapter), ignore_comments)
|
||||||
, allow_exceptions(allow_exceptions_)
|
, allow_exceptions(allow_exceptions_)
|
||||||
, ignore_trailing_commas(ignore_trailing_commas_)
|
, ignore_trailing_commas(ignore_trailing_commas_)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -164,12 +164,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
detail::parser_callback_t<basic_json>cb = nullptr,
|
detail::parser_callback_t<basic_json>cb = nullptr,
|
||||||
const bool allow_exceptions = true,
|
const bool allow_exceptions = true,
|
||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false,
|
const bool ignore_trailing_commas = false
|
||||||
const bool discard_number_values = false
|
|
||||||
)
|
)
|
||||||
{
|
{
|
||||||
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
||||||
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas, discard_number_values);
|
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas);
|
||||||
}
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
@@ -4134,7 +4133,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||||
@@ -4145,7 +4144,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||||
@@ -4154,7 +4153,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief generate SAX events
|
/// @brief generate SAX events
|
||||||
|
|||||||
@@ -7932,11 +7932,10 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
public:
|
public:
|
||||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||||
|
|
||||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept
|
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept
|
||||||
: ia(std::move(adapter))
|
: ia(std::move(adapter))
|
||||||
, ignore_comments(ignore_comments_)
|
, ignore_comments(ignore_comments_)
|
||||||
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
||||||
, discard_number_values(discard_number_values_)
|
|
||||||
{}
|
{}
|
||||||
|
|
||||||
// deleted because of pointer members
|
// deleted because of pointer members
|
||||||
@@ -9063,58 +9062,6 @@ scan_number_done:
|
|||||||
// we are done scanning a number)
|
// we are done scanning a number)
|
||||||
unget();
|
unget();
|
||||||
|
|
||||||
// If the caller does not need the converted value (only whether the
|
|
||||||
// input is syntactically valid; see json_sax_acceptor/accept()), an
|
|
||||||
// unsigned/integer token can be reported without calling
|
|
||||||
// strtoull()/strtoll() at all, *provided* we can already tell from
|
|
||||||
// the digit count alone that the conversion cannot overflow 64 bits.
|
|
||||||
// Such tokens are always finite and are accepted unconditionally by
|
|
||||||
// the parser regardless of their actual value (parser::sax_parse_internal()
|
|
||||||
// never checks finiteness for value_unsigned/value_integer), so the
|
|
||||||
// classification below is all that is needed.
|
|
||||||
//
|
|
||||||
// A decimal number with up to 18 digits is always representable in
|
|
||||||
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
|
|
||||||
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
|
|
||||||
// could not have set errno to ERANGE for it. Numbers with more digits
|
|
||||||
// (rare in practice) fall through to the exact code below, unchanged,
|
|
||||||
// so their handling -- including reclassification to value_float when
|
|
||||||
// the value overflows 64 bits, and rejection when it is not even
|
|
||||||
// finite as a double -- is bit-for-bit identical to before this
|
|
||||||
// optimization.
|
|
||||||
//
|
|
||||||
// Note this reasons about std::uint64_t/std::int64_t, not about
|
|
||||||
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
|
|
||||||
// narrower, template parameters -- e.g. std::uint32_t). That is fine
|
|
||||||
// *only* because discard_number_values is exclusively set by
|
|
||||||
// accept() (see json.hpp), and accept() always parses through the
|
|
||||||
// library's own json_sax_acceptor -- never a user-supplied SAX
|
|
||||||
// consumer -- whose number_unsigned()/number_integer()/number_float()
|
|
||||||
// callbacks unconditionally discard their argument and return true.
|
|
||||||
// So for every caller that can reach this branch, neither the token
|
|
||||||
// classification below nor the eventual (possibly narrowed, and on
|
|
||||||
// this fast path left stale/unset) value_unsigned/value_integer is
|
|
||||||
// ever consulted -- an unsigned/integer token is accepted outright,
|
|
||||||
// and even a >18-digit token that this fast path deliberately falls
|
|
||||||
// through for is, once reclassified to value_float, still finite
|
|
||||||
// (and thus accepted) for any digit count that fits in number_unsigned_t
|
|
||||||
// or number_integer_t regardless of that type's width. If this
|
|
||||||
// function is ever taught to run with discard_number_values true for
|
|
||||||
// a caller that *does* read the converted value, this reasoning (and
|
|
||||||
// the fast path below) would need to be revisited.
|
|
||||||
if (discard_number_values)
|
|
||||||
{
|
|
||||||
constexpr std::size_t safe_digit_count = 18;
|
|
||||||
if (number_type == token_type::value_unsigned && token_buffer.size() <= safe_digit_count)
|
|
||||||
{
|
|
||||||
return token_type::value_unsigned;
|
|
||||||
}
|
|
||||||
if (number_type == token_type::value_integer && token_buffer.size() - 1 <= safe_digit_count)
|
|
||||||
{
|
|
||||||
return token_type::value_integer;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||||
errno = 0;
|
errno = 0;
|
||||||
|
|
||||||
@@ -9229,7 +9176,8 @@ scan_number_done:
|
|||||||
*/
|
*/
|
||||||
char_int_type get()
|
char_int_type get()
|
||||||
{
|
{
|
||||||
advance_position();
|
++position.chars_read_total;
|
||||||
|
++position.chars_read_current_line;
|
||||||
|
|
||||||
if (next_unget)
|
if (next_unget)
|
||||||
{
|
{
|
||||||
@@ -9241,23 +9189,6 @@ scan_number_done:
|
|||||||
current = ia.get_character();
|
current = ia.get_character();
|
||||||
}
|
}
|
||||||
|
|
||||||
return track_after_read();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// shared head of get() / get_ignoring_pending_unget(): bump the
|
|
||||||
/// per-character position counters (line-count-on-'\n' bookkeeping is
|
|
||||||
/// handled afterwards, in track_after_read(), once `current` is known)
|
|
||||||
void advance_position() noexcept
|
|
||||||
{
|
|
||||||
++position.chars_read_total;
|
|
||||||
++position.chars_read_current_line;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// shared tail of get() / get_ignoring_pending_unget(): capture the
|
|
||||||
/// character for error messages (if needed) and update line/column
|
|
||||||
/// bookkeeping for the character now in `current`
|
|
||||||
char_int_type track_after_read()
|
|
||||||
{
|
|
||||||
// seekable adapters reconstruct the token lazily on error (see
|
// seekable adapters reconstruct the token lazily on error (see
|
||||||
// get_token_string), so the eager per-character copy is skipped
|
// get_token_string), so the eager per-character copy is skipped
|
||||||
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
||||||
@@ -9271,29 +9202,6 @@ scan_number_done:
|
|||||||
return current;
|
return current;
|
||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
|
||||||
@brief like get(), but for call sites that can prove no unget() is pending
|
|
||||||
|
|
||||||
get() has to check the `next_unget` flag on every call, because a
|
|
||||||
previous token may have ended with unget() (e.g. scan_number() always
|
|
||||||
ungets the character that terminated the number, so the next call to
|
|
||||||
scan() can see it again). skip_whitespace() reads that first,
|
|
||||||
possibly-ungotten character via a plain get(), but every further
|
|
||||||
character it reads is guaranteed to be a fresh read: nothing between
|
|
||||||
those calls invokes unget(). This variant skips the (otherwise always
|
|
||||||
false) next_unget branch for those calls; it is not a general
|
|
||||||
replacement for get().
|
|
||||||
*/
|
|
||||||
char_int_type get_ignoring_pending_unget()
|
|
||||||
{
|
|
||||||
JSON_ASSERT(!next_unget);
|
|
||||||
|
|
||||||
advance_position();
|
|
||||||
current = ia.get_character();
|
|
||||||
|
|
||||||
return track_after_read();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
||||||
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
||||||
|
|
||||||
@@ -9487,37 +9395,13 @@ scan_number_done:
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// whether `current` is one of the four JSON whitespace characters
|
|
||||||
bool current_is_whitespace() const noexcept
|
|
||||||
{
|
|
||||||
return current == ' ' || current == '\t' || current == '\n' || current == '\r';
|
|
||||||
}
|
|
||||||
|
|
||||||
void skip_whitespace()
|
void skip_whitespace()
|
||||||
{
|
{
|
||||||
// the first character may be a pending unget() left over from the
|
|
||||||
// previous token (see get_ignoring_pending_unget()); every
|
|
||||||
// subsequent character read by this loop is guaranteed fresh, since
|
|
||||||
// nothing below calls unget()
|
|
||||||
get();
|
|
||||||
|
|
||||||
if (!current_is_whitespace())
|
|
||||||
{
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
// this is written as an if-guarded do-while (rather than a plain
|
|
||||||
// while loop) because that shape is what lets both GCC and Clang
|
|
||||||
// keep the input adapter's read pointer in a register across
|
|
||||||
// iterations; the equivalent while-loop measurably defeated that
|
|
||||||
// optimization in testing, turning long whitespace runs (e.g. the
|
|
||||||
// indentation of pretty-printed JSON) from a register-only loop
|
|
||||||
// into one that reloads the pointer from memory every character
|
|
||||||
do
|
do
|
||||||
{
|
{
|
||||||
get_ignoring_pending_unget();
|
get();
|
||||||
}
|
}
|
||||||
while (current_is_whitespace());
|
while (current == ' ' || current == '\t' || current == '\n' || current == '\r');
|
||||||
}
|
}
|
||||||
|
|
||||||
token_type scan()
|
token_type scan()
|
||||||
@@ -9653,13 +9537,6 @@ scan_number_done:
|
|||||||
const char_int_type decimal_point_char = '.';
|
const char_int_type decimal_point_char = '.';
|
||||||
/// the position of the decimal point in the input
|
/// the position of the decimal point in the input
|
||||||
std::size_t decimal_point_position = std::string::npos;
|
std::size_t decimal_point_position = std::string::npos;
|
||||||
|
|
||||||
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
|
|
||||||
/// token classification and never looks at the converted numeric value;
|
|
||||||
/// when set, scan_number() may skip strtoull()/strtoll() for
|
|
||||||
/// value_unsigned/value_integer tokens whose digit count guarantees they
|
|
||||||
/// fit into 64 bits (see scan_number())
|
|
||||||
const bool discard_number_values = false;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
} // namespace detail
|
} // namespace detail
|
||||||
@@ -9952,6 +9829,16 @@ class json_sax_dom_parser
|
|||||||
JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
|
JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (len != detail::unknown_size())
|
||||||
|
{
|
||||||
|
// reserve upfront to avoid repeated reallocations while adding elements,
|
||||||
|
// but cap the reservation so a bogus/hostile length (which is not bounded
|
||||||
|
// by max_size(), unlike e.g. std::vector) cannot trigger an oversized
|
||||||
|
// allocation for a small or truncated input
|
||||||
|
constexpr std::size_t reserve_cap = 16384;
|
||||||
|
ref_stack.back()->m_data.m_value.array->reserve(len < reserve_cap ? len : reserve_cap);
|
||||||
|
}
|
||||||
|
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -10312,6 +10199,16 @@ class json_sax_dom_callback_parser
|
|||||||
{
|
{
|
||||||
JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
|
JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (len != detail::unknown_size())
|
||||||
|
{
|
||||||
|
// reserve upfront to avoid repeated reallocations while adding elements,
|
||||||
|
// but cap the reservation so a bogus/hostile length (which is not bounded
|
||||||
|
// by max_size(), unlike e.g. std::vector) cannot trigger an oversized
|
||||||
|
// allocation for a small or truncated input
|
||||||
|
constexpr std::size_t reserve_cap = 16384;
|
||||||
|
ref_stack.back()->m_data.m_value.array->reserve(len < reserve_cap ? len : reserve_cap);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return true;
|
return true;
|
||||||
@@ -14166,10 +14063,9 @@ class parser
|
|||||||
parser_callback_t<BasicJsonType> cb = nullptr,
|
parser_callback_t<BasicJsonType> cb = nullptr,
|
||||||
const bool allow_exceptions_ = true,
|
const bool allow_exceptions_ = true,
|
||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas_ = false,
|
const bool ignore_trailing_commas_ = false)
|
||||||
const bool discard_number_values_ = false)
|
|
||||||
: callback(std::move(cb))
|
: callback(std::move(cb))
|
||||||
, m_lexer(std::move(adapter), ignore_comments, discard_number_values_)
|
, m_lexer(std::move(adapter), ignore_comments)
|
||||||
, allow_exceptions(allow_exceptions_)
|
, allow_exceptions(allow_exceptions_)
|
||||||
, ignore_trailing_commas(ignore_trailing_commas_)
|
, ignore_trailing_commas(ignore_trailing_commas_)
|
||||||
{
|
{
|
||||||
@@ -21716,12 +21612,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
detail::parser_callback_t<basic_json>cb = nullptr,
|
detail::parser_callback_t<basic_json>cb = nullptr,
|
||||||
const bool allow_exceptions = true,
|
const bool allow_exceptions = true,
|
||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false,
|
const bool ignore_trailing_commas = false
|
||||||
const bool discard_number_values = false
|
|
||||||
)
|
)
|
||||||
{
|
{
|
||||||
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
||||||
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas, discard_number_values);
|
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas);
|
||||||
}
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
@@ -25686,7 +25581,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||||
@@ -25697,7 +25592,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||||
@@ -25706,7 +25601,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief generate SAX events
|
/// @brief generate SAX events
|
||||||
|
|||||||
@@ -3489,6 +3489,90 @@ TEST_CASE("BJData")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("issue #5405 - array reserve for definite-length BJData arrays")
|
||||||
|
{
|
||||||
|
SECTION("a huge claimed length with no element data must not over-allocate")
|
||||||
|
{
|
||||||
|
// optimized form [$type#count: type 'i' (int8), count as a four-byte
|
||||||
|
// little-endian 'l' (int32) of 0x7FFFFFFF (2147483647), but no
|
||||||
|
// element data at all. max_size() for a std::vector is far larger
|
||||||
|
// than this count, so it does not reject the header outright; the
|
||||||
|
// (capped) reservation must not attempt to allocate space for
|
||||||
|
// billions of elements before the missing data is detected.
|
||||||
|
json _;
|
||||||
|
const std::vector<uint8_t> input = {'[', '$', 'i', '#', 'l', 0xFF, 0xFF, 0xFF, 0x7F};
|
||||||
|
// On a platform where std::vector<json>::max_size() is smaller than
|
||||||
|
// the claimed count (e.g. 32-bit, where max_size() is bounded by a
|
||||||
|
// 32-bit SIZE_MAX divided by sizeof(json)), the SAX consumer's own
|
||||||
|
// check rejects the header outright (out_of_range.408, with the
|
||||||
|
// claimed count in the message) instead of accepting it and only
|
||||||
|
// finding it short of data once the (capped) reservation looks for
|
||||||
|
// element bytes that were never provided (parse_error.110). Either
|
||||||
|
// is an acceptable, bounded rejection of the hostile header -- the
|
||||||
|
// property under test is that no path attempts to allocate space
|
||||||
|
// for billions of elements.
|
||||||
|
bool threw = false;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
_ = json::from_bjdata(input);
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
threw = true;
|
||||||
|
CHECK(e.id == 110);
|
||||||
|
CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 10: syntax error while parsing BJData number: unexpected end of input");
|
||||||
|
}
|
||||||
|
catch (const json::out_of_range& e)
|
||||||
|
{
|
||||||
|
threw = true;
|
||||||
|
CHECK(e.id == 408);
|
||||||
|
CHECK(std::string(e.what()).find("excessive array size") != std::string::npos);
|
||||||
|
}
|
||||||
|
CHECK(threw);
|
||||||
|
CHECK(json::from_bjdata(input, true, false).is_discarded());
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("arrays of various sizes decode to the same value as before the reserve optimization")
|
||||||
|
{
|
||||||
|
for (const auto size :
|
||||||
|
{
|
||||||
|
std::size_t(0), std::size_t(1), std::size_t(5), // small
|
||||||
|
std::size_t(16384), // exactly at the reserve cap
|
||||||
|
std::size_t(20000) // above the reserve cap
|
||||||
|
})
|
||||||
|
{
|
||||||
|
CAPTURE(size)
|
||||||
|
json j = json::array();
|
||||||
|
for (std::size_t i = 0; i < size; ++i)
|
||||||
|
{
|
||||||
|
j.push_back(static_cast<int>(i % 1000));
|
||||||
|
}
|
||||||
|
|
||||||
|
// exercise both the plain and the optimized [$type#count encoding
|
||||||
|
const auto packed_plain = json::to_bjdata(j);
|
||||||
|
CHECK(json::from_bjdata(packed_plain) == j);
|
||||||
|
|
||||||
|
const auto packed_optimized = json::to_bjdata(j, true, true);
|
||||||
|
CHECK(json::from_bjdata(packed_optimized) == j);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
|
||||||
|
{
|
||||||
|
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
|
||||||
|
// a custom SAX consumer that does not touch a DOM array sees identical events
|
||||||
|
json j = json::array();
|
||||||
|
for (int i = 0; i < 100; ++i)
|
||||||
|
{
|
||||||
|
j.push_back(i);
|
||||||
|
}
|
||||||
|
const auto packed = json::to_bjdata(j, true, true);
|
||||||
|
|
||||||
|
SaxCountdown scp(1000000); // large enough to never trigger an abort
|
||||||
|
CHECK(json::sax_parse(packed, &scp, json::input_format_t::bjdata));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("Universal Binary JSON Specification Examples 1")
|
TEST_CASE("Universal Binary JSON Specification Examples 1")
|
||||||
{
|
{
|
||||||
SECTION("Null Value")
|
SECTION("Null Value")
|
||||||
|
|||||||
@@ -2035,6 +2035,86 @@ TEST_CASE("CBOR definite length equal to the indefinite-length sentinel")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("issue #5405 - array reserve for definite-length CBOR arrays")
|
||||||
|
{
|
||||||
|
SECTION("a huge claimed length with no element data must not over-allocate")
|
||||||
|
{
|
||||||
|
// 0x9A: array with a four-byte length; claims 0xFFFFFFFF (4294967295)
|
||||||
|
// elements but provides none. max_size() for a std::vector is far
|
||||||
|
// larger than this count, so it does not reject the header outright;
|
||||||
|
// the (capped) reservation must not attempt to allocate space for
|
||||||
|
// billions of elements before the missing data is detected.
|
||||||
|
json _;
|
||||||
|
const std::vector<uint8_t> input = {0x9A, 0xFF, 0xFF, 0xFF, 0xFF};
|
||||||
|
// On a platform where std::size_t is narrower than 64 bits (e.g.
|
||||||
|
// 32-bit), the claimed count 0xFFFFFFFF coincides with that
|
||||||
|
// platform's detail::unknown_size() sentinel (SIZE_MAX), so the
|
||||||
|
// format-level size check rejects it outright (out_of_range.408,
|
||||||
|
// "excessive ... size") before the SAX consumer's own max_size()
|
||||||
|
// check would even run; on a 64-bit platform it passes both of
|
||||||
|
// those checks and is only found short of data once the (capped)
|
||||||
|
// reservation looks for element bytes that were never provided
|
||||||
|
// (parse_error.110). Either is an acceptable, bounded rejection of
|
||||||
|
// the hostile header -- the property under test is that no path
|
||||||
|
// attempts to allocate space for billions of elements.
|
||||||
|
bool threw = false;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
_ = json::from_cbor(input);
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
threw = true;
|
||||||
|
CHECK(e.id == 110);
|
||||||
|
CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 6: syntax error while parsing CBOR value: unexpected end of input");
|
||||||
|
}
|
||||||
|
catch (const json::out_of_range& e)
|
||||||
|
{
|
||||||
|
threw = true;
|
||||||
|
CHECK(e.id == 408);
|
||||||
|
CHECK(std::string(e.what()).find("excessive") != std::string::npos);
|
||||||
|
}
|
||||||
|
CHECK(threw);
|
||||||
|
CHECK(json::from_cbor(input, true, false).is_discarded());
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("arrays of various sizes decode to the same value as before the reserve optimization")
|
||||||
|
{
|
||||||
|
for (const auto size :
|
||||||
|
{
|
||||||
|
std::size_t{0}, std::size_t{1}, std::size_t{5}, // small
|
||||||
|
std::size_t{16384}, // exactly at the reserve cap
|
||||||
|
std::size_t{20000} // above the reserve cap
|
||||||
|
})
|
||||||
|
{
|
||||||
|
CAPTURE(size)
|
||||||
|
json j = json::array();
|
||||||
|
for (std::size_t i = 0; i < size; ++i)
|
||||||
|
{
|
||||||
|
j.push_back(static_cast<int>(i % 1000));
|
||||||
|
}
|
||||||
|
|
||||||
|
const auto packed = json::to_cbor(j);
|
||||||
|
CHECK(json::from_cbor(packed) == j);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
|
||||||
|
{
|
||||||
|
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
|
||||||
|
// a custom SAX consumer that does not touch a DOM array sees identical events
|
||||||
|
json j = json::array();
|
||||||
|
for (int i = 0; i < 100; ++i)
|
||||||
|
{
|
||||||
|
j.push_back(i);
|
||||||
|
}
|
||||||
|
const auto packed = json::to_cbor(j);
|
||||||
|
|
||||||
|
SaxCountdown scp(1000000); // large enough to never trigger an abort
|
||||||
|
CHECK(json::sax_parse(packed, &scp, json::input_format_t::cbor));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("CBOR roundtrips" * doctest::skip())
|
TEST_CASE("CBOR roundtrips" * doctest::skip())
|
||||||
{
|
{
|
||||||
SECTION("input from flynn")
|
SECTION("input from flynn")
|
||||||
|
|||||||
@@ -930,94 +930,6 @@ TEST_CASE("parser class")
|
|||||||
CHECK(accept_helper("+1") == false);
|
CHECK(accept_helper("+1") == false);
|
||||||
CHECK(accept_helper("+0") == false);
|
CHECK(accept_helper("+0") == false);
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("issue #5411 - skip conversion when accept() does not need the numeric value")
|
|
||||||
{
|
|
||||||
// lexer::scan_number() may skip strtoull()/strtoll() for
|
|
||||||
// value_unsigned/value_integer tokens when the caller (e.g.
|
|
||||||
// json::accept()) does not need the converted value, as long
|
|
||||||
// as the digit count alone guarantees no 64-bit overflow (see
|
|
||||||
// the "safe_digit_count" fast path in scan_number()). This
|
|
||||||
// differential test checks that json::accept() (which enables
|
|
||||||
// the fast path) and json::parse() (which never does) always
|
|
||||||
// agree, over a corpus that exercises both the fast path
|
|
||||||
// (<=18 digits) and the untouched, exact fallback path (>=19
|
|
||||||
// digits) -- including reclassification of huge digit-only
|
|
||||||
// integers to a (possibly non-finite) floating-point value.
|
|
||||||
const std::vector<std::pair<std::string, bool>> cases =
|
|
||||||
{
|
|
||||||
// normal small/large integers, both signs
|
|
||||||
{"0", true}, {"1", true}, {"-1", true}, {"42", true}, {"-42", true},
|
|
||||||
{"123456789", true}, {"-123456789", true},
|
|
||||||
|
|
||||||
// digit-count boundary around the 18-digit safe cutoff (both signs)
|
|
||||||
{std::string(17, '9'), true},
|
|
||||||
{std::string(18, '9'), true},
|
|
||||||
{std::string(19, '9'), true},
|
|
||||||
{std::string(20, '9'), true},
|
|
||||||
{"-" + std::string(17, '9'), true},
|
|
||||||
{"-" + std::string(18, '9'), true},
|
|
||||||
{"-" + std::string(19, '9'), true},
|
|
||||||
{"-" + std::string(20, '9'), true},
|
|
||||||
|
|
||||||
// 64-bit boundaries
|
|
||||||
{"9223372036854775807", true}, // INT64_MAX
|
|
||||||
{"-9223372036854775808", true}, // INT64_MIN
|
|
||||||
{"18446744073709551615", true}, // UINT64_MAX
|
|
||||||
{"18446744073709551616", true}, // UINT64_MAX + 1 (overflows uint64_t, finite double)
|
|
||||||
|
|
||||||
// the 28-digit example from the issue: overflows uint64_t
|
|
||||||
// but is finite as a double, so the scanner reclassifies
|
|
||||||
// it to value_float and it is accepted
|
|
||||||
{"9999999999999999999999999999", true},
|
|
||||||
|
|
||||||
// huge digit-only integers that overflow even a double -> rejected
|
|
||||||
{std::string(309, '9'), false},
|
|
||||||
{std::string(400, '9'), false},
|
|
||||||
{"1" + std::string(400, '0'), false},
|
|
||||||
|
|
||||||
// 1e999 / 1e400 style overflow -> rejected
|
|
||||||
{"1e999", false},
|
|
||||||
{"1e400", false},
|
|
||||||
{"-1e999", false},
|
|
||||||
{"1E999", false},
|
|
||||||
|
|
||||||
// values straddling DBL_MAX
|
|
||||||
{"1.7976931348623157e308", true}, // <= DBL_MAX, finite
|
|
||||||
{"1.7976931348623159e308", false}, // > DBL_MAX, overflows to inf
|
|
||||||
|
|
||||||
// a mix of other valid/invalid numeric syntax
|
|
||||||
{"3.14159", true},
|
|
||||||
{"-0.0", true},
|
|
||||||
{"1.0e10", true},
|
|
||||||
{"01", false},
|
|
||||||
{"-", false},
|
|
||||||
{"1.", false},
|
|
||||||
{"1e", false},
|
|
||||||
{"+1", false},
|
|
||||||
};
|
|
||||||
|
|
||||||
for (const auto& c : cases)
|
|
||||||
{
|
|
||||||
const std::string& number = c.first;
|
|
||||||
const bool expected = c.second;
|
|
||||||
CAPTURE(number)
|
|
||||||
CAPTURE(expected)
|
|
||||||
|
|
||||||
// accept() takes the fast path (skips conversion when possible)
|
|
||||||
CHECK(json::accept(number) == expected);
|
|
||||||
|
|
||||||
// parse() always performs the full conversion; it must agree
|
|
||||||
json j;
|
|
||||||
CHECK_NOTHROW(json::parser(nlohmann::detail::input_adapter(number), nullptr, false).parse(true, j));
|
|
||||||
CHECK(!j.is_discarded() == expected);
|
|
||||||
|
|
||||||
// wrap in an array so get_token() is exercised beyond the
|
|
||||||
// very first (constructor-time) scan as well
|
|
||||||
const std::string wrapped = "[" + number + "," + number + "]";
|
|
||||||
CHECK(json::accept(wrapped) == expected);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1482,62 +1394,6 @@ TEST_CASE("parser class")
|
|||||||
CHECK(accept_helper("\"\\uD80C\\uFFFF\"") == false);
|
CHECK(accept_helper("\"\\uD80C\\uFFFF\"") == false);
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("issue #5412 - whitespace skipping bookkeeping (compact vs. pretty-printed)")
|
|
||||||
{
|
|
||||||
// lexer::skip_whitespace() reads its first character with get() (to
|
|
||||||
// honor a possibly pending unget() from the previous token) and every
|
|
||||||
// further whitespace character with get_ignoring_pending_unget() (a
|
|
||||||
// get() variant that skips the then-always-false next_unget check).
|
|
||||||
// This must not change the reported byte offset, line, or column of
|
|
||||||
// a syntax error, even when a long run of whitespace containing
|
|
||||||
// multiple newlines is skipped beforehand (as with pretty-printed
|
|
||||||
// input). The expected values below were captured from the
|
|
||||||
// unmodified do-while(get()) loop, so any regression that miscounts
|
|
||||||
// characters or newlines while skipping whitespace changes them.
|
|
||||||
const auto check_error = [](const std::string & input, std::size_t expected_byte,
|
|
||||||
const std::string & expected_what)
|
|
||||||
{
|
|
||||||
CAPTURE(input)
|
|
||||||
try
|
|
||||||
{
|
|
||||||
json _ = json::parse(input);
|
|
||||||
FAIL_CHECK("expected a parse_error, but parsing succeeded");
|
|
||||||
}
|
|
||||||
catch (const json::parse_error& e)
|
|
||||||
{
|
|
||||||
CHECK(e.byte == expected_byte);
|
|
||||||
CHECK(std::string(e.what()) == expected_what);
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
// a nested document, serialized both compactly and pretty-printed
|
|
||||||
// (dump(4)), each truncated right before the final closing '}' so
|
|
||||||
// that the parser hits EOF after skipping all of the (in the
|
|
||||||
// pretty-printed case, substantial) indentation whitespace
|
|
||||||
const json doc =
|
|
||||||
{
|
|
||||||
{"a", 1},
|
|
||||||
{"b", json::array({true, false, nullptr, "x"})},
|
|
||||||
{"c", json::object({{"d", 3.14}, {"e", json::array({1, 2, 3})}})}
|
|
||||||
};
|
|
||||||
|
|
||||||
const std::string compact = doc.dump();
|
|
||||||
const std::string pretty = doc.dump(4);
|
|
||||||
|
|
||||||
check_error(compact.substr(0, compact.size() - 1), 60,
|
|
||||||
"[json.exception.parse_error.101] parse error at line 1, column 60: syntax error while parsing object - unexpected end of input; expected '}'");
|
|
||||||
check_error(pretty.substr(0, pretty.size() - 1), 193,
|
|
||||||
"[json.exception.parse_error.101] parse error at line 17, column 1: syntax error while parsing object - unexpected end of input; expected '}'");
|
|
||||||
|
|
||||||
// an invalid token appearing after several indented, multi-line
|
|
||||||
// whitespace runs vs. the same document without any of that
|
|
||||||
// whitespace
|
|
||||||
check_error("{\n \"a\": 1,\n \"b\": [\n true,\n false\n ],\n \"c\": @\n}", 70,
|
|
||||||
"[json.exception.parse_error.101] parse error at line 7, column 10: syntax error while parsing value - invalid literal; last read: '\"c\": @'");
|
|
||||||
check_error("{\"a\":1,\"b\":[true,false],\"c\":@}", 29,
|
|
||||||
"[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing value - invalid literal; last read: '\"c\":@'");
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("tests found by mutate++")
|
SECTION("tests found by mutate++")
|
||||||
{
|
{
|
||||||
// test case to make sure no comma precedes the first key
|
// test case to make sure no comma precedes the first key
|
||||||
|
|||||||
@@ -1597,6 +1597,85 @@ TEST_CASE("MessagePack")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("issue #5405 - array reserve for definite-length MessagePack arrays")
|
||||||
|
{
|
||||||
|
SECTION("a huge claimed length with no element data must not over-allocate")
|
||||||
|
{
|
||||||
|
// 0xdd: array 32 (four-byte length); claims 0xFFFFFFFF (4294967295)
|
||||||
|
// elements but provides none. max_size() for a std::vector is far
|
||||||
|
// larger than this count, so it does not reject the header outright;
|
||||||
|
// the (capped) reservation must not attempt to allocate space for
|
||||||
|
// billions of elements before the missing data is detected.
|
||||||
|
json _;
|
||||||
|
const std::vector<uint8_t> input = {0xdd, 0xFF, 0xFF, 0xFF, 0xFF};
|
||||||
|
// On a platform where std::size_t is narrower than 64 bits (e.g.
|
||||||
|
// 32-bit), the claimed count 0xFFFFFFFF coincides with that
|
||||||
|
// platform's SIZE_MAX, which some size-narrowing checks treat the
|
||||||
|
// same as detail::unknown_size(); it may then be rejected before
|
||||||
|
// the SAX consumer's own max_size() check (out_of_range.408) rather
|
||||||
|
// than being accepted and only found short of data once the
|
||||||
|
// (capped) reservation looks for element bytes that were never
|
||||||
|
// provided (parse_error.110). Either is an acceptable, bounded
|
||||||
|
// rejection of the hostile header -- the property under test is
|
||||||
|
// that no path attempts to allocate space for billions of elements.
|
||||||
|
bool threw = false;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
_ = json::from_msgpack(input);
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
threw = true;
|
||||||
|
CHECK(e.id == 110);
|
||||||
|
CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 6: syntax error while parsing MessagePack value: unexpected end of input");
|
||||||
|
}
|
||||||
|
catch (const json::out_of_range& e)
|
||||||
|
{
|
||||||
|
threw = true;
|
||||||
|
CHECK(e.id == 408);
|
||||||
|
CHECK(std::string(e.what()).find("excessive") != std::string::npos);
|
||||||
|
}
|
||||||
|
CHECK(threw);
|
||||||
|
CHECK(json::from_msgpack(input, true, false).is_discarded());
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("arrays of various sizes decode to the same value as before the reserve optimization")
|
||||||
|
{
|
||||||
|
for (const auto size :
|
||||||
|
{
|
||||||
|
std::size_t(0), std::size_t(1), std::size_t(5), // small
|
||||||
|
std::size_t(16384), // exactly at the reserve cap
|
||||||
|
std::size_t(20000) // above the reserve cap
|
||||||
|
})
|
||||||
|
{
|
||||||
|
CAPTURE(size)
|
||||||
|
json j = json::array();
|
||||||
|
for (std::size_t i = 0; i < size; ++i)
|
||||||
|
{
|
||||||
|
j.push_back(static_cast<int>(i % 1000));
|
||||||
|
}
|
||||||
|
|
||||||
|
const auto packed = json::to_msgpack(j);
|
||||||
|
CHECK(json::from_msgpack(packed) == j);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
|
||||||
|
{
|
||||||
|
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
|
||||||
|
// a custom SAX consumer that does not touch a DOM array sees identical events
|
||||||
|
json j = json::array();
|
||||||
|
for (int i = 0; i < 100; ++i)
|
||||||
|
{
|
||||||
|
j.push_back(i);
|
||||||
|
}
|
||||||
|
const auto packed = json::to_msgpack(j);
|
||||||
|
|
||||||
|
SaxCountdown scp(1000000); // large enough to never trigger an abort
|
||||||
|
CHECK(json::sax_parse(packed, &scp, json::input_format_t::msgpack));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// use this testcase outside [hide] to run it with Valgrind
|
// use this testcase outside [hide] to run it with Valgrind
|
||||||
TEST_CASE("single MessagePack roundtrip")
|
TEST_CASE("single MessagePack roundtrip")
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -2149,6 +2149,90 @@ TEST_CASE("UBJSON")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("issue #5405 - array reserve for definite-length UBJSON arrays")
|
||||||
|
{
|
||||||
|
SECTION("a huge claimed length with no element data must not over-allocate")
|
||||||
|
{
|
||||||
|
// optimized form [$type#count: type 'i' (int8), count as a four-byte
|
||||||
|
// 'l' (int32) of 0x7FFFFFFF (2147483647), but no element data at all.
|
||||||
|
// max_size() for a std::vector is far larger than this count, so it
|
||||||
|
// does not reject the header outright; the (capped) reservation must
|
||||||
|
// not attempt to allocate space for billions of elements before the
|
||||||
|
// missing data is detected.
|
||||||
|
json _;
|
||||||
|
const std::vector<uint8_t> input = {'[', '$', 'i', '#', 'l', 0x7F, 0xFF, 0xFF, 0xFF};
|
||||||
|
// On a platform where std::vector<json>::max_size() is smaller than
|
||||||
|
// the claimed count (e.g. 32-bit, where max_size() is bounded by a
|
||||||
|
// 32-bit SIZE_MAX divided by sizeof(json)), the SAX consumer's own
|
||||||
|
// check rejects the header outright (out_of_range.408, with the
|
||||||
|
// claimed count in the message) instead of accepting it and only
|
||||||
|
// finding it short of data once the (capped) reservation looks for
|
||||||
|
// element bytes that were never provided (parse_error.110). Either
|
||||||
|
// is an acceptable, bounded rejection of the hostile header -- the
|
||||||
|
// property under test is that no path attempts to allocate space
|
||||||
|
// for billions of elements.
|
||||||
|
bool threw = false;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
_ = json::from_ubjson(input);
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
threw = true;
|
||||||
|
CHECK(e.id == 110);
|
||||||
|
CHECK(std::string(e.what()) == "[json.exception.parse_error.110] parse error at byte 10: syntax error while parsing UBJSON number: unexpected end of input");
|
||||||
|
}
|
||||||
|
catch (const json::out_of_range& e)
|
||||||
|
{
|
||||||
|
threw = true;
|
||||||
|
CHECK(e.id == 408);
|
||||||
|
CHECK(std::string(e.what()).find("excessive array size") != std::string::npos);
|
||||||
|
}
|
||||||
|
CHECK(threw);
|
||||||
|
CHECK(json::from_ubjson(input, true, false).is_discarded());
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("arrays of various sizes decode to the same value as before the reserve optimization")
|
||||||
|
{
|
||||||
|
for (const auto size :
|
||||||
|
{
|
||||||
|
std::size_t(0), std::size_t(1), std::size_t(5), // small
|
||||||
|
std::size_t(16384), // exactly at the reserve cap
|
||||||
|
std::size_t(20000) // above the reserve cap
|
||||||
|
})
|
||||||
|
{
|
||||||
|
CAPTURE(size)
|
||||||
|
json j = json::array();
|
||||||
|
for (std::size_t i = 0; i < size; ++i)
|
||||||
|
{
|
||||||
|
j.push_back(static_cast<int>(i % 1000));
|
||||||
|
}
|
||||||
|
|
||||||
|
// exercise both the plain and the optimized [$type#count encoding
|
||||||
|
const auto packed_plain = json::to_ubjson(j);
|
||||||
|
CHECK(json::from_ubjson(packed_plain) == j);
|
||||||
|
|
||||||
|
const auto packed_optimized = json::to_ubjson(j, true, true);
|
||||||
|
CHECK(json::from_ubjson(packed_optimized) == j);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
|
||||||
|
{
|
||||||
|
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
|
||||||
|
// a custom SAX consumer that does not touch a DOM array sees identical events
|
||||||
|
json j = json::array();
|
||||||
|
for (int i = 0; i < 100; ++i)
|
||||||
|
{
|
||||||
|
j.push_back(i);
|
||||||
|
}
|
||||||
|
const auto packed = json::to_ubjson(j, true, true);
|
||||||
|
|
||||||
|
SaxCountdown scp(1000000); // large enough to never trigger an abort
|
||||||
|
CHECK(json::sax_parse(packed, &scp, json::input_format_t::ubjson));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("Universal Binary JSON Specification Examples 1")
|
TEST_CASE("Universal Binary JSON Specification Examples 1")
|
||||||
{
|
{
|
||||||
SECTION("Null Value")
|
SECTION("Null Value")
|
||||||
|
|||||||
Reference in New Issue
Block a user