mirror of
https://github.com/nlohmann/json.git
synced 2026-09-07 08:47:57 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7d15336745 | ||
|
|
ff0aab0053 | ||
|
|
aeb33695c9 | ||
|
|
b6750e9911 | ||
|
|
5d8789e2f7 | ||
|
|
68c698194b | ||
|
|
ef97a360c8 | ||
|
|
7b648c7dc4 |
@@ -149,10 +149,11 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
public:
|
public:
|
||||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||||
|
|
||||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept
|
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept
|
||||||
: ia(std::move(adapter))
|
: ia(std::move(adapter))
|
||||||
, ignore_comments(ignore_comments_)
|
, ignore_comments(ignore_comments_)
|
||||||
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
||||||
|
, discard_number_values(discard_number_values_)
|
||||||
{}
|
{}
|
||||||
|
|
||||||
// deleted because of pointer members
|
// deleted because of pointer members
|
||||||
@@ -1279,6 +1280,58 @@ scan_number_done:
|
|||||||
// we are done scanning a number)
|
// we are done scanning a number)
|
||||||
unget();
|
unget();
|
||||||
|
|
||||||
|
// If the caller does not need the converted value (only whether the
|
||||||
|
// input is syntactically valid; see json_sax_acceptor/accept()), an
|
||||||
|
// unsigned/integer token can be reported without calling
|
||||||
|
// strtoull()/strtoll() at all, *provided* we can already tell from
|
||||||
|
// the digit count alone that the conversion cannot overflow 64 bits.
|
||||||
|
// Such tokens are always finite and are accepted unconditionally by
|
||||||
|
// the parser regardless of their actual value (parser::sax_parse_internal()
|
||||||
|
// never checks finiteness for value_unsigned/value_integer), so the
|
||||||
|
// classification below is all that is needed.
|
||||||
|
//
|
||||||
|
// A decimal number with up to 18 digits is always representable in
|
||||||
|
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
|
||||||
|
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
|
||||||
|
// could not have set errno to ERANGE for it. Numbers with more digits
|
||||||
|
// (rare in practice) fall through to the exact code below, unchanged,
|
||||||
|
// so their handling -- including reclassification to value_float when
|
||||||
|
// the value overflows 64 bits, and rejection when it is not even
|
||||||
|
// finite as a double -- is bit-for-bit identical to before this
|
||||||
|
// optimization.
|
||||||
|
//
|
||||||
|
// Note this reasons about std::uint64_t/std::int64_t, not about
|
||||||
|
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
|
||||||
|
// narrower, template parameters -- e.g. std::uint32_t). That is fine
|
||||||
|
// *only* because discard_number_values is exclusively set by
|
||||||
|
// accept() (see json.hpp), and accept() always parses through the
|
||||||
|
// library's own json_sax_acceptor -- never a user-supplied SAX
|
||||||
|
// consumer -- whose number_unsigned()/number_integer()/number_float()
|
||||||
|
// callbacks unconditionally discard their argument and return true.
|
||||||
|
// So for every caller that can reach this branch, neither the token
|
||||||
|
// classification below nor the eventual (possibly narrowed, and on
|
||||||
|
// this fast path left stale/unset) value_unsigned/value_integer is
|
||||||
|
// ever consulted -- an unsigned/integer token is accepted outright,
|
||||||
|
// and even a >18-digit token that this fast path deliberately falls
|
||||||
|
// through for is, once reclassified to value_float, still finite
|
||||||
|
// (and thus accepted) for any digit count that fits in number_unsigned_t
|
||||||
|
// or number_integer_t regardless of that type's width. If this
|
||||||
|
// function is ever taught to run with discard_number_values true for
|
||||||
|
// a caller that *does* read the converted value, this reasoning (and
|
||||||
|
// the fast path below) would need to be revisited.
|
||||||
|
if (discard_number_values)
|
||||||
|
{
|
||||||
|
constexpr std::size_t safe_digit_count = 18;
|
||||||
|
if (number_type == token_type::value_unsigned && token_buffer.size() <= safe_digit_count)
|
||||||
|
{
|
||||||
|
return token_type::value_unsigned;
|
||||||
|
}
|
||||||
|
if (number_type == token_type::value_integer && token_buffer.size() - 1 <= safe_digit_count)
|
||||||
|
{
|
||||||
|
return token_type::value_integer;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||||
errno = 0;
|
errno = 0;
|
||||||
|
|
||||||
@@ -1393,8 +1446,7 @@ scan_number_done:
|
|||||||
*/
|
*/
|
||||||
char_int_type get()
|
char_int_type get()
|
||||||
{
|
{
|
||||||
++position.chars_read_total;
|
advance_position();
|
||||||
++position.chars_read_current_line;
|
|
||||||
|
|
||||||
if (next_unget)
|
if (next_unget)
|
||||||
{
|
{
|
||||||
@@ -1406,6 +1458,23 @@ scan_number_done:
|
|||||||
current = ia.get_character();
|
current = ia.get_character();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
return track_after_read();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// shared head of get() / get_ignoring_pending_unget(): bump the
|
||||||
|
/// per-character position counters (line-count-on-'\n' bookkeeping is
|
||||||
|
/// handled afterwards, in track_after_read(), once `current` is known)
|
||||||
|
void advance_position() noexcept
|
||||||
|
{
|
||||||
|
++position.chars_read_total;
|
||||||
|
++position.chars_read_current_line;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// shared tail of get() / get_ignoring_pending_unget(): capture the
|
||||||
|
/// character for error messages (if needed) and update line/column
|
||||||
|
/// bookkeeping for the character now in `current`
|
||||||
|
char_int_type track_after_read()
|
||||||
|
{
|
||||||
// seekable adapters reconstruct the token lazily on error (see
|
// seekable adapters reconstruct the token lazily on error (see
|
||||||
// get_token_string), so the eager per-character copy is skipped
|
// get_token_string), so the eager per-character copy is skipped
|
||||||
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
||||||
@@ -1419,6 +1488,29 @@ scan_number_done:
|
|||||||
return current;
|
return current;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief like get(), but for call sites that can prove no unget() is pending
|
||||||
|
|
||||||
|
get() has to check the `next_unget` flag on every call, because a
|
||||||
|
previous token may have ended with unget() (e.g. scan_number() always
|
||||||
|
ungets the character that terminated the number, so the next call to
|
||||||
|
scan() can see it again). skip_whitespace() reads that first,
|
||||||
|
possibly-ungotten character via a plain get(), but every further
|
||||||
|
character it reads is guaranteed to be a fresh read: nothing between
|
||||||
|
those calls invokes unget(). This variant skips the (otherwise always
|
||||||
|
false) next_unget branch for those calls; it is not a general
|
||||||
|
replacement for get().
|
||||||
|
*/
|
||||||
|
char_int_type get_ignoring_pending_unget()
|
||||||
|
{
|
||||||
|
JSON_ASSERT(!next_unget);
|
||||||
|
|
||||||
|
advance_position();
|
||||||
|
current = ia.get_character();
|
||||||
|
|
||||||
|
return track_after_read();
|
||||||
|
}
|
||||||
|
|
||||||
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
||||||
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
||||||
|
|
||||||
@@ -1612,13 +1704,37 @@ scan_number_done:
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// whether `current` is one of the four JSON whitespace characters
|
||||||
|
bool current_is_whitespace() const noexcept
|
||||||
|
{
|
||||||
|
return current == ' ' || current == '\t' || current == '\n' || current == '\r';
|
||||||
|
}
|
||||||
|
|
||||||
void skip_whitespace()
|
void skip_whitespace()
|
||||||
{
|
{
|
||||||
|
// the first character may be a pending unget() left over from the
|
||||||
|
// previous token (see get_ignoring_pending_unget()); every
|
||||||
|
// subsequent character read by this loop is guaranteed fresh, since
|
||||||
|
// nothing below calls unget()
|
||||||
|
get();
|
||||||
|
|
||||||
|
if (!current_is_whitespace())
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// this is written as an if-guarded do-while (rather than a plain
|
||||||
|
// while loop) because that shape is what lets both GCC and Clang
|
||||||
|
// keep the input adapter's read pointer in a register across
|
||||||
|
// iterations; the equivalent while-loop measurably defeated that
|
||||||
|
// optimization in testing, turning long whitespace runs (e.g. the
|
||||||
|
// indentation of pretty-printed JSON) from a register-only loop
|
||||||
|
// into one that reloads the pointer from memory every character
|
||||||
do
|
do
|
||||||
{
|
{
|
||||||
get();
|
get_ignoring_pending_unget();
|
||||||
}
|
}
|
||||||
while (current == ' ' || current == '\t' || current == '\n' || current == '\r');
|
while (current_is_whitespace());
|
||||||
}
|
}
|
||||||
|
|
||||||
token_type scan()
|
token_type scan()
|
||||||
@@ -1754,6 +1870,13 @@ scan_number_done:
|
|||||||
const char_int_type decimal_point_char = '.';
|
const char_int_type decimal_point_char = '.';
|
||||||
/// the position of the decimal point in the input
|
/// the position of the decimal point in the input
|
||||||
std::size_t decimal_point_position = std::string::npos;
|
std::size_t decimal_point_position = std::string::npos;
|
||||||
|
|
||||||
|
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
|
||||||
|
/// token classification and never looks at the converted numeric value;
|
||||||
|
/// when set, scan_number() may skip strtoull()/strtoll() for
|
||||||
|
/// value_unsigned/value_integer tokens whose digit count guarantees they
|
||||||
|
/// fit into 64 bits (see scan_number())
|
||||||
|
const bool discard_number_values = false;
|
||||||
};
|
};
|
||||||
|
|
||||||
} // namespace detail
|
} // namespace detail
|
||||||
|
|||||||
@@ -72,9 +72,10 @@ class parser
|
|||||||
parser_callback_t<BasicJsonType> cb = nullptr,
|
parser_callback_t<BasicJsonType> cb = nullptr,
|
||||||
const bool allow_exceptions_ = true,
|
const bool allow_exceptions_ = true,
|
||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas_ = false)
|
const bool ignore_trailing_commas_ = false,
|
||||||
|
const bool discard_number_values_ = false)
|
||||||
: callback(std::move(cb))
|
: callback(std::move(cb))
|
||||||
, m_lexer(std::move(adapter), ignore_comments)
|
, m_lexer(std::move(adapter), ignore_comments, discard_number_values_)
|
||||||
, allow_exceptions(allow_exceptions_)
|
, allow_exceptions(allow_exceptions_)
|
||||||
, ignore_trailing_commas(ignore_trailing_commas_)
|
, ignore_trailing_commas(ignore_trailing_commas_)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -164,11 +164,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
detail::parser_callback_t<basic_json>cb = nullptr,
|
detail::parser_callback_t<basic_json>cb = nullptr,
|
||||||
const bool allow_exceptions = true,
|
const bool allow_exceptions = true,
|
||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false
|
const bool ignore_trailing_commas = false,
|
||||||
|
const bool discard_number_values = false
|
||||||
)
|
)
|
||||||
{
|
{
|
||||||
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
||||||
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas);
|
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas, discard_number_values);
|
||||||
}
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
@@ -4133,7 +4134,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||||
@@ -4144,7 +4145,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||||
@@ -4153,7 +4154,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief generate SAX events
|
/// @brief generate SAX events
|
||||||
|
|||||||
@@ -7932,10 +7932,11 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
public:
|
public:
|
||||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||||
|
|
||||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept
|
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept
|
||||||
: ia(std::move(adapter))
|
: ia(std::move(adapter))
|
||||||
, ignore_comments(ignore_comments_)
|
, ignore_comments(ignore_comments_)
|
||||||
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
||||||
|
, discard_number_values(discard_number_values_)
|
||||||
{}
|
{}
|
||||||
|
|
||||||
// deleted because of pointer members
|
// deleted because of pointer members
|
||||||
@@ -9062,6 +9063,58 @@ scan_number_done:
|
|||||||
// we are done scanning a number)
|
// we are done scanning a number)
|
||||||
unget();
|
unget();
|
||||||
|
|
||||||
|
// If the caller does not need the converted value (only whether the
|
||||||
|
// input is syntactically valid; see json_sax_acceptor/accept()), an
|
||||||
|
// unsigned/integer token can be reported without calling
|
||||||
|
// strtoull()/strtoll() at all, *provided* we can already tell from
|
||||||
|
// the digit count alone that the conversion cannot overflow 64 bits.
|
||||||
|
// Such tokens are always finite and are accepted unconditionally by
|
||||||
|
// the parser regardless of their actual value (parser::sax_parse_internal()
|
||||||
|
// never checks finiteness for value_unsigned/value_integer), so the
|
||||||
|
// classification below is all that is needed.
|
||||||
|
//
|
||||||
|
// A decimal number with up to 18 digits is always representable in
|
||||||
|
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
|
||||||
|
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
|
||||||
|
// could not have set errno to ERANGE for it. Numbers with more digits
|
||||||
|
// (rare in practice) fall through to the exact code below, unchanged,
|
||||||
|
// so their handling -- including reclassification to value_float when
|
||||||
|
// the value overflows 64 bits, and rejection when it is not even
|
||||||
|
// finite as a double -- is bit-for-bit identical to before this
|
||||||
|
// optimization.
|
||||||
|
//
|
||||||
|
// Note this reasons about std::uint64_t/std::int64_t, not about
|
||||||
|
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
|
||||||
|
// narrower, template parameters -- e.g. std::uint32_t). That is fine
|
||||||
|
// *only* because discard_number_values is exclusively set by
|
||||||
|
// accept() (see json.hpp), and accept() always parses through the
|
||||||
|
// library's own json_sax_acceptor -- never a user-supplied SAX
|
||||||
|
// consumer -- whose number_unsigned()/number_integer()/number_float()
|
||||||
|
// callbacks unconditionally discard their argument and return true.
|
||||||
|
// So for every caller that can reach this branch, neither the token
|
||||||
|
// classification below nor the eventual (possibly narrowed, and on
|
||||||
|
// this fast path left stale/unset) value_unsigned/value_integer is
|
||||||
|
// ever consulted -- an unsigned/integer token is accepted outright,
|
||||||
|
// and even a >18-digit token that this fast path deliberately falls
|
||||||
|
// through for is, once reclassified to value_float, still finite
|
||||||
|
// (and thus accepted) for any digit count that fits in number_unsigned_t
|
||||||
|
// or number_integer_t regardless of that type's width. If this
|
||||||
|
// function is ever taught to run with discard_number_values true for
|
||||||
|
// a caller that *does* read the converted value, this reasoning (and
|
||||||
|
// the fast path below) would need to be revisited.
|
||||||
|
if (discard_number_values)
|
||||||
|
{
|
||||||
|
constexpr std::size_t safe_digit_count = 18;
|
||||||
|
if (number_type == token_type::value_unsigned && token_buffer.size() <= safe_digit_count)
|
||||||
|
{
|
||||||
|
return token_type::value_unsigned;
|
||||||
|
}
|
||||||
|
if (number_type == token_type::value_integer && token_buffer.size() - 1 <= safe_digit_count)
|
||||||
|
{
|
||||||
|
return token_type::value_integer;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||||
errno = 0;
|
errno = 0;
|
||||||
|
|
||||||
@@ -9176,8 +9229,7 @@ scan_number_done:
|
|||||||
*/
|
*/
|
||||||
char_int_type get()
|
char_int_type get()
|
||||||
{
|
{
|
||||||
++position.chars_read_total;
|
advance_position();
|
||||||
++position.chars_read_current_line;
|
|
||||||
|
|
||||||
if (next_unget)
|
if (next_unget)
|
||||||
{
|
{
|
||||||
@@ -9189,6 +9241,23 @@ scan_number_done:
|
|||||||
current = ia.get_character();
|
current = ia.get_character();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
return track_after_read();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// shared head of get() / get_ignoring_pending_unget(): bump the
|
||||||
|
/// per-character position counters (line-count-on-'\n' bookkeeping is
|
||||||
|
/// handled afterwards, in track_after_read(), once `current` is known)
|
||||||
|
void advance_position() noexcept
|
||||||
|
{
|
||||||
|
++position.chars_read_total;
|
||||||
|
++position.chars_read_current_line;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// shared tail of get() / get_ignoring_pending_unget(): capture the
|
||||||
|
/// character for error messages (if needed) and update line/column
|
||||||
|
/// bookkeeping for the character now in `current`
|
||||||
|
char_int_type track_after_read()
|
||||||
|
{
|
||||||
// seekable adapters reconstruct the token lazily on error (see
|
// seekable adapters reconstruct the token lazily on error (see
|
||||||
// get_token_string), so the eager per-character copy is skipped
|
// get_token_string), so the eager per-character copy is skipped
|
||||||
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
||||||
@@ -9202,6 +9271,29 @@ scan_number_done:
|
|||||||
return current;
|
return current;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief like get(), but for call sites that can prove no unget() is pending
|
||||||
|
|
||||||
|
get() has to check the `next_unget` flag on every call, because a
|
||||||
|
previous token may have ended with unget() (e.g. scan_number() always
|
||||||
|
ungets the character that terminated the number, so the next call to
|
||||||
|
scan() can see it again). skip_whitespace() reads that first,
|
||||||
|
possibly-ungotten character via a plain get(), but every further
|
||||||
|
character it reads is guaranteed to be a fresh read: nothing between
|
||||||
|
those calls invokes unget(). This variant skips the (otherwise always
|
||||||
|
false) next_unget branch for those calls; it is not a general
|
||||||
|
replacement for get().
|
||||||
|
*/
|
||||||
|
char_int_type get_ignoring_pending_unget()
|
||||||
|
{
|
||||||
|
JSON_ASSERT(!next_unget);
|
||||||
|
|
||||||
|
advance_position();
|
||||||
|
current = ia.get_character();
|
||||||
|
|
||||||
|
return track_after_read();
|
||||||
|
}
|
||||||
|
|
||||||
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
||||||
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
||||||
|
|
||||||
@@ -9395,13 +9487,37 @@ scan_number_done:
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// whether `current` is one of the four JSON whitespace characters
|
||||||
|
bool current_is_whitespace() const noexcept
|
||||||
|
{
|
||||||
|
return current == ' ' || current == '\t' || current == '\n' || current == '\r';
|
||||||
|
}
|
||||||
|
|
||||||
void skip_whitespace()
|
void skip_whitespace()
|
||||||
{
|
{
|
||||||
|
// the first character may be a pending unget() left over from the
|
||||||
|
// previous token (see get_ignoring_pending_unget()); every
|
||||||
|
// subsequent character read by this loop is guaranteed fresh, since
|
||||||
|
// nothing below calls unget()
|
||||||
|
get();
|
||||||
|
|
||||||
|
if (!current_is_whitespace())
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// this is written as an if-guarded do-while (rather than a plain
|
||||||
|
// while loop) because that shape is what lets both GCC and Clang
|
||||||
|
// keep the input adapter's read pointer in a register across
|
||||||
|
// iterations; the equivalent while-loop measurably defeated that
|
||||||
|
// optimization in testing, turning long whitespace runs (e.g. the
|
||||||
|
// indentation of pretty-printed JSON) from a register-only loop
|
||||||
|
// into one that reloads the pointer from memory every character
|
||||||
do
|
do
|
||||||
{
|
{
|
||||||
get();
|
get_ignoring_pending_unget();
|
||||||
}
|
}
|
||||||
while (current == ' ' || current == '\t' || current == '\n' || current == '\r');
|
while (current_is_whitespace());
|
||||||
}
|
}
|
||||||
|
|
||||||
token_type scan()
|
token_type scan()
|
||||||
@@ -9537,6 +9653,13 @@ scan_number_done:
|
|||||||
const char_int_type decimal_point_char = '.';
|
const char_int_type decimal_point_char = '.';
|
||||||
/// the position of the decimal point in the input
|
/// the position of the decimal point in the input
|
||||||
std::size_t decimal_point_position = std::string::npos;
|
std::size_t decimal_point_position = std::string::npos;
|
||||||
|
|
||||||
|
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
|
||||||
|
/// token classification and never looks at the converted numeric value;
|
||||||
|
/// when set, scan_number() may skip strtoull()/strtoll() for
|
||||||
|
/// value_unsigned/value_integer tokens whose digit count guarantees they
|
||||||
|
/// fit into 64 bits (see scan_number())
|
||||||
|
const bool discard_number_values = false;
|
||||||
};
|
};
|
||||||
|
|
||||||
} // namespace detail
|
} // namespace detail
|
||||||
@@ -14043,9 +14166,10 @@ class parser
|
|||||||
parser_callback_t<BasicJsonType> cb = nullptr,
|
parser_callback_t<BasicJsonType> cb = nullptr,
|
||||||
const bool allow_exceptions_ = true,
|
const bool allow_exceptions_ = true,
|
||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas_ = false)
|
const bool ignore_trailing_commas_ = false,
|
||||||
|
const bool discard_number_values_ = false)
|
||||||
: callback(std::move(cb))
|
: callback(std::move(cb))
|
||||||
, m_lexer(std::move(adapter), ignore_comments)
|
, m_lexer(std::move(adapter), ignore_comments, discard_number_values_)
|
||||||
, allow_exceptions(allow_exceptions_)
|
, allow_exceptions(allow_exceptions_)
|
||||||
, ignore_trailing_commas(ignore_trailing_commas_)
|
, ignore_trailing_commas(ignore_trailing_commas_)
|
||||||
{
|
{
|
||||||
@@ -21592,11 +21716,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
detail::parser_callback_t<basic_json>cb = nullptr,
|
detail::parser_callback_t<basic_json>cb = nullptr,
|
||||||
const bool allow_exceptions = true,
|
const bool allow_exceptions = true,
|
||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false
|
const bool ignore_trailing_commas = false,
|
||||||
|
const bool discard_number_values = false
|
||||||
)
|
)
|
||||||
{
|
{
|
||||||
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
||||||
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas);
|
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas, discard_number_values);
|
||||||
}
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
@@ -25561,7 +25686,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||||
@@ -25572,7 +25697,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||||
@@ -25581,7 +25706,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const bool ignore_comments = false,
|
const bool ignore_comments = false,
|
||||||
const bool ignore_trailing_commas = false)
|
const bool ignore_trailing_commas = false)
|
||||||
{
|
{
|
||||||
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// @brief generate SAX events
|
/// @brief generate SAX events
|
||||||
|
|||||||
@@ -930,6 +930,98 @@ TEST_CASE("parser class")
|
|||||||
CHECK(accept_helper("+1") == false);
|
CHECK(accept_helper("+1") == false);
|
||||||
CHECK(accept_helper("+0") == false);
|
CHECK(accept_helper("+0") == false);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SECTION("issue #5411 - skip conversion when accept() does not need the numeric value")
|
||||||
|
{
|
||||||
|
// lexer::scan_number() may skip strtoull()/strtoll() for
|
||||||
|
// value_unsigned/value_integer tokens when the caller (e.g.
|
||||||
|
// json::accept()) does not need the converted value, as long
|
||||||
|
// as the digit count alone guarantees no 64-bit overflow (see
|
||||||
|
// the "safe_digit_count" fast path in scan_number()). This
|
||||||
|
// differential test checks that json::accept() (which enables
|
||||||
|
// the fast path) and json::parse() (which never does) always
|
||||||
|
// agree, over a corpus that exercises both the fast path
|
||||||
|
// (<=18 digits) and the untouched, exact fallback path (>=19
|
||||||
|
// digits) -- including reclassification of huge digit-only
|
||||||
|
// integers to a (possibly non-finite) floating-point value.
|
||||||
|
const std::vector<std::pair<std::string, bool>> cases =
|
||||||
|
{
|
||||||
|
// normal small/large integers, both signs
|
||||||
|
{"0", true}, {"1", true}, {"-1", true}, {"42", true}, {"-42", true},
|
||||||
|
{"123456789", true}, {"-123456789", true},
|
||||||
|
|
||||||
|
// digit-count boundary around the 18-digit safe cutoff (both signs)
|
||||||
|
{std::string(17, '9'), true},
|
||||||
|
{std::string(18, '9'), true},
|
||||||
|
{std::string(19, '9'), true},
|
||||||
|
{std::string(20, '9'), true},
|
||||||
|
{"-" + std::string(17, '9'), true},
|
||||||
|
{"-" + std::string(18, '9'), true},
|
||||||
|
{"-" + std::string(19, '9'), true},
|
||||||
|
{"-" + std::string(20, '9'), true},
|
||||||
|
|
||||||
|
// 64-bit boundaries
|
||||||
|
{"9223372036854775807", true}, // INT64_MAX
|
||||||
|
{"-9223372036854775808", true}, // INT64_MIN
|
||||||
|
{"18446744073709551615", true}, // UINT64_MAX
|
||||||
|
{"18446744073709551616", true}, // UINT64_MAX + 1 (overflows uint64_t, finite double)
|
||||||
|
|
||||||
|
// the 28-digit example from the issue: overflows uint64_t
|
||||||
|
// but is finite as a double, so the scanner reclassifies
|
||||||
|
// it to value_float and it is accepted
|
||||||
|
{"9999999999999999999999999999", true},
|
||||||
|
|
||||||
|
// huge digit-only integers that overflow even a double -> rejected
|
||||||
|
{std::string(309, '9'), false},
|
||||||
|
{std::string(400, '9'), false},
|
||||||
|
{"1" + std::string(400, '0'), false},
|
||||||
|
|
||||||
|
// 1e999 / 1e400 style overflow -> rejected
|
||||||
|
{"1e999", false},
|
||||||
|
{"1e400", false},
|
||||||
|
{"-1e999", false},
|
||||||
|
{"1E999", false},
|
||||||
|
|
||||||
|
// values straddling DBL_MAX
|
||||||
|
{"1.7976931348623157e308", true}, // <= DBL_MAX, finite
|
||||||
|
{"1.7976931348623159e308", false}, // > DBL_MAX, overflows to inf
|
||||||
|
|
||||||
|
// a mix of other valid/invalid numeric syntax
|
||||||
|
{"3.14159", true},
|
||||||
|
{"-0.0", true},
|
||||||
|
{"1.0e10", true},
|
||||||
|
{"01", false},
|
||||||
|
{"-", false},
|
||||||
|
{"1.", false},
|
||||||
|
{"1e", false},
|
||||||
|
{"+1", false},
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const auto& c : cases)
|
||||||
|
{
|
||||||
|
const std::string& number = c.first;
|
||||||
|
const bool expected = c.second;
|
||||||
|
CAPTURE(number)
|
||||||
|
CAPTURE(expected)
|
||||||
|
|
||||||
|
// accept() takes the fast path (skips conversion when possible)
|
||||||
|
CHECK(json::accept(number) == expected);
|
||||||
|
|
||||||
|
// parse() always performs the full conversion; it must agree
|
||||||
|
json j;
|
||||||
|
CHECK_NOTHROW(json::parser(nlohmann::detail::input_adapter(number), nullptr, false).parse(true, j));
|
||||||
|
CHECK(!j.is_discarded() == expected);
|
||||||
|
|
||||||
|
// wrap in an array so get_token() is exercised beyond the
|
||||||
|
// very first (constructor-time) scan as well
|
||||||
|
std::string wrapped = "[";
|
||||||
|
wrapped += number;
|
||||||
|
wrapped += ",";
|
||||||
|
wrapped += number;
|
||||||
|
wrapped += "]";
|
||||||
|
CHECK(json::accept(wrapped) == expected);
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1394,6 +1486,69 @@ TEST_CASE("parser class")
|
|||||||
CHECK(accept_helper("\"\\uD80C\\uFFFF\"") == false);
|
CHECK(accept_helper("\"\\uD80C\\uFFFF\"") == false);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SECTION("issue #5412 - whitespace skipping bookkeeping (compact vs. pretty-printed)")
|
||||||
|
{
|
||||||
|
// lexer::skip_whitespace() reads its first character with get() (to
|
||||||
|
// honor a possibly pending unget() from the previous token) and every
|
||||||
|
// further whitespace character with get_ignoring_pending_unget() (a
|
||||||
|
// get() variant that skips the then-always-false next_unget check).
|
||||||
|
// This must not change the reported byte offset, line, or column of
|
||||||
|
// a syntax error, even when a long run of whitespace containing
|
||||||
|
// multiple newlines is skipped beforehand (as with pretty-printed
|
||||||
|
// input). The expected values below were captured from the
|
||||||
|
// unmodified do-while(get()) loop, so any regression that miscounts
|
||||||
|
// characters or newlines while skipping whitespace changes them.
|
||||||
|
const auto check_error = [](const std::string & input, std::size_t expected_byte,
|
||||||
|
const std::string & expected_what)
|
||||||
|
{
|
||||||
|
CAPTURE(input)
|
||||||
|
try
|
||||||
|
{
|
||||||
|
json _ = json::parse(input);
|
||||||
|
FAIL_CHECK("expected a parse_error, but parsing succeeded");
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
CHECK(e.byte == expected_byte);
|
||||||
|
CHECK(std::string(e.what()) == expected_what);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// a nested document, serialized both compactly and pretty-printed
|
||||||
|
// (dump(4)), each truncated right before the final closing '}' so
|
||||||
|
// that the parser hits EOF after skipping all of the (in the
|
||||||
|
// pretty-printed case, substantial) indentation whitespace
|
||||||
|
const json doc =
|
||||||
|
{
|
||||||
|
{"a", 1},
|
||||||
|
{"b", json::array({true, false, nullptr, "x"})},
|
||||||
|
{"c", json::object({{"d", 3.14}, {"e", json::array({1, 2, 3})}})}
|
||||||
|
};
|
||||||
|
|
||||||
|
const std::string compact = doc.dump();
|
||||||
|
const std::string pretty = doc.dump(4);
|
||||||
|
|
||||||
|
check_error(compact.substr(0, compact.size() - 1), 60,
|
||||||
|
"[json.exception.parse_error.101] parse error at line 1, column 60: syntax error while parsing object - unexpected end of input; expected '}'");
|
||||||
|
check_error(pretty.substr(0, pretty.size() - 1), 193,
|
||||||
|
"[json.exception.parse_error.101] parse error at line 17, column 1: syntax error while parsing object - unexpected end of input; expected '}'");
|
||||||
|
|
||||||
|
// an invalid token appearing after several indented, multi-line
|
||||||
|
// whitespace runs vs. the same document without any of that
|
||||||
|
// whitespace
|
||||||
|
check_error(R"({
|
||||||
|
"a": 1,
|
||||||
|
"b": [
|
||||||
|
true,
|
||||||
|
false
|
||||||
|
],
|
||||||
|
"c": @
|
||||||
|
})", 70,
|
||||||
|
"[json.exception.parse_error.101] parse error at line 7, column 10: syntax error while parsing value - invalid literal; last read: '\"c\": @'");
|
||||||
|
check_error("{\"a\":1,\"b\":[true,false],\"c\":@}", 29,
|
||||||
|
"[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing value - invalid literal; last read: '\"c\":@'");
|
||||||
|
}
|
||||||
|
|
||||||
SECTION("tests found by mutate++")
|
SECTION("tests found by mutate++")
|
||||||
{
|
{
|
||||||
// test case to make sure no comma precedes the first key
|
// test case to make sure no comma precedes the first key
|
||||||
|
|||||||
@@ -1,489 +0,0 @@
|
|||||||
// __ _____ _____ _____
|
|
||||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
|
||||||
// | | |__ | | | | | | version 3.12.0
|
|
||||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
|
||||||
//
|
|
||||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
|
||||||
// SPDX-FileCopyrightText: 2018 Vitaliy Manushkin <agri@akamo.info>
|
|
||||||
// SPDX-License-Identifier: MIT
|
|
||||||
|
|
||||||
// This file closes a test-coverage gap described in GitHub issue #5421:
|
|
||||||
// nlohmann::ordered_json (and other non-default basic_json specializations,
|
|
||||||
// such as the alt_string-based one from unit-alt-string.cpp) were never
|
|
||||||
// exercised through the binary formats (CBOR/MessagePack/UBJSON/BSON/BJData)
|
|
||||||
// or through flatten()/unflatten()/diff()/patch()/merge_patch().
|
|
||||||
|
|
||||||
#include "doctest_compatibility.h"
|
|
||||||
|
|
||||||
#include <nlohmann/json.hpp>
|
|
||||||
|
|
||||||
#include <cstdint>
|
|
||||||
#include <string>
|
|
||||||
#include <utility>
|
|
||||||
#include <vector>
|
|
||||||
|
|
||||||
using nlohmann::json;
|
|
||||||
using nlohmann::ordered_json;
|
|
||||||
|
|
||||||
/////////////////////////////////////////////////////////////////////////////
|
|
||||||
// alt_json: a second, independent copy of the custom-string_t basic_json
|
|
||||||
// specialization defined in unit-alt-string.cpp.
|
|
||||||
//
|
|
||||||
// It is duplicated here (rather than shared via a header) because every
|
|
||||||
// unit-*.cpp file in this test suite is compiled into its own standalone
|
|
||||||
// executable (see tests/CMakeLists.txt), so there is no ODR concern in
|
|
||||||
// having the same class name defined in multiple translation units.
|
|
||||||
//
|
|
||||||
// Two members had to be added relative to the original alt_string
|
|
||||||
// (a constructor from std::string, and a find(char, pos) overload) because
|
|
||||||
// the original type was never used with the binary writers/readers before
|
|
||||||
// this file: BSON's array/document writer converts std::to_string() results
|
|
||||||
// and checks for embedded NUL characters via find(char), and the UBJSON/BSON
|
|
||||||
// high-precision-number path constructs the SAX string_t argument from a
|
|
||||||
// std::string. Neither path is exercised anywhere else in the test suite for
|
|
||||||
// this type, which is presumably why the gap was never noticed.
|
|
||||||
/////////////////////////////////////////////////////////////////////////////
|
|
||||||
|
|
||||||
class alt_string;
|
|
||||||
bool operator<(const char* op1, const alt_string& op2) noexcept; // NOLINT(misc-use-internal-linkage)
|
|
||||||
void int_to_string(alt_string& target, std::size_t value); // NOLINT(misc-use-internal-linkage)
|
|
||||||
|
|
||||||
class alt_string
|
|
||||||
{
|
|
||||||
public:
|
|
||||||
using value_type = std::string::value_type;
|
|
||||||
|
|
||||||
static constexpr auto npos = (std::numeric_limits<std::size_t>::max)();
|
|
||||||
|
|
||||||
alt_string(const char* str): str_impl(str) {}
|
|
||||||
alt_string(const char* str, std::size_t count): str_impl(str, count) {}
|
|
||||||
alt_string(std::string str): str_impl(std::move(str)) {}
|
|
||||||
alt_string(size_t count, char chr): str_impl(count, chr) {}
|
|
||||||
alt_string() = default;
|
|
||||||
|
|
||||||
alt_string& append(char ch)
|
|
||||||
{
|
|
||||||
str_impl.push_back(ch);
|
|
||||||
return *this;
|
|
||||||
}
|
|
||||||
|
|
||||||
alt_string& append(const alt_string& str)
|
|
||||||
{
|
|
||||||
str_impl.append(str.str_impl);
|
|
||||||
return *this;
|
|
||||||
}
|
|
||||||
|
|
||||||
alt_string& append(const char* s, std::size_t length)
|
|
||||||
{
|
|
||||||
str_impl.append(s, length);
|
|
||||||
return *this;
|
|
||||||
}
|
|
||||||
|
|
||||||
void push_back(char c)
|
|
||||||
{
|
|
||||||
str_impl.push_back(c);
|
|
||||||
}
|
|
||||||
|
|
||||||
template <typename op_type>
|
|
||||||
bool operator==(const op_type& op) const
|
|
||||||
{
|
|
||||||
return str_impl == op;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool operator==(const alt_string& op) const
|
|
||||||
{
|
|
||||||
return str_impl == op.str_impl;
|
|
||||||
}
|
|
||||||
|
|
||||||
template <typename op_type>
|
|
||||||
bool operator!=(const op_type& op) const
|
|
||||||
{
|
|
||||||
return str_impl != op;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool operator!=(const alt_string& op) const
|
|
||||||
{
|
|
||||||
return str_impl != op.str_impl;
|
|
||||||
}
|
|
||||||
|
|
||||||
std::size_t size() const noexcept
|
|
||||||
{
|
|
||||||
return str_impl.size();
|
|
||||||
}
|
|
||||||
|
|
||||||
void resize(std::size_t n)
|
|
||||||
{
|
|
||||||
str_impl.resize(n);
|
|
||||||
}
|
|
||||||
|
|
||||||
void resize(std::size_t n, char c)
|
|
||||||
{
|
|
||||||
str_impl.resize(n, c);
|
|
||||||
}
|
|
||||||
|
|
||||||
template <typename op_type>
|
|
||||||
bool operator<(const op_type& op) const noexcept
|
|
||||||
{
|
|
||||||
return str_impl < op;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool operator<(const alt_string& op) const noexcept
|
|
||||||
{
|
|
||||||
return str_impl < op.str_impl;
|
|
||||||
}
|
|
||||||
|
|
||||||
const char* c_str() const
|
|
||||||
{
|
|
||||||
return str_impl.c_str();
|
|
||||||
}
|
|
||||||
|
|
||||||
char& operator[](std::size_t index)
|
|
||||||
{
|
|
||||||
return str_impl[index];
|
|
||||||
}
|
|
||||||
|
|
||||||
const char& operator[](std::size_t index) const
|
|
||||||
{
|
|
||||||
return str_impl[index];
|
|
||||||
}
|
|
||||||
|
|
||||||
char& back()
|
|
||||||
{
|
|
||||||
return str_impl.back();
|
|
||||||
}
|
|
||||||
|
|
||||||
const char& back() const
|
|
||||||
{
|
|
||||||
return str_impl.back();
|
|
||||||
}
|
|
||||||
|
|
||||||
void clear()
|
|
||||||
{
|
|
||||||
str_impl.clear();
|
|
||||||
}
|
|
||||||
|
|
||||||
const value_type* data() const
|
|
||||||
{
|
|
||||||
return str_impl.data();
|
|
||||||
}
|
|
||||||
|
|
||||||
bool empty() const
|
|
||||||
{
|
|
||||||
return str_impl.empty();
|
|
||||||
}
|
|
||||||
|
|
||||||
std::size_t find(const alt_string& str, std::size_t pos = 0) const
|
|
||||||
{
|
|
||||||
return str_impl.find(str.str_impl, pos);
|
|
||||||
}
|
|
||||||
|
|
||||||
// needed by binary_writer's BSON support, which probes string keys for
|
|
||||||
// embedded NUL characters via find(char)
|
|
||||||
std::size_t find(char c, std::size_t pos = 0) const
|
|
||||||
{
|
|
||||||
return str_impl.find(c, pos);
|
|
||||||
}
|
|
||||||
|
|
||||||
std::size_t find_first_of(char c, std::size_t pos = 0) const
|
|
||||||
{
|
|
||||||
return str_impl.find_first_of(c, pos);
|
|
||||||
}
|
|
||||||
|
|
||||||
alt_string substr(std::size_t pos = 0, std::size_t count = npos) const
|
|
||||||
{
|
|
||||||
const std::string s = str_impl.substr(pos, count);
|
|
||||||
return {s.data(), s.size()};
|
|
||||||
}
|
|
||||||
|
|
||||||
alt_string& replace(std::size_t pos, std::size_t count, const alt_string& str)
|
|
||||||
{
|
|
||||||
str_impl.replace(pos, count, str.str_impl);
|
|
||||||
return *this;
|
|
||||||
}
|
|
||||||
|
|
||||||
void reserve(std::size_t new_cap = 0)
|
|
||||||
{
|
|
||||||
str_impl.reserve(new_cap);
|
|
||||||
}
|
|
||||||
|
|
||||||
private:
|
|
||||||
std::string str_impl {}; // NOLINT(readability-redundant-member-init)
|
|
||||||
|
|
||||||
friend bool operator<(const char* /*op1*/, const alt_string& /*op2*/) noexcept;
|
|
||||||
};
|
|
||||||
|
|
||||||
void int_to_string(alt_string& target, std::size_t value)
|
|
||||||
{
|
|
||||||
target = std::to_string(value).c_str();
|
|
||||||
}
|
|
||||||
|
|
||||||
using alt_json = nlohmann::basic_json <
|
|
||||||
std::map,
|
|
||||||
std::vector,
|
|
||||||
alt_string,
|
|
||||||
bool,
|
|
||||||
std::int64_t,
|
|
||||||
std::uint64_t,
|
|
||||||
double,
|
|
||||||
std::allocator,
|
|
||||||
nlohmann::adl_serializer >;
|
|
||||||
|
|
||||||
bool operator<(const char* op1, const alt_string& op2) noexcept
|
|
||||||
{
|
|
||||||
return op1 < op2.str_impl;
|
|
||||||
}
|
|
||||||
|
|
||||||
namespace
|
|
||||||
{
|
|
||||||
|
|
||||||
// collects the object keys of j, in iteration order
|
|
||||||
std::vector<std::string> collect_keys(const ordered_json& j)
|
|
||||||
{
|
|
||||||
std::vector<std::string> result;
|
|
||||||
for (auto it = j.cbegin(); it != j.cend(); ++it)
|
|
||||||
{
|
|
||||||
result.push_back(it.key());
|
|
||||||
}
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
|
|
||||||
// a nested object/array value with keys inserted in non-alphabetical order,
|
|
||||||
// used to check both round-trip equality and (for ordered_json) that
|
|
||||||
// insertion order survives a trip through a binary format
|
|
||||||
ordered_json make_rich_ordered_json()
|
|
||||||
{
|
|
||||||
ordered_json j;
|
|
||||||
j["zebra"] = 1;
|
|
||||||
j["apple"] = ordered_json::array({1, 2, 3});
|
|
||||||
j["mango"]["z_nested"] = true;
|
|
||||||
j["mango"]["a_nested"] = nullptr;
|
|
||||||
j["banana"] = "some text";
|
|
||||||
j["cherry"] = 3.14;
|
|
||||||
return j;
|
|
||||||
}
|
|
||||||
|
|
||||||
alt_json make_rich_alt_json()
|
|
||||||
{
|
|
||||||
alt_json j;
|
|
||||||
j["zebra"] = 1;
|
|
||||||
j["apple"] = alt_json::array({1, 2, 3});
|
|
||||||
j["mango"]["z_nested"] = true;
|
|
||||||
j["mango"]["a_nested"] = nullptr;
|
|
||||||
j["banana"] = "some text";
|
|
||||||
j["cherry"] = 3.14;
|
|
||||||
return j;
|
|
||||||
}
|
|
||||||
|
|
||||||
} // namespace
|
|
||||||
|
|
||||||
TEST_CASE("ordered_json across binary formats")
|
|
||||||
{
|
|
||||||
const ordered_json original = make_rich_ordered_json();
|
|
||||||
const std::vector<std::string> original_keys = collect_keys(original);
|
|
||||||
const std::vector<std::string> original_mango_keys = collect_keys(original["mango"]);
|
|
||||||
|
|
||||||
SECTION("CBOR")
|
|
||||||
{
|
|
||||||
const auto bytes = ordered_json::to_cbor(original);
|
|
||||||
const auto restored = ordered_json::from_cbor(bytes);
|
|
||||||
CHECK(restored == original);
|
|
||||||
CHECK(collect_keys(restored) == original_keys);
|
|
||||||
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("MessagePack")
|
|
||||||
{
|
|
||||||
const auto bytes = ordered_json::to_msgpack(original);
|
|
||||||
const auto restored = ordered_json::from_msgpack(bytes);
|
|
||||||
CHECK(restored == original);
|
|
||||||
CHECK(collect_keys(restored) == original_keys);
|
|
||||||
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("UBJSON")
|
|
||||||
{
|
|
||||||
const auto bytes = ordered_json::to_ubjson(original);
|
|
||||||
const auto restored = ordered_json::from_ubjson(bytes);
|
|
||||||
CHECK(restored == original);
|
|
||||||
CHECK(collect_keys(restored) == original_keys);
|
|
||||||
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("BSON")
|
|
||||||
{
|
|
||||||
const auto bytes = ordered_json::to_bson(original);
|
|
||||||
const auto restored = ordered_json::from_bson(bytes);
|
|
||||||
CHECK(restored == original);
|
|
||||||
CHECK(collect_keys(restored) == original_keys);
|
|
||||||
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("BJData")
|
|
||||||
{
|
|
||||||
const auto bytes = ordered_json::to_bjdata(original);
|
|
||||||
const auto restored = ordered_json::from_bjdata(bytes);
|
|
||||||
CHECK(restored == original);
|
|
||||||
CHECK(collect_keys(restored) == original_keys);
|
|
||||||
CHECK(collect_keys(restored["mango"]) == original_mango_keys);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST_CASE("alt_json (custom string_t) across binary formats")
|
|
||||||
{
|
|
||||||
const alt_json original = make_rich_alt_json();
|
|
||||||
|
|
||||||
SECTION("CBOR")
|
|
||||||
{
|
|
||||||
const auto bytes = alt_json::to_cbor(original);
|
|
||||||
const auto restored = alt_json::from_cbor(bytes);
|
|
||||||
CHECK(restored == original);
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("MessagePack")
|
|
||||||
{
|
|
||||||
const auto bytes = alt_json::to_msgpack(original);
|
|
||||||
const auto restored = alt_json::from_msgpack(bytes);
|
|
||||||
CHECK(restored == original);
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("UBJSON")
|
|
||||||
{
|
|
||||||
const auto bytes = alt_json::to_ubjson(original);
|
|
||||||
const auto restored = alt_json::from_ubjson(bytes);
|
|
||||||
CHECK(restored == original);
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("BSON")
|
|
||||||
{
|
|
||||||
const auto bytes = alt_json::to_bson(original);
|
|
||||||
const auto restored = alt_json::from_bson(bytes);
|
|
||||||
CHECK(restored == original);
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("BJData")
|
|
||||||
{
|
|
||||||
const auto bytes = alt_json::to_bjdata(original);
|
|
||||||
const auto restored = alt_json::from_bjdata(bytes);
|
|
||||||
CHECK(restored == original);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST_CASE("ordered_json operator== is sensitive to key order")
|
|
||||||
{
|
|
||||||
// Unlike nlohmann::json (whose object_t is a std::map, so equality never
|
|
||||||
// depends on insertion order), ordered_json's object_t (ordered_map) is a
|
|
||||||
// std::vector<std::pair<Key, T>> under the hood, and does not define its
|
|
||||||
// own operator==: it inherits std::vector's element-wise comparison. As a
|
|
||||||
// result, two ordered_json objects holding the very same key/value pairs
|
|
||||||
// in different insertion order compare *unequal*. This is the property
|
|
||||||
// that makes the round-trip `CHECK(restored == original)` checks above a
|
|
||||||
// meaningful order-preservation check by themselves (the explicit
|
|
||||||
// collect_keys() comparisons make that check explicit/readable, and
|
|
||||||
// guard against this operator== behavior ever changing).
|
|
||||||
ordered_json a;
|
|
||||||
a["x"] = 1;
|
|
||||||
a["y"] = 2;
|
|
||||||
|
|
||||||
ordered_json b;
|
|
||||||
b["y"] = 2;
|
|
||||||
b["x"] = 1;
|
|
||||||
|
|
||||||
CHECK(a.size() == b.size());
|
|
||||||
CHECK(a["x"] == b["x"]);
|
|
||||||
CHECK(a["y"] == b["y"]);
|
|
||||||
CHECK_FALSE(a == b);
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST_CASE("duplicate keys in a binary-encoded object")
|
|
||||||
{
|
|
||||||
// CBOR encoding of a map with two entries under the same key "a": {"a": 1, "a": 2}
|
|
||||||
const std::vector<std::uint8_t> cbor_bytes
|
|
||||||
{
|
|
||||||
0xA2, 0x61, 'a', 0x01, 0x61, 'a', 0x02
|
|
||||||
};
|
|
||||||
|
|
||||||
// Both json (std::map, via operator[]) and ordered_json (ordered_map, via
|
|
||||||
// operator[]) build binary-decoded objects by looking up/creating the
|
|
||||||
// entry for each incoming key and then assigning the value into it. This
|
|
||||||
// means a repeated key does *not* produce two entries in either case;
|
|
||||||
// instead, the *first* occurrence's position is kept (relevant only for
|
|
||||||
// ordered_json) while the *last* occurrence's value wins (for both) --
|
|
||||||
// this matches operator[]'s "assign the referenced slot" semantics, and
|
|
||||||
// is worth noting because it differs from the initializer-list
|
|
||||||
// construction path (`ordered_json{{"a",1},{"a",2}}`), which builds
|
|
||||||
// through insert()/emplace() and therefore keeps the *first* value, not
|
|
||||||
// the last (see the "There are no dup keys..." case in
|
|
||||||
// unit-ordered_json.cpp).
|
|
||||||
const auto j = json::from_cbor(cbor_bytes);
|
|
||||||
const auto oj = ordered_json::from_cbor(cbor_bytes);
|
|
||||||
|
|
||||||
CHECK(j.size() == 1);
|
|
||||||
CHECK(oj.size() == 1);
|
|
||||||
CHECK(j["a"] == 2);
|
|
||||||
CHECK(oj["a"] == 2);
|
|
||||||
CHECK(j == json(oj));
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST_CASE("ordered_json through flatten/unflatten")
|
|
||||||
{
|
|
||||||
const ordered_json original = make_rich_ordered_json();
|
|
||||||
const std::vector<std::string> original_keys = collect_keys(original);
|
|
||||||
const std::vector<std::string> original_mango_keys = collect_keys(original["mango"]);
|
|
||||||
|
|
||||||
const ordered_json flat = original.flatten();
|
|
||||||
const ordered_json unflattened = flat.unflatten();
|
|
||||||
|
|
||||||
CHECK(unflattened == original);
|
|
||||||
// flatten() walks the value depth-first in iteration order and
|
|
||||||
// unflatten() re-inserts each flattened key via operator[] in the flat
|
|
||||||
// object's iteration order, so for ordered_json the original key order
|
|
||||||
// (both top-level and nested) is preserved end-to-end.
|
|
||||||
CHECK(collect_keys(unflattened) == original_keys);
|
|
||||||
CHECK(collect_keys(unflattened["mango"]) == original_mango_keys);
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST_CASE("ordered_json through diff/patch/patch_inplace")
|
|
||||||
{
|
|
||||||
ordered_json original;
|
|
||||||
original["one"] = 1;
|
|
||||||
original["two"] = 2;
|
|
||||||
original["three"] = 3;
|
|
||||||
|
|
||||||
ordered_json target = original;
|
|
||||||
target["one"] = 100; // replace
|
|
||||||
target.erase("two"); // remove
|
|
||||||
target["four"] = 4; // add
|
|
||||||
|
|
||||||
const ordered_json patch = ordered_json::diff(original, target);
|
|
||||||
|
|
||||||
SECTION("patch")
|
|
||||||
{
|
|
||||||
const ordered_json patched = original.patch(patch);
|
|
||||||
CHECK(patched == target);
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("patch_inplace")
|
|
||||||
{
|
|
||||||
ordered_json copy = original;
|
|
||||||
copy.patch_inplace(patch);
|
|
||||||
CHECK(copy == target);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
TEST_CASE("ordered_json through merge_patch")
|
|
||||||
{
|
|
||||||
ordered_json original;
|
|
||||||
original["a"] = 1;
|
|
||||||
original["b"] = 2;
|
|
||||||
|
|
||||||
const ordered_json patch = {{"b", nullptr}, {"c", 3}};
|
|
||||||
|
|
||||||
original.merge_patch(patch);
|
|
||||||
|
|
||||||
ordered_json expected;
|
|
||||||
expected["a"] = 1;
|
|
||||||
expected["c"] = 3;
|
|
||||||
|
|
||||||
CHECK(original == expected);
|
|
||||||
CHECK(collect_keys(original) == collect_keys(expected));
|
|
||||||
}
|
|
||||||
@@ -17,7 +17,6 @@
|
|||||||
|
|
||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
using json = nlohmann::json;
|
using json = nlohmann::json;
|
||||||
using ordered_json = nlohmann::ordered_json;
|
|
||||||
|
|
||||||
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
|
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
|
||||||
#if JSON_HAS_STD_FORMAT
|
#if JSON_HAS_STD_FORMAT
|
||||||
@@ -94,16 +93,4 @@ TEST_CASE("std::formatter<nlohmann::json>")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST_CASE("std::formatter<nlohmann::ordered_json>")
|
|
||||||
{
|
|
||||||
// spot-check a non-default basic_json instantiation, since the formatter
|
|
||||||
// is written against the generic NLOHMANN_BASIC_JSON_TPL_DECLARATION
|
|
||||||
// template and must actually instantiate (and behave correctly) for
|
|
||||||
// template arguments other than nlohmann::json
|
|
||||||
const ordered_json j = {{"foo", 1}, {"bar", {1, 2, 3}}};
|
|
||||||
CHECK(std::format("{}", j) == j.dump());
|
|
||||||
CHECK(std::format("{:#}", j) == j.dump(4));
|
|
||||||
CHECK(std::format("{:2}", j) == j.dump(2));
|
|
||||||
}
|
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
Reference in New Issue
Block a user