mirror of
https://github.com/nlohmann/json.git
synced 2026-09-06 00:08:00 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
784c3ad13c | ||
|
|
6fb73ee33e |
@@ -149,11 +149,10 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
public:
|
||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||
|
||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept
|
||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept
|
||||
: ia(std::move(adapter))
|
||||
, ignore_comments(ignore_comments_)
|
||||
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
||||
, discard_number_values(discard_number_values_)
|
||||
{}
|
||||
|
||||
// deleted because of pointer members
|
||||
@@ -1280,58 +1279,6 @@ scan_number_done:
|
||||
// we are done scanning a number)
|
||||
unget();
|
||||
|
||||
// If the caller does not need the converted value (only whether the
|
||||
// input is syntactically valid; see json_sax_acceptor/accept()), an
|
||||
// unsigned/integer token can be reported without calling
|
||||
// strtoull()/strtoll() at all, *provided* we can already tell from
|
||||
// the digit count alone that the conversion cannot overflow 64 bits.
|
||||
// Such tokens are always finite and are accepted unconditionally by
|
||||
// the parser regardless of their actual value (parser::sax_parse_internal()
|
||||
// never checks finiteness for value_unsigned/value_integer), so the
|
||||
// classification below is all that is needed.
|
||||
//
|
||||
// A decimal number with up to 18 digits is always representable in
|
||||
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
|
||||
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
|
||||
// could not have set errno to ERANGE for it. Numbers with more digits
|
||||
// (rare in practice) fall through to the exact code below, unchanged,
|
||||
// so their handling -- including reclassification to value_float when
|
||||
// the value overflows 64 bits, and rejection when it is not even
|
||||
// finite as a double -- is bit-for-bit identical to before this
|
||||
// optimization.
|
||||
//
|
||||
// Note this reasons about std::uint64_t/std::int64_t, not about
|
||||
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
|
||||
// narrower, template parameters -- e.g. std::uint32_t). That is fine
|
||||
// *only* because discard_number_values is exclusively set by
|
||||
// accept() (see json.hpp), and accept() always parses through the
|
||||
// library's own json_sax_acceptor -- never a user-supplied SAX
|
||||
// consumer -- whose number_unsigned()/number_integer()/number_float()
|
||||
// callbacks unconditionally discard their argument and return true.
|
||||
// So for every caller that can reach this branch, neither the token
|
||||
// classification below nor the eventual (possibly narrowed, and on
|
||||
// this fast path left stale/unset) value_unsigned/value_integer is
|
||||
// ever consulted -- an unsigned/integer token is accepted outright,
|
||||
// and even a >18-digit token that this fast path deliberately falls
|
||||
// through for is, once reclassified to value_float, still finite
|
||||
// (and thus accepted) for any digit count that fits in number_unsigned_t
|
||||
// or number_integer_t regardless of that type's width. If this
|
||||
// function is ever taught to run with discard_number_values true for
|
||||
// a caller that *does* read the converted value, this reasoning (and
|
||||
// the fast path below) would need to be revisited.
|
||||
if (discard_number_values)
|
||||
{
|
||||
constexpr std::size_t safe_digit_count = 18;
|
||||
if (number_type == token_type::value_unsigned && token_buffer.size() <= safe_digit_count)
|
||||
{
|
||||
return token_type::value_unsigned;
|
||||
}
|
||||
if (number_type == token_type::value_integer && token_buffer.size() - 1 <= safe_digit_count)
|
||||
{
|
||||
return token_type::value_integer;
|
||||
}
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
errno = 0;
|
||||
|
||||
@@ -1446,7 +1393,8 @@ scan_number_done:
|
||||
*/
|
||||
char_int_type get()
|
||||
{
|
||||
advance_position();
|
||||
++position.chars_read_total;
|
||||
++position.chars_read_current_line;
|
||||
|
||||
if (next_unget)
|
||||
{
|
||||
@@ -1458,23 +1406,6 @@ scan_number_done:
|
||||
current = ia.get_character();
|
||||
}
|
||||
|
||||
return track_after_read();
|
||||
}
|
||||
|
||||
/// shared head of get() / get_ignoring_pending_unget(): bump the
|
||||
/// per-character position counters (line-count-on-'\n' bookkeeping is
|
||||
/// handled afterwards, in track_after_read(), once `current` is known)
|
||||
void advance_position() noexcept
|
||||
{
|
||||
++position.chars_read_total;
|
||||
++position.chars_read_current_line;
|
||||
}
|
||||
|
||||
/// shared tail of get() / get_ignoring_pending_unget(): capture the
|
||||
/// character for error messages (if needed) and update line/column
|
||||
/// bookkeeping for the character now in `current`
|
||||
char_int_type track_after_read()
|
||||
{
|
||||
// seekable adapters reconstruct the token lazily on error (see
|
||||
// get_token_string), so the eager per-character copy is skipped
|
||||
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
||||
@@ -1488,29 +1419,6 @@ scan_number_done:
|
||||
return current;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief like get(), but for call sites that can prove no unget() is pending
|
||||
|
||||
get() has to check the `next_unget` flag on every call, because a
|
||||
previous token may have ended with unget() (e.g. scan_number() always
|
||||
ungets the character that terminated the number, so the next call to
|
||||
scan() can see it again). skip_whitespace() reads that first,
|
||||
possibly-ungotten character via a plain get(), but every further
|
||||
character it reads is guaranteed to be a fresh read: nothing between
|
||||
those calls invokes unget(). This variant skips the (otherwise always
|
||||
false) next_unget branch for those calls; it is not a general
|
||||
replacement for get().
|
||||
*/
|
||||
char_int_type get_ignoring_pending_unget()
|
||||
{
|
||||
JSON_ASSERT(!next_unget);
|
||||
|
||||
advance_position();
|
||||
current = ia.get_character();
|
||||
|
||||
return track_after_read();
|
||||
}
|
||||
|
||||
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
||||
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
||||
|
||||
@@ -1704,37 +1612,13 @@ scan_number_done:
|
||||
return true;
|
||||
}
|
||||
|
||||
/// whether `current` is one of the four JSON whitespace characters
|
||||
bool current_is_whitespace() const noexcept
|
||||
{
|
||||
return current == ' ' || current == '\t' || current == '\n' || current == '\r';
|
||||
}
|
||||
|
||||
void skip_whitespace()
|
||||
{
|
||||
// the first character may be a pending unget() left over from the
|
||||
// previous token (see get_ignoring_pending_unget()); every
|
||||
// subsequent character read by this loop is guaranteed fresh, since
|
||||
// nothing below calls unget()
|
||||
get();
|
||||
|
||||
if (!current_is_whitespace())
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// this is written as an if-guarded do-while (rather than a plain
|
||||
// while loop) because that shape is what lets both GCC and Clang
|
||||
// keep the input adapter's read pointer in a register across
|
||||
// iterations; the equivalent while-loop measurably defeated that
|
||||
// optimization in testing, turning long whitespace runs (e.g. the
|
||||
// indentation of pretty-printed JSON) from a register-only loop
|
||||
// into one that reloads the pointer from memory every character
|
||||
do
|
||||
{
|
||||
get_ignoring_pending_unget();
|
||||
get();
|
||||
}
|
||||
while (current_is_whitespace());
|
||||
while (current == ' ' || current == '\t' || current == '\n' || current == '\r');
|
||||
}
|
||||
|
||||
token_type scan()
|
||||
@@ -1870,13 +1754,6 @@ scan_number_done:
|
||||
const char_int_type decimal_point_char = '.';
|
||||
/// the position of the decimal point in the input
|
||||
std::size_t decimal_point_position = std::string::npos;
|
||||
|
||||
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
|
||||
/// token classification and never looks at the converted numeric value;
|
||||
/// when set, scan_number() may skip strtoull()/strtoll() for
|
||||
/// value_unsigned/value_integer tokens whose digit count guarantees they
|
||||
/// fit into 64 bits (see scan_number())
|
||||
const bool discard_number_values = false;
|
||||
};
|
||||
|
||||
} // namespace detail
|
||||
|
||||
@@ -72,10 +72,9 @@ class parser
|
||||
parser_callback_t<BasicJsonType> cb = nullptr,
|
||||
const bool allow_exceptions_ = true,
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas_ = false,
|
||||
const bool discard_number_values_ = false)
|
||||
const bool ignore_trailing_commas_ = false)
|
||||
: callback(std::move(cb))
|
||||
, m_lexer(std::move(adapter), ignore_comments, discard_number_values_)
|
||||
, m_lexer(std::move(adapter), ignore_comments)
|
||||
, allow_exceptions(allow_exceptions_)
|
||||
, ignore_trailing_commas(ignore_trailing_commas_)
|
||||
{
|
||||
|
||||
@@ -164,12 +164,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
detail::parser_callback_t<basic_json>cb = nullptr,
|
||||
const bool allow_exceptions = true,
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas = false,
|
||||
const bool discard_number_values = false
|
||||
const bool ignore_trailing_commas = false
|
||||
)
|
||||
{
|
||||
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
||||
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas, discard_number_values);
|
||||
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas);
|
||||
}
|
||||
|
||||
private:
|
||||
@@ -4134,7 +4133,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas = false)
|
||||
{
|
||||
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
||||
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||
}
|
||||
|
||||
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||
@@ -4145,7 +4144,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas = false)
|
||||
{
|
||||
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
||||
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
@@ -4154,7 +4153,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas = false)
|
||||
{
|
||||
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
||||
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||
}
|
||||
|
||||
/// @brief generate SAX events
|
||||
|
||||
@@ -7932,11 +7932,10 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
public:
|
||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||
|
||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept
|
||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept
|
||||
: ia(std::move(adapter))
|
||||
, ignore_comments(ignore_comments_)
|
||||
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
||||
, discard_number_values(discard_number_values_)
|
||||
{}
|
||||
|
||||
// deleted because of pointer members
|
||||
@@ -9063,58 +9062,6 @@ scan_number_done:
|
||||
// we are done scanning a number)
|
||||
unget();
|
||||
|
||||
// If the caller does not need the converted value (only whether the
|
||||
// input is syntactically valid; see json_sax_acceptor/accept()), an
|
||||
// unsigned/integer token can be reported without calling
|
||||
// strtoull()/strtoll() at all, *provided* we can already tell from
|
||||
// the digit count alone that the conversion cannot overflow 64 bits.
|
||||
// Such tokens are always finite and are accepted unconditionally by
|
||||
// the parser regardless of their actual value (parser::sax_parse_internal()
|
||||
// never checks finiteness for value_unsigned/value_integer), so the
|
||||
// classification below is all that is needed.
|
||||
//
|
||||
// A decimal number with up to 18 digits is always representable in
|
||||
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
|
||||
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
|
||||
// could not have set errno to ERANGE for it. Numbers with more digits
|
||||
// (rare in practice) fall through to the exact code below, unchanged,
|
||||
// so their handling -- including reclassification to value_float when
|
||||
// the value overflows 64 bits, and rejection when it is not even
|
||||
// finite as a double -- is bit-for-bit identical to before this
|
||||
// optimization.
|
||||
//
|
||||
// Note this reasons about std::uint64_t/std::int64_t, not about
|
||||
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
|
||||
// narrower, template parameters -- e.g. std::uint32_t). That is fine
|
||||
// *only* because discard_number_values is exclusively set by
|
||||
// accept() (see json.hpp), and accept() always parses through the
|
||||
// library's own json_sax_acceptor -- never a user-supplied SAX
|
||||
// consumer -- whose number_unsigned()/number_integer()/number_float()
|
||||
// callbacks unconditionally discard their argument and return true.
|
||||
// So for every caller that can reach this branch, neither the token
|
||||
// classification below nor the eventual (possibly narrowed, and on
|
||||
// this fast path left stale/unset) value_unsigned/value_integer is
|
||||
// ever consulted -- an unsigned/integer token is accepted outright,
|
||||
// and even a >18-digit token that this fast path deliberately falls
|
||||
// through for is, once reclassified to value_float, still finite
|
||||
// (and thus accepted) for any digit count that fits in number_unsigned_t
|
||||
// or number_integer_t regardless of that type's width. If this
|
||||
// function is ever taught to run with discard_number_values true for
|
||||
// a caller that *does* read the converted value, this reasoning (and
|
||||
// the fast path below) would need to be revisited.
|
||||
if (discard_number_values)
|
||||
{
|
||||
constexpr std::size_t safe_digit_count = 18;
|
||||
if (number_type == token_type::value_unsigned && token_buffer.size() <= safe_digit_count)
|
||||
{
|
||||
return token_type::value_unsigned;
|
||||
}
|
||||
if (number_type == token_type::value_integer && token_buffer.size() - 1 <= safe_digit_count)
|
||||
{
|
||||
return token_type::value_integer;
|
||||
}
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
errno = 0;
|
||||
|
||||
@@ -9229,7 +9176,8 @@ scan_number_done:
|
||||
*/
|
||||
char_int_type get()
|
||||
{
|
||||
advance_position();
|
||||
++position.chars_read_total;
|
||||
++position.chars_read_current_line;
|
||||
|
||||
if (next_unget)
|
||||
{
|
||||
@@ -9241,23 +9189,6 @@ scan_number_done:
|
||||
current = ia.get_character();
|
||||
}
|
||||
|
||||
return track_after_read();
|
||||
}
|
||||
|
||||
/// shared head of get() / get_ignoring_pending_unget(): bump the
|
||||
/// per-character position counters (line-count-on-'\n' bookkeeping is
|
||||
/// handled afterwards, in track_after_read(), once `current` is known)
|
||||
void advance_position() noexcept
|
||||
{
|
||||
++position.chars_read_total;
|
||||
++position.chars_read_current_line;
|
||||
}
|
||||
|
||||
/// shared tail of get() / get_ignoring_pending_unget(): capture the
|
||||
/// character for error messages (if needed) and update line/column
|
||||
/// bookkeeping for the character now in `current`
|
||||
char_int_type track_after_read()
|
||||
{
|
||||
// seekable adapters reconstruct the token lazily on error (see
|
||||
// get_token_string), so the eager per-character copy is skipped
|
||||
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
||||
@@ -9271,29 +9202,6 @@ scan_number_done:
|
||||
return current;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief like get(), but for call sites that can prove no unget() is pending
|
||||
|
||||
get() has to check the `next_unget` flag on every call, because a
|
||||
previous token may have ended with unget() (e.g. scan_number() always
|
||||
ungets the character that terminated the number, so the next call to
|
||||
scan() can see it again). skip_whitespace() reads that first,
|
||||
possibly-ungotten character via a plain get(), but every further
|
||||
character it reads is guaranteed to be a fresh read: nothing between
|
||||
those calls invokes unget(). This variant skips the (otherwise always
|
||||
false) next_unget branch for those calls; it is not a general
|
||||
replacement for get().
|
||||
*/
|
||||
char_int_type get_ignoring_pending_unget()
|
||||
{
|
||||
JSON_ASSERT(!next_unget);
|
||||
|
||||
advance_position();
|
||||
current = ia.get_character();
|
||||
|
||||
return track_after_read();
|
||||
}
|
||||
|
||||
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
||||
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
||||
|
||||
@@ -9487,37 +9395,13 @@ scan_number_done:
|
||||
return true;
|
||||
}
|
||||
|
||||
/// whether `current` is one of the four JSON whitespace characters
|
||||
bool current_is_whitespace() const noexcept
|
||||
{
|
||||
return current == ' ' || current == '\t' || current == '\n' || current == '\r';
|
||||
}
|
||||
|
||||
void skip_whitespace()
|
||||
{
|
||||
// the first character may be a pending unget() left over from the
|
||||
// previous token (see get_ignoring_pending_unget()); every
|
||||
// subsequent character read by this loop is guaranteed fresh, since
|
||||
// nothing below calls unget()
|
||||
get();
|
||||
|
||||
if (!current_is_whitespace())
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// this is written as an if-guarded do-while (rather than a plain
|
||||
// while loop) because that shape is what lets both GCC and Clang
|
||||
// keep the input adapter's read pointer in a register across
|
||||
// iterations; the equivalent while-loop measurably defeated that
|
||||
// optimization in testing, turning long whitespace runs (e.g. the
|
||||
// indentation of pretty-printed JSON) from a register-only loop
|
||||
// into one that reloads the pointer from memory every character
|
||||
do
|
||||
{
|
||||
get_ignoring_pending_unget();
|
||||
get();
|
||||
}
|
||||
while (current_is_whitespace());
|
||||
while (current == ' ' || current == '\t' || current == '\n' || current == '\r');
|
||||
}
|
||||
|
||||
token_type scan()
|
||||
@@ -9653,13 +9537,6 @@ scan_number_done:
|
||||
const char_int_type decimal_point_char = '.';
|
||||
/// the position of the decimal point in the input
|
||||
std::size_t decimal_point_position = std::string::npos;
|
||||
|
||||
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
|
||||
/// token classification and never looks at the converted numeric value;
|
||||
/// when set, scan_number() may skip strtoull()/strtoll() for
|
||||
/// value_unsigned/value_integer tokens whose digit count guarantees they
|
||||
/// fit into 64 bits (see scan_number())
|
||||
const bool discard_number_values = false;
|
||||
};
|
||||
|
||||
} // namespace detail
|
||||
@@ -14166,10 +14043,9 @@ class parser
|
||||
parser_callback_t<BasicJsonType> cb = nullptr,
|
||||
const bool allow_exceptions_ = true,
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas_ = false,
|
||||
const bool discard_number_values_ = false)
|
||||
const bool ignore_trailing_commas_ = false)
|
||||
: callback(std::move(cb))
|
||||
, m_lexer(std::move(adapter), ignore_comments, discard_number_values_)
|
||||
, m_lexer(std::move(adapter), ignore_comments)
|
||||
, allow_exceptions(allow_exceptions_)
|
||||
, ignore_trailing_commas(ignore_trailing_commas_)
|
||||
{
|
||||
@@ -21716,12 +21592,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
detail::parser_callback_t<basic_json>cb = nullptr,
|
||||
const bool allow_exceptions = true,
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas = false,
|
||||
const bool discard_number_values = false
|
||||
const bool ignore_trailing_commas = false
|
||||
)
|
||||
{
|
||||
return ::nlohmann::detail::parser<basic_json, InputAdapterType>(std::move(adapter),
|
||||
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas, discard_number_values);
|
||||
std::move(cb), allow_exceptions, ignore_comments, ignore_trailing_commas);
|
||||
}
|
||||
|
||||
private:
|
||||
@@ -25686,7 +25561,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas = false)
|
||||
{
|
||||
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
||||
return parser(detail::input_adapter(std::forward<InputType>(i)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||
}
|
||||
|
||||
/// @brief check if the input is valid JSON (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||
@@ -25697,7 +25572,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas = false)
|
||||
{
|
||||
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
||||
return parser(detail::input_adapter(std::move(first), std::move(last)), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
@@ -25706,7 +25581,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas = false)
|
||||
{
|
||||
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas, true).accept(true);
|
||||
return parser(i.get(), nullptr, false, ignore_comments, ignore_trailing_commas).accept(true);
|
||||
}
|
||||
|
||||
/// @brief generate SAX events
|
||||
|
||||
@@ -177,6 +177,24 @@ json_test_add_test_for(src/unit-comparison.cpp
|
||||
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
|
||||
)
|
||||
|
||||
# test the parser again with JSON_DIAGNOSTIC_POSITIONS enabled
|
||||
json_test_set_test_options(test-class_parser_diagnostic_positions
|
||||
COMPILE_DEFINITIONS JSON_DIAGNOSTIC_POSITIONS=1
|
||||
)
|
||||
json_test_add_test_for(src/unit-class_parser.cpp
|
||||
NAME test-class_parser_diagnostic_positions
|
||||
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
|
||||
)
|
||||
|
||||
# test diagnostic positions again without regular diagnostics (JSON pointer paths)
|
||||
json_test_set_test_options(test-diagnostic-positions_only
|
||||
COMPILE_DEFINITIONS JSON_DIAGNOSTICS=0
|
||||
)
|
||||
json_test_add_test_for(src/unit-diagnostic-positions.cpp
|
||||
NAME test-diagnostic-positions_only
|
||||
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
|
||||
)
|
||||
|
||||
# *DO NOT* use json_test_set_test_options() below this line
|
||||
|
||||
#############################################################################
|
||||
|
||||
+580
-144
@@ -17,6 +17,8 @@ using nlohmann::json;
|
||||
|
||||
#include <valarray>
|
||||
#include <algorithm>
|
||||
#include <cstdio>
|
||||
#include <fstream>
|
||||
#include <list>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
@@ -344,6 +346,50 @@ void trailing_comma_helper(const std::string& s)
|
||||
}
|
||||
}
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
/**
|
||||
* Validates that the generated JSON object is the same as expected
|
||||
* Validates that the start position and end position match the start and end of the string
|
||||
*
|
||||
* This check assumes that there is no whitespace around the json object in the original string.
|
||||
*/
|
||||
void validate_generated_json_and_start_end_pos_helper(const std::string& original_string, const json& j, const json& check)
|
||||
{
|
||||
CHECK(j == check);
|
||||
CHECK(j.start_pos() == 0);
|
||||
CHECK(j.end_pos() == original_string.size());
|
||||
}
|
||||
|
||||
/**
|
||||
* Parses the root object from the given root string and validates that the start and end positions for the nested object are correct.
|
||||
*
|
||||
* This checks that whitespace around the nested object is included in the start and end positions of the root object.
|
||||
*/
|
||||
void validate_start_end_pos_for_nested_obj_helper(const std::string& nested_type_json_str, const std::string& root_type_json_str, const json& expected_json, const json::parser_callback_t& cb = nullptr)
|
||||
{
|
||||
json j;
|
||||
|
||||
// 1. If callback is provided, use callback version of parse()
|
||||
if (cb)
|
||||
{
|
||||
j = json::parse(root_type_json_str, cb);
|
||||
}
|
||||
else
|
||||
{
|
||||
j = json::parse(root_type_json_str);
|
||||
}
|
||||
|
||||
// 2. Check if the generated JSON is as expected
|
||||
// Assumptions: The root_type_json_str does not have any whitespace around the json object
|
||||
validate_generated_json_and_start_end_pos_helper(root_type_json_str, j, expected_json);
|
||||
|
||||
// 3. Get the nested object
|
||||
const auto& nested = j["nested"];
|
||||
// 4. Check if the start and end positions are generated correctly for nested objects and arrays
|
||||
CHECK(nested_type_json_str == root_type_json_str.substr(nested.start_pos(), nested.end_pos() - nested.start_pos()));
|
||||
}
|
||||
#endif
|
||||
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("parser class")
|
||||
@@ -930,94 +976,6 @@ TEST_CASE("parser class")
|
||||
CHECK(accept_helper("+1") == false);
|
||||
CHECK(accept_helper("+0") == false);
|
||||
}
|
||||
|
||||
SECTION("issue #5411 - skip conversion when accept() does not need the numeric value")
|
||||
{
|
||||
// lexer::scan_number() may skip strtoull()/strtoll() for
|
||||
// value_unsigned/value_integer tokens when the caller (e.g.
|
||||
// json::accept()) does not need the converted value, as long
|
||||
// as the digit count alone guarantees no 64-bit overflow (see
|
||||
// the "safe_digit_count" fast path in scan_number()). This
|
||||
// differential test checks that json::accept() (which enables
|
||||
// the fast path) and json::parse() (which never does) always
|
||||
// agree, over a corpus that exercises both the fast path
|
||||
// (<=18 digits) and the untouched, exact fallback path (>=19
|
||||
// digits) -- including reclassification of huge digit-only
|
||||
// integers to a (possibly non-finite) floating-point value.
|
||||
const std::vector<std::pair<std::string, bool>> cases =
|
||||
{
|
||||
// normal small/large integers, both signs
|
||||
{"0", true}, {"1", true}, {"-1", true}, {"42", true}, {"-42", true},
|
||||
{"123456789", true}, {"-123456789", true},
|
||||
|
||||
// digit-count boundary around the 18-digit safe cutoff (both signs)
|
||||
{std::string(17, '9'), true},
|
||||
{std::string(18, '9'), true},
|
||||
{std::string(19, '9'), true},
|
||||
{std::string(20, '9'), true},
|
||||
{"-" + std::string(17, '9'), true},
|
||||
{"-" + std::string(18, '9'), true},
|
||||
{"-" + std::string(19, '9'), true},
|
||||
{"-" + std::string(20, '9'), true},
|
||||
|
||||
// 64-bit boundaries
|
||||
{"9223372036854775807", true}, // INT64_MAX
|
||||
{"-9223372036854775808", true}, // INT64_MIN
|
||||
{"18446744073709551615", true}, // UINT64_MAX
|
||||
{"18446744073709551616", true}, // UINT64_MAX + 1 (overflows uint64_t, finite double)
|
||||
|
||||
// the 28-digit example from the issue: overflows uint64_t
|
||||
// but is finite as a double, so the scanner reclassifies
|
||||
// it to value_float and it is accepted
|
||||
{"9999999999999999999999999999", true},
|
||||
|
||||
// huge digit-only integers that overflow even a double -> rejected
|
||||
{std::string(309, '9'), false},
|
||||
{std::string(400, '9'), false},
|
||||
{"1" + std::string(400, '0'), false},
|
||||
|
||||
// 1e999 / 1e400 style overflow -> rejected
|
||||
{"1e999", false},
|
||||
{"1e400", false},
|
||||
{"-1e999", false},
|
||||
{"1E999", false},
|
||||
|
||||
// values straddling DBL_MAX
|
||||
{"1.7976931348623157e308", true}, // <= DBL_MAX, finite
|
||||
{"1.7976931348623159e308", false}, // > DBL_MAX, overflows to inf
|
||||
|
||||
// a mix of other valid/invalid numeric syntax
|
||||
{"3.14159", true},
|
||||
{"-0.0", true},
|
||||
{"1.0e10", true},
|
||||
{"01", false},
|
||||
{"-", false},
|
||||
{"1.", false},
|
||||
{"1e", false},
|
||||
{"+1", false},
|
||||
};
|
||||
|
||||
for (const auto& c : cases)
|
||||
{
|
||||
const std::string& number = c.first;
|
||||
const bool expected = c.second;
|
||||
CAPTURE(number)
|
||||
CAPTURE(expected)
|
||||
|
||||
// accept() takes the fast path (skips conversion when possible)
|
||||
CHECK(json::accept(number) == expected);
|
||||
|
||||
// parse() always performs the full conversion; it must agree
|
||||
json j;
|
||||
CHECK_NOTHROW(json::parser(nlohmann::detail::input_adapter(number), nullptr, false).parse(true, j));
|
||||
CHECK(!j.is_discarded() == expected);
|
||||
|
||||
// wrap in an array so get_token() is exercised beyond the
|
||||
// very first (constructor-time) scan as well
|
||||
const std::string wrapped = "[" + number + "," + number + "]";
|
||||
CHECK(json::accept(wrapped) == expected);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1482,62 +1440,6 @@ TEST_CASE("parser class")
|
||||
CHECK(accept_helper("\"\\uD80C\\uFFFF\"") == false);
|
||||
}
|
||||
|
||||
SECTION("issue #5412 - whitespace skipping bookkeeping (compact vs. pretty-printed)")
|
||||
{
|
||||
// lexer::skip_whitespace() reads its first character with get() (to
|
||||
// honor a possibly pending unget() from the previous token) and every
|
||||
// further whitespace character with get_ignoring_pending_unget() (a
|
||||
// get() variant that skips the then-always-false next_unget check).
|
||||
// This must not change the reported byte offset, line, or column of
|
||||
// a syntax error, even when a long run of whitespace containing
|
||||
// multiple newlines is skipped beforehand (as with pretty-printed
|
||||
// input). The expected values below were captured from the
|
||||
// unmodified do-while(get()) loop, so any regression that miscounts
|
||||
// characters or newlines while skipping whitespace changes them.
|
||||
const auto check_error = [](const std::string & input, std::size_t expected_byte,
|
||||
const std::string & expected_what)
|
||||
{
|
||||
CAPTURE(input)
|
||||
try
|
||||
{
|
||||
json _ = json::parse(input);
|
||||
FAIL_CHECK("expected a parse_error, but parsing succeeded");
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
CHECK(e.byte == expected_byte);
|
||||
CHECK(std::string(e.what()) == expected_what);
|
||||
}
|
||||
};
|
||||
|
||||
// a nested document, serialized both compactly and pretty-printed
|
||||
// (dump(4)), each truncated right before the final closing '}' so
|
||||
// that the parser hits EOF after skipping all of the (in the
|
||||
// pretty-printed case, substantial) indentation whitespace
|
||||
const json doc =
|
||||
{
|
||||
{"a", 1},
|
||||
{"b", json::array({true, false, nullptr, "x"})},
|
||||
{"c", json::object({{"d", 3.14}, {"e", json::array({1, 2, 3})}})}
|
||||
};
|
||||
|
||||
const std::string compact = doc.dump();
|
||||
const std::string pretty = doc.dump(4);
|
||||
|
||||
check_error(compact.substr(0, compact.size() - 1), 60,
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 60: syntax error while parsing object - unexpected end of input; expected '}'");
|
||||
check_error(pretty.substr(0, pretty.size() - 1), 193,
|
||||
"[json.exception.parse_error.101] parse error at line 17, column 1: syntax error while parsing object - unexpected end of input; expected '}'");
|
||||
|
||||
// an invalid token appearing after several indented, multi-line
|
||||
// whitespace runs vs. the same document without any of that
|
||||
// whitespace
|
||||
check_error("{\n \"a\": 1,\n \"b\": [\n true,\n false\n ],\n \"c\": @\n}", 70,
|
||||
"[json.exception.parse_error.101] parse error at line 7, column 10: syntax error while parsing value - invalid literal; last read: '\"c\": @'");
|
||||
check_error("{\"a\":1,\"b\":[true,false],\"c\":@}", 29,
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing value - invalid literal; last read: '\"c\":@'");
|
||||
}
|
||||
|
||||
SECTION("tests found by mutate++")
|
||||
{
|
||||
// test case to make sure no comma precedes the first key
|
||||
@@ -1923,6 +1825,228 @@ TEST_CASE("parser class")
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse("/a", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid comment; expecting '/' or '*' after '/'; last read: '/a'", json::parse_error);
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse("/*", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid comment; missing closing '*/'; last read: '/*<U+0000>'", json::parse_error);
|
||||
}
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
// Macro for all test cases for start_pos and end_pos
|
||||
#define SETUP_TESTCASES() \
|
||||
SECTION("with callback") \
|
||||
{ \
|
||||
SECTION("filter nothing") \
|
||||
{ \
|
||||
json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t /*unused*/, json& /*unused*/) noexcept \
|
||||
{ \
|
||||
return true; \
|
||||
}; \
|
||||
validate_start_end_pos_for_nested_obj_helper(nested_type_json_str, root_type_json_str, expected, cb); \
|
||||
} \
|
||||
SECTION("filter element") \
|
||||
{ \
|
||||
json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t event, json& j) noexcept \
|
||||
{ \
|
||||
return (event != json::parse_event_t::key && event != json::parse_event_t::value) || j != json("a"); \
|
||||
}; \
|
||||
validate_start_end_pos_for_nested_obj_helper(nested_type_json_str, root_type_json_str, filteredExpected, cb); \
|
||||
} \
|
||||
} \
|
||||
SECTION("without callback") \
|
||||
{ \
|
||||
validate_start_end_pos_for_nested_obj_helper(nested_type_json_str, root_type_json_str, expected); \
|
||||
}
|
||||
|
||||
SECTION("retrieve start position and end position")
|
||||
{
|
||||
SECTION("for object")
|
||||
{
|
||||
// Create an object with spaces to test the start and end positions. Spaces will not be included in the
|
||||
// JSON object, however, the start and end positions should include the spaces from the input JSON string.
|
||||
const std::string nested_type_json_str = R"({ "a": 1,"b" : "test1"})";
|
||||
const std::string root_type_json_str = R"({ "nested": )" + nested_type_json_str + R"(, "anotherValue": "test2"})";
|
||||
auto expected = json({{"nested", {{"a", 1}, {"b", "test1"}}}, {"anotherValue", "test2"}});
|
||||
auto filteredExpected = expected;
|
||||
filteredExpected["nested"].erase("a");
|
||||
|
||||
SETUP_TESTCASES()
|
||||
}
|
||||
|
||||
SECTION("for array")
|
||||
{
|
||||
const std::string nested_type_json_str = R"(["a", "test", 45])";
|
||||
const std::string root_type_json_str = R"({ "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
|
||||
auto expected = json({{"nested", {"a", "test", 45}}, {"anotherValue", "test"}});
|
||||
auto filteredExpected = expected;
|
||||
filteredExpected["nested"] = json({"test", 45});
|
||||
SETUP_TESTCASES()
|
||||
}
|
||||
|
||||
SECTION("for array with objects")
|
||||
{
|
||||
const std::string nested_type_json_str = R"([{"a": 1, "b": "test"}, {"c": 2, "d": "test2"}])";
|
||||
const std::string root_type_json_str = R"({ "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
|
||||
auto expected = json({{"nested", {{{"a", 1}, {"b", "test"}}, {{"c", 2}, {"d", "test2"}}}}, {"anotherValue", "test"}});
|
||||
auto filteredExpected = expected;
|
||||
filteredExpected["nested"][0].erase("a");
|
||||
SETUP_TESTCASES()
|
||||
|
||||
auto j = json::parse(root_type_json_str);
|
||||
auto nested_array = j["nested"];
|
||||
const auto& nested_obj = nested_array[0];
|
||||
CHECK(nested_type_json_str.substr(1, 21) == root_type_json_str.substr(nested_obj.start_pos(), nested_obj.end_pos() - nested_obj.start_pos()));
|
||||
CHECK(nested_type_json_str.substr(24, 22) == root_type_json_str.substr(nested_array[1].start_pos(), nested_array[1].end_pos() - nested_array[1].start_pos()));
|
||||
}
|
||||
|
||||
SECTION("for two levels of nesting objects")
|
||||
{
|
||||
const std::string nested_type_json_str = R"({"nested2": {"b": "test"}})";
|
||||
const std::string root_type_json_str = R"({ "a": 2, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
|
||||
auto expected = json({{"a", 2}, {"nested", {{"nested2", {{"b", "test"}}}}}, {"anotherValue", "test"}});
|
||||
auto filteredExpected = expected;
|
||||
filteredExpected.erase("a");
|
||||
SETUP_TESTCASES()
|
||||
|
||||
auto j = json::parse(root_type_json_str);
|
||||
auto nested_obj = j["nested"]["nested2"];
|
||||
CHECK(nested_type_json_str.substr(12, 13) == root_type_json_str.substr(nested_obj.start_pos(), nested_obj.end_pos() - nested_obj.start_pos()));
|
||||
}
|
||||
|
||||
SECTION("for simple types")
|
||||
{
|
||||
SECTION("no nested")
|
||||
{
|
||||
SECTION("with callback")
|
||||
{
|
||||
json::parser_callback_t const cb = [](int /*unused*/, json::parse_event_t /*unused*/, json& /*unused*/) noexcept
|
||||
{
|
||||
return true;
|
||||
};
|
||||
|
||||
// 1. string type
|
||||
std::string json_str = R"("test")";
|
||||
auto j = json::parse(json_str, cb);
|
||||
validate_generated_json_and_start_end_pos_helper(json_str, j, "test");
|
||||
|
||||
// 2. number type
|
||||
json_str = R"(1)";
|
||||
j = json::parse(json_str, cb);
|
||||
validate_generated_json_and_start_end_pos_helper(json_str, j, 1);
|
||||
|
||||
// 3. boolean type
|
||||
json_str = R"(true)";
|
||||
j = json::parse(json_str, cb);
|
||||
validate_generated_json_and_start_end_pos_helper(json_str, j, true);
|
||||
|
||||
// 4. null type
|
||||
json_str = R"(null)";
|
||||
j = json::parse(json_str, cb);
|
||||
validate_generated_json_and_start_end_pos_helper(json_str, j, nullptr);
|
||||
}
|
||||
|
||||
SECTION("without callback")
|
||||
{
|
||||
// 1. string type
|
||||
std::string json_str = R"("test")";
|
||||
auto j = json::parse(json_str);
|
||||
validate_generated_json_and_start_end_pos_helper(json_str, j, "test");
|
||||
|
||||
// 2. number type
|
||||
json_str = R"(1)";
|
||||
j = json::parse(json_str);
|
||||
validate_generated_json_and_start_end_pos_helper(json_str, j, 1);
|
||||
|
||||
json_str = R"(1.001239923)";
|
||||
j = json::parse(json_str);
|
||||
validate_generated_json_and_start_end_pos_helper(json_str, j, 1.001239923);
|
||||
|
||||
json_str = R"(1.123812389000000)";
|
||||
j = json::parse(json_str);
|
||||
validate_generated_json_and_start_end_pos_helper(json_str, j, 1.123812389);
|
||||
|
||||
// 3. boolean type
|
||||
json_str = R"(true)";
|
||||
j = json::parse(json_str);
|
||||
validate_generated_json_and_start_end_pos_helper(json_str, j, true);
|
||||
|
||||
json_str = R"(false)";
|
||||
j = json::parse(json_str);
|
||||
validate_generated_json_and_start_end_pos_helper(json_str, j, false);
|
||||
|
||||
// 4. null type
|
||||
json_str = R"(null)";
|
||||
j = json::parse(json_str);
|
||||
validate_generated_json_and_start_end_pos_helper(json_str, j, nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("string type")
|
||||
{
|
||||
const std::string nested_type_json_str = R"("test")";
|
||||
const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
|
||||
auto expected = json({{"nested", "test"}, {"anotherValue", "test"}, {"a", 1}});
|
||||
auto filteredExpected = expected;
|
||||
filteredExpected.erase("a");
|
||||
SETUP_TESTCASES()
|
||||
}
|
||||
|
||||
SECTION("number type")
|
||||
{
|
||||
const std::string nested_type_json_str = R"(2)";
|
||||
const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
|
||||
auto expected = json({{"nested", 2}, {"anotherValue", "test"}, {"a", 1}});
|
||||
auto filteredExpected = expected;
|
||||
filteredExpected.erase("a");
|
||||
SETUP_TESTCASES()
|
||||
}
|
||||
|
||||
SECTION("boolean type")
|
||||
{
|
||||
const std::string nested_type_json_str = R"(true)";
|
||||
const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
|
||||
auto expected = json({{"nested", true}, {"anotherValue", "test"}, {"a", 1}});
|
||||
auto filteredExpected = expected;
|
||||
filteredExpected.erase("a");
|
||||
SETUP_TESTCASES()
|
||||
}
|
||||
|
||||
SECTION("null type")
|
||||
{
|
||||
const std::string nested_type_json_str = R"(null)";
|
||||
const std::string root_type_json_str = R"({ "a": 1, "nested": )" + nested_type_json_str + R"(, "anotherValue": "test" })";
|
||||
auto expected = json({{"nested", nullptr}, {"anotherValue", "test"}, {"a", 1}});
|
||||
auto filteredExpected = expected;
|
||||
filteredExpected.erase("a");
|
||||
SETUP_TESTCASES()
|
||||
}
|
||||
}
|
||||
SECTION("with leading whitespace and newlines around root JSON")
|
||||
{
|
||||
const std::string initial_whitespace = R"(
|
||||
|
||||
)";
|
||||
const std::string nested_type_json_str = R"({
|
||||
"a": 1,
|
||||
"nested": {
|
||||
"b": "test"
|
||||
},
|
||||
"anotherValue": "test"
|
||||
})";
|
||||
const std::string end_whitespace = R"(
|
||||
|
||||
)";
|
||||
const std::string root_type_json_str = initial_whitespace + nested_type_json_str + end_whitespace;
|
||||
|
||||
auto expected = json({{"a", 1}, {"nested", {{"b", "test"}}}, {"anotherValue", "test"}});
|
||||
|
||||
auto j = json::parse(root_type_json_str);
|
||||
|
||||
// 2. Check if the generated JSON is as expected
|
||||
CHECK(j == expected);
|
||||
|
||||
// 3. Check if the start and end positions do not include the surrounding whitespace
|
||||
CHECK(j.start_pos() == initial_whitespace.size());
|
||||
CHECK(j.end_pos() == root_type_json_str.size() - end_whitespace.size());
|
||||
}
|
||||
}
|
||||
#undef SETUP_TESTCASES
|
||||
#endif
|
||||
}
|
||||
|
||||
// this test relies on parse errors being thrown, so it is skipped when
|
||||
@@ -2031,3 +2155,315 @@ TEST_CASE("last-read diagnostics are identical across input adapters")
|
||||
}
|
||||
}
|
||||
#endif // !defined(JSON_NOEXCEPTION)
|
||||
|
||||
// this test characterizes the current (documented-by-example, not otherwise
|
||||
// specified) behavior of JSON_DIAGNOSTIC_POSITIONS positions with respect to
|
||||
// value lifetime (copy/move/swap/mutation), the various input adapters, and
|
||||
// user-driven SAX usage. It is regression protection, not a behavior
|
||||
// specification: if any of these checks fail after a change to json.hpp,
|
||||
// that change deliberately altered observable behavior and the test (and
|
||||
// this comment) should be updated accordingly, rather than "fixed" blindly.
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX")
|
||||
{
|
||||
SECTION("value lifetime")
|
||||
{
|
||||
SECTION("copy constructor copies positions, recursively")
|
||||
{
|
||||
// basic_json(const basic_json&) (json.hpp, around line 1192) copies
|
||||
// start_position/end_position for the value itself; nested values
|
||||
// are copied via their own copy constructor (through the copied
|
||||
// object/array container), so positions are preserved throughout
|
||||
// the whole tree.
|
||||
const std::string s = R"({"a":1,"b":[1,2,3]})";
|
||||
const json a = json::parse(s);
|
||||
const json b = a; // NOLINT(performance-unnecessary-copy-initialization)
|
||||
|
||||
CHECK(b.start_pos() == a.start_pos());
|
||||
CHECK(b.end_pos() == a.end_pos());
|
||||
CHECK(b["b"].start_pos() == a["b"].start_pos());
|
||||
CHECK(b["b"].end_pos() == a["b"].end_pos());
|
||||
CHECK(b["b"][0].start_pos() == a["b"][0].start_pos());
|
||||
CHECK(b["b"][0].end_pos() == a["b"][0].end_pos());
|
||||
|
||||
// sanity: the positions are meaningful (not all npos)
|
||||
CHECK(b.start_pos() == 0);
|
||||
CHECK(b.end_pos() == s.size());
|
||||
}
|
||||
|
||||
SECTION("move constructor resets the moved-from value to npos")
|
||||
{
|
||||
// basic_json(basic_json&&) (json.hpp, around line 1265) copies
|
||||
// other's start_position/end_position into *this and then resets
|
||||
// other's to npos (see the "// cppcheck-suppress[accessForwarded]
|
||||
// TODO check" comments there). Only the top-level moved-from value
|
||||
// is affected; its (moved-away) children are gone along with it.
|
||||
const std::string s = R"({"a":1,"b":[1,2,3]})";
|
||||
json a = json::parse(s);
|
||||
const auto a_start = a.start_pos();
|
||||
const auto a_end = a.end_pos();
|
||||
const auto nested_start = a["b"].start_pos();
|
||||
const auto nested_end = a["b"].end_pos();
|
||||
|
||||
const json b(std::move(a));
|
||||
|
||||
// the destination retains the original positions, recursively
|
||||
CHECK(b.start_pos() == a_start);
|
||||
CHECK(b.end_pos() == a_end);
|
||||
CHECK(b["b"].start_pos() == nested_start);
|
||||
CHECK(b["b"].end_pos() == nested_end);
|
||||
|
||||
// the moved-from value is reset to a null and reports npos
|
||||
CHECK(a.is_null()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move)
|
||||
CHECK(a.start_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move)
|
||||
CHECK(a.end_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move)
|
||||
}
|
||||
|
||||
SECTION("swap() does NOT exchange positions (likely a real bug, see below)")
|
||||
{
|
||||
// NOTE (characterizing, not fixing, for #5420): basic_json::swap()
|
||||
// (json.hpp, around line 3540, and the friend swap() that forwards
|
||||
// to it) swaps m_data.m_type and m_data.m_value but -- unlike
|
||||
// copy-assignment's operator=(basic_json) (json.hpp, around line
|
||||
// 1291), which swaps start_position/end_position as part of its
|
||||
// copy-and-swap implementation -- it never touches
|
||||
// start_position/end_position. So after swap(a, b), the *values*
|
||||
// of a and b are exchanged, but their *positions* are not: each
|
||||
// ends up with its own original position describing the other's
|
||||
// new content. This looks like an oversight/inconsistency rather
|
||||
// than intended behavior, and is flagged to the maintainer; this
|
||||
// test only pins the current (surprising) behavior so a fix (or a
|
||||
// deliberate decision to keep it) shows up here as an intentional
|
||||
// change rather than a silent regression.
|
||||
json a = json::parse(R"({"a":1})");
|
||||
json b = json::parse(R"([1,2,3,4,5])");
|
||||
const auto a_start = a.start_pos();
|
||||
const auto a_end = a.end_pos();
|
||||
const auto b_start = b.start_pos();
|
||||
const auto b_end = b.end_pos();
|
||||
// both start at 0 (root values start right away), but their
|
||||
// lengths (and thus end positions) differ, which is enough to
|
||||
// tell after the swap whether positions actually moved with
|
||||
// the values
|
||||
CHECK(a_end != b_end);
|
||||
|
||||
using std::swap;
|
||||
swap(a, b);
|
||||
|
||||
// values were exchanged as expected ...
|
||||
CHECK(a == json::parse(R"([1,2,3,4,5])"));
|
||||
CHECK(b == json::parse(R"({"a":1})"));
|
||||
|
||||
// ... but positions were NOT: each variable kept its own
|
||||
// original position, now describing the other's content
|
||||
CHECK(a.start_pos() == a_start);
|
||||
CHECK(a.end_pos() == a_end);
|
||||
CHECK(b.start_pos() == b_start);
|
||||
CHECK(b.end_pos() == b_end);
|
||||
}
|
||||
|
||||
SECTION("mutating a parsed document leaves positions of unrelated values untouched")
|
||||
{
|
||||
// Positions are recorded once, during parsing, and are not
|
||||
// recomputed on mutation. As a consequence, after a mutation the
|
||||
// parent's own recorded span may no longer describe its current
|
||||
// (serialized) content -- it still describes what was originally
|
||||
// parsed. This is characterized here as current behavior, not
|
||||
// asserted to be desirable or specified.
|
||||
SECTION("operator[] adding a new object key")
|
||||
{
|
||||
const std::string s = R"({"a":1})";
|
||||
json j = json::parse(s);
|
||||
const auto root_start = j.start_pos();
|
||||
const auto root_end = j.end_pos();
|
||||
const auto a_start = j["a"].start_pos();
|
||||
const auto a_end = j["a"].end_pos();
|
||||
|
||||
j["c"] = 42;
|
||||
|
||||
// the newly-added value was never parsed, so it has no position
|
||||
CHECK(j["c"].start_pos() == std::string::npos);
|
||||
CHECK(j["c"].end_pos() == std::string::npos);
|
||||
|
||||
// the existing sibling's position is unaffected
|
||||
CHECK(j["a"].start_pos() == a_start);
|
||||
CHECK(j["a"].end_pos() == a_end);
|
||||
|
||||
// the parent's own recorded span is left as-is (now stale:
|
||||
// it still reflects the original, shorter `{"a":1}` string)
|
||||
CHECK(j.start_pos() == root_start);
|
||||
CHECK(j.end_pos() == root_end);
|
||||
}
|
||||
|
||||
SECTION("push_back on a parsed array")
|
||||
{
|
||||
const std::string s = R"([1,2,3])";
|
||||
json j = json::parse(s);
|
||||
const auto root_start = j.start_pos();
|
||||
const auto root_end = j.end_pos();
|
||||
const auto first_start = j[0].start_pos();
|
||||
|
||||
j.push_back(4);
|
||||
|
||||
CHECK(j.back().start_pos() == std::string::npos);
|
||||
CHECK(j.back().end_pos() == std::string::npos);
|
||||
CHECK(j[0].start_pos() == first_start);
|
||||
CHECK(j.start_pos() == root_start);
|
||||
CHECK(j.end_pos() == root_end);
|
||||
}
|
||||
|
||||
SECTION("erase on a parsed array shifts elements but keeps their own positions")
|
||||
{
|
||||
const std::string s = R"([1,2,3])";
|
||||
json j = json::parse(s);
|
||||
const auto second_start = j[1].start_pos();
|
||||
const auto third_start = j[2].start_pos();
|
||||
const auto root_start = j.start_pos();
|
||||
const auto root_end = j.end_pos();
|
||||
|
||||
j.erase(0);
|
||||
|
||||
// remaining elements moved down an index, but each one still
|
||||
// reports the position it had *before* the erase (i.e. its
|
||||
// position in the original source string, not a
|
||||
// recalculated one)
|
||||
CHECK(j[0].start_pos() == second_start);
|
||||
CHECK(j[1].start_pos() == third_start);
|
||||
|
||||
// the parent's own recorded span is again left as-is
|
||||
CHECK(j.start_pos() == root_start);
|
||||
CHECK(j.end_pos() == root_end);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("input adapters")
|
||||
{
|
||||
SECTION("wide string input: positions count transcoded UTF-8 bytes, not wide characters")
|
||||
{
|
||||
// 'é' (U+00E9) is a single code unit in a wchar_t/UTF-16 string, but
|
||||
// transcodes to 2 bytes in UTF-8; the lexer only ever sees the
|
||||
// transcoded UTF-8 byte stream, so reported positions are byte
|
||||
// offsets into that UTF-8 stream, not indices into the original
|
||||
// std::wstring.
|
||||
const std::wstring ws = L"{\"a\":\"éé\"}";
|
||||
CHECK(ws.size() == 10); // 10 wide characters
|
||||
|
||||
const json j = json::parse(ws);
|
||||
CHECK(j.start_pos() == 0);
|
||||
// the transcoded UTF-8 form is 2 bytes longer than the wide string,
|
||||
// because each of the two 'é' characters becomes 2 UTF-8 bytes
|
||||
CHECK(j.end_pos() == 12);
|
||||
CHECK(j.end_pos() != ws.size());
|
||||
|
||||
const json& a = j["a"];
|
||||
CHECK(a.start_pos() == 5);
|
||||
CHECK(a.end_pos() == 11);
|
||||
}
|
||||
|
||||
SECTION("BOM-prefixed input: start_pos() reflects the skipped 3-byte BOM")
|
||||
{
|
||||
const std::string s = "\xEF\xBB\xBF{\"a\":1}";
|
||||
const json j = json::parse(s);
|
||||
|
||||
// the lexer silently skips the BOM before parsing the value, so
|
||||
// the root value's recorded span starts right after it
|
||||
CHECK(j.start_pos() == 3);
|
||||
CHECK(j.end_pos() == s.size());
|
||||
}
|
||||
|
||||
SECTION("std::istringstream: positions are consistent, not npos")
|
||||
{
|
||||
const std::string s = R"({"a":1,"b":2})";
|
||||
std::istringstream ss(s);
|
||||
const json j = json::parse(ss);
|
||||
|
||||
CHECK(j.start_pos() == 0);
|
||||
CHECK(j.end_pos() == s.size());
|
||||
CHECK(j["a"].start_pos() == 5);
|
||||
}
|
||||
|
||||
SECTION("std::ifstream: positions are consistent, not npos")
|
||||
{
|
||||
const std::string s = R"({"a":1,"b":2})";
|
||||
{
|
||||
std::ofstream file("unit-class_parser_diagnostic_positions.tmp");
|
||||
file << s;
|
||||
}
|
||||
|
||||
{
|
||||
std::ifstream f("unit-class_parser_diagnostic_positions.tmp");
|
||||
const json j = json::parse(f);
|
||||
|
||||
CHECK(j.start_pos() == 0);
|
||||
CHECK(j.end_pos() == s.size());
|
||||
CHECK(j["a"].start_pos() == 5);
|
||||
}
|
||||
|
||||
static_cast<void>(std::remove("unit-class_parser_diagnostic_positions.tmp"));
|
||||
}
|
||||
|
||||
SECTION("iterator-pair input: positions are consistent, not npos")
|
||||
{
|
||||
const std::string s = R"({"a":1,"b":2})";
|
||||
const json j = json::parse(s.begin(), s.end());
|
||||
|
||||
CHECK(j.start_pos() == 0);
|
||||
CHECK(j.end_pos() == s.size());
|
||||
CHECK(j["a"].start_pos() == 5);
|
||||
}
|
||||
|
||||
SECTION("binary formats have no text positions")
|
||||
{
|
||||
// binary formats (CBOR, MessagePack, UBJSON, BSON, BJData) are
|
||||
// parsed via detail::binary_reader, which never sets
|
||||
// start_position/end_position on the values it produces (they
|
||||
// have no notion of a text offset), so every value's position
|
||||
// stays at its default of npos.
|
||||
const json src = json::parse(R"({"a":1,"b":[1,2]})");
|
||||
|
||||
const json from_cbor = json::from_cbor(json::to_cbor(src));
|
||||
CHECK(from_cbor.start_pos() == std::string::npos);
|
||||
CHECK(from_cbor.end_pos() == std::string::npos);
|
||||
CHECK(from_cbor["a"].start_pos() == std::string::npos);
|
||||
CHECK(from_cbor["b"][0].start_pos() == std::string::npos);
|
||||
|
||||
const json from_msgpack = json::from_msgpack(json::to_msgpack(src));
|
||||
CHECK(from_msgpack.start_pos() == std::string::npos);
|
||||
CHECK(from_msgpack.end_pos() == std::string::npos);
|
||||
|
||||
const json from_ubjson = json::from_ubjson(json::to_ubjson(src));
|
||||
CHECK(from_ubjson.start_pos() == std::string::npos);
|
||||
CHECK(from_ubjson.end_pos() == std::string::npos);
|
||||
|
||||
const json from_bson_val = json::from_bson(json::to_bson(src));
|
||||
CHECK(from_bson_val.start_pos() == std::string::npos);
|
||||
CHECK(from_bson_val.end_pos() == std::string::npos);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("user-driven SAX consumers with no lexer report npos")
|
||||
{
|
||||
// json::parse() internally wires up its json_sax_dom_parser with a
|
||||
// pointer to its own lexer (see parser.hpp), which is how positions
|
||||
// get set at all. A user who constructs a json_sax_dom_parser
|
||||
// directly (e.g. to drive it via json::sax_parse()) and does not
|
||||
// supply a lexer pointer gets a consumer with m_lexer_ref == nullptr;
|
||||
// every "if (m_lexer_ref)" guard in json_sax.hpp is then skipped, so
|
||||
// every value it produces keeps its default, unset position (npos).
|
||||
// This was previously true but silently unasserted (operator==
|
||||
// ignores positions), see #5420.
|
||||
json result;
|
||||
nlohmann::detail::json_sax_dom_parser<json, nlohmann::detail::string_input_adapter_type> sdp(result);
|
||||
const std::string s = R"({"a":1,"b":[1,2,3]})";
|
||||
CHECK(json::sax_parse(s, &sdp));
|
||||
|
||||
CHECK(result.start_pos() == std::string::npos);
|
||||
CHECK(result.end_pos() == std::string::npos);
|
||||
CHECK(result["a"].start_pos() == std::string::npos);
|
||||
CHECK(result["a"].end_pos() == std::string::npos);
|
||||
CHECK(result["b"][0].start_pos() == std::string::npos);
|
||||
CHECK(result["b"][0].end_pos() == std::string::npos);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,44 +0,0 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
|
||||
#ifdef JSON_DIAGNOSTICS
|
||||
#undef JSON_DIAGNOSTICS
|
||||
#endif
|
||||
|
||||
#define JSON_DIAGNOSTICS 0
|
||||
#define JSON_DIAGNOSTIC_POSITIONS 1
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
TEST_CASE("Better diagnostics with positions only")
|
||||
{
|
||||
SECTION("invalid type")
|
||||
{
|
||||
const std::string json_invalid_string = R"(
|
||||
{
|
||||
"address": {
|
||||
"street": "Fake Street",
|
||||
"housenumber": "1"
|
||||
}
|
||||
}
|
||||
)";
|
||||
json j = json::parse(json_invalid_string);
|
||||
CHECK_THROWS_WITH_AS(j.at("address").at("housenumber").get<int>(),
|
||||
"[json.exception.type_error.302] (bytes 108-111) type must be number, but is string", json::type_error);
|
||||
}
|
||||
|
||||
SECTION("invalid type without positions")
|
||||
{
|
||||
const json j = "foo";
|
||||
CHECK_THROWS_WITH_AS(j.get<int>(),
|
||||
"[json.exception.type_error.302] type must be number, but is string", json::type_error);
|
||||
}
|
||||
}
|
||||
@@ -8,7 +8,9 @@
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
|
||||
#define JSON_DIAGNOSTICS 1
|
||||
#ifndef JSON_DIAGNOSTICS
|
||||
#define JSON_DIAGNOSTICS 1
|
||||
#endif
|
||||
#define JSON_DIAGNOSTIC_POSITIONS 1
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
@@ -27,8 +29,13 @@ TEST_CASE("Better diagnostics with positions")
|
||||
}
|
||||
)";
|
||||
json j = json::parse(json_invalid_string);
|
||||
#if JSON_DIAGNOSTICS
|
||||
CHECK_THROWS_WITH_AS(j.at("address").at("housenumber").get<int>(),
|
||||
"[json.exception.type_error.302] (/address/housenumber) (bytes 108-111) type must be number, but is string", json::type_error);
|
||||
#else
|
||||
CHECK_THROWS_WITH_AS(j.at("address").at("housenumber").get<int>(),
|
||||
"[json.exception.type_error.302] (bytes 108-111) type must be number, but is string", json::type_error);
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("invalid type without positions")
|
||||
@@ -74,7 +81,12 @@ TEST_CASE("Better diagnostics with positions")
|
||||
// (/foo/bar); the position of that parent is reported in the message
|
||||
const json doc = json::parse(R"({"foo":{"bar":"a string"}})");
|
||||
const json patch = json::parse(R"([{"op":"add","path":"/foo/bar/baz","value":1}])");
|
||||
#if JSON_DIAGNOSTICS
|
||||
CHECK_THROWS_WITH_AS(doc.patch(patch),
|
||||
"[json.exception.out_of_range.411] (/foo/bar) (bytes 14-24) cannot add value: the JSON Patch 'add' target's parent is of type string, but must be an object or array", json::out_of_range);
|
||||
#else
|
||||
CHECK_THROWS_WITH_AS(doc.patch(patch),
|
||||
"[json.exception.out_of_range.411] (bytes 14-24) cannot add value: the JSON Patch 'add' target's parent is of type string, but must be an object or array", json::out_of_range);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user