mirror of
https://github.com/nlohmann/json.git
synced 2026-09-11 18:57:58 +00:00
Merge remote-tracking branch 'origin/develop' into claude/parser-performance-research-4hkdf8
Resolves the number-parsing conflict between the discard_number_values fast path (an early return for accept()-only calls that skips value conversion for short digit runs) and the contiguous number scanner's convert_number()/convert_integer() split: the discard check now runs as the first step of convert_number(), so both the byte-at-a-time scanner and the bulk contiguous path benefit from it. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -178,10 +178,11 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
public:
|
||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||
|
||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept
|
||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept
|
||||
: ia(std::move(adapter))
|
||||
, ignore_comments(ignore_comments_)
|
||||
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
||||
, discard_number_values(discard_number_values_)
|
||||
{}
|
||||
|
||||
// deleted because of pointer members
|
||||
@@ -1465,6 +1466,58 @@ scan_number_done:
|
||||
*/
|
||||
token_type convert_number(token_type number_type, std::size_t mantissa_end)
|
||||
{
|
||||
// If the caller does not need the converted value (only whether the
|
||||
// input is syntactically valid; see json_sax_acceptor/accept()), an
|
||||
// unsigned/integer token can be reported without calling
|
||||
// strtoull()/strtoll() at all, *provided* we can already tell from
|
||||
// the digit count alone that the conversion cannot overflow 64 bits.
|
||||
// Such tokens are always finite and are accepted unconditionally by
|
||||
// the parser regardless of their actual value (parser::sax_parse_internal()
|
||||
// never checks finiteness for value_unsigned/value_integer), so the
|
||||
// classification below is all that is needed.
|
||||
//
|
||||
// A decimal number with up to 18 digits is always representable in
|
||||
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
|
||||
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
|
||||
// could not have set errno to ERANGE for it. Numbers with more digits
|
||||
// (rare in practice) fall through to the exact code below, unchanged,
|
||||
// so their handling -- including reclassification to value_float when
|
||||
// the value overflows 64 bits, and rejection when it is not even
|
||||
// finite as a double -- is bit-for-bit identical to before this
|
||||
// optimization.
|
||||
//
|
||||
// Note this reasons about std::uint64_t/std::int64_t, not about
|
||||
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
|
||||
// narrower, template parameters -- e.g. std::uint32_t). That is fine
|
||||
// *only* because discard_number_values is exclusively set by
|
||||
// accept() (see json.hpp), and accept() always parses through the
|
||||
// library's own json_sax_acceptor -- never a user-supplied SAX
|
||||
// consumer -- whose number_unsigned()/number_integer()/number_float()
|
||||
// callbacks unconditionally discard their argument and return true.
|
||||
// So for every caller that can reach this branch, neither the token
|
||||
// classification below nor the eventual (possibly narrowed, and on
|
||||
// this fast path left stale/unset) value_unsigned/value_integer is
|
||||
// ever consulted -- an unsigned/integer token is accepted outright,
|
||||
// and even a >18-digit token that this fast path deliberately falls
|
||||
// through for is, once reclassified to value_float, still finite
|
||||
// (and thus accepted) for any digit count that fits in number_unsigned_t
|
||||
// or number_integer_t regardless of that type's width. If this
|
||||
// function is ever taught to run with discard_number_values true for
|
||||
// a caller that *does* read the converted value, this reasoning (and
|
||||
// the fast path below) would need to be revisited.
|
||||
if (discard_number_values)
|
||||
{
|
||||
constexpr std::size_t safe_digit_count = 18;
|
||||
if (number_type == token_type::value_unsigned && token_buffer.size() <= safe_digit_count)
|
||||
{
|
||||
return token_type::value_unsigned;
|
||||
}
|
||||
if (number_type == token_type::value_integer && token_buffer.size() - 1 <= safe_digit_count)
|
||||
{
|
||||
return token_type::value_integer;
|
||||
}
|
||||
}
|
||||
|
||||
const char* const num_begin = token_buffer.data();
|
||||
const char* const num_end = num_begin + token_buffer.size();
|
||||
|
||||
@@ -1723,8 +1776,7 @@ scan_number_done:
|
||||
*/
|
||||
char_int_type get()
|
||||
{
|
||||
++position.chars_read_total;
|
||||
++position.chars_read_current_line;
|
||||
advance_position();
|
||||
|
||||
if (next_unget)
|
||||
{
|
||||
@@ -1736,6 +1788,23 @@ scan_number_done:
|
||||
current = ia.get_character();
|
||||
}
|
||||
|
||||
return track_after_read();
|
||||
}
|
||||
|
||||
/// shared head of get() / get_ignoring_pending_unget(): bump the
|
||||
/// per-character position counters (line-count-on-'\n' bookkeeping is
|
||||
/// handled afterwards, in track_after_read(), once `current` is known)
|
||||
void advance_position() noexcept
|
||||
{
|
||||
++position.chars_read_total;
|
||||
++position.chars_read_current_line;
|
||||
}
|
||||
|
||||
/// shared tail of get() / get_ignoring_pending_unget(): capture the
|
||||
/// character for error messages (if needed) and update line/column
|
||||
/// bookkeeping for the character now in `current`
|
||||
char_int_type track_after_read()
|
||||
{
|
||||
// seekable adapters reconstruct the token lazily on error (see
|
||||
// get_token_string), so the eager per-character copy is skipped
|
||||
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
||||
@@ -1752,6 +1821,29 @@ scan_number_done:
|
||||
return current;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief like get(), but for call sites that can prove no unget() is pending
|
||||
|
||||
get() has to check the `next_unget` flag on every call, because a
|
||||
previous token may have ended with unget() (e.g. scan_number() always
|
||||
ungets the character that terminated the number, so the next call to
|
||||
scan() can see it again). skip_whitespace() reads that first,
|
||||
possibly-ungotten character via a plain get(), but every further
|
||||
character it reads is guaranteed to be a fresh read: nothing between
|
||||
those calls invokes unget(). This variant skips the (otherwise always
|
||||
false) next_unget branch for those calls; it is not a general
|
||||
replacement for get().
|
||||
*/
|
||||
char_int_type get_ignoring_pending_unget()
|
||||
{
|
||||
JSON_ASSERT(!next_unget);
|
||||
|
||||
advance_position();
|
||||
current = ia.get_character();
|
||||
|
||||
return track_after_read();
|
||||
}
|
||||
|
||||
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
||||
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
||||
|
||||
@@ -1953,13 +2045,37 @@ scan_number_done:
|
||||
return true;
|
||||
}
|
||||
|
||||
/// whether `current` is one of the four JSON whitespace characters
|
||||
bool current_is_whitespace() const noexcept
|
||||
{
|
||||
return current == ' ' || current == '\t' || current == '\n' || current == '\r';
|
||||
}
|
||||
|
||||
void skip_whitespace()
|
||||
{
|
||||
// the first character may be a pending unget() left over from the
|
||||
// previous token (see get_ignoring_pending_unget()); every
|
||||
// subsequent character read by this loop is guaranteed fresh, since
|
||||
// nothing below calls unget()
|
||||
get();
|
||||
|
||||
if (!current_is_whitespace())
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
// this is written as an if-guarded do-while (rather than a plain
|
||||
// while loop) because that shape is what lets both GCC and Clang
|
||||
// keep the input adapter's read pointer in a register across
|
||||
// iterations; the equivalent while-loop measurably defeated that
|
||||
// optimization in testing, turning long whitespace runs (e.g. the
|
||||
// indentation of pretty-printed JSON) from a register-only loop
|
||||
// into one that reloads the pointer from memory every character
|
||||
do
|
||||
{
|
||||
get();
|
||||
get_ignoring_pending_unget();
|
||||
}
|
||||
while (current == ' ' || current == '\t' || current == '\n' || current == '\r');
|
||||
while (current_is_whitespace());
|
||||
}
|
||||
|
||||
token_type scan()
|
||||
@@ -2099,6 +2215,13 @@ scan_number_done:
|
||||
const char_int_type decimal_point_char = '.';
|
||||
/// the position of the decimal point in the input
|
||||
std::size_t decimal_point_position = std::string::npos;
|
||||
|
||||
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
|
||||
/// token classification and never looks at the converted numeric value;
|
||||
/// when set, scan_number() may skip strtoull()/strtoll() for
|
||||
/// value_unsigned/value_integer tokens whose digit count guarantees they
|
||||
/// fit into 64 bits (see scan_number())
|
||||
const bool discard_number_values = false;
|
||||
};
|
||||
|
||||
} // namespace detail
|
||||
|
||||
@@ -72,9 +72,10 @@ class parser
|
||||
parser_callback_t<BasicJsonType> cb = nullptr,
|
||||
const bool allow_exceptions_ = true,
|
||||
const bool ignore_comments = false,
|
||||
const bool ignore_trailing_commas_ = false)
|
||||
const bool ignore_trailing_commas_ = false,
|
||||
const bool discard_number_values_ = false)
|
||||
: callback(std::move(cb))
|
||||
, m_lexer(std::move(adapter), ignore_comments)
|
||||
, m_lexer(std::move(adapter), ignore_comments, discard_number_values_)
|
||||
, allow_exceptions(allow_exceptions_)
|
||||
, ignore_trailing_commas(ignore_trailing_commas_)
|
||||
{
|
||||
|
||||
Reference in New Issue
Block a user