mirror of
https://github.com/nlohmann/json.git
synced 2026-09-07 00:37:58 +00:00
Speed up whitespace skipping in the lexer
lexer::skip_whitespace() called get() for every whitespace byte, and get() checks the (almost always false, once past the first character) next_unget flag on every call. skip_whitespace() now reads its first character with get() (needed to honor a pending unget() left over from finishing the previous token, e.g. scan_number() always ungets the character that terminated the number) and every further whitespace character with a new get_ignoring_pending_unget() variant that skips that branch, since nothing in the loop calls unget(). This is a narrower fix than the full contiguous-buffer bulk-skip suggested in the issue (scan a run of whitespace directly in the adapter's buffer and update position counters once per run). That approach depends on bulk-scan adapter infrastructure (supports_bulk_scan/bulk_data()/bulk_skip()) introduced by the open, unmerged parser-performance PR #5283, which this change intentionally does not depend on or replicate. Building new bulk-scan adapter infrastructure from scratch was judged out of scope/riskier than warranted here, so this change is limited to the safe, always-correct improvement of removing redundant per-character bookkeeping from the existing byte-at-a-time loop; full bulk-skipping is left as future work once #5283 (or equivalent adapter support) lands. Line/column/byte-offset bookkeeping is untouched and verified bit-for-bit identical before and after this change, including for pretty-printed (dump(4)) input with embedded newlines. Fixes #5412 Stacked on top of the PR for #5411 (branch issue-5411-lexer-skip-conversion). Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -1459,6 +1459,14 @@ scan_number_done:
|
||||
current = ia.get_character();
|
||||
}
|
||||
|
||||
return track_after_read();
|
||||
}
|
||||
|
||||
/// shared tail of get() / get_ignoring_pending_unget(): capture the
|
||||
/// character for error messages (if needed) and update line/column
|
||||
/// bookkeeping for the character now in `current`
|
||||
char_int_type track_after_read()
|
||||
{
|
||||
// seekable adapters reconstruct the token lazily on error (see
|
||||
// get_token_string), so the eager per-character copy is skipped
|
||||
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
||||
@@ -1472,6 +1480,30 @@ scan_number_done:
|
||||
return current;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief like get(), but for call sites that can prove no unget() is pending
|
||||
|
||||
get() has to check the `next_unget` flag on every call, because a
|
||||
previous token may have ended with unget() (e.g. scan_number() always
|
||||
ungets the character that terminated the number, so the next call to
|
||||
scan() can see it again). skip_whitespace() reads that first,
|
||||
possibly-ungotten character via a plain get(), but every further
|
||||
character it reads is guaranteed to be a fresh read: nothing between
|
||||
those calls invokes unget(). This variant skips the (otherwise always
|
||||
false) next_unget branch for those calls; it is not a general
|
||||
replacement for get().
|
||||
*/
|
||||
char_int_type get_ignoring_pending_unget()
|
||||
{
|
||||
JSON_ASSERT(!next_unget);
|
||||
|
||||
++position.chars_read_total;
|
||||
++position.chars_read_current_line;
|
||||
current = ia.get_character();
|
||||
|
||||
return track_after_read();
|
||||
}
|
||||
|
||||
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
||||
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
||||
|
||||
@@ -1667,11 +1699,16 @@ scan_number_done:
|
||||
|
||||
void skip_whitespace()
|
||||
{
|
||||
do
|
||||
// the first character may be a pending unget() left over from the
|
||||
// previous token (see get_ignoring_pending_unget()); every
|
||||
// subsequent character read by this loop is guaranteed fresh, since
|
||||
// nothing below calls unget()
|
||||
get();
|
||||
|
||||
while (current == ' ' || current == '\t' || current == '\n' || current == '\r')
|
||||
{
|
||||
get();
|
||||
get_ignoring_pending_unget();
|
||||
}
|
||||
while (current == ' ' || current == '\t' || current == '\n' || current == '\r');
|
||||
}
|
||||
|
||||
token_type scan()
|
||||
|
||||
@@ -9242,6 +9242,14 @@ scan_number_done:
|
||||
current = ia.get_character();
|
||||
}
|
||||
|
||||
return track_after_read();
|
||||
}
|
||||
|
||||
/// shared tail of get() / get_ignoring_pending_unget(): capture the
|
||||
/// character for error messages (if needed) and update line/column
|
||||
/// bookkeeping for the character now in `current`
|
||||
char_int_type track_after_read()
|
||||
{
|
||||
// seekable adapters reconstruct the token lazily on error (see
|
||||
// get_token_string), so the eager per-character copy is skipped
|
||||
capture_char(std::integral_constant<bool, lazy_token_string> {});
|
||||
@@ -9255,6 +9263,30 @@ scan_number_done:
|
||||
return current;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief like get(), but for call sites that can prove no unget() is pending
|
||||
|
||||
get() has to check the `next_unget` flag on every call, because a
|
||||
previous token may have ended with unget() (e.g. scan_number() always
|
||||
ungets the character that terminated the number, so the next call to
|
||||
scan() can see it again). skip_whitespace() reads that first,
|
||||
possibly-ungotten character via a plain get(), but every further
|
||||
character it reads is guaranteed to be a fresh read: nothing between
|
||||
those calls invokes unget(). This variant skips the (otherwise always
|
||||
false) next_unget branch for those calls; it is not a general
|
||||
replacement for get().
|
||||
*/
|
||||
char_int_type get_ignoring_pending_unget()
|
||||
{
|
||||
JSON_ASSERT(!next_unget);
|
||||
|
||||
++position.chars_read_total;
|
||||
++position.chars_read_current_line;
|
||||
current = ia.get_character();
|
||||
|
||||
return track_after_read();
|
||||
}
|
||||
|
||||
/// seekable adapter: nothing to capture, the token is rebuilt on error
|
||||
void capture_char(std::true_type /*lazy*/) const noexcept {}
|
||||
|
||||
@@ -9450,11 +9482,16 @@ scan_number_done:
|
||||
|
||||
void skip_whitespace()
|
||||
{
|
||||
do
|
||||
// the first character may be a pending unget() left over from the
|
||||
// previous token (see get_ignoring_pending_unget()); every
|
||||
// subsequent character read by this loop is guaranteed fresh, since
|
||||
// nothing below calls unget()
|
||||
get();
|
||||
|
||||
while (current == ' ' || current == '\t' || current == '\n' || current == '\r')
|
||||
{
|
||||
get();
|
||||
get_ignoring_pending_unget();
|
||||
}
|
||||
while (current == ' ' || current == '\t' || current == '\n' || current == '\r');
|
||||
}
|
||||
|
||||
token_type scan()
|
||||
|
||||
@@ -1486,6 +1486,62 @@ TEST_CASE("parser class")
|
||||
CHECK(accept_helper("\"\\uD80C\\uFFFF\"") == false);
|
||||
}
|
||||
|
||||
SECTION("issue #5412 - whitespace skipping bookkeeping (compact vs. pretty-printed)")
|
||||
{
|
||||
// lexer::skip_whitespace() reads its first character with get() (to
|
||||
// honor a possibly pending unget() from the previous token) and every
|
||||
// further whitespace character with get_ignoring_pending_unget() (a
|
||||
// get() variant that skips the then-always-false next_unget check).
|
||||
// This must not change the reported byte offset, line, or column of
|
||||
// a syntax error, even when a long run of whitespace containing
|
||||
// multiple newlines is skipped beforehand (as with pretty-printed
|
||||
// input). The expected values below were captured from the
|
||||
// unmodified do-while(get()) loop, so any regression that miscounts
|
||||
// characters or newlines while skipping whitespace changes them.
|
||||
const auto check_error = [](const std::string & input, std::size_t expected_byte,
|
||||
const std::string & expected_what)
|
||||
{
|
||||
CAPTURE(input)
|
||||
try
|
||||
{
|
||||
json _ = json::parse(input);
|
||||
FAIL_CHECK("expected a parse_error, but parsing succeeded");
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
CHECK(e.byte == expected_byte);
|
||||
CHECK(std::string(e.what()) == expected_what);
|
||||
}
|
||||
};
|
||||
|
||||
// a nested document, serialized both compactly and pretty-printed
|
||||
// (dump(4)), each truncated right before the final closing '}' so
|
||||
// that the parser hits EOF after skipping all of the (in the
|
||||
// pretty-printed case, substantial) indentation whitespace
|
||||
const json doc =
|
||||
{
|
||||
{"a", 1},
|
||||
{"b", json::array({true, false, nullptr, "x"})},
|
||||
{"c", json::object({{"d", 3.14}, {"e", json::array({1, 2, 3})}})}
|
||||
};
|
||||
|
||||
const std::string compact = doc.dump();
|
||||
const std::string pretty = doc.dump(4);
|
||||
|
||||
check_error(compact.substr(0, compact.size() - 1), 60,
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 60: syntax error while parsing object - unexpected end of input; expected '}'");
|
||||
check_error(pretty.substr(0, pretty.size() - 1), 193,
|
||||
"[json.exception.parse_error.101] parse error at line 17, column 1: syntax error while parsing object - unexpected end of input; expected '}'");
|
||||
|
||||
// an invalid token appearing after several indented, multi-line
|
||||
// whitespace runs vs. the same document without any of that
|
||||
// whitespace
|
||||
check_error("{\n \"a\": 1,\n \"b\": [\n true,\n false\n ],\n \"c\": @\n}", 70,
|
||||
"[json.exception.parse_error.101] parse error at line 7, column 10: syntax error while parsing value - invalid literal; last read: '\"c\": @'");
|
||||
check_error("{\"a\":1,\"b\":[true,false],\"c\":@}", 29,
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing value - invalid literal; last read: '\"c\":@'");
|
||||
}
|
||||
|
||||
SECTION("tests found by mutate++")
|
||||
{
|
||||
// test case to make sure no comma precedes the first key
|
||||
|
||||
Reference in New Issue
Block a user