mirror of
https://github.com/nlohmann/json.git
synced 2026-10-06 22:47:13 +00:00
Speedup; check for the expected separator before the lexer's token switch (#5592)
* Check for the expected separator before the lexer's token switch After a key the parser expects ':', after a value usually ','. Test for that character first instead of going through scan()'s switch, which compiles to an indirect jump. Any other character takes the old path, so tokens and error messages are unchanged. Parsing 6.3% faster with GCC 15.2 and 2.7% with Clang 22.1 (geomean of the ParseString, ParseFile and ParseIndented benchmarks). Signed-off-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com> * Improvement: address PR comments Signed-off-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com> * fix: address comments Signed-off-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com> * Fix clang-tidy bugprone-signed-char-misuse in scan_expecting Convert the expected separator through unsigned char before storing it as char_int_type. The generated code is unchanged. Signed-off-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com> * Use raw string literals in the separator comment tests Signed-off-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com> --------- Signed-off-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com> Co-authored-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com>
This commit is contained in:
1 parent
0490778fc3
commit
5ecb704f6b
4 files changed
+140
-10
No files matched your search
@@ -12313,6 +12313,39 @@ scan_number_done:
|
||||
// read the next character and ignore whitespace
|
||||
skip_whitespace();
|
||||
|
||||
return scan_after_whitespace();
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief scan the next token when the caller expects a separator (':' or
|
||||
',') most of the time
|
||||
|
||||
After an object key the next token is almost always ':', after a value
|
||||
inside an object or array almost always ','. Testing for that character
|
||||
first is a compare and a well-predicted branch, where the switch in
|
||||
scan_after_whitespace() is an indirect jump through a table. Anything else
|
||||
goes through the switch, so the result is the same as scan()'s.
|
||||
|
||||
May only be called after scan() has run once (the BOM check is skipped).
|
||||
*/
|
||||
token_type scan_expecting(token_type expected_type)
|
||||
{
|
||||
JSON_ASSERT(expected_type == token_type::name_separator || expected_type == token_type::value_separator);
|
||||
JSON_ASSERT(position.chars_read_total > 0);
|
||||
const char_int_type expected_char = static_cast<unsigned char>((expected_type == token_type::name_separator) ? ':' : ',');
|
||||
skip_whitespace();
|
||||
if (JSON_HEDLEY_LIKELY(current == expected_char))
|
||||
{
|
||||
return expected_type;
|
||||
}
|
||||
return scan_after_whitespace();
|
||||
}
|
||||
|
||||
private:
|
||||
/// the part of scan() after the leading whitespace: skip comments and
|
||||
/// scan the token that starts with current
|
||||
token_type scan_after_whitespace()
|
||||
{
|
||||
// ignore comments
|
||||
while (ignore_comments && current == '/')
|
||||
{
|
||||
@@ -12392,7 +12425,6 @@ scan_number_done:
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
/// input adapter
|
||||
InputAdapterType ia;
|
||||
|
||||
@@ -18422,7 +18454,7 @@ class parser
|
||||
}
|
||||
|
||||
// parse separator (:)
|
||||
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
|
||||
if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator)))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
@@ -18585,7 +18617,7 @@ class parser
|
||||
{
|
||||
// comma -> next value
|
||||
// or end of array (ignore_trailing_commas = true)
|
||||
if (get_token() == token_type::value_separator)
|
||||
if (get_token_expecting(token_type::value_separator))
|
||||
{
|
||||
// parse a new value
|
||||
get_token();
|
||||
@@ -18625,7 +18657,7 @@ class parser
|
||||
|
||||
// comma -> next value
|
||||
// or end of object (ignore_trailing_commas = true)
|
||||
if (get_token() == token_type::value_separator)
|
||||
if (get_token_expecting(token_type::value_separator))
|
||||
{
|
||||
get_token();
|
||||
|
||||
@@ -18646,7 +18678,7 @@ class parser
|
||||
}
|
||||
|
||||
// parse separator (:)
|
||||
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
|
||||
if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator)))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
@@ -18690,6 +18722,13 @@ class parser
|
||||
return last_token = m_lexer.scan();
|
||||
}
|
||||
|
||||
/// get next token from lexer; true if it is the separator @a expected_type
|
||||
/// (name_separator or value_separator), which it usually is
|
||||
bool get_token_expecting(token_type expected_type)
|
||||
{
|
||||
return (last_token = m_lexer.scan_expecting(expected_type)) == expected_type;
|
||||
}
|
||||
|
||||
std::string exception_message(const token_type expected, const std::string& context)
|
||||
{
|
||||
std::string error_msg = "syntax error ";
|
||||
|
||||
Reference in new issue
Block a user