Compare commits

..
Author SHA1 Message Date
Michiel van SlobbeandMichiel van Slobbe 5ecb704f6b Speedup; check for the expected separator before the lexer's token switch (#5592)
* Check for the expected separator before the lexer's token switch

After a key the parser expects ':', after a value usually ','. Test for
that character first instead of going through scan()'s switch, which
compiles to an indirect jump. Any other character takes the old path,
so tokens and error messages are unchanged.

Parsing 6.3% faster with GCC 15.2 and 2.7% with Clang 22.1 (geomean of
the ParseString, ParseFile and ParseIndented benchmarks).

Signed-off-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com>

* Improvement: address PR comments

Signed-off-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com>

* fix: address comments

Signed-off-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com>

* Fix clang-tidy bugprone-signed-char-misuse in scan_expecting

Convert the expected separator through unsigned char before storing it as
char_int_type. The generated code is unchanged.

Signed-off-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com>

* Use raw string literals in the separator comment tests

Signed-off-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com>

---------

Signed-off-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com>
Co-authored-by: Michiel van Slobbe <michiel.van.slobbe@gmail.com>
2026-10-06 08:36:40 +02:00
8 changed files with 232 additions and 141 deletions

No files matched your search

@@ -131,7 +131,6 @@ The library maps CBOR types to JSON value types as follows:
| Byte string | binary | 0x59 |
| Byte string | binary | 0x5A |
| Byte string | binary | 0x5B |
| Byte string | binary | 0x5F |
| UTF-8 string | string | 0x60..0x77 |
| UTF-8 string | string | 0x78 |
| UTF-8 string | string | 0x79 |
@@ -157,9 +156,6 @@ The library maps CBOR types to JSON value types as follows:
| Single-Precision Float | number_float | 0xFA |
| Double-Precision Float | number_float | 0xFB |
Indefinite-length UTF-8 strings (0x7F) and byte strings (0x5F) are supported. Each chunk must be a definite-length
string of the same major type, as required by [RFC 8949, Section 3.2.3](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.2.3).
!!! warning "Incomplete mapping"
The mapping is **incomplete** in the sense that not all CBOR types can be converted to a JSON value. The following CBOR types are not supported and will yield parse errors:
+38 -48
View File
@@ -1072,20 +1072,6 @@ class binary_reader
}
}
/*!
@brief reports a nested indefinite-length CBOR string or byte array
@param[in] type_name name of the rejected string type
@param[in] context parsing context for the error message
@return whether the SAX consumer accepts the parse error
*/
bool cbor_indefinite_string_error(const char* type_name, const char* context)
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("indefinite-length ", type_name,
" is not allowed inside indefinite-length ", type_name, "; last byte: 0x", last_token), context), nullptr));
}
/*!
@brief reads a definite-length CBOR string
@@ -1095,13 +1081,12 @@ class binary_reader
into the same string.
@param[out] result string the bytes are appended to
@param[in] inside_indefinite whether the bytes belong to an indefinite-length string
@return whether string creation completed
@pre @a current is not EOF
*/
bool get_cbor_string_chunk(string_t& result, const bool inside_indefinite)
bool get_cbor_string_chunk(string_t& result)
{
switch (current)
{
@@ -1162,7 +1147,7 @@ class binary_reader
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("expected length specification (0x60-0x7B)", inside_indefinite ? "" : " or indefinite string type (0x7F)", "; last byte: 0x", last_token), "string"), nullptr));
exception_message(concat("expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x", last_token), "string"), nullptr));
}
}
}
@@ -1180,9 +1165,13 @@ class binary_reader
*/
bool get_cbor_string(string_t& result, const char* context = "string")
{
// read chunks iteratively, but reject a second indefinite-length
// level as required by RFC 8949, Section 3.2.3
bool indefinite = false;
// number of indefinite-length strings that have been opened and not
// closed yet. RFC 8949, Section 3.2.3 does not permit nesting them,
// but this reader has always accepted it, so the open levels are
// counted instead of recursed through, which overflowed the stack for
// an input of repeated 0x7F bytes (see #5104). Every chunk is appended
// to the same result, so no per-level state is needed.
std::size_t open = 0;
while (true)
{
@@ -1193,28 +1182,29 @@ class binary_reader
if (current == 0x7F) // UTF-8 string (indefinite length)
{
if (JSON_HEDLEY_UNLIKELY(indefinite))
{
return cbor_indefinite_string_error("string", "string");
}
indefinite = true;
++open;
get();
continue;
}
// a break marker closes the indefinite-length string; outside
// of one it falls through to the error below
if (indefinite && current == 0xFF)
// a break marker closes the innermost indefinite-length string;
// outside of one it is not a string and falls through to the error
if (open != 0 && current == 0xFF)
{
return check_string_utf8(result, context);
if (--open == 0)
{
return check_string_utf8(result, context);
}
get();
continue;
}
if (JSON_HEDLEY_UNLIKELY(!get_cbor_string_chunk(result, indefinite)))
if (JSON_HEDLEY_UNLIKELY(!get_cbor_string_chunk(result)))
{
return false;
}
if (!indefinite)
if (open == 0)
{
return check_string_utf8(result, context);
}
@@ -1306,13 +1296,12 @@ class binary_reader
read into the same byte array.
@param[out] result byte array the bytes are appended to
@param[in] inside_indefinite whether the bytes belong to an indefinite-length string
@return whether byte array creation completed
@pre @a current is not EOF
*/
bool get_cbor_binary_chunk(binary_t& result, const bool inside_indefinite)
bool get_cbor_binary_chunk(binary_t& result)
{
switch (current)
{
@@ -1377,7 +1366,7 @@ class binary_reader
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("expected length specification (0x40-0x5B)", inside_indefinite ? "" : " or indefinite binary array type (0x5F)", "; last byte: 0x", last_token), "binary"), nullptr));
exception_message(concat("expected length specification (0x40-0x5B) or indefinite binary array type (0x5F); last byte: 0x", last_token), "binary"), nullptr));
}
}
}
@@ -1395,9 +1384,9 @@ class binary_reader
*/
bool get_cbor_binary(binary_t& result)
{
// read chunks iteratively, but reject a second indefinite-length
// level as required by RFC 8949, Section 3.2.3
bool indefinite = false;
// the open indefinite-length byte arrays are counted rather than
// recursed through, for the reason given in @ref get_cbor_string
std::size_t open = 0;
while (true)
{
@@ -1408,28 +1397,29 @@ class binary_reader
if (current == 0x5F) // Binary data (indefinite length)
{
if (JSON_HEDLEY_UNLIKELY(indefinite))
{
return cbor_indefinite_string_error("binary array", "binary");
}
indefinite = true;
++open;
get();
continue;
}
// a break marker closes the indefinite-length string; outside
// of one it falls through to the error below
if (indefinite && current == 0xFF)
// a break marker closes the innermost indefinite-length byte
// array; outside of one it falls through to the error below
if (open != 0 && current == 0xFF)
{
return true;
if (--open == 0)
{
return true;
}
get();
continue;
}
if (JSON_HEDLEY_UNLIKELY(!get_cbor_binary_chunk(result, indefinite)))
if (JSON_HEDLEY_UNLIKELY(!get_cbor_binary_chunk(result)))
{
return false;
}
if (!indefinite)
if (open == 0)
{
return true;
}
+33 -1
View File
@@ -2021,6 +2021,39 @@ scan_number_done:
// read the next character and ignore whitespace
skip_whitespace();
return scan_after_whitespace();
}
/*!
@brief scan the next token when the caller expects a separator (':' or
',') most of the time
After an object key the next token is almost always ':', after a value
inside an object or array almost always ','. Testing for that character
first is a compare and a well-predicted branch, where the switch in
scan_after_whitespace() is an indirect jump through a table. Anything else
goes through the switch, so the result is the same as scan()'s.
May only be called after scan() has run once (the BOM check is skipped).
*/
token_type scan_expecting(token_type expected_type)
{
JSON_ASSERT(expected_type == token_type::name_separator || expected_type == token_type::value_separator);
JSON_ASSERT(position.chars_read_total > 0);
const char_int_type expected_char = static_cast<unsigned char>((expected_type == token_type::name_separator) ? ':' : ',');
skip_whitespace();
if (JSON_HEDLEY_LIKELY(current == expected_char))
{
return expected_type;
}
return scan_after_whitespace();
}
private:
/// the part of scan() after the leading whitespace: skip comments and
/// scan the token that starts with current
token_type scan_after_whitespace()
{
// ignore comments
while (ignore_comments && current == '/')
{
@@ -2100,7 +2133,6 @@ scan_number_done:
}
}
private:
/// input adapter
InputAdapterType ia;
+11 -4
View File
@@ -260,7 +260,7 @@ class parser
}
// parse separator (:)
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator)))
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
@@ -423,7 +423,7 @@ class parser
{
// comma -> next value
// or end of array (ignore_trailing_commas = true)
if (get_token() == token_type::value_separator)
if (get_token_expecting(token_type::value_separator))
{
// parse a new value
get_token();
@@ -463,7 +463,7 @@ class parser
// comma -> next value
// or end of object (ignore_trailing_commas = true)
if (get_token() == token_type::value_separator)
if (get_token_expecting(token_type::value_separator))
{
get_token();
@@ -484,7 +484,7 @@ class parser
}
// parse separator (:)
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator)))
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
@@ -528,6 +528,13 @@ class parser
return last_token = m_lexer.scan();
}
/// get next token from lexer; true if it is the separator @a expected_type
/// (name_separator or value_separator), which it usually is
bool get_token_expecting(token_type expected_type)
{
return (last_token = m_lexer.scan_expecting(expected_type)) == expected_type;
}
std::string exception_message(const token_type expected, const std::string& context)
{
std::string error_msg = "syntax error ";
+82 -53
View File
@@ -12313,6 +12313,39 @@ scan_number_done:
// read the next character and ignore whitespace
skip_whitespace();
return scan_after_whitespace();
}
/*!
@brief scan the next token when the caller expects a separator (':' or
',') most of the time
After an object key the next token is almost always ':', after a value
inside an object or array almost always ','. Testing for that character
first is a compare and a well-predicted branch, where the switch in
scan_after_whitespace() is an indirect jump through a table. Anything else
goes through the switch, so the result is the same as scan()'s.
May only be called after scan() has run once (the BOM check is skipped).
*/
token_type scan_expecting(token_type expected_type)
{
JSON_ASSERT(expected_type == token_type::name_separator || expected_type == token_type::value_separator);
JSON_ASSERT(position.chars_read_total > 0);
const char_int_type expected_char = static_cast<unsigned char>((expected_type == token_type::name_separator) ? ':' : ',');
skip_whitespace();
if (JSON_HEDLEY_LIKELY(current == expected_char))
{
return expected_type;
}
return scan_after_whitespace();
}
private:
/// the part of scan() after the leading whitespace: skip comments and
/// scan the token that starts with current
token_type scan_after_whitespace()
{
// ignore comments
while (ignore_comments && current == '/')
{
@@ -12392,7 +12425,6 @@ scan_number_done:
}
}
private:
/// input adapter
InputAdapterType ia;
@@ -14789,20 +14821,6 @@ class binary_reader
}
}
/*!
@brief reports a nested indefinite-length CBOR string or byte array
@param[in] type_name name of the rejected string type
@param[in] context parsing context for the error message
@return whether the SAX consumer accepts the parse error
*/
bool cbor_indefinite_string_error(const char* type_name, const char* context)
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("indefinite-length ", type_name,
" is not allowed inside indefinite-length ", type_name, "; last byte: 0x", last_token), context), nullptr));
}
/*!
@brief reads a definite-length CBOR string
@@ -14812,13 +14830,12 @@ class binary_reader
into the same string.
@param[out] result string the bytes are appended to
@param[in] inside_indefinite whether the bytes belong to an indefinite-length string
@return whether string creation completed
@pre @a current is not EOF
*/
bool get_cbor_string_chunk(string_t& result, const bool inside_indefinite)
bool get_cbor_string_chunk(string_t& result)
{
switch (current)
{
@@ -14879,7 +14896,7 @@ class binary_reader
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("expected length specification (0x60-0x7B)", inside_indefinite ? "" : " or indefinite string type (0x7F)", "; last byte: 0x", last_token), "string"), nullptr));
exception_message(concat("expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x", last_token), "string"), nullptr));
}
}
}
@@ -14897,9 +14914,13 @@ class binary_reader
*/
bool get_cbor_string(string_t& result, const char* context = "string")
{
// read chunks iteratively, but reject a second indefinite-length
// level as required by RFC 8949, Section 3.2.3
bool indefinite = false;
// number of indefinite-length strings that have been opened and not
// closed yet. RFC 8949, Section 3.2.3 does not permit nesting them,
// but this reader has always accepted it, so the open levels are
// counted instead of recursed through, which overflowed the stack for
// an input of repeated 0x7F bytes (see #5104). Every chunk is appended
// to the same result, so no per-level state is needed.
std::size_t open = 0;
while (true)
{
@@ -14910,28 +14931,29 @@ class binary_reader
if (current == 0x7F) // UTF-8 string (indefinite length)
{
if (JSON_HEDLEY_UNLIKELY(indefinite))
{
return cbor_indefinite_string_error("string", "string");
}
indefinite = true;
++open;
get();
continue;
}
// a break marker closes the indefinite-length string; outside
// of one it falls through to the error below
if (indefinite && current == 0xFF)
// a break marker closes the innermost indefinite-length string;
// outside of one it is not a string and falls through to the error
if (open != 0 && current == 0xFF)
{
return check_string_utf8(result, context);
if (--open == 0)
{
return check_string_utf8(result, context);
}
get();
continue;
}
if (JSON_HEDLEY_UNLIKELY(!get_cbor_string_chunk(result, indefinite)))
if (JSON_HEDLEY_UNLIKELY(!get_cbor_string_chunk(result)))
{
return false;
}
if (!indefinite)
if (open == 0)
{
return check_string_utf8(result, context);
}
@@ -15023,13 +15045,12 @@ class binary_reader
read into the same byte array.
@param[out] result byte array the bytes are appended to
@param[in] inside_indefinite whether the bytes belong to an indefinite-length string
@return whether byte array creation completed
@pre @a current is not EOF
*/
bool get_cbor_binary_chunk(binary_t& result, const bool inside_indefinite)
bool get_cbor_binary_chunk(binary_t& result)
{
switch (current)
{
@@ -15094,7 +15115,7 @@ class binary_reader
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("expected length specification (0x40-0x5B)", inside_indefinite ? "" : " or indefinite binary array type (0x5F)", "; last byte: 0x", last_token), "binary"), nullptr));
exception_message(concat("expected length specification (0x40-0x5B) or indefinite binary array type (0x5F); last byte: 0x", last_token), "binary"), nullptr));
}
}
}
@@ -15112,9 +15133,9 @@ class binary_reader
*/
bool get_cbor_binary(binary_t& result)
{
// read chunks iteratively, but reject a second indefinite-length
// level as required by RFC 8949, Section 3.2.3
bool indefinite = false;
// the open indefinite-length byte arrays are counted rather than
// recursed through, for the reason given in @ref get_cbor_string
std::size_t open = 0;
while (true)
{
@@ -15125,28 +15146,29 @@ class binary_reader
if (current == 0x5F) // Binary data (indefinite length)
{
if (JSON_HEDLEY_UNLIKELY(indefinite))
{
return cbor_indefinite_string_error("binary array", "binary");
}
indefinite = true;
++open;
get();
continue;
}
// a break marker closes the indefinite-length string; outside
// of one it falls through to the error below
if (indefinite && current == 0xFF)
// a break marker closes the innermost indefinite-length byte
// array; outside of one it falls through to the error below
if (open != 0 && current == 0xFF)
{
return true;
if (--open == 0)
{
return true;
}
get();
continue;
}
if (JSON_HEDLEY_UNLIKELY(!get_cbor_binary_chunk(result, indefinite)))
if (JSON_HEDLEY_UNLIKELY(!get_cbor_binary_chunk(result)))
{
return false;
}
if (!indefinite)
if (open == 0)
{
return true;
}
@@ -18432,7 +18454,7 @@ class parser
}
// parse separator (:)
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator)))
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
@@ -18595,7 +18617,7 @@ class parser
{
// comma -> next value
// or end of array (ignore_trailing_commas = true)
if (get_token() == token_type::value_separator)
if (get_token_expecting(token_type::value_separator))
{
// parse a new value
get_token();
@@ -18635,7 +18657,7 @@ class parser
// comma -> next value
// or end of object (ignore_trailing_commas = true)
if (get_token() == token_type::value_separator)
if (get_token_expecting(token_type::value_separator))
{
get_token();
@@ -18656,7 +18678,7 @@ class parser
}
// parse separator (:)
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator)))
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
@@ -18700,6 +18722,13 @@ class parser
return last_token = m_lexer.scan();
}
/// get next token from lexer; true if it is the separator @a expected_type
/// (name_separator or value_separator), which it usually is
bool get_token_expecting(token_type expected_type)
{
return (last_token = m_lexer.scan_expecting(expected_type)) == expected_type;
}
std::string exception_message(const token_type expected, const std::string& context)
{
std::string error_msg = "syntax error ";
+16 -23
View File
@@ -1699,7 +1699,7 @@ TEST_CASE("CBOR")
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0x61, 0X61})), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xBF, 0x61, 0X61})), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x5F})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing CBOR binary: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x5F, 0x00})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR binary: expected length specification (0x40-0x5B); last byte: 0x00", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x5F, 0x00})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR binary: expected length specification (0x40-0x5B) or indefinite binary array type (0x5F); last byte: 0x00", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x41})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing CBOR binary: unexpected end of input", json::parse_error&);
CHECK(json::from_cbor(std::vector<uint8_t>({0x18}), true, false).is_discarded());
@@ -2305,21 +2305,22 @@ TEST_CASE("CBOR indefinite-length strings do not recurse per chunk")
{
// Reading an indefinite-length string or byte array used to call itself
// once per chunk, so a payload of repeated 0x7F (or 0x5F) bytes exhausted
// the call stack before any of the input was rejected. Nested indefinite
// chunks are now rejected at the second byte, without recursing.
// the call stack before any of the input was rejected. The open levels are
// counted now, and the levels below prove the reader still reads the same
// values and reports the same errors at the same byte offsets.
json _;
SECTION("nested levels are rejected, not crashed on")
SECTION("many open levels are reported, not crashed on")
{
const std::vector<uint8_t> input(200000, 0x7F);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: indefinite-length string is not allowed inside indefinite-length string; last byte: 0x7F", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.110] parse error at byte 200001: syntax error while parsing CBOR string: unexpected end of input", json::parse_error&);
CHECK(json::from_cbor(input, true, false).is_discarded());
}
SECTION("nested levels are rejected, not crashed on (binary)")
SECTION("many open levels are reported, not crashed on (binary)")
{
const std::vector<uint8_t> input(200000, 0x5F);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR binary: indefinite-length binary array is not allowed inside indefinite-length binary array; last byte: 0x5F", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.parse_error.110] parse error at byte 200001: syntax error while parsing CBOR binary: unexpected end of input", json::parse_error&);
CHECK(json::from_cbor(input, true, false).is_discarded());
}
@@ -2327,22 +2328,22 @@ TEST_CASE("CBOR indefinite-length strings do not recurse per chunk")
{
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0xFF})) == json(""));
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0x61, 0x61, 0xFF})) == json("a"));
// empty and nonempty definite-length chunks concatenate in order
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0x61, 'a', 0x60, 0x61, 'b', 0x61, 'c', 0xFF})) == json("abc"));
// nested indefinite-length strings are concatenated across levels
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0x7F, 0x61, 0x61, 0xFF, 0x61, 0x62, 0xFF})) == json("ab"));
CHECK(json::from_cbor(std::vector<uint8_t>({0x7F, 0x7F, 0x7F, 0x61, 0x7A, 0xFF, 0xFF, 0xFF})) == json("z"));
CHECK(json::from_cbor(std::vector<uint8_t>({0xA1, 0x7F, 0x61, 0x61, 0xFF, 0x01})) == json({{"a", 1}}));
}
SECTION("chunks are still concatenated (binary)")
{
CHECK(json::from_cbor(std::vector<uint8_t>({0x5F, 0x41, 0x61, 0xFF})) == json::binary({0x61}));
CHECK(json::from_cbor(std::vector<uint8_t>({0x5F, 0xFF})) == json::binary({}));
CHECK(json::from_cbor(std::vector<uint8_t>({0x5F, 0x41, 0x61, 0x40, 0x41, 0x62, 0x41, 0x63, 0xFF})) == json::binary({0x61, 0x62, 0x63}));
CHECK(json::from_cbor(std::vector<uint8_t>({0x5F, 0x5F, 0x41, 0x61, 0xFF, 0x41, 0x62, 0xFF})) == json::binary({0x61, 0x62}));
}
SECTION("a chunk that is not a string is still rejected")
{
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7F, 0x00})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B); last byte: 0x00", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x5F, 0x00})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR binary: expected length specification (0x40-0x5B); last byte: 0x00", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7F, 0x7F, 0x00})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x00", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x5F, 0x5F, 0x00})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR binary: expected length specification (0x40-0x5B) or indefinite binary array type (0x5F); last byte: 0x00", json::parse_error&);
}
SECTION("a break marker outside an indefinite-length string is not a string")
@@ -2895,17 +2896,9 @@ TEST_CASE("examples from RFC 8949 Appendix A")
{
const auto packed = utils::read_binary_file(TEST_DATA_DIRECTORY "/binary_data/cbor_binary.cbor");
json j;
// the fixture's tail contains nested indefinite-length byte strings.
CHECK_THROWS_WITH_AS(j = json::from_cbor(packed), "[json.exception.parse_error.113] parse error at byte 513: syntax error while parsing CBOR binary: indefinite-length binary array is not allowed inside indefinite-length binary array; last byte: 0x5F", json::parse_error&);
CHECK_NOTHROW(j = json::from_cbor(packed));
// keep the byte-for-byte decoding check for its valid prefix: the first
// 512 encoded bytes contain 468 payload bytes in definite-length chunks.
auto valid_prefix = packed;
valid_prefix.resize(512);
valid_prefix.push_back(0xFF);
auto expected = utils::read_binary_file(TEST_DATA_DIRECTORY "/binary_data/cbor_binary.out");
expected.resize(468);
CHECK_NOTHROW(j = json::from_cbor(valid_prefix));
const auto expected = utils::read_binary_file(TEST_DATA_DIRECTORY "/binary_data/cbor_binary.out");
CHECK(j == json::binary(expected));
// 0xd8
+52
View File
@@ -2317,6 +2317,58 @@ TEST_CASE("parser class")
#endif
}
SECTION("comments before separators")
{
// The parser first checks for the expected ':' or ',' and only then
// falls back to the full token switch, which skips comments. A comment
// directly before a separator takes that fallback.
json _;
SECTION("ignored")
{
const std::vector<std::pair<std::string, json>> inputs =
{
{"{\"a\" /* c */ : 1}", {{"a", 1}}},
{"{\"a\" // c\n: 1}", {{"a", 1}}},
{R"({"a": 1, "b" /* c */ : 2})", {{"a", 1}, {"b", 2}}},
{R"({"a": 1 /* c */ , "b": 2})", {{"a", 1}, {"b", 2}}},
{"{\"a\": 1 // c\n, \"b\": 2}", {{"a", 1}, {"b", 2}}},
{"[1 /* c */ , 2]", {1, 2}},
{"[1 // c\n, 2]", {1, 2}},
{"{\"a\" /* c */ /* d */ : [1 // c\n , 2 /**/ ] /**/ , \"b\" : 3}", {{"a", {1, 2}}, {"b", 3}}}
};
for (const auto& input : inputs)
{
CAPTURE(input.first)
CHECK(json::parse(input.first, nullptr, true, true) == input.second);
CHECK(json::accept(input.first, true));
}
}
SECTION("ignored, with trailing commas")
{
CHECK(json::parse(std::string("[1 /* c */ , ]"), nullptr, true, true, true) == json({1}));
CHECK(json::parse(std::string("{\"a\": 1 /* c */ , }"), nullptr, true, true, true) == json({{"a", 1}}));
CHECK_THROWS_WITH_AS(_ = json::parse(std::string("[1 /* c */ , ]"), nullptr, true, true),
"[json.exception.parse_error.101] parse error at line 1, column 14: syntax error while parsing value - unexpected ']'; expected '[', '{', or a literal", json::parse_error);
CHECK_THROWS_WITH_AS(_ = json::parse(std::string("{\"a\": 1 /* c */ , }"), nullptr, true, true),
"[json.exception.parse_error.101] parse error at line 1, column 19: syntax error while parsing object key - unexpected '}'; expected string literal", json::parse_error);
}
SECTION("not ignored")
{
CHECK_THROWS_WITH_AS(_ = json::parse(std::string("{\"a\" /* c */ : 1}")),
"[json.exception.parse_error.101] parse error at line 1, column 6: syntax error while parsing object separator - invalid literal; last read: '\"a\" /'; expected ':'", json::parse_error);
CHECK_THROWS_WITH_AS(_ = json::parse(std::string("{\"a\": 1, \"b\" /* c */ : 2}")),
"[json.exception.parse_error.101] parse error at line 1, column 14: syntax error while parsing object separator - invalid literal; last read: '\"b\" /'; expected ':'", json::parse_error);
CHECK_THROWS_WITH_AS(_ = json::parse(std::string("{\"a\": 1 /* c */ , \"b\": 2}")),
"[json.exception.parse_error.101] parse error at line 1, column 9: syntax error while parsing object - invalid literal; last read: '1 /'; expected '}'", json::parse_error);
CHECK_THROWS_WITH_AS(_ = json::parse(std::string("[1 /* c */ , 2]")),
"[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing array - invalid literal; last read: '1 /'; expected ']'", json::parse_error);
CHECK(!json::accept(std::string("[1 /* c */ , 2]")));
}
}
#if JSON_DIAGNOSTIC_POSITIONS
// Macro for all test cases for start_pos and end_pos
#define SETUP_TESTCASES() \
-8
View File
@@ -920,12 +920,4 @@ TEST_CASE("regression test #5476 - array type without reserve()")
}
}
TEST_CASE("issue #5317 - nested indefinite-length CBOR string chunks are rejected")
{
json _;
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<std::uint8_t>({0x7F, 0x7F, 0x61, 0x61, 0xFF, 0xFF})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: indefinite-length string is not allowed inside indefinite-length string; last byte: 0x7F", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<std::uint8_t>({0x5F, 0x5F, 0x41, 0x61, 0xFF, 0xFF})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR binary: indefinite-length binary array is not allowed inside indefinite-length binary array; last byte: 0x5F", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<std::uint8_t>({0xA1, 0x7F, 0x7F, 0xFF, 0xFF, 0x01})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: indefinite-length string is not allowed inside indefinite-length string; last byte: 0x7F", json::parse_error&);
}
DOCTEST_CLANG_SUPPRESS_WARNING_POP