Use and extend the detail helpers to remove duplicated code (#5783)

* Use and extend the detail helpers to remove duplicated code

Library:
- binary_reader: format bytes with hex_byte() instead of snprintf
- add throw_type_must_be() for the 20 copies of type_error.302
- binary_reader: add last_byte_error()/unexpected_byte() for the 32
  "parse error at the last read byte" sites (replaces bon8_error)
- json_sax: add check_container_size() for out_of_range.408 and
  diagnostic_positions::set_container_start/_end()
- json_pointer: add throw_no_parent() (405) and throw_unresolved() (404)

Tests:
- unit-class_parser uses the shared utils::SaxCountdown
- move SaxEventLogger (and its ExitAfter* variants) from unit-class_parser
  and unit-deserialization into the new tests/src/test_sax.hpp

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

* Factor out more repeated error paths and test boilerplate

Library:
- add throw_cannot_use_with() for the 34 copies of type_error.304-312
  "cannot use X with Y"
- iter_impl: add throw_cannot_get_value() (invalid_iterator.214)
- parser: add syntax_error() for the 12 parse_error.101 sites
- ordered_map: share the four at() bodies via at_impl()
- json_sax_dom_callback_parser: add pop_container() for end_object()
  and end_array()
- binary_reader: build the two UBJSON/BJData length-type messages with
  concat() and last_byte_error()

Tests:
- move same_value(), the NDEBUG guard, and step 0 (parse without
  exceptions) of the seven fuzzer drivers into tests/src/fuzzer_common.hpp

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

* Name the exception ids and address review comments

Add detail::exception_id, a scoped enum with one named enumerator per
documented exception id, and use it for every id in the library. The
create() functions get an overload for it; the int overloads stay for
user code.

Following the review of #5783: add binary_reader::invalid_byte() and
length_type_error(), basic_json::throw_subscript_wrong_type(), move the
fuzzer includes and the using-declaration into fuzzer_common.hpp, and
rename test_sax.hpp to sax_event_loggers.hpp.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

* Silence -Wweak-vtables for the SAX event loggers

The loggers moved from anonymous namespaces in the test files into
sax_event_loggers.hpp, so clang now warns that their vtables are
emitted in every translation unit.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

* Silence MSVC 2015 C4100 in parser::syntax_error for static SAX::parse_error

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

* Inline invalid_byte() into its call sites

The wrapper only fixed the message string of unexpected_byte(), which is
the same kind of per-argument helper that was declined for
throw_type_must_be(). Call unexpected_byte("invalid byte", ...) directly.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

---------

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann authored and GitHub committed 2026-10-11 08:21:13 +02:00
1 parent f8b47ff6f6
commit 1649d9eda4
26 files changed
+1329 -1391

No files matched your search

+84 -135
View File
@@ -12,7 +12,6 @@
#include <cmath> // ldexp
#include <cstddef> // size_t
#include <cstdint> // uint8_t, uint16_t, uint32_t, uint64_t, uintmax_t
#include <cstdio> // snprintf
#include <cstring> // memcpy
#include <iterator> // back_inserter
#include <limits> // numeric_limits
@@ -196,8 +195,7 @@ class binary_reader
if (JSON_HEDLEY_UNLIKELY(current != char_traits<char_type>::eof()))
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(110, chars_read,
exception_message(concat("expected end of input; last byte: 0x", get_token_string()), "value"), nullptr));
return last_byte_error(exception_id::unexpected_end_of_input, concat("expected end of input; last byte: 0x", get_token_string()), "value");
}
}
@@ -319,8 +317,7 @@ class binary_reader
{
if (JSON_HEDLEY_UNLIKELY(document_size < 0 || static_cast<std::size_t>(document_size) != chars_read - document_start))
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read,
exception_message(concat("document size ", std::to_string(document_size), " does not match the number of bytes read (", std::to_string(chars_read - document_start), ")"), "document"), nullptr));
return last_byte_error(exception_id::unexpected_byte, concat("document size ", std::to_string(document_size), " does not match the number of bytes read (", std::to_string(chars_read - document_start), ")"), "document");
}
return true;
}
@@ -512,9 +509,7 @@ class binary_reader
{
if (JSON_HEDLEY_UNLIKELY(len < 1))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(concat("string length must be at least 1, is ", std::to_string(len)), "string"), nullptr));
return last_byte_error(exception_id::unexpected_byte, concat("string length must be at least 1, is ", std::to_string(len)), "string");
}
if (JSON_HEDLEY_UNLIKELY(!get_string(len - static_cast<NumberType>(1), result)))
@@ -524,10 +519,7 @@ class binary_reader
if (JSON_HEDLEY_UNLIKELY(get() != 0x00))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message("BSON string is not null-terminated",
"string"), nullptr));
return last_byte_error(exception_id::unexpected_byte, "BSON string is not null-terminated", "string");
}
return check_string_utf8(result, "string");
@@ -547,9 +539,7 @@ class binary_reader
{
if (JSON_HEDLEY_UNLIKELY(len < 0))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(concat("byte array length cannot be negative, is ", std::to_string(len)), "binary"), nullptr));
return last_byte_error(exception_id::unexpected_byte, concat("byte array length cannot be negative, is ", std::to_string(len)), "binary");
}
// All BSON binary values have a subtype
@@ -639,11 +629,9 @@ class binary_reader
default: // anything else is not supported (yet)
{
std::array<char, 3> cr{{}};
static_cast<void>((std::snprintf)(cr.data(), cr.size(), "%.2hhX", static_cast<unsigned char>(element_type))); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg)
const std::string cr_str{cr.data()};
const std::string cr_str = hex_byte(static_cast<std::uint8_t>(element_type));
return sax->parse_error(element_type_parse_position, cr_str,
parse_error::create(114, element_type_parse_position, concat("Unsupported BSON record type 0x", cr_str), nullptr));
parse_error::create(exception_id::bson_unsupported_type, element_type_parse_position, concat("Unsupported BSON record type 0x", cr_str), nullptr));
}
}
}
@@ -967,9 +955,7 @@ class binary_reader
{
if (tag_handler == cbor_tag_handler_t::error)
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(concat("invalid byte: 0x", last_token), "value"), nullptr));
return unexpected_byte("invalid byte", "value");
}
// ignore and store: the tag value is already in the head, so
@@ -989,9 +975,7 @@ class binary_reader
{
case cbor_tag_handler_t::error:
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(concat("invalid byte: 0x", last_token), "value"), nullptr));
return unexpected_byte("invalid byte", "value");
}
case cbor_tag_handler_t::ignore:
@@ -1065,9 +1049,7 @@ class binary_reader
default: // anything else (0xFF is handled inside the other types)
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(concat("invalid byte: 0x", last_token), "value"), nullptr));
return unexpected_byte("invalid byte", "value");
}
}
}
@@ -1080,10 +1062,7 @@ class binary_reader
*/
bool cbor_indefinite_string_error(const char* type_name, const char* context)
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("indefinite-length ", type_name,
" is not allowed inside indefinite-length ", type_name, "; last byte: 0x", last_token), context), nullptr));
return last_byte_error(exception_id::invalid_string_or_size, concat("indefinite-length ", type_name, " is not allowed inside indefinite-length ", type_name, "; last byte: 0x", get_token_string()), context);
}
/*!
@@ -1160,9 +1139,7 @@ class binary_reader
default:
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("expected length specification (0x60-0x7B)", inside_indefinite ? "" : " or indefinite string type (0x7F)", "; last byte: 0x", last_token), "string"), nullptr));
return last_byte_error(exception_id::invalid_string_or_size, concat("expected length specification (0x60-0x7B)", inside_indefinite ? "" : " or indefinite string type (0x7F)", "; last byte: 0x", get_token_string()), "string");
}
}
}
@@ -1292,9 +1269,7 @@ class binary_reader
break;
}
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr));
return last_byte_error(exception_id::invalid_string_or_size, concat("only string keys are supported, but found ", found, "; last byte: 0x", get_token_string()), "object key");
}
/*!
@@ -1375,9 +1350,7 @@ class binary_reader
default:
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("expected length specification (0x40-0x5B)", inside_indefinite ? "" : " or indefinite binary array type (0x5F)", "; last byte: 0x", last_token), "binary"), nullptr));
return last_byte_error(exception_id::invalid_string_or_size, concat("expected length specification (0x40-0x5B)", inside_indefinite ? "" : " or indefinite binary array type (0x5F)", "; last byte: 0x", get_token_string()), "binary");
}
}
}
@@ -1523,7 +1496,7 @@ class binary_reader
{
if (JSON_HEDLEY_UNLIKELY(!value_in_range_of<std::size_t>(len) || len == detail::unknown_size()))
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408,
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(exception_id::container_too_large,
exception_message(concat("excessive ", context, " size"), "size"), nullptr));
}
result = conditional_static_cast<std::size_t>(len);
@@ -2011,9 +1984,7 @@ class binary_reader
default: // anything else
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(concat("invalid byte: 0x", last_token), "value"), nullptr));
return unexpected_byte("invalid byte", "value");
}
}
}
@@ -2094,9 +2065,7 @@ class binary_reader
default:
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0x", last_token), "string"), nullptr));
return last_byte_error(exception_id::invalid_string_or_size, concat("expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0x", get_token_string()), "string");
}
}
}
@@ -2188,9 +2157,7 @@ class binary_reader
break;
}
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr));
return last_byte_error(exception_id::invalid_string_or_size, concat("only string keys are supported, but found ", found, "; last byte: 0x", get_token_string()), "object key");
}
/*!
@@ -2499,8 +2466,7 @@ class binary_reader
{
if (JSON_HEDLEY_UNLIKELY(len < 0))
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message("string length must not be negative", "string"), nullptr));
return last_byte_error(exception_id::invalid_string_or_size, "string length must not be negative", "string");
}
return true;
}
@@ -2600,18 +2566,7 @@ class binary_reader
default:
break;
}
auto last_token = get_token_string();
std::string message;
if (input_format != input_format_t::bjdata)
{
message = "expected length type specification (U, i, I, l, L); last byte: 0x" + last_token;
}
else
{
message = "expected length type specification (U, i, u, I, m, l, M, L); last byte: 0x" + last_token;
}
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, exception_message(message, "string"), nullptr));
return length_type_error("", "string");
}
/*!
@@ -2696,12 +2651,11 @@ class binary_reader
}
if (JSON_HEDLEY_UNLIKELY(number < 0))
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message("count in an optimized container must be positive", "size"), nullptr));
return last_byte_error(exception_id::invalid_string_or_size, "count in an optimized container must be positive", "size");
}
if (JSON_HEDLEY_UNLIKELY(!value_in_range_of<std::size_t>(number)))
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408,
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(exception_id::container_too_large,
exception_message("integer value overflow", "size"), nullptr));
}
result = static_cast<std::size_t>(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char
@@ -2799,7 +2753,7 @@ class binary_reader
}
if (!value_in_range_of<std::size_t>(number))
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408,
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(exception_id::container_too_large,
exception_message("integer value overflow", "size"), nullptr));
}
result = detail::conditional_static_cast<std::size_t>(number);
@@ -2814,7 +2768,7 @@ class binary_reader
}
if (is_ndarray) // ndarray dimensional vector can only contain integers and cannot embed another array
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, exception_message("ndarray dimensional vector is not allowed", "size"), nullptr));
return last_byte_error(exception_id::invalid_string_or_size, "ndarray dimensional vector is not allowed", "size");
}
std::vector<size_t> dim;
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_ndarray_size(dim)))
@@ -2850,9 +2804,7 @@ class binary_reader
const char* type_name = bjd_type_name(ndarray_dtype);
if (JSON_HEDLEY_UNLIKELY(type_name == nullptr))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message("invalid byte: 0x" + last_token, "type"), nullptr));
return unexpected_byte("invalid byte", "type");
}
string_t type_key = "_ArrayType_";
@@ -2880,7 +2832,7 @@ class binary_reader
// or SIZE_MAX.
if (JSON_HEDLEY_UNLIKELY(result > (std::numeric_limits<std::size_t>::max)() / i))
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message("excessive ndarray size caused overflow", "size"), nullptr));
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(exception_id::container_too_large, exception_message("excessive ndarray size caused overflow", "size"), nullptr));
}
result *= i;
// the pre-check above already rules out result becoming 0
@@ -2889,7 +2841,7 @@ class binary_reader
// unknown-size container (see get_ubjson_size_type())
if (result == npos)
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message("excessive ndarray size caused overflow", "size"), nullptr));
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(exception_id::container_too_large, exception_message("excessive ndarray size caused overflow", "size"), nullptr));
}
if (JSON_HEDLEY_UNLIKELY(!emit_unsigned(i)))
{
@@ -2906,18 +2858,7 @@ class binary_reader
default:
break;
}
auto last_token = get_token_string();
std::string message;
if (input_format != input_format_t::bjdata)
{
message = "expected length type specification (U, i, I, l, L) after '#'; last byte: 0x" + last_token;
}
else
{
message = "expected length type specification (U, i, u, I, m, l, M, L) after '#'; last byte: 0x" + last_token;
}
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, exception_message(message, "size"), nullptr));
return length_type_error(" after '#'", "size");
}
/*!
@@ -2950,9 +2891,7 @@ class binary_reader
if (input_format == input_format_t::bjdata
&& JSON_HEDLEY_UNLIKELY(is_bjd_excluded_optimized_type(result.second)))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(concat("marker 0x", last_token, " is not a permitted optimized array type"), "type"), nullptr));
return last_byte_error(exception_id::unexpected_byte, concat("marker 0x", get_token_string(), " is not a permitted optimized array type"), "type");
}
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof("type")))
@@ -2967,9 +2906,7 @@ class binary_reader
{
return false;
}
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(concat("expected '#' after type information; last byte: 0x", last_token), "size"), nullptr));
return last_byte_error(exception_id::unexpected_byte, concat("expected '#' after type information; last byte: 0x", get_token_string()), "size");
}
const bool is_error = get_ubjson_size_value(result.first, is_ndarray, 0, result.second);
@@ -2988,8 +2925,7 @@ class binary_reader
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
if (input_format == input_format_t::bjdata && is_ndarray && !inside_ndarray)
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read,
exception_message("ndarray requires both type and size", "size"), nullptr));
return last_byte_error(exception_id::unexpected_byte, "ndarray requires both type and size", "size");
}
return is_error;
}
@@ -3121,9 +3057,7 @@ class binary_reader
}
if (JSON_HEDLEY_UNLIKELY(current > 127))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(concat("byte after 'C' must be in range 0x00..0x7F; last byte: 0x", last_token), "char"), nullptr));
return last_byte_error(exception_id::invalid_string_or_size, concat("byte after 'C' must be in range 0x00..0x7F; last byte: 0x", get_token_string()), "char");
}
string_t s(1, static_cast<typename string_t::value_type>(current));
return sax->string(s);
@@ -3144,8 +3078,7 @@ class binary_reader
default: // anything else
break;
}
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, exception_message("invalid byte: 0x" + last_token, "value"), nullptr));
return unexpected_byte("invalid byte", "value");
}
/*!
@@ -3208,7 +3141,7 @@ class binary_reader
if (JSON_HEDLEY_UNLIKELY((size_and_type.second == 'Z' || size_and_type.second == 'T' || size_and_type.second == 'F')
&& size_and_type.first > max_valueless_container_size))
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408,
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(exception_id::container_too_large,
exception_message("excessive array size", "size"), nullptr));
}
@@ -3244,9 +3177,7 @@ class binary_reader
// do not accept ND-array size in objects in BJData
if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0)
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message("BJData object does not support ND-array size in optimized format", "object"), nullptr));
return last_byte_error(exception_id::unexpected_byte, "BJData object does not support ND-array size in optimized format", "object");
}
if (size_and_type.first != npos)
@@ -3283,7 +3214,7 @@ class binary_reader
// the lexer would stop at a NUL and accept the digits before it
if (JSON_HEDLEY_UNLIKELY(current == '\0'))
{
return sax->parse_error(chars_read, "00", parse_error::create(115, chars_read,
return sax->parse_error(chars_read, "00", parse_error::create(exception_id::invalid_high_precision_number, chars_read,
exception_message("invalid number text; last byte: 0x00", "high-precision number"), nullptr));
}
number_vector.push_back(static_cast<char>(current));
@@ -3300,7 +3231,7 @@ class binary_reader
if (JSON_HEDLEY_UNLIKELY(result_remainder != token_type::end_of_input))
{
return sax->parse_error(chars_read, number_string, parse_error::create(115, chars_read,
return sax->parse_error(chars_read, number_string, parse_error::create(exception_id::invalid_high_precision_number, chars_read,
exception_message(concat("invalid number text: ", number_lexer.get_token_string()), "high-precision number"), nullptr));
}
@@ -3318,7 +3249,7 @@ class binary_reader
return sax->parse_error(
chars_read,
number_string,
out_of_range::create(406, concat("number overflow parsing '", number_string, '\''), nullptr));
out_of_range::create(exception_id::number_overflow, concat("number overflow parsing '", number_string, '\''), nullptr));
}
// number_string is a std::string, while the SAX interface takes a
// string_t; convert explicitly, as the two are only implicitly
@@ -3340,7 +3271,7 @@ class binary_reader
case token_type::end_of_input:
case token_type::literal_or_value:
default:
return sax->parse_error(chars_read, number_string, parse_error::create(115, chars_read,
return sax->parse_error(chars_read, number_string, parse_error::create(exception_id::invalid_high_precision_number, chars_read,
exception_message(concat("invalid number text: ", number_lexer.get_token_string()), "high-precision number"), nullptr));
}
}
@@ -3399,20 +3330,6 @@ class binary_reader
return 0x80 <= c && c <= 0xBF;
}
/*!
@brief report a parse error at the last read byte
@param[in] detail a detailed error message
@param[in] context further context information
@return false
*/
bool bon8_error(const std::string& detail, const char* context)
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(concat(detail, ": 0x", last_token), context), nullptr));
}
/*!
@brief read a BON8 value and everything nested inside it
@@ -3628,7 +3545,7 @@ class binary_reader
}
// 0xFE: end of container where a value is expected
return bon8_error("invalid byte", "value");
return unexpected_byte("invalid byte", "value");
}
/*!
@@ -3755,7 +3672,7 @@ class binary_reader
current = byte;
}
return bon8_error("expected a string; last byte", "key");
return unexpected_byte("expected a string; last byte", "key");
}
/*!
@@ -3878,7 +3795,7 @@ class binary_reader
if (JSON_HEDLEY_UNLIKELY(!valid_second))
{
return bon8_error("invalid UTF-8 byte", "string");
return unexpected_byte("invalid UTF-8 byte", "string");
}
result.push_back(static_cast<typename string_t::value_type>(byte));
@@ -3892,7 +3809,7 @@ class binary_reader
}
if (JSON_HEDLEY_UNLIKELY(!is_bon8_continuation(current)))
{
return bon8_error("invalid UTF-8 byte", "string");
return unexpected_byte("invalid UTF-8 byte", "string");
}
result.push_back(static_cast<typename string_t::value_type>(current));
}
@@ -3937,7 +3854,7 @@ class binary_reader
{
// in case of failure, advance position by 1 to report the failing location
++chars_read;
sax->parse_error(chars_read, "<end of file>", parse_error::create(110, chars_read, exception_message("unexpected end of input", context), nullptr));
sax->parse_error(chars_read, "<end of file>", parse_error::create(exception_id::unexpected_end_of_input, chars_read, exception_message("unexpected end of input", context), nullptr));
return false;
}
return true;
@@ -4091,7 +4008,7 @@ class binary_reader
if (JSON_HEDLEY_UNLIKELY(std::isfinite(number) && !std::isfinite(result)))
{
return sax->parse_error(chars_read, get_token_string(),
out_of_range::create(406, exception_message("number overflow", "value"), nullptr));
out_of_range::create(exception_id::number_overflow, exception_message("number overflow", "value"), nullptr));
}
return sax->number_float(result, "");
}
@@ -4211,9 +4128,7 @@ class binary_reader
if (error_handler == error_handler_t::strict)
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message("invalid string: ill-formed UTF-8 byte", context), nullptr));
return last_byte_error(exception_id::invalid_string_or_size, "invalid string: ill-formed UTF-8 byte", context);
}
result = sanitize_utf8(result, error_handler);
@@ -4309,19 +4224,53 @@ class binary_reader
if (JSON_HEDLEY_UNLIKELY(current == char_traits<char_type>::eof()))
{
return sax->parse_error(chars_read, "<end of file>",
parse_error::create(110, chars_read, exception_message("unexpected end of input", context), nullptr));
parse_error::create(exception_id::unexpected_end_of_input, chars_read, exception_message("unexpected end of input", context), nullptr));
}
return true;
}
/*!
@brief reports a parse error at the last read byte
@param[in] id_ the id of the parse_error exception
@param[in] detail a detailed error message
@param[in] context further context information
@return the result of the SAX parser's parse_error()
*/
bool last_byte_error(const exception_id id_, const std::string& detail, const std::string& context) const
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(id_, chars_read, exception_message(detail, context), nullptr));
}
/*!
@brief reports the last read byte as unexpected (parse_error.112)
@param[in] detail what is wrong with the byte, e.g. "invalid byte"
@param[in] context further context information
@return the result of the SAX parser's parse_error()
*/
bool unexpected_byte(const char* detail, const std::string& context) const
{
return last_byte_error(exception_id::unexpected_byte, concat(detail, ": 0x", get_token_string()), context);
}
/*!
@brief reports that the last read byte is not a UBJSON/BJData length type (parse_error.113)
@param[in] position where the length type was expected, e.g. " after '#'"
@param[in] context further context information
@return the result of the SAX parser's parse_error()
*/
bool length_type_error(const char* position, const char* context) const
{
const char* types = input_format == input_format_t::bjdata ? "U, i, u, I, m, l, M, L" : "U, i, I, l, L";
return last_byte_error(exception_id::invalid_string_or_size,
concat("expected length type specification (", types, ")", position, "; last byte: 0x", get_token_string()), context);
}
/*!
@return a string representation of the last read byte
*/
std::string get_token_string() const
{
std::array<char, 3> cr{{}};
static_cast<void>((std::snprintf)(cr.data(), cr.size(), "%.2hhX", static_cast<unsigned char>(current))); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg)
return std::string{cr.data()};
return hex_byte(static_cast<std::uint8_t>(current));
}
/*!
@@ -581,7 +581,7 @@ class wide_string_input_adapter
template<class T>
JSON_HEDLEY_NO_RETURN std::size_t get_elements(T* /*dest*/, std::size_t /*count*/ = 1)
{
JSON_THROW(parse_error::create(112, 1, "wide string type cannot be interpreted as binary data", nullptr));
JSON_THROW(parse_error::create(exception_id::unexpected_byte, 1, "wide string type cannot be interpreted as binary data", nullptr));
}
private:
@@ -793,7 +793,7 @@ inline file_input_adapter input_adapter(std::FILE* file)
{
if (file == nullptr)
{
JSON_THROW(parse_error::create(101, 0, "attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr));
JSON_THROW(parse_error::create(exception_id::syntax_error, 0, "attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr));
}
return file_input_adapter(file);
}
@@ -802,7 +802,7 @@ inline input_stream_adapter input_adapter(std::istream& stream)
{
if (stream.rdbuf() == nullptr)
{
JSON_THROW(parse_error::create(101, 0, "attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr));
JSON_THROW(parse_error::create(exception_id::syntax_error, 0, "attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr));
}
return input_stream_adapter(stream);
}
@@ -827,7 +827,7 @@ contiguous_bytes_input_adapter input_adapter(CharT b)
{
if (b == nullptr)
{
JSON_THROW(parse_error::create(101, 0, "attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr));
JSON_THROW(parse_error::create(exception_id::syntax_error, 0, "attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr));
}
auto length = std::strlen(reinterpret_cast<const char*>(b));
const auto* ptr = reinterpret_cast<const char*>(b);
+83 -76
View File
@@ -176,6 +176,27 @@ template<typename ArrayType>
inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/)
{}
/*!
@brief reports an object or array whose announced size exceeds max_size()
Shared by json_sax_dom_parser and json_sax_dom_callback_parser.
@param[in] sax the SAX parser to report the error to
@param[in] len the number of elements announced by the input, or unknown_size()
@param[in] kind "object" or "array"
@param[in] ref the object or array that was just created
@return whether parsing should continue (false after reporting out_of_range.408)
*/
template<typename SAX, typename BasicJsonType>
bool check_container_size(SAX& sax, std::size_t len, const char* kind, BasicJsonType* ref)
{
if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref->max_size()))
{
return sax.parse_error(0, "", out_of_range::create(exception_id::container_too_large, concat("excessive ", kind, " size: ", std::to_string(len)), ref));
}
return true;
}
#if JSON_DIAGNOSTIC_POSITIONS
/*!
@brief set the diagnostic positions of a value the DOM SAX parsers just stored
@@ -185,6 +206,35 @@ befriends this struct, as the position members are private.
*/
struct diagnostic_positions
{
/*!
@param[in,out] v the object or array whose opening brace or bracket was just read
@param[in] lexer the lexer that read it, or nullptr to leave @a v alone
*/
template<typename BasicJsonType, typename LexerType>
static void set_container_start(BasicJsonType& v, LexerType* lexer)
{
if (lexer)
{
// Lexer has read the first character of the container, so
// subtract 1 from the position to get the correct start position.
v.start_position = lexer->get_position() - 1;
}
}
/*!
@param[in,out] v the object or array whose closing brace or bracket was just read
@param[in] lexer the lexer that read it, or nullptr to leave @a v alone
*/
template<typename BasicJsonType, typename LexerType>
static void set_container_end(BasicJsonType& v, LexerType* lexer)
{
if (lexer)
{
// Lexer's position is past the closing brace or bracket, so set that as the end position.
v.end_position = lexer->get_position();
}
}
/*!
@param[in,out] v the value that was just parsed
@param[in] lexer the lexer that read it, or nullptr to leave @a v alone
@@ -349,22 +399,11 @@ class json_sax_dom_parser
ref_stack.push_back(handle_value(BasicJsonType::value_t::object));
#if JSON_DIAGNOSTIC_POSITIONS
// Manually set the start position of the object here.
// Ensure this is after the call to handle_value to ensure correct start position.
if (m_lexer_ref)
{
// Lexer has read the first character of the object, so
// subtract 1 from the position to get the correct start position.
ref_stack.back()->start_position = m_lexer_ref->get_position() - 1;
}
diagnostic_positions::set_container_start(*ref_stack.back(), m_lexer_ref);
#endif
if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size()))
{
return parse_error(0, "", out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back()));
}
return true;
return check_container_size(*this, len, "object", ref_stack.back());
}
bool key(string_t& val)
@@ -383,11 +422,7 @@ class json_sax_dom_parser
JSON_ASSERT(ref_stack.back()->is_object());
#if JSON_DIAGNOSTIC_POSITIONS
if (m_lexer_ref)
{
// Lexer's position is past the closing brace, so set that as the end position.
ref_stack.back()->end_position = m_lexer_ref->get_position();
}
diagnostic_positions::set_container_end(*ref_stack.back(), m_lexer_ref);
#endif
ref_stack.back()->set_parents();
@@ -400,17 +435,13 @@ class json_sax_dom_parser
ref_stack.push_back(handle_value(BasicJsonType::value_t::array));
#if JSON_DIAGNOSTIC_POSITIONS
// Manually set the start position of the array here.
// Ensure this is after the call to handle_value to ensure correct start position.
if (m_lexer_ref)
{
ref_stack.back()->start_position = m_lexer_ref->get_position() - 1;
}
diagnostic_positions::set_container_start(*ref_stack.back(), m_lexer_ref);
#endif
if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size()))
if (JSON_HEDLEY_UNLIKELY(!check_container_size(*this, len, "array", ref_stack.back())))
{
return parse_error(0, "", out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
return false;
}
if (len != detail::unknown_size())
@@ -427,11 +458,7 @@ class json_sax_dom_parser
JSON_ASSERT(ref_stack.back()->is_array());
#if JSON_DIAGNOSTIC_POSITIONS
if (m_lexer_ref)
{
// Lexer's position is past the closing bracket, so set that as the end position.
ref_stack.back()->end_position = m_lexer_ref->get_position();
}
diagnostic_positions::set_container_end(*ref_stack.back(), m_lexer_ref);
#endif
ref_stack.back()->set_parents();
@@ -611,21 +638,11 @@ class json_sax_dom_callback_parser
{
#if JSON_DIAGNOSTIC_POSITIONS
// Manually set the start position of the object here.
// Ensure this is after the call to handle_value to ensure correct start position.
if (m_lexer_ref)
{
// Lexer has read the first character of the object, so
// subtract 1 from the position to get the correct start position.
ref_stack.back()->start_position = m_lexer_ref->get_position() - 1;
}
diagnostic_positions::set_container_start(*ref_stack.back(), m_lexer_ref);
#endif
// check object limit
if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size()))
{
return parse_error(0, "", out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back()));
}
return check_container_size(*this, len, "object", ref_stack.back());
}
return true;
}
@@ -695,11 +712,7 @@ class json_sax_dom_callback_parser
{
#if JSON_DIAGNOSTIC_POSITIONS
if (m_lexer_ref)
{
// Lexer's position is past the closing brace, so set that as the end position.
ref_stack.back()->end_position = m_lexer_ref->get_position();
}
diagnostic_positions::set_container_end(*ref_stack.back(), m_lexer_ref);
#endif
ref_stack.back()->set_parents();
@@ -710,13 +723,7 @@ class json_sax_dom_callback_parser
}
}
JSON_ASSERT(!ref_stack.empty());
JSON_ASSERT(!keep_stack.empty());
JSON_ASSERT(!container_key_stack.empty());
ref_stack.pop_back();
keep_stack.pop_back();
const string_t object_key = std::move(container_key_stack.back());
container_key_stack.pop_back();
const string_t object_key = pop_container();
if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_structured())
{
@@ -742,20 +749,13 @@ class json_sax_dom_callback_parser
{
#if JSON_DIAGNOSTIC_POSITIONS
// Manually set the start position of the array here.
// Ensure this is after the call to handle_value to ensure correct start position.
if (m_lexer_ref)
{
// Lexer has read the first character of the array, so
// subtract 1 from the position to get the correct start position.
ref_stack.back()->start_position = m_lexer_ref->get_position() - 1;
}
diagnostic_positions::set_container_start(*ref_stack.back(), m_lexer_ref);
#endif
// check array limit
if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size()))
if (JSON_HEDLEY_UNLIKELY(!check_container_size(*this, len, "array", ref_stack.back())))
{
return parse_error(0, "", out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back()));
return false;
}
if (len != detail::unknown_size())
@@ -779,11 +779,7 @@ class json_sax_dom_callback_parser
{
#if JSON_DIAGNOSTIC_POSITIONS
if (m_lexer_ref)
{
// Lexer's position is past the closing bracket, so set that as the end position.
ref_stack.back()->end_position = m_lexer_ref->get_position();
}
diagnostic_positions::set_container_end(*ref_stack.back(), m_lexer_ref);
#endif
ref_stack.back()->set_parents();
@@ -809,13 +805,7 @@ class json_sax_dom_callback_parser
}
}
JSON_ASSERT(!ref_stack.empty());
JSON_ASSERT(!keep_stack.empty());
JSON_ASSERT(!container_key_stack.empty());
ref_stack.pop_back();
keep_stack.pop_back();
const string_t object_key = std::move(container_key_stack.back());
container_key_stack.pop_back();
const string_t object_key = pop_container();
// remove discarded value
if (!ref_stack.empty() && ref_stack.back())
@@ -902,6 +892,23 @@ class json_sax_dom_callback_parser
return string_t{};
}
/*!
@brief leave the object or array that end_object()/end_array() just closed
@return the key the container is stored under in its parent (see
container_key_stack)
*/
string_t pop_container()
{
JSON_ASSERT(!ref_stack.empty());
JSON_ASSERT(!keep_stack.empty());
JSON_ASSERT(!container_key_stack.empty());
ref_stack.pop_back();
keep_stack.pop_back();
string_t object_key = std::move(container_key_stack.back());
container_key_stack.pop_back();
return object_key;
}
/*!
@brief remove the discarded value the callback rejected from its parent,
unless it is a duplicate key's slot with a stashed previous value, in
+28 -39
View File
@@ -156,9 +156,7 @@ class parser
// strict mode: next byte must be EOF
if (get_token() != token_type::end_of_input)
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
return syntax_error(*sax, exception_message(token_type::end_of_input, "value"));
}
}
else
@@ -197,10 +195,7 @@ class parser
// in strict mode, input must be completely read
if (get_token() != token_type::end_of_input)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(),
exception_message(token_type::end_of_input, "value"), nullptr));
syntax_error(sdp, exception_message(token_type::end_of_input, "value"));
}
}
else
@@ -250,9 +245,7 @@ class parser
// parse key
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr));
return syntax_error(*sax, exception_message(token_type::value_string, "object key"));
}
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
{
@@ -262,9 +255,7 @@ class parser
// parse separator (:)
if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator)))
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr));
return syntax_error(*sax, exception_message(token_type::name_separator, "object separator"));
}
// remember we are now inside an object
@@ -307,7 +298,7 @@ class parser
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr));
out_of_range::create(exception_id::number_overflow, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr));
}
if (JSON_HEDLEY_UNLIKELY(!sax->number_float(res, m_lexer.get_string())))
@@ -375,23 +366,16 @@ class parser
case token_type::parse_error:
{
// using "uninitialized" to avoid an "expected" message
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::uninitialized, "value"), nullptr));
return syntax_error(*sax, exception_message(token_type::uninitialized, "value"));
}
case token_type::end_of_input:
{
if (JSON_HEDLEY_UNLIKELY(m_lexer.get_position().chars_read_total == 1))
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(),
"attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr));
return syntax_error(*sax, "attempting to parse an empty input; check that your input string or stream contains the expected JSON");
}
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr));
return syntax_error(*sax, exception_message(token_type::literal_or_value, "value"));
}
case token_type::uninitialized:
case token_type::end_array:
@@ -401,9 +385,7 @@ class parser
case token_type::literal_or_value:
default: // the last token was unexpected
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr));
return syntax_error(*sax, exception_message(token_type::literal_or_value, "value"));
}
}
}
@@ -454,9 +436,7 @@ class parser
continue;
}
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_array, "array"), nullptr));
return syntax_error(*sax, exception_message(token_type::end_array, "array"));
}
// states.back() is false -> object
@@ -473,9 +453,7 @@ class parser
// parse key
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr));
return syntax_error(*sax, exception_message(token_type::value_string, "object key"));
}
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
@@ -486,9 +464,7 @@ class parser
// parse separator (:)
if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator)))
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr));
return syntax_error(*sax, exception_message(token_type::name_separator, "object separator"));
}
// parse values
@@ -516,9 +492,7 @@ class parser
continue;
}
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_object, "object"), nullptr));
return syntax_error(*sax, exception_message(token_type::end_object, "object"));
}
}
@@ -535,6 +509,21 @@ class parser
return (last_token = m_lexer.scan_expecting(expected_type)) == expected_type;
}
/*!
@brief reports a syntax error at the current token (parse_error.101)
@param[in] sax the SAX parser to report the error to
@param[in] message the error message
@return the result of the SAX parser's parse_error()
*/
template<typename SAX>
bool syntax_error(SAX& sax, const std::string& message)
{
// MSVC 2015 reports C4100 (unreferenced parameter) when SAX::parse_error is static
static_cast<void>(sax);
return sax.parse_error(m_lexer.get_position(), m_lexer.get_token_string(),
parse_error::create(exception_id::syntax_error, m_lexer.get_position(), message, nullptr));
}
std::string exception_message(const token_type expected, const std::string& context)
{
std::string error_msg = "syntax error ";