Merge branch 'json-view/03-string-scan' into json-view/04-unicode-escapes

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-10-01 10:18:33 +02:00
306 changed files with 6550 additions and 15197 deletions
+311 -359
View File
@@ -8,7 +8,6 @@
#pragma once
#include <algorithm> // generate_n
#include <array> // array
#include <cmath> // ldexp
#include <cstddef> // size_t
@@ -123,16 +122,16 @@ class binary_reader
~binary_reader() = default;
/*!
@param[in] format the binary format to parse
@brief parse in the format the constructor was given
@param[in] sax_ a SAX event processor
@param[in] strict whether to expect the input to be consumed completed
@param[in] tag_handler how to treat CBOR tags
@return whether parsing was successful
*/
JSON_HEDLEY_NON_NULL(3)
bool sax_parse(const input_format_t format,
json_sax_t* sax_,
JSON_HEDLEY_NON_NULL(2)
bool sax_parse(json_sax_t* sax_,
const bool strict = true,
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error)
{
@@ -141,14 +140,14 @@ class binary_reader
bon8_pushback_size = 0;
bool result = false;
switch (format)
switch (input_format)
{
case input_format_t::bson:
result = parse_bson_internal();
break;
case input_format_t::cbor:
result = parse_cbor_internal(true, tag_handler);
result = parse_cbor_internal(tag_handler);
break;
case input_format_t::msgpack:
@@ -271,6 +270,22 @@ class binary_reader
return enter_container(/*is_object*/true, len, type_marker);
}
/*!
@brief close the innermost open array or object
Pops the container opened by the matching @ref enter_container call and
emits the SAX end event. Every format-specific driver otherwise repeated
the same pop-then-dispatch sequence at its own close site.
@return whether the SAX parser accepted the end event
*/
bool leave_container()
{
const bool is_object = container_stack.back().is_object;
container_stack.pop_back();
return is_object ? sax->end_object() : sax->end_array();
}
//////////
// BSON //
//////////
@@ -355,8 +370,8 @@ class binary_reader
if (element_type == 0) // end of the innermost document
{
// a copy, not a reference: it must stay valid across the
// pop_back() below, which destroys the container_stack
// element it would otherwise alias
// pop_back() inside leave_container() below, which destroys
// the container_stack element it would otherwise alias
const container_frame top = container_stack.back();
if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(top.start_position, top.declared_size)))
@@ -364,8 +379,7 @@ class binary_reader
return false;
}
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -407,7 +421,7 @@ class binary_reader
@brief Parses a C-style string from the BSON input.
@param[in,out] result A reference to the string variable where the read
string is to be stored.
@return `true` if the \x00-byte indicating the end of the string was
@return `true` if the \\x00-byte indicating the end of the string was
encountered before the EOF; false` indicates an unexpected EOF.
*/
bool get_bson_cstr(string_t& result)
@@ -437,7 +451,7 @@ class binary_reader
@brief read a C-style string from contiguous input in one step
@param[in,out] result the string to append to
@return whether the string was read; if the input has no \x00-byte, nothing
@return whether the string was read; if the input has no \\x00-byte, nothing
is read, and @ref get_bson_cstr reports the end of the input
*/
bool get_bson_cstr_bulk(string_t& result, std::true_type /*bulk*/)
@@ -860,29 +874,13 @@ class binary_reader
return enter_array(conditional_static_cast<std::size_t>(static_cast<unsigned int>(current) & 0x1Fu));
case 0x98: // array (one-byte uint8_t for n follows)
{
std::uint8_t len{};
return get_number(input_format_t::cbor, len) && enter_array(static_cast<std::size_t>(len));
}
case 0x99: // array (two-byte uint16_t for n follow)
{
std::uint16_t len{};
return get_number(input_format_t::cbor, len) && enter_array(static_cast<std::size_t>(len));
}
case 0x9A: // array (four-byte uint32_t for n follow)
{
std::uint32_t len{};
std::size_t size{};
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size);
}
case 0x9B: // array (eight-byte uint64_t for n follow)
{
std::uint64_t len{};
std::size_t size{};
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size);
return get_cbor_argument(len) && get_cbor_container_size(len, size, "array") && enter_array(size);
}
case 0x9F: // array (indefinite length)
@@ -916,35 +914,19 @@ class binary_reader
return enter_object(conditional_static_cast<std::size_t>(static_cast<unsigned int>(current) & 0x1Fu));
case 0xB8: // map (one-byte uint8_t for n follows)
{
std::uint8_t len{};
return get_number(input_format_t::cbor, len) && enter_object(static_cast<std::size_t>(len));
}
case 0xB9: // map (two-byte uint16_t for n follow)
{
std::uint16_t len{};
return get_number(input_format_t::cbor, len) && enter_object(static_cast<std::size_t>(len));
}
case 0xBA: // map (four-byte uint32_t for n follow)
{
std::uint32_t len{};
std::size_t size{};
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size);
}
case 0xBB: // map (eight-byte uint64_t for n follow)
{
std::uint64_t len{};
std::size_t size{};
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size);
return get_cbor_argument(len) && get_cbor_container_size(len, size, "map") && enter_object(size);
}
case 0xBF: // map (indefinite length)
return enter_object(detail::unknown_size());
case 0xC0: // tagged item
case 0xC0: // tagged item (tag value 0-23, in the head itself)
case 0xC1:
case 0xC2:
case 0xC3:
@@ -968,6 +950,22 @@ class binary_reader
case 0xD5:
case 0xD6:
case 0xD7:
{
if (tag_handler == cbor_tag_handler_t::error)
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr));
}
// ignore and store: the tag value is already in the head, so
// there is nothing left to read here; the tagged value that
// follows is read by the loop in parse_cbor_internal() rather
// than by recursing here
tag_pending = true;
return true;
}
case 0xD8: // tagged item (1 byte follows)
case 0xD9: // tagged item (2 bytes follow)
case 0xDA: // tagged item (4 bytes follow)
@@ -984,47 +982,11 @@ class binary_reader
case cbor_tag_handler_t::ignore:
{
// ignore binary subtype
switch (current)
// ignore the tag's binary subtype argument
std::uint64_t subtype_to_ignore{};
if (!get_cbor_argument(subtype_to_ignore))
{
case 0xD8:
{
std::uint8_t subtype_to_ignore{};
if (!get_number(input_format_t::cbor, subtype_to_ignore))
{
return false;
}
break;
}
case 0xD9:
{
std::uint16_t subtype_to_ignore{};
if (!get_number(input_format_t::cbor, subtype_to_ignore))
{
return false;
}
break;
}
case 0xDA:
{
std::uint32_t subtype_to_ignore{};
if (!get_number(input_format_t::cbor, subtype_to_ignore))
{
return false;
}
break;
}
case 0xDB:
{
std::uint64_t subtype_to_ignore{};
if (!get_number(input_format_t::cbor, subtype_to_ignore))
{
return false;
}
break;
}
default:
break;
return false;
}
// the tagged value follows; it is read by the loop in
// parse_cbor_internal() rather than by recursing here
@@ -1034,57 +996,15 @@ class binary_reader
case cbor_tag_handler_t::store:
{
binary_t b;
// use binary subtype and store in a binary container
switch (current)
std::uint64_t subtype{};
if (!get_cbor_argument(subtype))
{
case 0xD8:
{
std::uint8_t subtype{};
if (!get_number(input_format_t::cbor, subtype))
{
return false;
}
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
break;
}
case 0xD9:
{
std::uint16_t subtype{};
if (!get_number(input_format_t::cbor, subtype))
{
return false;
}
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
break;
}
case 0xDA:
{
std::uint32_t subtype{};
if (!get_number(input_format_t::cbor, subtype))
{
return false;
}
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
break;
}
case 0xDB:
{
std::uint64_t subtype{};
if (!get_number(input_format_t::cbor, subtype))
{
return false;
}
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
break;
}
default:
{
// as above, the tagged value is read by the caller
tag_pending = true;
return true;
}
return false;
}
binary_t b;
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
get();
// a byte string (the heads accepted by get_cbor_binary) keeps the tag as subtype
if ((current >= 0x40 && current <= 0x5B) || current == 0x5F)
@@ -1115,52 +1035,7 @@ class binary_reader
return sax->null();
case 0xF9: // Half-Precision Float (two-byte IEEE 754)
{
const auto byte1_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number")))
{
return false;
}
const auto byte2_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number")))
{
return false;
}
const auto byte1 = static_cast<unsigned char>(byte1_raw);
const auto byte2 = static_cast<unsigned char>(byte2_raw);
// Code from RFC 8949, Appendix D, Figure 3:
// As half-precision floating-point numbers were only added
// to IEEE 754 in 2008, today's programming platforms often
// still only have limited support for them. It is very
// easy to include at least decoding support for them even
// without such support. An example of a small decoder for
// half-precision floating-point numbers in the C language
// is shown in Fig. 3.
const auto half = static_cast<unsigned int>((byte1 << 8u) + byte2);
const double val = [&half]
{
const int exp = (half >> 10u) & 0x1Fu;
const unsigned int mant = half & 0x3FFu;
JSON_ASSERT(exp <= 31);
JSON_ASSERT(mant <= 1023);
switch (exp)
{
case 0:
return std::ldexp(mant, -24);
case 31:
return (mant == 0)
? std::numeric_limits<double>::infinity()
: std::numeric_limits<double>::quiet_NaN();
default:
return std::ldexp(mant + 1024, exp - 25);
}
}();
return sax->number_float((half & 0x8000u) != 0
? static_cast<number_float_t>(-val)
: static_cast<number_float_t>(val), "");
}
return get_half_float(input_format_t::cbor, false);
case 0xFA: // Single-Precision Float (four-byte IEEE 754)
{
@@ -1539,6 +1414,73 @@ class binary_reader
}
}
/*!
@brief read a CBOR argument (additional information 24-27) of the width
@ref current announces
The lower 5 bits of @a current (0x18-0x1B) select a 1/2/4/8-byte
big-endian unsigned integer that follows the head byte; this is shared by
every major type that uses this encoding (unsigned/negative integers,
strings, arrays, maps, tags). Reading always goes through @ref get_number,
so EOF is reported the same way as before this helper existed.
@param[out] value the decoded argument
@return whether reading succeeded
*/
bool get_cbor_argument(std::uint64_t& value)
{
switch (current & 0x1F)
{
case 0x18: // 1 byte
{
std::uint8_t n{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n)))
{
return false;
}
value = n;
return true;
}
case 0x19: // 2 bytes
{
std::uint16_t n{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n)))
{
return false;
}
value = n;
return true;
}
case 0x1A: // 4 bytes
{
std::uint32_t n{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n)))
{
return false;
}
value = n;
return true;
}
case 0x1B: // 8 bytes
{
std::uint64_t n{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n)))
{
return false;
}
value = n;
return true;
}
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
return false; // LCOV_EXCL_LINE
}
}
/*!
@brief narrow a definite CBOR array/map length to std::size_t
@@ -1571,19 +1513,15 @@ class binary_reader
enclosing container after each element, so that the nesting depth of the
input costs heap rather than native stack (see #5104).
@param[in] get_char whether a new character should be retrieved from the
input (true) or whether the last read character
@a current should be considered instead
@param[in] tag_handler how CBOR tags should be treated
@return whether reading the value succeeded
*/
bool parse_cbor_internal(const bool get_char,
const cbor_tag_handler_t tag_handler)
bool parse_cbor_internal(const cbor_tag_handler_t tag_handler)
{
// whether the next value starts at a fresh byte or at the one already
// read into `current`
bool fetch = get_char;
bool fetch = true;
// the key currently being read; hoisted out of the loop so that its
// capacity is reused across elements and across nesting levels
@@ -1626,8 +1564,7 @@ class binary_reader
if (at_end)
{
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -1676,9 +1613,6 @@ class binary_reader
// MsgPack //
/////////////
/*!
@return whether a valid MessagePack value was passed to the SAX parser
*/
/*!
@brief read one MessagePack value
@@ -2377,8 +2311,7 @@ class binary_reader
if (container_stack.back().remaining == 0)
{
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -2423,20 +2356,16 @@ class binary_reader
////////////
/*!
@param[in] get_char whether a new character should be retrieved from the
input (true, default) or whether the last read
character should be considered instead
@return whether a valid UBJSON value was passed to the SAX parser
*/
bool parse_ubjson_internal(const bool get_char = true)
bool parse_ubjson_internal()
{
// the key currently being read; hoisted out of the loop so that its
// capacity is reused across elements and across nesting levels
string_t key;
// the type marker of the value to read next
char_int_type prefix = get_char ? get_ignore_noop() : current;
char_int_type prefix = get_ignore_noop();
while (true)
{
@@ -2511,8 +2440,7 @@ class binary_reader
break;
}
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -2720,6 +2648,42 @@ class binary_reader
return true;
}
/*!
@brief read a UBJSON/BJData optimized-container count of a signed marker
type ('i', 'I', 'l', 'L') and narrow it to std::size_t
Every signed count marker rejects a negative value the same way (error
113); the value_in_range_of check additionally needed for 'L' is only
ever live when @a SignedType is std::int64_t on a target where
std::size_t is narrower (e.g. 32-bit), since 'i'/'I'/'l' can never exceed
std::size_t there.
@tparam SignedType std::int8_t, std::int16_t, std::int32_t or std::int64_t
@param[out] result the count narrowed to std::size_t
@return whether reading and validating succeeded
*/
template<typename SignedType>
bool get_ubjson_signed_count(std::size_t& result)
{
SignedType number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (JSON_HEDLEY_UNLIKELY(number < 0))
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
if (JSON_HEDLEY_UNLIKELY(!value_in_range_of<std::size_t>(number)))
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408,
exception_message(input_format, "integer value overflow", "size"), nullptr));
}
result = static_cast<std::size_t>(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char
return true;
}
/*!
@param[out] result determined size
@param[in,out] is_ndarray for input, `true` means already inside an ndarray vector
@@ -2752,73 +2716,16 @@ class binary_reader
}
case 'i':
{
std::int8_t number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (number < 0)
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
result = static_cast<std::size_t>(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char
return true;
}
return get_ubjson_signed_count<std::int8_t>(result);
case 'I':
{
std::int16_t number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (number < 0)
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
result = static_cast<std::size_t>(number);
return true;
}
return get_ubjson_signed_count<std::int16_t>(result);
case 'l':
{
std::int32_t number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (number < 0)
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
result = static_cast<std::size_t>(number);
return true;
}
return get_ubjson_signed_count<std::int32_t>(result);
case 'L':
{
std::int64_t number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (number < 0)
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
if (!value_in_range_of<std::size_t>(number))
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408,
exception_message(input_format, "integer value overflow", "size"), nullptr));
}
result = static_cast<std::size_t>(number);
return true;
}
return get_ubjson_signed_count<std::int64_t>(result);
case 'u':
{
@@ -2909,16 +2816,23 @@ class binary_reader
result = 1;
for (auto i : dim)
{
// Pre-multiplication overflow check: if i > 0 and result > SIZE_MAX/i, then result*i would overflow.
// This check must happen before multiplication since overflow detection after the fact is unreliable
// as modular arithmetic can produce any value, not just 0 or SIZE_MAX.
if (JSON_HEDLEY_UNLIKELY(i > 0 && result > (std::numeric_limits<std::size_t>::max)() / i))
// Pre-multiplication overflow check: since the loop above
// already rejected any zero dimension, i is always > 0
// here, so result > SIZE_MAX/i means result*i would
// overflow. This check must happen before multiplication
// since overflow detection after the fact is unreliable,
// as modular arithmetic can produce any value, not just 0
// or SIZE_MAX.
if (JSON_HEDLEY_UNLIKELY(result > (std::numeric_limits<std::size_t>::max)() / i))
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr));
}
result *= i;
// Additional post-multiplication check to catch any edge cases the pre-check might miss
if (result == 0 || result == npos)
// the pre-check above already rules out result becoming 0
// by overflow; the only value it cannot rule out is an
// exact match with npos, the sentinel reserved for an
// unknown-size container (see get_ubjson_size_type())
if (result == npos)
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr));
}
@@ -2979,7 +2893,7 @@ class binary_reader
{
result.second = get(); // must not ignore 'N', because 'N' maybe the type
if (input_format == input_format_t::bjdata
&& JSON_HEDLEY_UNLIKELY(std::binary_search(bjd_optimized_type_markers.begin(), bjd_optimized_type_markers.end(), result.second)))
&& JSON_HEDLEY_UNLIKELY(is_bjd_excluded_optimized_type(result.second)))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
@@ -3123,50 +3037,7 @@ class binary_reader
{
break;
}
const auto byte1_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number")))
{
return false;
}
const auto byte2_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number")))
{
return false;
}
const auto byte1 = static_cast<unsigned char>(byte1_raw);
const auto byte2 = static_cast<unsigned char>(byte2_raw);
// Code from RFC 8949, Appendix D, Figure 3:
// As half-precision floating-point numbers were only added
// to IEEE 754 in 2008, today's programming platforms often
// still only have limited support for them. It is very
// easy to include at least decoding support for them even
// without such support. An example of a small decoder for
// half-precision floating-point numbers in the C language
// is shown in Fig. 3.
const auto half = static_cast<unsigned int>((byte2 << 8u) + byte1);
const double val = [&half]
{
const int exp = (half >> 10u) & 0x1Fu;
const unsigned int mant = half & 0x3FFu;
JSON_ASSERT(exp <= 31);
JSON_ASSERT(mant <= 1023);
switch (exp)
{
case 0:
return std::ldexp(mant, -24);
case 31:
return (mant == 0)
? std::numeric_limits<double>::infinity()
: std::numeric_limits<double>::quiet_NaN();
default:
return std::ldexp(mant + 1024, exp - 25);
}
}();
return sax->number_float((half & 0x8000u) != 0
? static_cast<number_float_t>(-val)
: static_cast<number_float_t>(val), "");
return get_half_float(input_format, true);
}
case 'd':
@@ -3239,19 +3110,16 @@ class binary_reader
if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0)
{
size_and_type.second &= ~(static_cast<char_int_type>(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t)
{
return p.first < t;
});
const char* type_name = bjd_type_name(size_and_type.second);
string_t key = "_ArrayType_";
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second))
if (JSON_HEDLEY_UNLIKELY(type_name == nullptr))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
}
string_t type = it->second; // sax->string() takes a reference
string_t type = type_name; // sax->string() takes a reference
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type)))
{
return false;
@@ -3344,8 +3212,8 @@ class binary_reader
return enter_object(detail::unknown_size());
}
// Note, no reader for UBJSON binary types is implemented because they do
// not exist
// Note, UBJSON has no binary type of its own; BJData, which shares this
// reader, decodes optimized 'B' arrays as binary in get_ubjson_array().
bool get_ubjson_high_precision_number()
{
@@ -3543,8 +3411,7 @@ class binary_reader
if (at_end)
{
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -4052,7 +3919,7 @@ class binary_reader
#endif
}
/*
/*!
@brief read a number from the input
@tparam NumberType the type of the number
@@ -4062,10 +3929,10 @@ class binary_reader
@return whether conversion completed
@note This function needs to respect the system's endianness, because
bytes in CBOR, MessagePack, and UBJSON are stored in network order
(big endian) and therefore need reordering on little endian systems.
On the other hand, BSON and BJData use little endian and should reorder
on big endian systems.
bytes in CBOR, MessagePack, UBJSON, and BON8 are stored in network
order (big endian) and therefore need reordering on little endian
systems. On the other hand, BSON and BJData use little endian and
should reorder on big endian systems.
*/
template<typename NumberType, bool InputIsLittleEndian = false>
bool get_number(const input_format_t format, NumberType& result)
@@ -4083,6 +3950,68 @@ class binary_reader
return true;
}
/*!
@brief read and decode an IEEE 754 half-precision (16-bit) float
Used by CBOR (big endian) and BJData (little endian); the two formats
only differ in the byte order of the two bytes that make up the half.
@param[in] format the current format (for diagnostics)
@param[in] little_endian whether the two bytes are little endian (BJData)
or big endian (CBOR)
@return whether reading and decoding succeeded
*/
bool get_half_float(const input_format_t format, const bool little_endian)
{
const auto byte1_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number")))
{
return false;
}
const auto byte2_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number")))
{
return false;
}
const auto byte1 = static_cast<unsigned char>(byte1_raw);
const auto byte2 = static_cast<unsigned char>(byte2_raw);
// Code from RFC 8949, Appendix D, Figure 3:
// As half-precision floating-point numbers were only added
// to IEEE 754 in 2008, today's programming platforms often
// still only have limited support for them. It is very
// easy to include at least decoding support for them even
// without such support. An example of a small decoder for
// half-precision floating-point numbers in the C language
// is shown in Fig. 3.
const auto half = little_endian
? static_cast<unsigned int>((byte2 << 8u) + byte1)
: static_cast<unsigned int>((byte1 << 8u) + byte2);
const double val = [&half]
{
const int exp = (half >> 10u) & 0x1Fu;
const unsigned int mant = half & 0x3FFu;
JSON_ASSERT(exp <= 31);
JSON_ASSERT(mant <= 1023);
switch (exp)
{
case 0:
return std::ldexp(mant, -24);
case 31:
return (mant == 0)
? std::numeric_limits<double>::infinity()
: std::numeric_limits<double>::quiet_NaN();
default:
return std::ldexp(mant + 1024, exp - 25);
}
}();
return sax->number_float((half & 0x8000u) != 0
? static_cast<number_float_t>(-val)
: static_cast<number_float_t>(val), "");
}
/*!
@brief create a string by reading characters from the input
@@ -4308,38 +4237,61 @@ class binary_reader
/// BON8: number of bytes in @ref bon8_pushback
std::size_t bon8_pushback_size = 0;
// excluded markers in bjdata optimized type
#define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \
make_array<char_int_type>('F', 'H', 'N', 'S', 'T', 'Z', '[', '{')
#define JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ \
make_array<bjd_type>( \
bjd_type{'B', "byte"}, \
bjd_type{'C', "char"}, \
bjd_type{'D', "double"}, \
bjd_type{'I', "int16"}, \
bjd_type{'L', "int64"}, \
bjd_type{'M', "uint64"}, \
bjd_type{'U', "uint8"}, \
bjd_type{'d', "single"}, \
bjd_type{'i', "int8"}, \
bjd_type{'l', "int32"}, \
bjd_type{'m', "uint32"}, \
bjd_type{'u', "uint16"})
JSON_PRIVATE_UNLESS_TESTED:
// lookup tables
// NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes)
const decltype(JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_) bjd_optimized_type_markers =
JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_;
/*!
@brief whether @a marker is excluded from BJData's optimized ND-array types
@return whether @a marker is one of 'F', 'H', 'N', 'S', 'T', 'Z', '[', '{'
using bjd_type = std::pair<char_int_type, string_t>;
// NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes)
const decltype(JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_) bjd_types_map =
JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_;
Mirrors binary_writer's @ref binary_writer::is_bjdata_excluded_type_marker
"is_bjdata_excluded_type_marker()`, which encodes the same list the other
way; keep the two in sync.
*/
static constexpr bool is_bjd_excluded_optimized_type(const char_int_type marker) noexcept
{
return marker == '[' || marker == '{' || marker == 'S' || marker == 'H'
|| marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z';
}
#undef JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_
#undef JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_
/*!
@brief look up the ND-array element type name for a BJData dtype marker
@return the type name ("uint8", "int8", ...), or nullptr if @a marker does
not name a known dtype
A C++11 `constexpr` function cannot contain a `switch`, so this is a
plain (non-constexpr) switch instead.
*/
static const char* bjd_type_name(const char_int_type marker)
{
switch (marker)
{
case 'B':
return "byte";
case 'C':
return "char";
case 'D':
return "double";
case 'I':
return "int16";
case 'L':
return "int64";
case 'M':
return "uint64";
case 'U':
return "uint8";
case 'd':
return "single";
case 'i':
return "int8";
case 'l':
return "int32";
case 'm':
return "uint32";
case 'u':
return "uint16";
default:
return nullptr;
}
}
};
#ifndef JSON_HAS_CPP_17
@@ -8,12 +8,12 @@
#pragma once
#include <algorithm> // min
#include <array> // array
#include <cstddef> // size_t
#include <cstdint> // uint32_t
#include <cstring> // strlen
#include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next
#include <memory> // shared_ptr, make_shared, addressof
#include <numeric> // accumulate
#include <streambuf> // streambuf
#include <string> // string, char_traits
#include <type_traits> // enable_if, is_base_of, is_pointer, is_integral, remove_pointer
@@ -28,6 +28,7 @@
#include <nlohmann/detail/iterators/iterator_traits.hpp>
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/type_traits.hpp>
#include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -82,8 +83,9 @@ class file_input_adapter
};
/*!
Input adapter for a (caching) istream. Ignores a UFT Byte Order Mark at
beginning of input. Does not support changing the underlying std::streambuf
Input adapter for a (caching) istream. Does not skip a UTF Byte Order Mark
itself; that is done by the lexer's skip_bom(). Does not support changing
the underlying std::streambuf
in mid-input. Maintains underlying std::istream and std::streambuf to support
subsequent use of standard std::istream operations to process any input
characters following those used in parsing the JSON input. Clears the
@@ -454,32 +456,14 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
// get the current character
const auto wc = input.get_character();
// UTF-32 to UTF-8 encoding
if (wc < 0x80)
if (wc <= 0x10FFFF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
utf8_bytes_filled = 1;
}
else if (wc <= 0x7FF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u) & 0x1Fu));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 2;
}
else if (wc <= 0xFFFF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u) & 0x0Fu));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 3;
}
else if (wc <= 0x10FFFF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | ((static_cast<unsigned int>(wc) >> 18u) & 0x07u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 12u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 4;
// UTF-32 to UTF-8 encoding
utf8_bytes_filled = 0;
encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
{
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
});
}
else
{
@@ -516,24 +500,15 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
// get the current character
const auto wc = input.get_character();
// UTF-16 to UTF-8 encoding
if (wc < 0x80)
if (0xD800 > wc || wc >= 0xE000)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
utf8_bytes_filled = 1;
}
else if (wc <= 0x7FF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u)));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 2;
}
else if (0xD800 > wc || wc >= 0xE000)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u)));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 3;
// a UTF-16 code unit outside the surrogate range is a valid
// code point (at most U+FFFF) on its own
utf8_bytes_filled = 0;
encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
{
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
});
}
else
{
@@ -551,11 +526,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu));
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (charcode & 0x3Fu));
utf8_bytes_filled = 4;
utf8_bytes_filled = 0;
encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
{
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
});
valid_pair = true;
}
}
@@ -769,6 +744,9 @@ struct container_input_adapter_factory< ContainerType,
static adapter_type create(ContainerType&& container)
{
// container is forwarded twice on purpose: the resulting begin/end
// iterator types must match adapter_type, computed the same way
// NOLINTNEXTLINE(bugprone-use-after-move)
return input_adapter(begin(std::forward<ContainerType>(container)), end(std::forward<ContainerType>(container)));
}
};
@@ -884,9 +862,9 @@ auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) /
return input_adapter(array, array + N);
}
// This class only handles inputs of input_buffer_adapter type.
// It's required so that expressions like {ptr, len} can be implicitly cast
// to the correct adapter.
// This class only handles inputs that construct a contiguous_bytes_input_adapter
// (e.g. span_input_adapter). It's required so that expressions like {ptr, len}
// can be implicitly cast to the correct adapter.
class span_input_adapter
{
public:
+89 -142
View File
@@ -10,6 +10,7 @@
#include <algorithm> // find_if, min
#include <cstddef>
#include <limits> // numeric_limits
#include <string> // string
#include <type_traits> // enable_if_t
#include <utility> // move, pair
@@ -175,6 +176,88 @@ template<typename ArrayType>
inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/)
{}
#if JSON_DIAGNOSTIC_POSITIONS
/*!
@brief set the diagnostic positions of a value the DOM SAX parsers just stored
Shared by json_sax_dom_parser and json_sax_dom_callback_parser. basic_json
befriends this struct, as the position members are private.
*/
struct diagnostic_positions
{
/*!
@param[in,out] v the value that was just parsed
@param[in] lexer the lexer that read it, or nullptr to leave @a v alone
*/
template<typename BasicJsonType, typename LexerType>
static void set_from_lexer(BasicJsonType& v, LexerType* lexer)
{
if (lexer)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = lexer->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = lexer->get_token_start_position();
break;
}
case value_t::discarded:
{
// an object or array the callback of
// json_sax_dom_callback_parser rejected has no position
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - lexer->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
}
}
};
#endif
/*!
@brief SAX implementation to create a JSON value from SAX events
@@ -376,76 +459,6 @@ class json_sax_dom_parser
private:
#if JSON_DIAGNOSTIC_POSITIONS
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
{
if (m_lexer_ref)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = m_lexer_ref->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = m_lexer_ref->get_token_start_position();
break;
}
// As we handle the start and end positions for values created during parsing,
// we do not expect the following value type to be called. Regardless, set the positions
// in case this is created manually or through a different constructor. Exclude from lcov
// since the exact condition of this switch is esoteric.
// LCOV_EXCL_START
case value_t::discarded:
{
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
// LCOV_EXCL_STOP
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - m_lexer_ref->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
}
}
}
#endif
/*!
@invariant If the ref stack is empty, then the passed value will be the new
root.
@@ -461,7 +474,7 @@ class json_sax_dom_parser
root = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS
handle_diagnostic_positions_for_json_value(root);
diagnostic_positions::set_from_lexer(root, m_lexer_ref);
#endif
return &root;
@@ -474,7 +487,7 @@ class json_sax_dom_parser
ref_stack.back()->m_data.m_value.array->emplace_back(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS
handle_diagnostic_positions_for_json_value(ref_stack.back()->m_data.m_value.array->back());
diagnostic_positions::set_from_lexer(ref_stack.back()->m_data.m_value.array->back(), m_lexer_ref);
#endif
return &(ref_stack.back()->m_data.m_value.array->back());
@@ -485,7 +498,7 @@ class json_sax_dom_parser
*object_element = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS
handle_diagnostic_positions_for_json_value(*object_element);
diagnostic_positions::set_from_lexer(*object_element, m_lexer_ref);
#endif
return object_element;
@@ -674,7 +687,7 @@ class json_sax_dom_callback_parser
#if JSON_DIAGNOSTIC_POSITIONS
// Set start/end positions for discarded object.
handle_diagnostic_positions_for_json_value(*ref_stack.back());
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref);
#endif
}
}
@@ -790,7 +803,7 @@ class json_sax_dom_callback_parser
#if JSON_DIAGNOSTIC_POSITIONS
// Set start/end positions for discarded array.
handle_diagnostic_positions_for_json_value(*ref_stack.back());
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref);
#endif
}
}
@@ -843,72 +856,6 @@ class json_sax_dom_callback_parser
private:
#if JSON_DIAGNOSTIC_POSITIONS
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
{
if (m_lexer_ref)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = m_lexer_ref->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = m_lexer_ref->get_token_start_position();
break;
}
case value_t::discarded:
{
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - m_lexer_ref->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
}
}
}
#endif
/// if there is a pending duplicate-key stash entry for this exact slot,
/// remove it from the stash; if restore_value is true, the stashed
/// previous value is moved back into the slot first (use this when the
@@ -1030,7 +977,7 @@ class json_sax_dom_callback_parser
auto value = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS
handle_diagnostic_positions_for_json_value(value);
diagnostic_positions::set_from_lexer(value, m_lexer_ref);
#endif
// check callback
+35 -70
View File
@@ -10,6 +10,7 @@
#include <array> // array
#include <cstddef> // size_t
#include <cstdint> // uint32_t
#include <cstdio> // snprintf
#include <initializer_list> // initializer_list
#include <string> // char_traits, string
@@ -22,6 +23,7 @@
#include <nlohmann/detail/input/string_scan.hpp>
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/type_traits.hpp>
#include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -258,9 +260,9 @@ class lexer : public lexer_base<BasicJsonType>
}
/*!
@brief get codepoint from 4 hex characters following `\u`
@brief get codepoint from 4 hex characters following `\\u`
For input "\u c1 c2 c3 c4" the codepoint is:
For input "\\u c1 c2 c3 c4" the codepoint is:
(c1 * 0x1000) + (c2 * 0x0100) + (c3 * 0x0010) + c4
= (c1 << 12) + (c2 << 8) + (c3 << 4) + (c4 << 0)
@@ -532,32 +534,10 @@ class lexer : public lexer_base<BasicJsonType>
JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF);
// translate codepoint into bytes
if (codepoint < 0x80)
encode_utf8(static_cast<std::uint32_t>(codepoint), [this](std::uint32_t byte)
{
// 1-byte characters: 0xxxxxxx (ASCII)
add(static_cast<char_int_type>(codepoint));
}
else if (codepoint <= 0x7FF)
{
// 2-byte characters: 110xxxxx 10xxxxxx
add(static_cast<char_int_type>(0xC0u | (static_cast<unsigned int>(codepoint) >> 6u)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
else if (codepoint <= 0xFFFF)
{
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
add(static_cast<char_int_type>(0xE0u | (static_cast<unsigned int>(codepoint) >> 12u)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
else
{
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
add(static_cast<char_int_type>(0xF0u | (static_cast<unsigned int>(codepoint) >> 18u)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 12u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
add(static_cast<char_int_type>(byte));
});
break;
}
@@ -1457,45 +1437,30 @@ scan_number_done:
*/
token_type convert_number(token_type number_type, std::size_t mantissa_end)
{
// If the caller does not need the converted value (only whether the
// input is syntactically valid; see json_sax_acceptor/accept()), an
// unsigned/integer token can be reported without calling
// strtoull()/strtoll() at all, *provided* we can already tell from
// the digit count alone that the conversion cannot overflow 64 bits.
// Such tokens are always finite and are accepted unconditionally by
// the parser regardless of their actual value (parser::sax_parse_internal()
// never checks finiteness for value_unsigned/value_integer), so the
// classification below is all that is needed.
// accept() only needs to know whether the input is valid, so it sets
// discard_number_values (see json.hpp), and an integer token whose
// digit count shows that it fits is reported without calling
// convert_integer(). A number with up to 18 digits always fits into
// both std::uint64_t and std::int64_t (18 nines is about 1e18, below
// INT64_MAX, which is about 9.2e18). Longer tokens take the exact path
// below, including the fallback to floating point when the value does
// not fit.
//
// A decimal number with up to 18 digits is always representable in
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
// could not have set errno to ERANGE for it. Numbers with more digits
// (rare in practice) fall through to the exact code below, unchanged,
// so their handling -- including reclassification to value_float when
// the value overflows 64 bits, and rejection when it is not even
// finite as a double -- is bit-for-bit identical to before this
// optimization.
// With a narrower number_unsigned_t/number_integer_t (e.g.
// std::uint32_t), the exact path would reclassify some of these tokens
// as (finite) floats, while this check reports integers. That does not
// change the result of accept(): it always parses through
// json_sax_acceptor, whose number callbacks discard their argument and
// return true, and the parser rejects neither integers nor finite
// floats. value_unsigned/value_integer are left unset here, so a caller
// that reads the converted value must not set discard_number_values.
//
// Note this reasons about std::uint64_t/std::int64_t, not about
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
// narrower, template parameters -- e.g. std::uint32_t). That is fine
// *only* because discard_number_values is exclusively set by
// accept() (see json.hpp), and accept() always parses through the
// library's own json_sax_acceptor -- never a user-supplied SAX
// consumer -- whose number_unsigned()/number_integer()/number_float()
// callbacks unconditionally discard their argument and return true.
// So for every caller that can reach this branch, neither the token
// classification below nor the eventual (possibly narrowed, and on
// this fast path left stale/unset) value_unsigned/value_integer is
// ever consulted -- an unsigned/integer token is accepted outright,
// and even a >18-digit token that this fast path deliberately falls
// through for is, once reclassified to value_float, still finite
// (and thus accepted) for any digit count that fits in number_unsigned_t
// or number_integer_t regardless of that type's width. If this
// function is ever taught to run with discard_number_values true for
// a caller that *does* read the converted value, this reasoning (and
// the fast path below) would need to be revisited.
// On contiguous input, scan_number_bulk_contiguous() converts integer
// tokens itself and does not pass them to this function, unless
// JSON_DIAGNOSTIC_POSITIONS is enabled. This check is therefore only
// reached for input without bulk access (e.g. streams), with
// JSON_DIAGNOSTIC_POSITIONS, or when scan_number_bulk_contiguous()
// falls back to scan_number().
if (discard_number_values)
{
constexpr std::size_t safe_digit_count = 18;
@@ -1925,7 +1890,7 @@ scan_number_done:
return value_float;
}
/// return current string value (implicitly resets the token; useful only once)
/// return current string value
string_t& get_string()
{
// a number token holds '.' regardless of the locale (#4084)
@@ -2227,11 +2192,11 @@ scan_number_done:
/// the position of the decimal point in token_buffer
std::size_t decimal_point_position = std::string::npos;
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
/// token classification and never looks at the converted numeric value;
/// when set, scan_number() may skip strtoull()/strtoll() for
/// value_unsigned/value_integer tokens whose digit count guarantees they
/// fit into 64 bits (see scan_number())
/// whether the caller only needs the token types and never looks at the
/// converted numeric values; set only by accept(), which parses through
/// json_sax_acceptor. When set, convert_number() skips converting integer
/// tokens whose digit count guarantees that they fit into 64 bits (see
/// there)
const bool discard_number_values = false;
};
+50 -43
View File
@@ -54,7 +54,8 @@ using parser_callback_t =
/*!
@brief syntax analysis
This class implements a recursive descent parser.
This class implements an iterative parser that keeps the open containers on
an explicit stack and reports what it reads as SAX events.
*/
template<typename BasicJsonType, typename InputAdapterType>
class parser
@@ -98,28 +99,9 @@ class parser
if (callback)
{
json_sax_dom_callback_parser<BasicJsonType, InputAdapterType> sdp(result, callback, allow_exceptions, &m_lexer);
sax_parse_internal(&sdp);
if (strict)
{
// in strict mode, input must be completely read
if (get_token() != token_type::end_of_input)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(),
exception_message(token_type::end_of_input, "value"), nullptr));
}
}
else
{
// the caller keeps using the input: position it right after
// the value by leaving the character that terminated it
m_lexer.release_lookahead();
}
// in case of an error, return a discarded value
if (sdp.is_errored())
if (!parse_dom(sdp, strict))
{
result = value_t::discarded;
return;
@@ -135,26 +117,9 @@ class parser
else
{
json_sax_dom_parser<BasicJsonType, InputAdapterType> sdp(result, allow_exceptions, &m_lexer);
sax_parse_internal(&sdp);
if (strict)
{
// in strict mode, input must be completely read
if (get_token() != token_type::end_of_input)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
}
}
else
{
// see above
m_lexer.release_lookahead();
}
// in case of an error, return a discarded value
if (sdp.is_errored())
if (!parse_dom(sdp, strict))
{
result = value_t::discarded;
return;
@@ -207,6 +172,46 @@ class parser
}
private:
/*!
@brief run a DOM SAX parser to completion and position the lexer
Shared by both branches of @ref parse(): builds no SAX parser itself,
but drives an already-constructed @a json_sax_dom_parser or
@ref json_sax_dom_callback_parser through @ref sax_parse_internal(),
then applies the strict-EOF check (reporting parse_error.101 through
@a sdp on failure) or, in non-strict mode, releases the lookahead so
the caller can keep reading the input right after the parsed value.
@param[in,out] sdp the DOM SAX parser to run
@param[in] strict whether to expect the last token to be EOF
@return whether @a sdp did not report an error
*/
template<typename DomSax>
bool parse_dom(DomSax& sdp, const bool strict)
{
sax_parse_internal(&sdp);
if (strict)
{
// in strict mode, input must be completely read
if (get_token() != token_type::end_of_input)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(),
exception_message(token_type::end_of_input, "value"), nullptr));
}
}
else
{
// the caller keeps using the input: position it right after
// the value by leaving the character that terminated it
m_lexer.release_lookahead();
}
return !sdp.is_errored();
}
template<typename SAX>
JSON_HEDLEY_NON_NULL(2)
bool sax_parse_internal(SAX* sax)
@@ -439,8 +444,9 @@ class parser
// We are done with this array. Before we can parse a
// new value, we need to evaluate the new state first.
// By setting skip_to_state_evaluation to false, we
// are effectively jumping to the beginning of this if.
// By setting skip_to_state_evaluation to true, the next
// iteration skips parsing a value and evaluates the
// enclosing state directly.
JSON_ASSERT(!states.empty());
states.pop_back();
skip_to_state_evaluation = true;
@@ -500,8 +506,9 @@ class parser
// We are done with this object. Before we can parse a
// new value, we need to evaluate the new state first.
// By setting skip_to_state_evaluation to false, we
// are effectively jumping to the beginning of this if.
// By setting skip_to_state_evaluation to true, the next
// iteration skips parsing a value and evaluates the
// enclosing state directly.
JSON_ASSERT(!states.empty());
states.pop_back();
skip_to_state_evaluation = true;