Merge branch 'develop' into claude/binary-utf8-roundtrip-5651

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-10-01 07:44:29 +02:00
295 changed files with 5316 additions and 15043 deletions
+311 -359
View File
@@ -8,7 +8,6 @@
#pragma once
#include <algorithm> // generate_n
#include <array> // array
#include <cmath> // ldexp
#include <cstddef> // size_t
@@ -123,16 +122,16 @@ class binary_reader
~binary_reader() = default;
/*!
@param[in] format the binary format to parse
@brief parse in the format the constructor was given
@param[in] sax_ a SAX event processor
@param[in] strict whether to expect the input to be consumed completed
@param[in] tag_handler how to treat CBOR tags
@return whether parsing was successful
*/
JSON_HEDLEY_NON_NULL(3)
bool sax_parse(const input_format_t format,
json_sax_t* sax_,
JSON_HEDLEY_NON_NULL(2)
bool sax_parse(json_sax_t* sax_,
const bool strict = true,
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error)
{
@@ -141,14 +140,14 @@ class binary_reader
bon8_pushback_size = 0;
bool result = false;
switch (format)
switch (input_format)
{
case input_format_t::bson:
result = parse_bson_internal();
break;
case input_format_t::cbor:
result = parse_cbor_internal(true, tag_handler);
result = parse_cbor_internal(tag_handler);
break;
case input_format_t::msgpack:
@@ -271,6 +270,22 @@ class binary_reader
return enter_container(/*is_object*/true, len, type_marker);
}
/*!
@brief close the innermost open array or object
Pops the container opened by the matching @ref enter_container call and
emits the SAX end event. Every format-specific driver otherwise repeated
the same pop-then-dispatch sequence at its own close site.
@return whether the SAX parser accepted the end event
*/
bool leave_container()
{
const bool is_object = container_stack.back().is_object;
container_stack.pop_back();
return is_object ? sax->end_object() : sax->end_array();
}
//////////
// BSON //
//////////
@@ -355,8 +370,8 @@ class binary_reader
if (element_type == 0) // end of the innermost document
{
// a copy, not a reference: it must stay valid across the
// pop_back() below, which destroys the container_stack
// element it would otherwise alias
// pop_back() inside leave_container() below, which destroys
// the container_stack element it would otherwise alias
const container_frame top = container_stack.back();
if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(top.start_position, top.declared_size)))
@@ -364,8 +379,7 @@ class binary_reader
return false;
}
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -407,7 +421,7 @@ class binary_reader
@brief Parses a C-style string from the BSON input.
@param[in,out] result A reference to the string variable where the read
string is to be stored.
@return `true` if the \x00-byte indicating the end of the string was
@return `true` if the \\x00-byte indicating the end of the string was
encountered before the EOF; false` indicates an unexpected EOF.
*/
bool get_bson_cstr(string_t& result)
@@ -437,7 +451,7 @@ class binary_reader
@brief read a C-style string from contiguous input in one step
@param[in,out] result the string to append to
@return whether the string was read; if the input has no \x00-byte, nothing
@return whether the string was read; if the input has no \\x00-byte, nothing
is read, and @ref get_bson_cstr reports the end of the input
*/
bool get_bson_cstr_bulk(string_t& result, std::true_type /*bulk*/)
@@ -860,29 +874,13 @@ class binary_reader
return enter_array(conditional_static_cast<std::size_t>(static_cast<unsigned int>(current) & 0x1Fu));
case 0x98: // array (one-byte uint8_t for n follows)
{
std::uint8_t len{};
return get_number(input_format_t::cbor, len) && enter_array(static_cast<std::size_t>(len));
}
case 0x99: // array (two-byte uint16_t for n follow)
{
std::uint16_t len{};
return get_number(input_format_t::cbor, len) && enter_array(static_cast<std::size_t>(len));
}
case 0x9A: // array (four-byte uint32_t for n follow)
{
std::uint32_t len{};
std::size_t size{};
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size);
}
case 0x9B: // array (eight-byte uint64_t for n follow)
{
std::uint64_t len{};
std::size_t size{};
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size);
return get_cbor_argument(len) && get_cbor_container_size(len, size, "array") && enter_array(size);
}
case 0x9F: // array (indefinite length)
@@ -916,35 +914,19 @@ class binary_reader
return enter_object(conditional_static_cast<std::size_t>(static_cast<unsigned int>(current) & 0x1Fu));
case 0xB8: // map (one-byte uint8_t for n follows)
{
std::uint8_t len{};
return get_number(input_format_t::cbor, len) && enter_object(static_cast<std::size_t>(len));
}
case 0xB9: // map (two-byte uint16_t for n follow)
{
std::uint16_t len{};
return get_number(input_format_t::cbor, len) && enter_object(static_cast<std::size_t>(len));
}
case 0xBA: // map (four-byte uint32_t for n follow)
{
std::uint32_t len{};
std::size_t size{};
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size);
}
case 0xBB: // map (eight-byte uint64_t for n follow)
{
std::uint64_t len{};
std::size_t size{};
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size);
return get_cbor_argument(len) && get_cbor_container_size(len, size, "map") && enter_object(size);
}
case 0xBF: // map (indefinite length)
return enter_object(detail::unknown_size());
case 0xC0: // tagged item
case 0xC0: // tagged item (tag value 0-23, in the head itself)
case 0xC1:
case 0xC2:
case 0xC3:
@@ -968,6 +950,22 @@ class binary_reader
case 0xD5:
case 0xD6:
case 0xD7:
{
if (tag_handler == cbor_tag_handler_t::error)
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr));
}
// ignore and store: the tag value is already in the head, so
// there is nothing left to read here; the tagged value that
// follows is read by the loop in parse_cbor_internal() rather
// than by recursing here
tag_pending = true;
return true;
}
case 0xD8: // tagged item (1 byte follows)
case 0xD9: // tagged item (2 bytes follow)
case 0xDA: // tagged item (4 bytes follow)
@@ -984,47 +982,11 @@ class binary_reader
case cbor_tag_handler_t::ignore:
{
// ignore binary subtype
switch (current)
// ignore the tag's binary subtype argument
std::uint64_t subtype_to_ignore{};
if (!get_cbor_argument(subtype_to_ignore))
{
case 0xD8:
{
std::uint8_t subtype_to_ignore{};
if (!get_number(input_format_t::cbor, subtype_to_ignore))
{
return false;
}
break;
}
case 0xD9:
{
std::uint16_t subtype_to_ignore{};
if (!get_number(input_format_t::cbor, subtype_to_ignore))
{
return false;
}
break;
}
case 0xDA:
{
std::uint32_t subtype_to_ignore{};
if (!get_number(input_format_t::cbor, subtype_to_ignore))
{
return false;
}
break;
}
case 0xDB:
{
std::uint64_t subtype_to_ignore{};
if (!get_number(input_format_t::cbor, subtype_to_ignore))
{
return false;
}
break;
}
default:
break;
return false;
}
// the tagged value follows; it is read by the loop in
// parse_cbor_internal() rather than by recursing here
@@ -1034,57 +996,15 @@ class binary_reader
case cbor_tag_handler_t::store:
{
binary_t b;
// use binary subtype and store in a binary container
switch (current)
std::uint64_t subtype{};
if (!get_cbor_argument(subtype))
{
case 0xD8:
{
std::uint8_t subtype{};
if (!get_number(input_format_t::cbor, subtype))
{
return false;
}
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
break;
}
case 0xD9:
{
std::uint16_t subtype{};
if (!get_number(input_format_t::cbor, subtype))
{
return false;
}
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
break;
}
case 0xDA:
{
std::uint32_t subtype{};
if (!get_number(input_format_t::cbor, subtype))
{
return false;
}
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
break;
}
case 0xDB:
{
std::uint64_t subtype{};
if (!get_number(input_format_t::cbor, subtype))
{
return false;
}
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
break;
}
default:
{
// as above, the tagged value is read by the caller
tag_pending = true;
return true;
}
return false;
}
binary_t b;
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
get();
// a byte string (the heads accepted by get_cbor_binary) keeps the tag as subtype
if ((current >= 0x40 && current <= 0x5B) || current == 0x5F)
@@ -1115,52 +1035,7 @@ class binary_reader
return sax->null();
case 0xF9: // Half-Precision Float (two-byte IEEE 754)
{
const auto byte1_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number")))
{
return false;
}
const auto byte2_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number")))
{
return false;
}
const auto byte1 = static_cast<unsigned char>(byte1_raw);
const auto byte2 = static_cast<unsigned char>(byte2_raw);
// Code from RFC 8949, Appendix D, Figure 3:
// As half-precision floating-point numbers were only added
// to IEEE 754 in 2008, today's programming platforms often
// still only have limited support for them. It is very
// easy to include at least decoding support for them even
// without such support. An example of a small decoder for
// half-precision floating-point numbers in the C language
// is shown in Fig. 3.
const auto half = static_cast<unsigned int>((byte1 << 8u) + byte2);
const double val = [&half]
{
const int exp = (half >> 10u) & 0x1Fu;
const unsigned int mant = half & 0x3FFu;
JSON_ASSERT(exp <= 31);
JSON_ASSERT(mant <= 1023);
switch (exp)
{
case 0:
return std::ldexp(mant, -24);
case 31:
return (mant == 0)
? std::numeric_limits<double>::infinity()
: std::numeric_limits<double>::quiet_NaN();
default:
return std::ldexp(mant + 1024, exp - 25);
}
}();
return sax->number_float((half & 0x8000u) != 0
? static_cast<number_float_t>(-val)
: static_cast<number_float_t>(val), "");
}
return get_half_float(input_format_t::cbor, false);
case 0xFA: // Single-Precision Float (four-byte IEEE 754)
{
@@ -1539,6 +1414,73 @@ class binary_reader
}
}
/*!
@brief read a CBOR argument (additional information 24-27) of the width
@ref current announces
The lower 5 bits of @a current (0x18-0x1B) select a 1/2/4/8-byte
big-endian unsigned integer that follows the head byte; this is shared by
every major type that uses this encoding (unsigned/negative integers,
strings, arrays, maps, tags). Reading always goes through @ref get_number,
so EOF is reported the same way as before this helper existed.
@param[out] value the decoded argument
@return whether reading succeeded
*/
bool get_cbor_argument(std::uint64_t& value)
{
switch (current & 0x1F)
{
case 0x18: // 1 byte
{
std::uint8_t n{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n)))
{
return false;
}
value = n;
return true;
}
case 0x19: // 2 bytes
{
std::uint16_t n{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n)))
{
return false;
}
value = n;
return true;
}
case 0x1A: // 4 bytes
{
std::uint32_t n{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n)))
{
return false;
}
value = n;
return true;
}
case 0x1B: // 8 bytes
{
std::uint64_t n{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n)))
{
return false;
}
value = n;
return true;
}
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
return false; // LCOV_EXCL_LINE
}
}
/*!
@brief narrow a definite CBOR array/map length to std::size_t
@@ -1571,19 +1513,15 @@ class binary_reader
enclosing container after each element, so that the nesting depth of the
input costs heap rather than native stack (see #5104).
@param[in] get_char whether a new character should be retrieved from the
input (true) or whether the last read character
@a current should be considered instead
@param[in] tag_handler how CBOR tags should be treated
@return whether reading the value succeeded
*/
bool parse_cbor_internal(const bool get_char,
const cbor_tag_handler_t tag_handler)
bool parse_cbor_internal(const cbor_tag_handler_t tag_handler)
{
// whether the next value starts at a fresh byte or at the one already
// read into `current`
bool fetch = get_char;
bool fetch = true;
// the key currently being read; hoisted out of the loop so that its
// capacity is reused across elements and across nesting levels
@@ -1626,8 +1564,7 @@ class binary_reader
if (at_end)
{
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -1676,9 +1613,6 @@ class binary_reader
// MsgPack //
/////////////
/*!
@return whether a valid MessagePack value was passed to the SAX parser
*/
/*!
@brief read one MessagePack value
@@ -2377,8 +2311,7 @@ class binary_reader
if (container_stack.back().remaining == 0)
{
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -2423,20 +2356,16 @@ class binary_reader
////////////
/*!
@param[in] get_char whether a new character should be retrieved from the
input (true, default) or whether the last read
character should be considered instead
@return whether a valid UBJSON value was passed to the SAX parser
*/
bool parse_ubjson_internal(const bool get_char = true)
bool parse_ubjson_internal()
{
// the key currently being read; hoisted out of the loop so that its
// capacity is reused across elements and across nesting levels
string_t key;
// the type marker of the value to read next
char_int_type prefix = get_char ? get_ignore_noop() : current;
char_int_type prefix = get_ignore_noop();
while (true)
{
@@ -2511,8 +2440,7 @@ class binary_reader
break;
}
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -2720,6 +2648,42 @@ class binary_reader
return true;
}
/*!
@brief read a UBJSON/BJData optimized-container count of a signed marker
type ('i', 'I', 'l', 'L') and narrow it to std::size_t
Every signed count marker rejects a negative value the same way (error
113); the value_in_range_of check additionally needed for 'L' is only
ever live when @a SignedType is std::int64_t on a target where
std::size_t is narrower (e.g. 32-bit), since 'i'/'I'/'l' can never exceed
std::size_t there.
@tparam SignedType std::int8_t, std::int16_t, std::int32_t or std::int64_t
@param[out] result the count narrowed to std::size_t
@return whether reading and validating succeeded
*/
template<typename SignedType>
bool get_ubjson_signed_count(std::size_t& result)
{
SignedType number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (JSON_HEDLEY_UNLIKELY(number < 0))
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
if (JSON_HEDLEY_UNLIKELY(!value_in_range_of<std::size_t>(number)))
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408,
exception_message(input_format, "integer value overflow", "size"), nullptr));
}
result = static_cast<std::size_t>(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char
return true;
}
/*!
@param[out] result determined size
@param[in,out] is_ndarray for input, `true` means already inside an ndarray vector
@@ -2752,73 +2716,16 @@ class binary_reader
}
case 'i':
{
std::int8_t number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (number < 0)
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
result = static_cast<std::size_t>(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char
return true;
}
return get_ubjson_signed_count<std::int8_t>(result);
case 'I':
{
std::int16_t number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (number < 0)
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
result = static_cast<std::size_t>(number);
return true;
}
return get_ubjson_signed_count<std::int16_t>(result);
case 'l':
{
std::int32_t number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (number < 0)
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
result = static_cast<std::size_t>(number);
return true;
}
return get_ubjson_signed_count<std::int32_t>(result);
case 'L':
{
std::int64_t number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (number < 0)
{
return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
if (!value_in_range_of<std::size_t>(number))
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408,
exception_message(input_format, "integer value overflow", "size"), nullptr));
}
result = static_cast<std::size_t>(number);
return true;
}
return get_ubjson_signed_count<std::int64_t>(result);
case 'u':
{
@@ -2909,16 +2816,23 @@ class binary_reader
result = 1;
for (auto i : dim)
{
// Pre-multiplication overflow check: if i > 0 and result > SIZE_MAX/i, then result*i would overflow.
// This check must happen before multiplication since overflow detection after the fact is unreliable
// as modular arithmetic can produce any value, not just 0 or SIZE_MAX.
if (JSON_HEDLEY_UNLIKELY(i > 0 && result > (std::numeric_limits<std::size_t>::max)() / i))
// Pre-multiplication overflow check: since the loop above
// already rejected any zero dimension, i is always > 0
// here, so result > SIZE_MAX/i means result*i would
// overflow. This check must happen before multiplication
// since overflow detection after the fact is unreliable,
// as modular arithmetic can produce any value, not just 0
// or SIZE_MAX.
if (JSON_HEDLEY_UNLIKELY(result > (std::numeric_limits<std::size_t>::max)() / i))
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr));
}
result *= i;
// Additional post-multiplication check to catch any edge cases the pre-check might miss
if (result == 0 || result == npos)
// the pre-check above already rules out result becoming 0
// by overflow; the only value it cannot rule out is an
// exact match with npos, the sentinel reserved for an
// unknown-size container (see get_ubjson_size_type())
if (result == npos)
{
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr));
}
@@ -2979,7 +2893,7 @@ class binary_reader
{
result.second = get(); // must not ignore 'N', because 'N' maybe the type
if (input_format == input_format_t::bjdata
&& JSON_HEDLEY_UNLIKELY(std::binary_search(bjd_optimized_type_markers.begin(), bjd_optimized_type_markers.end(), result.second)))
&& JSON_HEDLEY_UNLIKELY(is_bjd_excluded_optimized_type(result.second)))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
@@ -3123,50 +3037,7 @@ class binary_reader
{
break;
}
const auto byte1_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number")))
{
return false;
}
const auto byte2_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number")))
{
return false;
}
const auto byte1 = static_cast<unsigned char>(byte1_raw);
const auto byte2 = static_cast<unsigned char>(byte2_raw);
// Code from RFC 8949, Appendix D, Figure 3:
// As half-precision floating-point numbers were only added
// to IEEE 754 in 2008, today's programming platforms often
// still only have limited support for them. It is very
// easy to include at least decoding support for them even
// without such support. An example of a small decoder for
// half-precision floating-point numbers in the C language
// is shown in Fig. 3.
const auto half = static_cast<unsigned int>((byte2 << 8u) + byte1);
const double val = [&half]
{
const int exp = (half >> 10u) & 0x1Fu;
const unsigned int mant = half & 0x3FFu;
JSON_ASSERT(exp <= 31);
JSON_ASSERT(mant <= 1023);
switch (exp)
{
case 0:
return std::ldexp(mant, -24);
case 31:
return (mant == 0)
? std::numeric_limits<double>::infinity()
: std::numeric_limits<double>::quiet_NaN();
default:
return std::ldexp(mant + 1024, exp - 25);
}
}();
return sax->number_float((half & 0x8000u) != 0
? static_cast<number_float_t>(-val)
: static_cast<number_float_t>(val), "");
return get_half_float(input_format, true);
}
case 'd':
@@ -3239,19 +3110,16 @@ class binary_reader
if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0)
{
size_and_type.second &= ~(static_cast<char_int_type>(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t)
{
return p.first < t;
});
const char* type_name = bjd_type_name(size_and_type.second);
string_t key = "_ArrayType_";
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second))
if (JSON_HEDLEY_UNLIKELY(type_name == nullptr))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
}
string_t type = it->second; // sax->string() takes a reference
string_t type = type_name; // sax->string() takes a reference
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type)))
{
return false;
@@ -3344,8 +3212,8 @@ class binary_reader
return enter_object(detail::unknown_size());
}
// Note, no reader for UBJSON binary types is implemented because they do
// not exist
// Note, UBJSON has no binary type of its own; BJData, which shares this
// reader, decodes optimized 'B' arrays as binary in get_ubjson_array().
bool get_ubjson_high_precision_number()
{
@@ -3543,8 +3411,7 @@ class binary_reader
if (at_end)
{
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -4052,7 +3919,7 @@ class binary_reader
#endif
}
/*
/*!
@brief read a number from the input
@tparam NumberType the type of the number
@@ -4062,10 +3929,10 @@ class binary_reader
@return whether conversion completed
@note This function needs to respect the system's endianness, because
bytes in CBOR, MessagePack, and UBJSON are stored in network order
(big endian) and therefore need reordering on little endian systems.
On the other hand, BSON and BJData use little endian and should reorder
on big endian systems.
bytes in CBOR, MessagePack, UBJSON, and BON8 are stored in network
order (big endian) and therefore need reordering on little endian
systems. On the other hand, BSON and BJData use little endian and
should reorder on big endian systems.
*/
template<typename NumberType, bool InputIsLittleEndian = false>
bool get_number(const input_format_t format, NumberType& result)
@@ -4083,6 +3950,68 @@ class binary_reader
return true;
}
/*!
@brief read and decode an IEEE 754 half-precision (16-bit) float
Used by CBOR (big endian) and BJData (little endian); the two formats
only differ in the byte order of the two bytes that make up the half.
@param[in] format the current format (for diagnostics)
@param[in] little_endian whether the two bytes are little endian (BJData)
or big endian (CBOR)
@return whether reading and decoding succeeded
*/
bool get_half_float(const input_format_t format, const bool little_endian)
{
const auto byte1_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number")))
{
return false;
}
const auto byte2_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number")))
{
return false;
}
const auto byte1 = static_cast<unsigned char>(byte1_raw);
const auto byte2 = static_cast<unsigned char>(byte2_raw);
// Code from RFC 8949, Appendix D, Figure 3:
// As half-precision floating-point numbers were only added
// to IEEE 754 in 2008, today's programming platforms often
// still only have limited support for them. It is very
// easy to include at least decoding support for them even
// without such support. An example of a small decoder for
// half-precision floating-point numbers in the C language
// is shown in Fig. 3.
const auto half = little_endian
? static_cast<unsigned int>((byte2 << 8u) + byte1)
: static_cast<unsigned int>((byte1 << 8u) + byte2);
const double val = [&half]
{
const int exp = (half >> 10u) & 0x1Fu;
const unsigned int mant = half & 0x3FFu;
JSON_ASSERT(exp <= 31);
JSON_ASSERT(mant <= 1023);
switch (exp)
{
case 0:
return std::ldexp(mant, -24);
case 31:
return (mant == 0)
? std::numeric_limits<double>::infinity()
: std::numeric_limits<double>::quiet_NaN();
default:
return std::ldexp(mant + 1024, exp - 25);
}
}();
return sax->number_float((half & 0x8000u) != 0
? static_cast<number_float_t>(-val)
: static_cast<number_float_t>(val), "");
}
/*!
@brief create a string by reading characters from the input
@@ -4312,38 +4241,61 @@ class binary_reader
/// BON8: number of bytes in @ref bon8_pushback
std::size_t bon8_pushback_size = 0;
// excluded markers in bjdata optimized type
#define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \
make_array<char_int_type>('F', 'H', 'N', 'S', 'T', 'Z', '[', '{')
#define JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ \
make_array<bjd_type>( \
bjd_type{'B', "byte"}, \
bjd_type{'C', "char"}, \
bjd_type{'D', "double"}, \
bjd_type{'I', "int16"}, \
bjd_type{'L', "int64"}, \
bjd_type{'M', "uint64"}, \
bjd_type{'U', "uint8"}, \
bjd_type{'d', "single"}, \
bjd_type{'i', "int8"}, \
bjd_type{'l', "int32"}, \
bjd_type{'m', "uint32"}, \
bjd_type{'u', "uint16"})
JSON_PRIVATE_UNLESS_TESTED:
// lookup tables
// NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes)
const decltype(JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_) bjd_optimized_type_markers =
JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_;
/*!
@brief whether @a marker is excluded from BJData's optimized ND-array types
@return whether @a marker is one of 'F', 'H', 'N', 'S', 'T', 'Z', '[', '{'
using bjd_type = std::pair<char_int_type, string_t>;
// NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes)
const decltype(JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_) bjd_types_map =
JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_;
Mirrors binary_writer's @ref binary_writer::is_bjdata_excluded_type_marker
"is_bjdata_excluded_type_marker()`, which encodes the same list the other
way; keep the two in sync.
*/
static constexpr bool is_bjd_excluded_optimized_type(const char_int_type marker) noexcept
{
return marker == '[' || marker == '{' || marker == 'S' || marker == 'H'
|| marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z';
}
#undef JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_
#undef JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_
/*!
@brief look up the ND-array element type name for a BJData dtype marker
@return the type name ("uint8", "int8", ...), or nullptr if @a marker does
not name a known dtype
A C++11 `constexpr` function cannot contain a `switch`, so this is a
plain (non-constexpr) switch instead.
*/
static const char* bjd_type_name(const char_int_type marker)
{
switch (marker)
{
case 'B':
return "byte";
case 'C':
return "char";
case 'D':
return "double";
case 'I':
return "int16";
case 'L':
return "int64";
case 'M':
return "uint64";
case 'U':
return "uint8";
case 'd':
return "single";
case 'i':
return "int8";
case 'l':
return "int32";
case 'm':
return "uint32";
case 'u':
return "uint16";
default:
return nullptr;
}
}
};
#ifndef JSON_HAS_CPP_17