Merge remote-tracking branch 'origin/develop' into claude/fix-issue-3989-db7e45

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

# Conflicts:
#	include/nlohmann/detail/input/binary_reader.hpp
#	include/nlohmann/json.hpp
#	single_include/nlohmann/json.hpp
This commit is contained in:
Niels Lohmann
2026-09-30 22:53:03 +02:00
55 changed files with 3174 additions and 2080 deletions
+307 -350
View File
@@ -8,7 +8,6 @@
#pragma once
#include <algorithm> // generate_n
#include <array> // array
#include <cmath> // ldexp
#include <cstddef> // size_t
@@ -131,7 +130,8 @@ class binary_reader
~binary_reader() = default;
/*!
@param[in] format the binary format to parse
@brief parse in the format the constructor was given
@param[in] sax_ a SAX event processor
@param[in] strict whether to expect the input to be consumed completed
@param[in] tag_handler how to treat CBOR tags
@@ -139,9 +139,8 @@ class binary_reader
@return whether parsing was successful: the input was read without errors,
and no SAX event returned false
*/
JSON_HEDLEY_NON_NULL(3)
bool sax_parse(const input_format_t format,
json_sax_t* sax_,
JSON_HEDLEY_NON_NULL(2)
bool sax_parse(json_sax_t* sax_,
const bool strict = true,
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error)
{
@@ -155,14 +154,14 @@ class binary_reader
ndarray_open = 0;
bool result = false;
switch (format)
switch (input_format)
{
case input_format_t::bson:
result = parse_bson_internal();
break;
case input_format_t::cbor:
result = parse_cbor_internal(true, tag_handler);
result = parse_cbor_internal(tag_handler);
break;
case input_format_t::msgpack:
@@ -290,6 +289,22 @@ class binary_reader
return enter_container(/*is_object*/true, len, type_marker);
}
/*!
@brief close the innermost open array or object
Pops the container opened by the matching @ref enter_container call and
emits the SAX end event. Every format-specific driver otherwise repeated
the same pop-then-dispatch sequence at its own close site.
@return whether the SAX parser accepted the end event
*/
bool leave_container()
{
const bool is_object = container_stack.back().is_object;
container_stack.pop_back();
return is_object ? sax->end_object() : sax->end_array();
}
//////////
// BSON //
//////////
@@ -548,8 +563,8 @@ class binary_reader
if (element_type == 0) // end of the innermost document
{
// a copy, not a reference: it must stay valid across the
// pop_back() below, which destroys the container_stack
// element it would otherwise alias
// pop_back() inside leave_container() below, which destroys
// the container_stack element it would otherwise alias
const container_frame top = container_stack.back();
if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(top.start_position, top.declared_size)))
@@ -557,8 +572,7 @@ class binary_reader
return false;
}
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -1057,29 +1071,13 @@ class binary_reader
return enter_array(conditional_static_cast<std::size_t>(static_cast<unsigned int>(current) & 0x1Fu));
case 0x98: // array (one-byte uint8_t for n follows)
{
std::uint8_t len{};
return get_number(input_format_t::cbor, len) && enter_array(static_cast<std::size_t>(len));
}
case 0x99: // array (two-byte uint16_t for n follow)
{
std::uint16_t len{};
return get_number(input_format_t::cbor, len) && enter_array(static_cast<std::size_t>(len));
}
case 0x9A: // array (four-byte uint32_t for n follow)
{
std::uint32_t len{};
std::size_t size{};
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size);
}
case 0x9B: // array (eight-byte uint64_t for n follow)
{
std::uint64_t len{};
std::size_t size{};
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size);
return get_cbor_argument(len) && get_cbor_container_size(len, size, "array") && enter_array(size);
}
case 0x9F: // array (indefinite length)
@@ -1113,35 +1111,19 @@ class binary_reader
return enter_object(conditional_static_cast<std::size_t>(static_cast<unsigned int>(current) & 0x1Fu));
case 0xB8: // map (one-byte uint8_t for n follows)
{
std::uint8_t len{};
return get_number(input_format_t::cbor, len) && enter_object(static_cast<std::size_t>(len));
}
case 0xB9: // map (two-byte uint16_t for n follow)
{
std::uint16_t len{};
return get_number(input_format_t::cbor, len) && enter_object(static_cast<std::size_t>(len));
}
case 0xBA: // map (four-byte uint32_t for n follow)
{
std::uint32_t len{};
std::size_t size{};
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size);
}
case 0xBB: // map (eight-byte uint64_t for n follow)
{
std::uint64_t len{};
std::size_t size{};
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size);
return get_cbor_argument(len) && get_cbor_container_size(len, size, "map") && enter_object(size);
}
case 0xBF: // map (indefinite length)
return enter_object(detail::unknown_size());
case 0xC0: // tagged item
case 0xC0: // tagged item (tag value 0-23, in the head itself)
case 0xC1:
case 0xC2:
case 0xC3:
@@ -1165,6 +1147,27 @@ class binary_reader
case 0xD5:
case 0xD6:
case 0xD7:
{
if (tag_handler == cbor_tag_handler_t::error)
{
auto last_token = get_token_string();
if (!report_repairable_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr)))
{
return false;
}
// when recovering, the tag is ignored, as RFC 8949,
// Section 6.1 suggests for converting to JSON
}
// ignore and store: the tag value is already in the head, so
// there is nothing left to read here; the tagged value that
// follows is read by the loop in parse_cbor_internal() rather
// than by recursing here
tag_pending = true;
return true;
}
case 0xD8: // tagged item (1 byte follows)
case 0xD9: // tagged item (2 bytes follow)
case 0xDA: // tagged item (4 bytes follow)
@@ -1187,47 +1190,11 @@ class binary_reader
case cbor_tag_handler_t::ignore:
{
// ignore binary subtype
switch (current)
// ignore the tag's binary subtype argument
std::uint64_t subtype_to_ignore{};
if (!get_cbor_argument(subtype_to_ignore))
{
case 0xD8:
{
std::uint8_t subtype_to_ignore{};
if (!get_number(input_format_t::cbor, subtype_to_ignore))
{
return false;
}
break;
}
case 0xD9:
{
std::uint16_t subtype_to_ignore{};
if (!get_number(input_format_t::cbor, subtype_to_ignore))
{
return false;
}
break;
}
case 0xDA:
{
std::uint32_t subtype_to_ignore{};
if (!get_number(input_format_t::cbor, subtype_to_ignore))
{
return false;
}
break;
}
case 0xDB:
{
std::uint64_t subtype_to_ignore{};
if (!get_number(input_format_t::cbor, subtype_to_ignore))
{
return false;
}
break;
}
default:
break;
return false;
}
// the tagged value follows; it is read by the loop in
// parse_cbor_internal() rather than by recursing here
@@ -1237,57 +1204,15 @@ class binary_reader
case cbor_tag_handler_t::store:
{
binary_t b;
// use binary subtype and store in a binary container
switch (current)
std::uint64_t subtype{};
if (!get_cbor_argument(subtype))
{
case 0xD8:
{
std::uint8_t subtype{};
if (!get_number(input_format_t::cbor, subtype))
{
return false;
}
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
break;
}
case 0xD9:
{
std::uint16_t subtype{};
if (!get_number(input_format_t::cbor, subtype))
{
return false;
}
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
break;
}
case 0xDA:
{
std::uint32_t subtype{};
if (!get_number(input_format_t::cbor, subtype))
{
return false;
}
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
break;
}
case 0xDB:
{
std::uint64_t subtype{};
if (!get_number(input_format_t::cbor, subtype))
{
return false;
}
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
break;
}
default:
{
// as above, the tagged value is read by the caller
tag_pending = true;
return true;
}
return false;
}
binary_t b;
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
get();
// a byte string (the heads accepted by get_cbor_binary) keeps the tag as subtype
if ((current >= 0x40 && current <= 0x5B) || current == 0x5F)
@@ -1318,52 +1243,7 @@ class binary_reader
return sax->null();
case 0xF9: // Half-Precision Float (two-byte IEEE 754)
{
const auto byte1_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number")))
{
return false;
}
const auto byte2_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number")))
{
return false;
}
const auto byte1 = static_cast<unsigned char>(byte1_raw);
const auto byte2 = static_cast<unsigned char>(byte2_raw);
// Code from RFC 8949, Appendix D, Figure 3:
// As half-precision floating-point numbers were only added
// to IEEE 754 in 2008, today's programming platforms often
// still only have limited support for them. It is very
// easy to include at least decoding support for them even
// without such support. An example of a small decoder for
// half-precision floating-point numbers in the C language
// is shown in Fig. 3.
const auto half = static_cast<unsigned int>((byte1 << 8u) + byte2);
const double val = [&half]
{
const int exp = (half >> 10u) & 0x1Fu;
const unsigned int mant = half & 0x3FFu;
JSON_ASSERT(exp <= 31);
JSON_ASSERT(mant <= 1023);
switch (exp)
{
case 0:
return std::ldexp(mant, -24);
case 31:
return (mant == 0)
? std::numeric_limits<double>::infinity()
: std::numeric_limits<double>::quiet_NaN();
default:
return std::ldexp(mant + 1024, exp - 25);
}
}();
return sax->number_float((half & 0x8000u) != 0
? static_cast<number_float_t>(-val)
: static_cast<number_float_t>(val), "");
}
return get_half_float(input_format_t::cbor, false);
case 0xFA: // Single-Precision Float (four-byte IEEE 754)
{
@@ -1759,6 +1639,73 @@ class binary_reader
}
}
/*!
@brief read a CBOR argument (additional information 24-27) of the width
@ref current announces
The lower 5 bits of @a current (0x18-0x1B) select a 1/2/4/8-byte
big-endian unsigned integer that follows the head byte; this is shared by
every major type that uses this encoding (unsigned/negative integers,
strings, arrays, maps, tags). Reading always goes through @ref get_number,
so EOF is reported the same way as before this helper existed.
@param[out] value the decoded argument
@return whether reading succeeded
*/
bool get_cbor_argument(std::uint64_t& value)
{
switch (current & 0x1F)
{
case 0x18: // 1 byte
{
std::uint8_t n{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n)))
{
return false;
}
value = n;
return true;
}
case 0x19: // 2 bytes
{
std::uint16_t n{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n)))
{
return false;
}
value = n;
return true;
}
case 0x1A: // 4 bytes
{
std::uint32_t n{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n)))
{
return false;
}
value = n;
return true;
}
case 0x1B: // 8 bytes
{
std::uint64_t n{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n)))
{
return false;
}
value = n;
return true;
}
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
return false; // LCOV_EXCL_LINE
}
}
/*!
@brief narrow a definite CBOR array/map length to std::size_t
@@ -1791,19 +1738,15 @@ class binary_reader
enclosing container after each element, so that the nesting depth of the
input costs heap rather than native stack (see #5104).
@param[in] get_char whether a new character should be retrieved from the
input (true) or whether the last read character
@a current should be considered instead
@param[in] tag_handler how CBOR tags should be treated
@return whether reading the value succeeded
*/
bool parse_cbor_internal(const bool get_char,
const cbor_tag_handler_t tag_handler)
bool parse_cbor_internal(const cbor_tag_handler_t tag_handler)
{
// whether the next value starts at a fresh byte or at the one already
// read into `current`
bool fetch = get_char;
bool fetch = true;
// the key currently being read; hoisted out of the loop so that its
// capacity is reused across elements and across nesting levels
@@ -1846,8 +1789,7 @@ class binary_reader
if (at_end)
{
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -2035,9 +1977,6 @@ class binary_reader
// MsgPack //
/////////////
/*!
@return whether a valid MessagePack value was passed to the SAX parser
*/
/*!
@brief read one MessagePack value
@@ -2741,8 +2680,7 @@ class binary_reader
if (container_stack.back().remaining == 0)
{
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -2936,20 +2874,16 @@ class binary_reader
////////////
/*!
@param[in] get_char whether a new character should be retrieved from the
input (true, default) or whether the last read
character should be considered instead
@return whether a valid UBJSON value was passed to the SAX parser
*/
bool parse_ubjson_internal(const bool get_char = true)
bool parse_ubjson_internal()
{
// the key currently being read; hoisted out of the loop so that its
// capacity is reused across elements and across nesting levels
string_t key;
// the type marker of the value to read next
char_int_type prefix = get_char ? get_ignore_noop() : current;
char_int_type prefix = get_ignore_noop();
while (true)
{
@@ -3024,8 +2958,7 @@ class binary_reader
break;
}
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -3233,6 +3166,42 @@ class binary_reader
return true;
}
/*!
@brief read a UBJSON/BJData optimized-container count of a signed marker
type ('i', 'I', 'l', 'L') and narrow it to std::size_t
Every signed count marker rejects a negative value the same way (error
113); the value_in_range_of check additionally needed for 'L' is only
ever live when @a SignedType is std::int64_t on a target where
std::size_t is narrower (e.g. 32-bit), since 'i'/'I'/'l' can never exceed
std::size_t there.
@tparam SignedType std::int8_t, std::int16_t, std::int32_t or std::int64_t
@param[out] result the count narrowed to std::size_t
@return whether reading and validating succeeded
*/
template<typename SignedType>
bool get_ubjson_signed_count(std::size_t& result)
{
SignedType number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (JSON_HEDLEY_UNLIKELY(number < 0))
{
return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
if (JSON_HEDLEY_UNLIKELY(!value_in_range_of<std::size_t>(number)))
{
return report_error(chars_read, get_token_string(), out_of_range::create(408,
exception_message(input_format, "integer value overflow", "size"), nullptr));
}
result = static_cast<std::size_t>(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char
return true;
}
/*!
@param[out] result determined size
@param[in,out] is_ndarray for input, `true` means already inside an ndarray vector
@@ -3265,73 +3234,16 @@ class binary_reader
}
case 'i':
{
std::int8_t number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (number < 0)
{
return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
result = static_cast<std::size_t>(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char
return true;
}
return get_ubjson_signed_count<std::int8_t>(result);
case 'I':
{
std::int16_t number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (number < 0)
{
return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
result = static_cast<std::size_t>(number);
return true;
}
return get_ubjson_signed_count<std::int16_t>(result);
case 'l':
{
std::int32_t number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (number < 0)
{
return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
result = static_cast<std::size_t>(number);
return true;
}
return get_ubjson_signed_count<std::int32_t>(result);
case 'L':
{
std::int64_t number{};
if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number)))
{
return false;
}
if (number < 0)
{
return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read,
exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr));
}
if (!value_in_range_of<std::size_t>(number))
{
return report_error(chars_read, get_token_string(), out_of_range::create(408,
exception_message(input_format, "integer value overflow", "size"), nullptr));
}
result = static_cast<std::size_t>(number);
return true;
}
return get_ubjson_signed_count<std::int64_t>(result);
case 'u':
{
@@ -3423,16 +3335,23 @@ class binary_reader
result = 1;
for (auto i : dim)
{
// Pre-multiplication overflow check: if i > 0 and result > SIZE_MAX/i, then result*i would overflow.
// This check must happen before multiplication since overflow detection after the fact is unreliable
// as modular arithmetic can produce any value, not just 0 or SIZE_MAX.
if (JSON_HEDLEY_UNLIKELY(i > 0 && result > (std::numeric_limits<std::size_t>::max)() / i))
// Pre-multiplication overflow check: since the loop above
// already rejected any zero dimension, i is always > 0
// here, so result > SIZE_MAX/i means result*i would
// overflow. This check must happen before multiplication
// since overflow detection after the fact is unreliable,
// as modular arithmetic can produce any value, not just 0
// or SIZE_MAX.
if (JSON_HEDLEY_UNLIKELY(result > (std::numeric_limits<std::size_t>::max)() / i))
{
return report_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr));
}
result *= i;
// Additional post-multiplication check to catch any edge cases the pre-check might miss
if (result == 0 || result == npos)
// the pre-check above already rules out result becoming 0
// by overflow; the only value it cannot rule out is an
// exact match with npos, the sentinel reserved for an
// unknown-size container (see get_ubjson_size_type())
if (result == npos)
{
return report_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr));
}
@@ -3494,7 +3413,7 @@ class binary_reader
{
result.second = get(); // must not ignore 'N', because 'N' maybe the type
if (input_format == input_format_t::bjdata
&& JSON_HEDLEY_UNLIKELY(std::binary_search(bjd_optimized_type_markers.begin(), bjd_optimized_type_markers.end(), result.second)))
&& JSON_HEDLEY_UNLIKELY(is_bjd_excluded_optimized_type(result.second)))
{
auto last_token = get_token_string();
return report_error(chars_read, last_token, parse_error::create(112, chars_read,
@@ -3638,50 +3557,7 @@ class binary_reader
{
break;
}
const auto byte1_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number")))
{
return false;
}
const auto byte2_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number")))
{
return false;
}
const auto byte1 = static_cast<unsigned char>(byte1_raw);
const auto byte2 = static_cast<unsigned char>(byte2_raw);
// Code from RFC 8949, Appendix D, Figure 3:
// As half-precision floating-point numbers were only added
// to IEEE 754 in 2008, today's programming platforms often
// still only have limited support for them. It is very
// easy to include at least decoding support for them even
// without such support. An example of a small decoder for
// half-precision floating-point numbers in the C language
// is shown in Fig. 3.
const auto half = static_cast<unsigned int>((byte2 << 8u) + byte1);
const double val = [&half]
{
const int exp = (half >> 10u) & 0x1Fu;
const unsigned int mant = half & 0x3FFu;
JSON_ASSERT(exp <= 31);
JSON_ASSERT(mant <= 1023);
switch (exp)
{
case 0:
return std::ldexp(mant, -24);
case 31:
return (mant == 0)
? std::numeric_limits<double>::infinity()
: std::numeric_limits<double>::quiet_NaN();
default:
return std::ldexp(mant + 1024, exp - 25);
}
}();
return sax->number_float((half & 0x8000u) != 0
? static_cast<number_float_t>(-val)
: static_cast<number_float_t>(val), "");
return get_half_float(input_format, true);
}
case 'd':
@@ -3762,19 +3638,16 @@ class binary_reader
if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0)
{
size_and_type.second &= ~(static_cast<char_int_type>(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t)
{
return p.first < t;
});
const char* type_name = bjd_type_name(size_and_type.second);
string_t key = "_ArrayType_";
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second))
if (JSON_HEDLEY_UNLIKELY(type_name == nullptr))
{
auto last_token = get_token_string();
return report_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
}
string_t type = it->second; // sax->string() takes a reference
string_t type = type_name; // sax->string() takes a reference
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type)))
{
return false;
@@ -4143,8 +4016,7 @@ class binary_reader
if (at_end)
{
container_stack.pop_back();
if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array()))
if (JSON_HEDLEY_UNLIKELY(!leave_container()))
{
return false;
}
@@ -4808,6 +4680,68 @@ class binary_reader
return true;
}
/*!
@brief read and decode an IEEE 754 half-precision (16-bit) float
Used by CBOR (big endian) and BJData (little endian); the two formats
only differ in the byte order of the two bytes that make up the half.
@param[in] format the current format (for diagnostics)
@param[in] little_endian whether the two bytes are little endian (BJData)
or big endian (CBOR)
@return whether reading and decoding succeeded
*/
bool get_half_float(const input_format_t format, const bool little_endian)
{
const auto byte1_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number")))
{
return false;
}
const auto byte2_raw = get();
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number")))
{
return false;
}
const auto byte1 = static_cast<unsigned char>(byte1_raw);
const auto byte2 = static_cast<unsigned char>(byte2_raw);
// Code from RFC 8949, Appendix D, Figure 3:
// As half-precision floating-point numbers were only added
// to IEEE 754 in 2008, today's programming platforms often
// still only have limited support for them. It is very
// easy to include at least decoding support for them even
// without such support. An example of a small decoder for
// half-precision floating-point numbers in the C language
// is shown in Fig. 3.
const auto half = little_endian
? static_cast<unsigned int>((byte2 << 8u) + byte1)
: static_cast<unsigned int>((byte1 << 8u) + byte2);
const double val = [&half]
{
const int exp = (half >> 10u) & 0x1Fu;
const unsigned int mant = half & 0x3FFu;
JSON_ASSERT(exp <= 31);
JSON_ASSERT(mant <= 1023);
switch (exp)
{
case 0:
return std::ldexp(mant, -24);
case 31:
return (mant == 0)
? std::numeric_limits<double>::infinity()
: std::numeric_limits<double>::quiet_NaN();
default:
return std::ldexp(mant + 1024, exp - 25);
}
}();
return sax->number_float((half & 0x8000u) != 0
? static_cast<number_float_t>(-val)
: static_cast<number_float_t>(val), "");
}
/*!
@brief create a string by reading characters from the input
@@ -5353,38 +5287,61 @@ class binary_reader
/// open: none, its object, or its object and an array inside it
std::uint8_t ndarray_open = 0;
// excluded markers in bjdata optimized type
#define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \
make_array<char_int_type>('F', 'H', 'N', 'S', 'T', 'Z', '[', '{')
#define JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ \
make_array<bjd_type>( \
bjd_type{'B', "byte"}, \
bjd_type{'C', "char"}, \
bjd_type{'D', "double"}, \
bjd_type{'I', "int16"}, \
bjd_type{'L', "int64"}, \
bjd_type{'M', "uint64"}, \
bjd_type{'U', "uint8"}, \
bjd_type{'d', "single"}, \
bjd_type{'i', "int8"}, \
bjd_type{'l', "int32"}, \
bjd_type{'m', "uint32"}, \
bjd_type{'u', "uint16"})
JSON_PRIVATE_UNLESS_TESTED:
// lookup tables
// NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes)
const decltype(JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_) bjd_optimized_type_markers =
JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_;
/*!
@brief whether @a marker is excluded from BJData's optimized ND-array types
@return whether @a marker is one of 'F', 'H', 'N', 'S', 'T', 'Z', '[', '{'
using bjd_type = std::pair<char_int_type, string_t>;
// NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes)
const decltype(JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_) bjd_types_map =
JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_;
Mirrors binary_writer's @ref binary_writer::is_bjdata_excluded_type_marker
"is_bjdata_excluded_type_marker()`, which encodes the same list the other
way; keep the two in sync.
*/
static constexpr bool is_bjd_excluded_optimized_type(const char_int_type marker) noexcept
{
return marker == '[' || marker == '{' || marker == 'S' || marker == 'H'
|| marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z';
}
#undef JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_
#undef JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_
/*!
@brief look up the ND-array element type name for a BJData dtype marker
@return the type name ("uint8", "int8", ...), or nullptr if @a marker does
not name a known dtype
A C++11 `constexpr` function cannot contain a `switch`, so this is a
plain (non-constexpr) switch instead.
*/
static const char* bjd_type_name(const char_int_type marker)
{
switch (marker)
{
case 'B':
return "byte";
case 'C':
return "char";
case 'D':
return "double";
case 'I':
return "int16";
case 'L':
return "int64";
case 'M':
return "uint64";
case 'U':
return "uint8";
case 'd':
return "single";
case 'i':
return "int8";
case 'l':
return "int32";
case 'm':
return "uint32";
case 'u':
return "uint16";
default:
return nullptr;
}
}
};
#ifndef JSON_HAS_CPP_17