Compare commits

..
Author SHA1 Message Date
Niels Lohmann b98aef8a07 Round-trip BJData ND-array annotations exactly (single precision, key order)
to_bjdata() encoded a JData-annotated object as a BJData ND-array in two
cases where from_bjdata() then returned a different value, breaking the
documented round-trip guarantee:

1. A "single" element that is finite and in range but not exactly
   representable as float (e.g. 0.1) or that underflows to 0 (e.g. 1e-300)
   was silently narrowed instead of falling back to a plain object, unlike
   out-of-range integer elements. write_bjdata_ndarray() now only accepts a
   "single" element if it survives the narrowing to float and back, the
   same criterion write_compact_float() already uses for CBOR/MessagePack.

2. from_bjdata() emitted the annotation keys as _ArraySize_, _ArrayType_,
   _ArrayData_ instead of the documented _ArrayType_, _ArraySize_,
   _ArrayData_, because the size key is written while the dimension vector
   is read, before the type key. For ordered_json, whose comparison takes
   key order into account, this made a round trip of the documented example
   compare unequal. The element type marker is known before the dimension
   vector is read (it precedes '#'), so it is now passed down and the
   "_ArrayType_" key is emitted first.

Fixes #5661.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 23:52:49 +02:00
11 changed files with 800 additions and 620 deletions
@@ -132,8 +132,14 @@ The library uses the following mapping from JSON values types to BJData types ac
parsed back as a regular array, parsed back as a regular array,
- every entry of `"_ArraySize_"` is a positive integer, and their product is representable as a `std::size_t`, - every entry of `"_ArraySize_"` is a positive integer, and their product is representable as a `std::size_t`,
- `"_ArrayData_"` is an array holding exactly that many elements, and - `"_ArrayData_"` is an array holding exactly that many elements, and
- every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"` (a floating-point number for - every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"`: for the integer types, a
`single` and `double`, an integer otherwise). value that fits the named width; for `double`, any value; for `single`, a value that survives narrowing to
`float` and back without change (for instance, `0.1` does not, since it is not exactly representable as
`float`).
An annotated object is always read back with its keys in the order shown above, `"_ArrayType_"`, `"_ArraySize_"`,
`"_ArrayData_"`, regardless of the order the ND-array's header stores them in on the wire. This matters for
`ordered_json`, whose comparison takes key order into account.
The current version of this library does not yet support automatic detection of and conversion from a nested JSON The current version of this library does not yet support automatic detection of and conversion from a nested JSON
array input to a BJData ND-array. array input to a BJData ND-array.
+49 -28
View File
@@ -2728,10 +2728,15 @@ class binary_reader
is_ndarray can only return `true` when its initial value is_ndarray can only return `true` when its initial value
is `false` is `false`
@param[in] prefix type marker if already read, otherwise set to 0 @param[in] prefix type marker if already read, otherwise set to 0
@param[in] ndarray_dtype the element type marker of the enclosing bjdata ndarray if
already known (it precedes the dimension vector read here),
otherwise 0; used to emit the "_ArrayType_" annotation key
before "_ArraySize_" if a dimension vector turns out to
describe an ndarray
@return whether size determination completed @return whether size determination completed
*/ */
bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0) bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0, char_int_type ndarray_dtype = 0)
{ {
if (prefix == 0) if (prefix == 0)
{ {
@@ -2901,8 +2906,37 @@ class binary_reader
} }
} }
if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3)))
{
return false;
}
// the element type precedes the dimension vector (see get_ubjson_size_type)
// and is passed down as ndarray_dtype; emit it here so the annotation keys
// follow the documented _ArrayType_, _ArraySize_, _ArrayData_ order
if (ndarray_dtype != 0)
{
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), ndarray_dtype, [](const bjd_type & p, char_int_type t)
{
return p.first < t;
});
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != ndarray_dtype))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
}
string_t type_key = "_ArrayType_";
string_t type = it->second; // sax->string() takes a reference
if (JSON_HEDLEY_UNLIKELY(!sax->key(type_key) || !sax->string(type)))
{
return false;
}
}
string_t key = "_ArraySize_"; string_t key = "_ArraySize_";
if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3) || !sax->key(key) || !sax->start_array(dim.size()))) if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(dim.size())))
{ {
return false; return false;
} }
@@ -3003,7 +3037,7 @@ class binary_reader
exception_message(input_format, concat("expected '#' after type information; last byte: 0x", last_token), "size"), nullptr)); exception_message(input_format, concat("expected '#' after type information; last byte: 0x", last_token), "size"), nullptr));
} }
const bool is_error = get_ubjson_size_value(result.first, is_ndarray); const bool is_error = get_ubjson_size_value(result.first, is_ndarray, 0, result.second);
// an ndarray was read here only if the flag flipped; when it was // an ndarray was read here only if the flag flipped; when it was
// seeded true, get_ubjson_size_value() already rejected the nested // seeded true, get_ubjson_size_value() already rejected the nested
// dimension vector // dimension vector
@@ -3239,30 +3273,17 @@ class binary_reader
if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0) if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0)
{ {
size_and_type.second &= ~(static_cast<char_int_type>(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker size_and_type.second &= ~(static_cast<char_int_type>(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t)
{
return p.first < t;
});
string_t key = "_ArrayType_";
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
}
string_t type = it->second; // sax->string() takes a reference
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type)))
{
return false;
}
// the "_ArrayType_" and "_ArraySize_" annotation keys were already emitted by
// get_ubjson_size_value() (the type marker is known before the dimension vector
// that determines size_and_type.first is read, so it is emitted first there to
// match the documented _ArrayType_, _ArraySize_, _ArrayData_ key order)
if (size_and_type.second == 'C' || size_and_type.second == 'B') if (size_and_type.second == 'C' || size_and_type.second == 'B')
{ {
size_and_type.second = 'U'; size_and_type.second = 'U';
} }
key = "_ArrayData_"; string_t key = "_ArrayData_";
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(size_and_type.first) )) if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(size_and_type.first) ))
{ {
return false; return false;
@@ -3344,8 +3365,8 @@ class binary_reader
return enter_object(detail::unknown_size()); return enter_object(detail::unknown_size());
} }
// Note, UBJSON has no binary type of its own; BJData, which shares this // Note, no reader for UBJSON binary types is implemented because they do
// reader, decodes optimized 'B' arrays as binary in get_ubjson_array(). // not exist
bool get_ubjson_high_precision_number() bool get_ubjson_high_precision_number()
{ {
@@ -4052,7 +4073,7 @@ class binary_reader
#endif #endif
} }
/*! /*
@brief read a number from the input @brief read a number from the input
@tparam NumberType the type of the number @tparam NumberType the type of the number
@@ -4062,10 +4083,10 @@ class binary_reader
@return whether conversion completed @return whether conversion completed
@note This function needs to respect the system's endianness, because @note This function needs to respect the system's endianness, because
bytes in CBOR, MessagePack, UBJSON, and BON8 are stored in network bytes in CBOR, MessagePack, and UBJSON are stored in network order
order (big endian) and therefore need reordering on little endian (big endian) and therefore need reordering on little endian systems.
systems. On the other hand, BSON and BJData use little endian and On the other hand, BSON and BJData use little endian and should reorder
should reorder on big endian systems. on big endian systems.
*/ */
template<typename NumberType, bool InputIsLittleEndian = false> template<typename NumberType, bool InputIsLittleEndian = false>
bool get_number(const input_format_t format, NumberType& result) bool get_number(const input_format_t format, NumberType& result)
@@ -8,12 +8,12 @@
#pragma once #pragma once
#include <algorithm> // min
#include <array> // array #include <array> // array
#include <cstddef> // size_t #include <cstddef> // size_t
#include <cstdint> // uint32_t
#include <cstring> // strlen #include <cstring> // strlen
#include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next #include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next
#include <memory> // shared_ptr, make_shared, addressof
#include <numeric> // accumulate
#include <streambuf> // streambuf #include <streambuf> // streambuf
#include <string> // string, char_traits #include <string> // string, char_traits
#include <type_traits> // enable_if, is_base_of, is_pointer, is_integral, remove_pointer #include <type_traits> // enable_if, is_base_of, is_pointer, is_integral, remove_pointer
@@ -28,7 +28,6 @@
#include <nlohmann/detail/iterators/iterator_traits.hpp> #include <nlohmann/detail/iterators/iterator_traits.hpp>
#include <nlohmann/detail/macro_scope.hpp> #include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/type_traits.hpp> #include <nlohmann/detail/meta/type_traits.hpp>
#include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail namespace detail
@@ -83,9 +82,8 @@ class file_input_adapter
}; };
/*! /*!
Input adapter for a (caching) istream. Does not skip a UTF Byte Order Mark Input adapter for a (caching) istream. Ignores a UFT Byte Order Mark at
itself; that is done by the lexer's skip_bom(). Does not support changing beginning of input. Does not support changing the underlying std::streambuf
the underlying std::streambuf
in mid-input. Maintains underlying std::istream and std::streambuf to support in mid-input. Maintains underlying std::istream and std::streambuf to support
subsequent use of standard std::istream operations to process any input subsequent use of standard std::istream operations to process any input
characters following those used in parsing the JSON input. Clears the characters following those used in parsing the JSON input. Clears the
@@ -450,14 +448,32 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
// get the current character // get the current character
const auto wc = input.get_character(); const auto wc = input.get_character();
if (wc <= 0x10FFFF) // UTF-32 to UTF-8 encoding
if (wc < 0x80)
{ {
// UTF-32 to UTF-8 encoding utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
utf8_bytes_filled = 0; utf8_bytes_filled = 1;
encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) }
{ else if (wc <= 0x7FF)
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte); {
}); utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u) & 0x1Fu));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 2;
}
else if (wc <= 0xFFFF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u) & 0x0Fu));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 3;
}
else if (wc <= 0x10FFFF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | ((static_cast<unsigned int>(wc) >> 18u) & 0x07u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 12u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 4;
} }
else else
{ {
@@ -494,15 +510,24 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
// get the current character // get the current character
const auto wc = input.get_character(); const auto wc = input.get_character();
if (0xD800 > wc || wc >= 0xE000) // UTF-16 to UTF-8 encoding
if (wc < 0x80)
{ {
// a UTF-16 code unit outside the surrogate range is a valid utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
// code point (at most U+FFFF) on its own utf8_bytes_filled = 1;
utf8_bytes_filled = 0; }
encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) else if (wc <= 0x7FF)
{ {
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte); utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u)));
}); utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 2;
}
else if (0xD800 > wc || wc >= 0xE000)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u)));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 3;
} }
else else
{ {
@@ -520,11 +545,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (0xDC00 <= wc2 && wc2 <= 0xDFFF) if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{ {
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes_filled = 0; utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
{ utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu));
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte); utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (charcode & 0x3Fu));
}); utf8_bytes_filled = 4;
valid_pair = true; valid_pair = true;
} }
} }
@@ -837,9 +862,9 @@ auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) /
return input_adapter(array, array + N); return input_adapter(array, array + N);
} }
// This class only handles inputs that construct a contiguous_bytes_input_adapter // This class only handles inputs of input_buffer_adapter type.
// (e.g. span_input_adapter). It's required so that expressions like {ptr, len} // It's required so that expressions like {ptr, len} can be implicitly cast
// can be implicitly cast to the correct adapter. // to the correct adapter.
class span_input_adapter class span_input_adapter
{ {
public: public:
+142 -89
View File
@@ -10,7 +10,6 @@
#include <algorithm> // find_if, min #include <algorithm> // find_if, min
#include <cstddef> #include <cstddef>
#include <limits> // numeric_limits
#include <string> // string #include <string> // string
#include <type_traits> // enable_if_t #include <type_traits> // enable_if_t
#include <utility> // move, pair #include <utility> // move, pair
@@ -176,88 +175,6 @@ template<typename ArrayType>
inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/) inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/)
{} {}
#if JSON_DIAGNOSTIC_POSITIONS
/*!
@brief set the diagnostic positions of a value the DOM SAX parsers just stored
Shared by json_sax_dom_parser and json_sax_dom_callback_parser. basic_json
befriends this struct, as the position members are private.
*/
struct diagnostic_positions
{
/*!
@param[in,out] v the value that was just parsed
@param[in] lexer the lexer that read it, or nullptr to leave @a v alone
*/
template<typename BasicJsonType, typename LexerType>
static void set_from_lexer(BasicJsonType& v, LexerType* lexer)
{
if (lexer)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = lexer->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = lexer->get_token_start_position();
break;
}
case value_t::discarded:
{
// an object or array the callback of
// json_sax_dom_callback_parser rejected has no position
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - lexer->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
}
}
};
#endif
/*! /*!
@brief SAX implementation to create a JSON value from SAX events @brief SAX implementation to create a JSON value from SAX events
@@ -459,6 +376,76 @@ class json_sax_dom_parser
private: private:
#if JSON_DIAGNOSTIC_POSITIONS
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
{
if (m_lexer_ref)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = m_lexer_ref->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = m_lexer_ref->get_token_start_position();
break;
}
// As we handle the start and end positions for values created during parsing,
// we do not expect the following value type to be called. Regardless, set the positions
// in case this is created manually or through a different constructor. Exclude from lcov
// since the exact condition of this switch is esoteric.
// LCOV_EXCL_START
case value_t::discarded:
{
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
// LCOV_EXCL_STOP
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - m_lexer_ref->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
}
}
}
#endif
/*! /*!
@invariant If the ref stack is empty, then the passed value will be the new @invariant If the ref stack is empty, then the passed value will be the new
root. root.
@@ -474,7 +461,7 @@ class json_sax_dom_parser
root = BasicJsonType(std::forward<Value>(v)); root = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS #if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(root, m_lexer_ref); handle_diagnostic_positions_for_json_value(root);
#endif #endif
return &root; return &root;
@@ -487,7 +474,7 @@ class json_sax_dom_parser
ref_stack.back()->m_data.m_value.array->emplace_back(std::forward<Value>(v)); ref_stack.back()->m_data.m_value.array->emplace_back(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS #if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(ref_stack.back()->m_data.m_value.array->back(), m_lexer_ref); handle_diagnostic_positions_for_json_value(ref_stack.back()->m_data.m_value.array->back());
#endif #endif
return &(ref_stack.back()->m_data.m_value.array->back()); return &(ref_stack.back()->m_data.m_value.array->back());
@@ -498,7 +485,7 @@ class json_sax_dom_parser
*object_element = BasicJsonType(std::forward<Value>(v)); *object_element = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS #if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(*object_element, m_lexer_ref); handle_diagnostic_positions_for_json_value(*object_element);
#endif #endif
return object_element; return object_element;
@@ -675,7 +662,7 @@ class json_sax_dom_callback_parser
#if JSON_DIAGNOSTIC_POSITIONS #if JSON_DIAGNOSTIC_POSITIONS
// Set start/end positions for discarded object. // Set start/end positions for discarded object.
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref); handle_diagnostic_positions_for_json_value(*ref_stack.back());
#endif #endif
} }
} }
@@ -791,7 +778,7 @@ class json_sax_dom_callback_parser
#if JSON_DIAGNOSTIC_POSITIONS #if JSON_DIAGNOSTIC_POSITIONS
// Set start/end positions for discarded array. // Set start/end positions for discarded array.
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref); handle_diagnostic_positions_for_json_value(*ref_stack.back());
#endif #endif
} }
} }
@@ -844,6 +831,72 @@ class json_sax_dom_callback_parser
private: private:
#if JSON_DIAGNOSTIC_POSITIONS
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
{
if (m_lexer_ref)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = m_lexer_ref->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = m_lexer_ref->get_token_start_position();
break;
}
case value_t::discarded:
{
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - m_lexer_ref->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
}
}
}
#endif
/// if there is a pending duplicate-key stash entry for this exact slot, /// if there is a pending duplicate-key stash entry for this exact slot,
/// remove it from the stash; if restore_value is true, the stashed /// remove it from the stash; if restore_value is true, the stashed
/// previous value is moved back into the slot first (use this when the /// previous value is moved back into the slot first (use this when the
@@ -965,7 +1018,7 @@ class json_sax_dom_callback_parser
auto value = BasicJsonType(std::forward<Value>(v)); auto value = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS #if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(value, m_lexer_ref); handle_diagnostic_positions_for_json_value(value);
#endif #endif
// check callback // check callback
+68 -33
View File
@@ -11,7 +11,6 @@
#include <array> // array #include <array> // array
#include <clocale> // localeconv #include <clocale> // localeconv
#include <cstddef> // size_t #include <cstddef> // size_t
#include <cstdint> // uint32_t
#include <cstdio> // snprintf #include <cstdio> // snprintf
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull #include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
#include <initializer_list> // initializer_list #include <initializer_list> // initializer_list
@@ -25,7 +24,6 @@
#include <nlohmann/detail/input/string_scan.hpp> #include <nlohmann/detail/input/string_scan.hpp>
#include <nlohmann/detail/macro_scope.hpp> #include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/type_traits.hpp> #include <nlohmann/detail/meta/type_traits.hpp>
#include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail namespace detail
@@ -502,10 +500,32 @@ class lexer : public lexer_base<BasicJsonType>
JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF); JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF);
// translate codepoint into bytes // translate codepoint into bytes
encode_utf8(static_cast<std::uint32_t>(codepoint), [this](std::uint32_t byte) if (codepoint < 0x80)
{ {
add(static_cast<char_int_type>(byte)); // 1-byte characters: 0xxxxxxx (ASCII)
}); add(static_cast<char_int_type>(codepoint));
}
else if (codepoint <= 0x7FF)
{
// 2-byte characters: 110xxxxx 10xxxxxx
add(static_cast<char_int_type>(0xC0u | (static_cast<unsigned int>(codepoint) >> 6u)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
else if (codepoint <= 0xFFFF)
{
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
add(static_cast<char_int_type>(0xE0u | (static_cast<unsigned int>(codepoint) >> 12u)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
else
{
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
add(static_cast<char_int_type>(0xF0u | (static_cast<unsigned int>(codepoint) >> 18u)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 12u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
break; break;
} }
@@ -1474,30 +1494,45 @@ scan_number_done:
*/ */
token_type convert_number(token_type number_type, std::size_t mantissa_end) token_type convert_number(token_type number_type, std::size_t mantissa_end)
{ {
// accept() only needs to know whether the input is valid, so it sets // If the caller does not need the converted value (only whether the
// discard_number_values (see json.hpp), and an integer token whose // input is syntactically valid; see json_sax_acceptor/accept()), an
// digit count shows that it fits is reported without calling // unsigned/integer token can be reported without calling
// convert_integer(). A number with up to 18 digits always fits into // strtoull()/strtoll() at all, *provided* we can already tell from
// both std::uint64_t and std::int64_t (18 nines is about 1e18, below // the digit count alone that the conversion cannot overflow 64 bits.
// INT64_MAX, which is about 9.2e18). Longer tokens take the exact path // Such tokens are always finite and are accepted unconditionally by
// below, including the fallback to floating point when the value does // the parser regardless of their actual value (parser::sax_parse_internal()
// not fit. // never checks finiteness for value_unsigned/value_integer), so the
// classification below is all that is needed.
// //
// With a narrower number_unsigned_t/number_integer_t (e.g. // A decimal number with up to 18 digits is always representable in
// std::uint32_t), the exact path would reclassify some of these tokens // both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
// as (finite) floats, while this check reports integers. That does not // both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
// change the result of accept(): it always parses through // could not have set errno to ERANGE for it. Numbers with more digits
// json_sax_acceptor, whose number callbacks discard their argument and // (rare in practice) fall through to the exact code below, unchanged,
// return true, and the parser rejects neither integers nor finite // so their handling -- including reclassification to value_float when
// floats. value_unsigned/value_integer are left unset here, so a caller // the value overflows 64 bits, and rejection when it is not even
// that reads the converted value must not set discard_number_values. // finite as a double -- is bit-for-bit identical to before this
// optimization.
// //
// On contiguous input, scan_number_bulk_contiguous() converts integer // Note this reasons about std::uint64_t/std::int64_t, not about
// tokens itself and does not pass them to this function, unless // number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
// JSON_DIAGNOSTIC_POSITIONS is enabled. This check is therefore only // narrower, template parameters -- e.g. std::uint32_t). That is fine
// reached for input without bulk access (e.g. streams), with // *only* because discard_number_values is exclusively set by
// JSON_DIAGNOSTIC_POSITIONS, or when scan_number_bulk_contiguous() // accept() (see json.hpp), and accept() always parses through the
// falls back to scan_number(). // library's own json_sax_acceptor -- never a user-supplied SAX
// consumer -- whose number_unsigned()/number_integer()/number_float()
// callbacks unconditionally discard their argument and return true.
// So for every caller that can reach this branch, neither the token
// classification below nor the eventual (possibly narrowed, and on
// this fast path left stale/unset) value_unsigned/value_integer is
// ever consulted -- an unsigned/integer token is accepted outright,
// and even a >18-digit token that this fast path deliberately falls
// through for is, once reclassified to value_float, still finite
// (and thus accepted) for any digit count that fits in number_unsigned_t
// or number_integer_t regardless of that type's width. If this
// function is ever taught to run with discard_number_values true for
// a caller that *does* read the converted value, this reasoning (and
// the fast path below) would need to be revisited.
if (discard_number_values) if (discard_number_values)
{ {
constexpr std::size_t safe_digit_count = 18; constexpr std::size_t safe_digit_count = 18;
@@ -1988,7 +2023,7 @@ scan_number_done:
return value_float; return value_float;
} }
/// return current string value /// return current string value (implicitly resets the token; useful only once)
string_t& get_string() string_t& get_string()
{ {
// a number token holds '.' regardless of the locale (#4084) // a number token holds '.' regardless of the locale (#4084)
@@ -2290,11 +2325,11 @@ scan_number_done:
/// the position of the decimal point in token_buffer /// the position of the decimal point in token_buffer
std::size_t decimal_point_position = std::string::npos; std::size_t decimal_point_position = std::string::npos;
/// whether the caller only needs the token types and never looks at the /// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
/// converted numeric values; set only by accept(), which parses through /// token classification and never looks at the converted numeric value;
/// json_sax_acceptor. When set, convert_number() skips converting integer /// when set, scan_number() may skip strtoull()/strtoll() for
/// tokens whose digit count guarantees that they fit into 64 bits (see /// value_unsigned/value_integer tokens whose digit count guarantees they
/// there) /// fit into 64 bits (see scan_number())
const bool discard_number_values = false; const bool discard_number_values = false;
}; };
+43 -50
View File
@@ -54,8 +54,7 @@ using parser_callback_t =
/*! /*!
@brief syntax analysis @brief syntax analysis
This class implements an iterative parser that keeps the open containers on This class implements a recursive descent parser.
an explicit stack and reports what it reads as SAX events.
*/ */
template<typename BasicJsonType, typename InputAdapterType> template<typename BasicJsonType, typename InputAdapterType>
class parser class parser
@@ -99,9 +98,28 @@ class parser
if (callback) if (callback)
{ {
json_sax_dom_callback_parser<BasicJsonType, InputAdapterType> sdp(result, callback, allow_exceptions, &m_lexer); json_sax_dom_callback_parser<BasicJsonType, InputAdapterType> sdp(result, callback, allow_exceptions, &m_lexer);
sax_parse_internal(&sdp);
if (strict)
{
// in strict mode, input must be completely read
if (get_token() != token_type::end_of_input)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(),
exception_message(token_type::end_of_input, "value"), nullptr));
}
}
else
{
// the caller keeps using the input: position it right after
// the value by leaving the character that terminated it
m_lexer.release_lookahead();
}
// in case of an error, return a discarded value // in case of an error, return a discarded value
if (!parse_dom(sdp, strict)) if (sdp.is_errored())
{ {
result = value_t::discarded; result = value_t::discarded;
return; return;
@@ -117,9 +135,26 @@ class parser
else else
{ {
json_sax_dom_parser<BasicJsonType, InputAdapterType> sdp(result, allow_exceptions, &m_lexer); json_sax_dom_parser<BasicJsonType, InputAdapterType> sdp(result, allow_exceptions, &m_lexer);
sax_parse_internal(&sdp);
if (strict)
{
// in strict mode, input must be completely read
if (get_token() != token_type::end_of_input)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
}
}
else
{
// see above
m_lexer.release_lookahead();
}
// in case of an error, return a discarded value // in case of an error, return a discarded value
if (!parse_dom(sdp, strict)) if (sdp.is_errored())
{ {
result = value_t::discarded; result = value_t::discarded;
return; return;
@@ -172,46 +207,6 @@ class parser
} }
private: private:
/*!
@brief run a DOM SAX parser to completion and position the lexer
Shared by both branches of @ref parse(): builds no SAX parser itself,
but drives an already-constructed @a json_sax_dom_parser or
@ref json_sax_dom_callback_parser through @ref sax_parse_internal(),
then applies the strict-EOF check (reporting parse_error.101 through
@a sdp on failure) or, in non-strict mode, releases the lookahead so
the caller can keep reading the input right after the parsed value.
@param[in,out] sdp the DOM SAX parser to run
@param[in] strict whether to expect the last token to be EOF
@return whether @a sdp did not report an error
*/
template<typename DomSax>
bool parse_dom(DomSax& sdp, const bool strict)
{
sax_parse_internal(&sdp);
if (strict)
{
// in strict mode, input must be completely read
if (get_token() != token_type::end_of_input)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(),
exception_message(token_type::end_of_input, "value"), nullptr));
}
}
else
{
// the caller keeps using the input: position it right after
// the value by leaving the character that terminated it
m_lexer.release_lookahead();
}
return !sdp.is_errored();
}
template<typename SAX> template<typename SAX>
JSON_HEDLEY_NON_NULL(2) JSON_HEDLEY_NON_NULL(2)
bool sax_parse_internal(SAX* sax) bool sax_parse_internal(SAX* sax)
@@ -444,9 +439,8 @@ class parser
// We are done with this array. Before we can parse a // We are done with this array. Before we can parse a
// new value, we need to evaluate the new state first. // new value, we need to evaluate the new state first.
// By setting skip_to_state_evaluation to true, the next // By setting skip_to_state_evaluation to false, we
// iteration skips parsing a value and evaluates the // are effectively jumping to the beginning of this if.
// enclosing state directly.
JSON_ASSERT(!states.empty()); JSON_ASSERT(!states.empty());
states.pop_back(); states.pop_back();
skip_to_state_evaluation = true; skip_to_state_evaluation = true;
@@ -506,9 +500,8 @@ class parser
// We are done with this object. Before we can parse a // We are done with this object. Before we can parse a
// new value, we need to evaluate the new state first. // new value, we need to evaluate the new state first.
// By setting skip_to_state_evaluation to true, the next // By setting skip_to_state_evaluation to false, we
// iteration skips parsing a value and evaluates the // are effectively jumping to the beginning of this if.
// enclosing state directly.
JSON_ASSERT(!states.empty()); JSON_ASSERT(!states.empty());
states.pop_back(); states.pop_back();
skip_to_state_evaluation = true; skip_to_state_evaluation = true;
@@ -1991,9 +1991,21 @@ class binary_writer
case 'd': case 'd':
{ {
const auto dval = el.template get<double>(); const auto dval = el.template get<double>();
in_range = !std::isfinite(dval) || #ifdef __GNUC__
JSON_HEDLEY_DIAGNOSTIC_PUSH
JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal")
#endif
// a value that would be rounded (rather than exactly represented) by the
// narrowing to float is treated like an out-of-range integer element above;
// this is the same criterion write_compact_float() uses for CBOR/MessagePack
in_range = std::isnan(dval) ||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) && (dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
dval <= static_cast<double>((std::numeric_limits<float>::max)())); dval <= static_cast<double>((std::numeric_limits<float>::max)()) &&
static_cast<double>(static_cast<float>(dval)) == dval) ||
std::isinf(dval);
#ifdef __GNUC__
JSON_HEDLEY_DIAGNOSTIC_POP
#endif
break; break;
} }
default: default:
+5 -72
View File
@@ -36,61 +36,6 @@ StringType to_string(std::size_t value)
return result; return result;
} }
///////////////////
// UTF-8 encoding //
///////////////////
/*!
@brief encode a Unicode code point as UTF-8
Used to turn a decoded code point back into bytes: by the wide-string input
adapters in input_adapters.hpp (one code point per UTF-32 unit, per UTF-16
unit outside the surrogate range, and per valid UTF-16 surrogate pair), and
by the lexer's `\uXXXX`/`\uXXXX\uYYYY` handling in lexer.hpp. Passing a
code point above U+10FFFF, or one in the surrogate range U+D800..U+DFFF, is
undefined behavior; callers are expected to have rejected those already
(the wide-string adapters pass malformed units through unencoded instead of
calling this function, and the lexer rejects unpaired surrogates before
reaching it).
@tparam Out a callable invoked with one byte (as std::uint32_t, 0x00..0xFF)
at a time, most significant byte first
@param[in] cp the code point to encode (at most U+10FFFF)
@param[in] out called once for each byte of the UTF-8 encoding of @a cp
*/
template<typename Out>
void encode_utf8(std::uint32_t cp, Out&& out)
{
JSON_ASSERT(cp <= 0x10FFFF);
if (cp < 0x80)
{
// 1-byte characters: 0xxxxxxx (ASCII)
out(cp);
}
else if (cp <= 0x7FF)
{
// 2-byte characters: 110xxxxx 10xxxxxx
out(0xC0u | (cp >> 6u));
out(0x80u | (cp & 0x3Fu));
}
else if (cp <= 0xFFFF)
{
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
out(0xE0u | (cp >> 12u));
out(0x80u | ((cp >> 6u) & 0x3Fu));
out(0x80u | (cp & 0x3Fu));
}
else
{
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
out(0xF0u | (cp >> 18u));
out(0x80u | ((cp >> 12u) & 0x3Fu));
out(0x80u | ((cp >> 6u) & 0x3Fu));
out(0x80u | (cp & 0x3Fu));
}
}
/////////////////// ///////////////////
// UTF-8 decoding // // UTF-8 decoding //
/////////////////// ///////////////////
@@ -106,23 +51,11 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally
written by Björn Hoehrmann. See written by Björn Hoehrmann. See
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details. http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
The library checks UTF-8 well-formedness (RFC 3629, section 4) in four This decoder is the single source of truth for UTF-8 validation in this
places, which differ in speed, diagnostics, and how they read the input: library: it is used both by the serializer (to escape and, in strict mode,
reject ill-formed UTF-8 when dumping a string) and by the binary readers
- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in (to reject ill-formed UTF-8 in CBOR/MessagePack/BSON/UBJSON text strings at
strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR, decode time; see @ref is_valid_utf8 below).
MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in
text strings at decode time).
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
diagnostic for each kind of error.
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
bulk string scan, the bulk path of the BON8 reader, and the BON8 writer.
They must accept exactly what the lexer's switch accepts.
- the byte path of binary_reader::get_bon8_string(): BON8 input without bulk
access, and the bytes the bulk path leaves to it.
All four must accept the same set of sequences, so a change to one needs a
matching change to the others.
@param[in,out] state the current decoder state @param[in,out] state the current decoder state
@param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT) @param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT)
-3
View File
@@ -149,9 +149,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
friend class ::nlohmann::detail::json_sax_dom_parser; friend class ::nlohmann::detail::json_sax_dom_parser;
template<typename BasicJsonType, typename InputAdapterType> template<typename BasicJsonType, typename InputAdapterType>
friend class ::nlohmann::detail::json_sax_dom_callback_parser; friend class ::nlohmann::detail::json_sax_dom_callback_parser;
#if JSON_DIAGNOSTIC_POSITIONS
friend struct ::nlohmann::detail::diagnostic_positions;
#endif
friend class ::nlohmann::detail::exception; friend class ::nlohmann::detail::exception;
/// workaround type for MSVC /// workaround type for MSVC
+375 -308
View File
@@ -6234,61 +6234,6 @@ StringType to_string(std::size_t value)
return result; return result;
} }
///////////////////
// UTF-8 encoding //
///////////////////
/*!
@brief encode a Unicode code point as UTF-8
Used to turn a decoded code point back into bytes: by the wide-string input
adapters in input_adapters.hpp (one code point per UTF-32 unit, per UTF-16
unit outside the surrogate range, and per valid UTF-16 surrogate pair), and
by the lexer's `\uXXXX`/`\uXXXX\uYYYY` handling in lexer.hpp. Passing a
code point above U+10FFFF, or one in the surrogate range U+D800..U+DFFF, is
undefined behavior; callers are expected to have rejected those already
(the wide-string adapters pass malformed units through unencoded instead of
calling this function, and the lexer rejects unpaired surrogates before
reaching it).
@tparam Out a callable invoked with one byte (as std::uint32_t, 0x00..0xFF)
at a time, most significant byte first
@param[in] cp the code point to encode (at most U+10FFFF)
@param[in] out called once for each byte of the UTF-8 encoding of @a cp
*/
template<typename Out>
void encode_utf8(std::uint32_t cp, Out&& out)
{
JSON_ASSERT(cp <= 0x10FFFF);
if (cp < 0x80)
{
// 1-byte characters: 0xxxxxxx (ASCII)
out(cp);
}
else if (cp <= 0x7FF)
{
// 2-byte characters: 110xxxxx 10xxxxxx
out(0xC0u | (cp >> 6u));
out(0x80u | (cp & 0x3Fu));
}
else if (cp <= 0xFFFF)
{
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
out(0xE0u | (cp >> 12u));
out(0x80u | ((cp >> 6u) & 0x3Fu));
out(0x80u | (cp & 0x3Fu));
}
else
{
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
out(0xF0u | (cp >> 18u));
out(0x80u | ((cp >> 12u) & 0x3Fu));
out(0x80u | ((cp >> 6u) & 0x3Fu));
out(0x80u | (cp & 0x3Fu));
}
}
/////////////////// ///////////////////
// UTF-8 decoding // // UTF-8 decoding //
/////////////////// ///////////////////
@@ -6304,23 +6249,11 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally
written by Björn Hoehrmann. See written by Björn Hoehrmann. See
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details. http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
The library checks UTF-8 well-formedness (RFC 3629, section 4) in four This decoder is the single source of truth for UTF-8 validation in this
places, which differ in speed, diagnostics, and how they read the input: library: it is used both by the serializer (to escape and, in strict mode,
reject ill-formed UTF-8 when dumping a string) and by the binary readers
- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in (to reject ill-formed UTF-8 in CBOR/MessagePack/BSON/UBJSON text strings at
strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR, decode time; see @ref is_valid_utf8 below).
MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in
text strings at decode time).
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
diagnostic for each kind of error.
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
bulk string scan, the bulk path of the BON8 reader, and the BON8 writer.
They must accept exactly what the lexer's switch accepts.
- the byte path of binary_reader::get_bon8_string(): BON8 input without bulk
access, and the bytes the bulk path leaves to it.
All four must accept the same set of sequences, so a change to one needs a
matching change to the others.
@param[in,out] state the current decoder state @param[in,out] state the current decoder state
@param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT) @param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT)
@@ -7615,12 +7548,12 @@ NLOHMANN_JSON_NAMESPACE_END
#include <algorithm> // min
#include <array> // array #include <array> // array
#include <cstddef> // size_t #include <cstddef> // size_t
#include <cstdint> // uint32_t
#include <cstring> // strlen #include <cstring> // strlen
#include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next #include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next
#include <memory> // shared_ptr, make_shared, addressof
#include <numeric> // accumulate
#include <streambuf> // streambuf #include <streambuf> // streambuf
#include <string> // string, char_traits #include <string> // string, char_traits
#include <type_traits> // enable_if, is_base_of, is_pointer, is_integral, remove_pointer #include <type_traits> // enable_if, is_base_of, is_pointer, is_integral, remove_pointer
@@ -7639,8 +7572,6 @@ NLOHMANN_JSON_NAMESPACE_END
// #include <nlohmann/detail/meta/type_traits.hpp> // #include <nlohmann/detail/meta/type_traits.hpp>
// #include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail namespace detail
@@ -7695,9 +7626,8 @@ class file_input_adapter
}; };
/*! /*!
Input adapter for a (caching) istream. Does not skip a UTF Byte Order Mark Input adapter for a (caching) istream. Ignores a UFT Byte Order Mark at
itself; that is done by the lexer's skip_bom(). Does not support changing beginning of input. Does not support changing the underlying std::streambuf
the underlying std::streambuf
in mid-input. Maintains underlying std::istream and std::streambuf to support in mid-input. Maintains underlying std::istream and std::streambuf to support
subsequent use of standard std::istream operations to process any input subsequent use of standard std::istream operations to process any input
characters following those used in parsing the JSON input. Clears the characters following those used in parsing the JSON input. Clears the
@@ -8062,14 +7992,32 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
// get the current character // get the current character
const auto wc = input.get_character(); const auto wc = input.get_character();
if (wc <= 0x10FFFF) // UTF-32 to UTF-8 encoding
if (wc < 0x80)
{ {
// UTF-32 to UTF-8 encoding utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
utf8_bytes_filled = 0; utf8_bytes_filled = 1;
encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) }
{ else if (wc <= 0x7FF)
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte); {
}); utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u) & 0x1Fu));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 2;
}
else if (wc <= 0xFFFF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u) & 0x0Fu));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 3;
}
else if (wc <= 0x10FFFF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | ((static_cast<unsigned int>(wc) >> 18u) & 0x07u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 12u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 4;
} }
else else
{ {
@@ -8106,15 +8054,24 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
// get the current character // get the current character
const auto wc = input.get_character(); const auto wc = input.get_character();
if (0xD800 > wc || wc >= 0xE000) // UTF-16 to UTF-8 encoding
if (wc < 0x80)
{ {
// a UTF-16 code unit outside the surrogate range is a valid utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
// code point (at most U+FFFF) on its own utf8_bytes_filled = 1;
utf8_bytes_filled = 0; }
encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) else if (wc <= 0x7FF)
{ {
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte); utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u)));
}); utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 2;
}
else if (0xD800 > wc || wc >= 0xE000)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u)));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 3;
} }
else else
{ {
@@ -8132,11 +8089,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (0xDC00 <= wc2 && wc2 <= 0xDFFF) if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{ {
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes_filled = 0; utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
{ utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu));
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte); utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (charcode & 0x3Fu));
}); utf8_bytes_filled = 4;
valid_pair = true; valid_pair = true;
} }
} }
@@ -8449,9 +8406,9 @@ auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) /
return input_adapter(array, array + N); return input_adapter(array, array + N);
} }
// This class only handles inputs that construct a contiguous_bytes_input_adapter // This class only handles inputs of input_buffer_adapter type.
// (e.g. span_input_adapter). It's required so that expressions like {ptr, len} // It's required so that expressions like {ptr, len} can be implicitly cast
// can be implicitly cast to the correct adapter. // to the correct adapter.
class span_input_adapter class span_input_adapter
{ {
public: public:
@@ -8496,7 +8453,6 @@ NLOHMANN_JSON_NAMESPACE_END
#include <algorithm> // find_if, min #include <algorithm> // find_if, min
#include <cstddef> #include <cstddef>
#include <limits> // numeric_limits
#include <string> // string #include <string> // string
#include <type_traits> // enable_if_t #include <type_traits> // enable_if_t
#include <utility> // move, pair #include <utility> // move, pair
@@ -8518,7 +8474,6 @@ NLOHMANN_JSON_NAMESPACE_END
#include <array> // array #include <array> // array
#include <clocale> // localeconv #include <clocale> // localeconv
#include <cstddef> // size_t #include <cstddef> // size_t
#include <cstdint> // uint32_t
#include <cstdio> // snprintf #include <cstdio> // snprintf
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull #include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
#include <initializer_list> // initializer_list #include <initializer_list> // initializer_list
@@ -9161,8 +9116,6 @@ NLOHMANN_JSON_NAMESPACE_END
// #include <nlohmann/detail/meta/type_traits.hpp> // #include <nlohmann/detail/meta/type_traits.hpp>
// #include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail namespace detail
@@ -9639,10 +9592,32 @@ class lexer : public lexer_base<BasicJsonType>
JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF); JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF);
// translate codepoint into bytes // translate codepoint into bytes
encode_utf8(static_cast<std::uint32_t>(codepoint), [this](std::uint32_t byte) if (codepoint < 0x80)
{ {
add(static_cast<char_int_type>(byte)); // 1-byte characters: 0xxxxxxx (ASCII)
}); add(static_cast<char_int_type>(codepoint));
}
else if (codepoint <= 0x7FF)
{
// 2-byte characters: 110xxxxx 10xxxxxx
add(static_cast<char_int_type>(0xC0u | (static_cast<unsigned int>(codepoint) >> 6u)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
else if (codepoint <= 0xFFFF)
{
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
add(static_cast<char_int_type>(0xE0u | (static_cast<unsigned int>(codepoint) >> 12u)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
else
{
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
add(static_cast<char_int_type>(0xF0u | (static_cast<unsigned int>(codepoint) >> 18u)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 12u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
break; break;
} }
@@ -10611,30 +10586,45 @@ scan_number_done:
*/ */
token_type convert_number(token_type number_type, std::size_t mantissa_end) token_type convert_number(token_type number_type, std::size_t mantissa_end)
{ {
// accept() only needs to know whether the input is valid, so it sets // If the caller does not need the converted value (only whether the
// discard_number_values (see json.hpp), and an integer token whose // input is syntactically valid; see json_sax_acceptor/accept()), an
// digit count shows that it fits is reported without calling // unsigned/integer token can be reported without calling
// convert_integer(). A number with up to 18 digits always fits into // strtoull()/strtoll() at all, *provided* we can already tell from
// both std::uint64_t and std::int64_t (18 nines is about 1e18, below // the digit count alone that the conversion cannot overflow 64 bits.
// INT64_MAX, which is about 9.2e18). Longer tokens take the exact path // Such tokens are always finite and are accepted unconditionally by
// below, including the fallback to floating point when the value does // the parser regardless of their actual value (parser::sax_parse_internal()
// not fit. // never checks finiteness for value_unsigned/value_integer), so the
// classification below is all that is needed.
// //
// With a narrower number_unsigned_t/number_integer_t (e.g. // A decimal number with up to 18 digits is always representable in
// std::uint32_t), the exact path would reclassify some of these tokens // both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
// as (finite) floats, while this check reports integers. That does not // both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
// change the result of accept(): it always parses through // could not have set errno to ERANGE for it. Numbers with more digits
// json_sax_acceptor, whose number callbacks discard their argument and // (rare in practice) fall through to the exact code below, unchanged,
// return true, and the parser rejects neither integers nor finite // so their handling -- including reclassification to value_float when
// floats. value_unsigned/value_integer are left unset here, so a caller // the value overflows 64 bits, and rejection when it is not even
// that reads the converted value must not set discard_number_values. // finite as a double -- is bit-for-bit identical to before this
// optimization.
// //
// On contiguous input, scan_number_bulk_contiguous() converts integer // Note this reasons about std::uint64_t/std::int64_t, not about
// tokens itself and does not pass them to this function, unless // number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
// JSON_DIAGNOSTIC_POSITIONS is enabled. This check is therefore only // narrower, template parameters -- e.g. std::uint32_t). That is fine
// reached for input without bulk access (e.g. streams), with // *only* because discard_number_values is exclusively set by
// JSON_DIAGNOSTIC_POSITIONS, or when scan_number_bulk_contiguous() // accept() (see json.hpp), and accept() always parses through the
// falls back to scan_number(). // library's own json_sax_acceptor -- never a user-supplied SAX
// consumer -- whose number_unsigned()/number_integer()/number_float()
// callbacks unconditionally discard their argument and return true.
// So for every caller that can reach this branch, neither the token
// classification below nor the eventual (possibly narrowed, and on
// this fast path left stale/unset) value_unsigned/value_integer is
// ever consulted -- an unsigned/integer token is accepted outright,
// and even a >18-digit token that this fast path deliberately falls
// through for is, once reclassified to value_float, still finite
// (and thus accepted) for any digit count that fits in number_unsigned_t
// or number_integer_t regardless of that type's width. If this
// function is ever taught to run with discard_number_values true for
// a caller that *does* read the converted value, this reasoning (and
// the fast path below) would need to be revisited.
if (discard_number_values) if (discard_number_values)
{ {
constexpr std::size_t safe_digit_count = 18; constexpr std::size_t safe_digit_count = 18;
@@ -11125,7 +11115,7 @@ scan_number_done:
return value_float; return value_float;
} }
/// return current string value /// return current string value (implicitly resets the token; useful only once)
string_t& get_string() string_t& get_string()
{ {
// a number token holds '.' regardless of the locale (#4084) // a number token holds '.' regardless of the locale (#4084)
@@ -11427,11 +11417,11 @@ scan_number_done:
/// the position of the decimal point in token_buffer /// the position of the decimal point in token_buffer
std::size_t decimal_point_position = std::string::npos; std::size_t decimal_point_position = std::string::npos;
/// whether the caller only needs the token types and never looks at the /// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
/// converted numeric values; set only by accept(), which parses through /// token classification and never looks at the converted numeric value;
/// json_sax_acceptor. When set, convert_number() skips converting integer /// when set, scan_number() may skip strtoull()/strtoll() for
/// tokens whose digit count guarantees that they fit into 64 bits (see /// value_unsigned/value_integer tokens whose digit count guarantees they
/// there) /// fit into 64 bits (see scan_number())
const bool discard_number_values = false; const bool discard_number_values = false;
}; };
@@ -11599,88 +11589,6 @@ template<typename ArrayType>
inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/) inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/)
{} {}
#if JSON_DIAGNOSTIC_POSITIONS
/*!
@brief set the diagnostic positions of a value the DOM SAX parsers just stored
Shared by json_sax_dom_parser and json_sax_dom_callback_parser. basic_json
befriends this struct, as the position members are private.
*/
struct diagnostic_positions
{
/*!
@param[in,out] v the value that was just parsed
@param[in] lexer the lexer that read it, or nullptr to leave @a v alone
*/
template<typename BasicJsonType, typename LexerType>
static void set_from_lexer(BasicJsonType& v, LexerType* lexer)
{
if (lexer)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = lexer->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = lexer->get_token_start_position();
break;
}
case value_t::discarded:
{
// an object or array the callback of
// json_sax_dom_callback_parser rejected has no position
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - lexer->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
}
}
};
#endif
/*! /*!
@brief SAX implementation to create a JSON value from SAX events @brief SAX implementation to create a JSON value from SAX events
@@ -11882,6 +11790,76 @@ class json_sax_dom_parser
private: private:
#if JSON_DIAGNOSTIC_POSITIONS
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
{
if (m_lexer_ref)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = m_lexer_ref->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = m_lexer_ref->get_token_start_position();
break;
}
// As we handle the start and end positions for values created during parsing,
// we do not expect the following value type to be called. Regardless, set the positions
// in case this is created manually or through a different constructor. Exclude from lcov
// since the exact condition of this switch is esoteric.
// LCOV_EXCL_START
case value_t::discarded:
{
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
// LCOV_EXCL_STOP
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - m_lexer_ref->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
}
}
}
#endif
/*! /*!
@invariant If the ref stack is empty, then the passed value will be the new @invariant If the ref stack is empty, then the passed value will be the new
root. root.
@@ -11897,7 +11875,7 @@ class json_sax_dom_parser
root = BasicJsonType(std::forward<Value>(v)); root = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS #if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(root, m_lexer_ref); handle_diagnostic_positions_for_json_value(root);
#endif #endif
return &root; return &root;
@@ -11910,7 +11888,7 @@ class json_sax_dom_parser
ref_stack.back()->m_data.m_value.array->emplace_back(std::forward<Value>(v)); ref_stack.back()->m_data.m_value.array->emplace_back(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS #if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(ref_stack.back()->m_data.m_value.array->back(), m_lexer_ref); handle_diagnostic_positions_for_json_value(ref_stack.back()->m_data.m_value.array->back());
#endif #endif
return &(ref_stack.back()->m_data.m_value.array->back()); return &(ref_stack.back()->m_data.m_value.array->back());
@@ -11921,7 +11899,7 @@ class json_sax_dom_parser
*object_element = BasicJsonType(std::forward<Value>(v)); *object_element = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS #if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(*object_element, m_lexer_ref); handle_diagnostic_positions_for_json_value(*object_element);
#endif #endif
return object_element; return object_element;
@@ -12098,7 +12076,7 @@ class json_sax_dom_callback_parser
#if JSON_DIAGNOSTIC_POSITIONS #if JSON_DIAGNOSTIC_POSITIONS
// Set start/end positions for discarded object. // Set start/end positions for discarded object.
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref); handle_diagnostic_positions_for_json_value(*ref_stack.back());
#endif #endif
} }
} }
@@ -12214,7 +12192,7 @@ class json_sax_dom_callback_parser
#if JSON_DIAGNOSTIC_POSITIONS #if JSON_DIAGNOSTIC_POSITIONS
// Set start/end positions for discarded array. // Set start/end positions for discarded array.
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref); handle_diagnostic_positions_for_json_value(*ref_stack.back());
#endif #endif
} }
} }
@@ -12267,6 +12245,72 @@ class json_sax_dom_callback_parser
private: private:
#if JSON_DIAGNOSTIC_POSITIONS
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
{
if (m_lexer_ref)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = m_lexer_ref->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = m_lexer_ref->get_token_start_position();
break;
}
case value_t::discarded:
{
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - m_lexer_ref->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
}
}
}
#endif
/// if there is a pending duplicate-key stash entry for this exact slot, /// if there is a pending duplicate-key stash entry for this exact slot,
/// remove it from the stash; if restore_value is true, the stashed /// remove it from the stash; if restore_value is true, the stashed
/// previous value is moved back into the slot first (use this when the /// previous value is moved back into the slot first (use this when the
@@ -12388,7 +12432,7 @@ class json_sax_dom_callback_parser
auto value = BasicJsonType(std::forward<Value>(v)); auto value = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS #if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(value, m_lexer_ref); handle_diagnostic_positions_for_json_value(value);
#endif #endif
// check callback // check callback
@@ -15451,10 +15495,15 @@ class binary_reader
is_ndarray can only return `true` when its initial value is_ndarray can only return `true` when its initial value
is `false` is `false`
@param[in] prefix type marker if already read, otherwise set to 0 @param[in] prefix type marker if already read, otherwise set to 0
@param[in] ndarray_dtype the element type marker of the enclosing bjdata ndarray if
already known (it precedes the dimension vector read here),
otherwise 0; used to emit the "_ArrayType_" annotation key
before "_ArraySize_" if a dimension vector turns out to
describe an ndarray
@return whether size determination completed @return whether size determination completed
*/ */
bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0) bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0, char_int_type ndarray_dtype = 0)
{ {
if (prefix == 0) if (prefix == 0)
{ {
@@ -15624,8 +15673,37 @@ class binary_reader
} }
} }
if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3)))
{
return false;
}
// the element type precedes the dimension vector (see get_ubjson_size_type)
// and is passed down as ndarray_dtype; emit it here so the annotation keys
// follow the documented _ArrayType_, _ArraySize_, _ArrayData_ order
if (ndarray_dtype != 0)
{
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), ndarray_dtype, [](const bjd_type & p, char_int_type t)
{
return p.first < t;
});
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != ndarray_dtype))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
}
string_t type_key = "_ArrayType_";
string_t type = it->second; // sax->string() takes a reference
if (JSON_HEDLEY_UNLIKELY(!sax->key(type_key) || !sax->string(type)))
{
return false;
}
}
string_t key = "_ArraySize_"; string_t key = "_ArraySize_";
if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3) || !sax->key(key) || !sax->start_array(dim.size()))) if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(dim.size())))
{ {
return false; return false;
} }
@@ -15726,7 +15804,7 @@ class binary_reader
exception_message(input_format, concat("expected '#' after type information; last byte: 0x", last_token), "size"), nullptr)); exception_message(input_format, concat("expected '#' after type information; last byte: 0x", last_token), "size"), nullptr));
} }
const bool is_error = get_ubjson_size_value(result.first, is_ndarray); const bool is_error = get_ubjson_size_value(result.first, is_ndarray, 0, result.second);
// an ndarray was read here only if the flag flipped; when it was // an ndarray was read here only if the flag flipped; when it was
// seeded true, get_ubjson_size_value() already rejected the nested // seeded true, get_ubjson_size_value() already rejected the nested
// dimension vector // dimension vector
@@ -15962,30 +16040,17 @@ class binary_reader
if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0) if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0)
{ {
size_and_type.second &= ~(static_cast<char_int_type>(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker size_and_type.second &= ~(static_cast<char_int_type>(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t)
{
return p.first < t;
});
string_t key = "_ArrayType_";
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
}
string_t type = it->second; // sax->string() takes a reference
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type)))
{
return false;
}
// the "_ArrayType_" and "_ArraySize_" annotation keys were already emitted by
// get_ubjson_size_value() (the type marker is known before the dimension vector
// that determines size_and_type.first is read, so it is emitted first there to
// match the documented _ArrayType_, _ArraySize_, _ArrayData_ key order)
if (size_and_type.second == 'C' || size_and_type.second == 'B') if (size_and_type.second == 'C' || size_and_type.second == 'B')
{ {
size_and_type.second = 'U'; size_and_type.second = 'U';
} }
key = "_ArrayData_"; string_t key = "_ArrayData_";
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(size_and_type.first) )) if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(size_and_type.first) ))
{ {
return false; return false;
@@ -16067,8 +16132,8 @@ class binary_reader
return enter_object(detail::unknown_size()); return enter_object(detail::unknown_size());
} }
// Note, UBJSON has no binary type of its own; BJData, which shares this // Note, no reader for UBJSON binary types is implemented because they do
// reader, decodes optimized 'B' arrays as binary in get_ubjson_array(). // not exist
bool get_ubjson_high_precision_number() bool get_ubjson_high_precision_number()
{ {
@@ -16775,7 +16840,7 @@ class binary_reader
#endif #endif
} }
/*! /*
@brief read a number from the input @brief read a number from the input
@tparam NumberType the type of the number @tparam NumberType the type of the number
@@ -16785,10 +16850,10 @@ class binary_reader
@return whether conversion completed @return whether conversion completed
@note This function needs to respect the system's endianness, because @note This function needs to respect the system's endianness, because
bytes in CBOR, MessagePack, UBJSON, and BON8 are stored in network bytes in CBOR, MessagePack, and UBJSON are stored in network order
order (big endian) and therefore need reordering on little endian (big endian) and therefore need reordering on little endian systems.
systems. On the other hand, BSON and BJData use little endian and On the other hand, BSON and BJData use little endian and should reorder
should reorder on big endian systems. on big endian systems.
*/ */
template<typename NumberType, bool InputIsLittleEndian = false> template<typename NumberType, bool InputIsLittleEndian = false>
bool get_number(const input_format_t format, NumberType& result) bool get_number(const input_format_t format, NumberType& result)
@@ -17142,8 +17207,7 @@ using parser_callback_t =
/*! /*!
@brief syntax analysis @brief syntax analysis
This class implements an iterative parser that keeps the open containers on This class implements a recursive descent parser.
an explicit stack and reports what it reads as SAX events.
*/ */
template<typename BasicJsonType, typename InputAdapterType> template<typename BasicJsonType, typename InputAdapterType>
class parser class parser
@@ -17187,9 +17251,28 @@ class parser
if (callback) if (callback)
{ {
json_sax_dom_callback_parser<BasicJsonType, InputAdapterType> sdp(result, callback, allow_exceptions, &m_lexer); json_sax_dom_callback_parser<BasicJsonType, InputAdapterType> sdp(result, callback, allow_exceptions, &m_lexer);
sax_parse_internal(&sdp);
if (strict)
{
// in strict mode, input must be completely read
if (get_token() != token_type::end_of_input)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(),
exception_message(token_type::end_of_input, "value"), nullptr));
}
}
else
{
// the caller keeps using the input: position it right after
// the value by leaving the character that terminated it
m_lexer.release_lookahead();
}
// in case of an error, return a discarded value // in case of an error, return a discarded value
if (!parse_dom(sdp, strict)) if (sdp.is_errored())
{ {
result = value_t::discarded; result = value_t::discarded;
return; return;
@@ -17205,9 +17288,26 @@ class parser
else else
{ {
json_sax_dom_parser<BasicJsonType, InputAdapterType> sdp(result, allow_exceptions, &m_lexer); json_sax_dom_parser<BasicJsonType, InputAdapterType> sdp(result, allow_exceptions, &m_lexer);
sax_parse_internal(&sdp);
if (strict)
{
// in strict mode, input must be completely read
if (get_token() != token_type::end_of_input)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
}
}
else
{
// see above
m_lexer.release_lookahead();
}
// in case of an error, return a discarded value // in case of an error, return a discarded value
if (!parse_dom(sdp, strict)) if (sdp.is_errored())
{ {
result = value_t::discarded; result = value_t::discarded;
return; return;
@@ -17260,46 +17360,6 @@ class parser
} }
private: private:
/*!
@brief run a DOM SAX parser to completion and position the lexer
Shared by both branches of @ref parse(): builds no SAX parser itself,
but drives an already-constructed @a json_sax_dom_parser or
@ref json_sax_dom_callback_parser through @ref sax_parse_internal(),
then applies the strict-EOF check (reporting parse_error.101 through
@a sdp on failure) or, in non-strict mode, releases the lookahead so
the caller can keep reading the input right after the parsed value.
@param[in,out] sdp the DOM SAX parser to run
@param[in] strict whether to expect the last token to be EOF
@return whether @a sdp did not report an error
*/
template<typename DomSax>
bool parse_dom(DomSax& sdp, const bool strict)
{
sax_parse_internal(&sdp);
if (strict)
{
// in strict mode, input must be completely read
if (get_token() != token_type::end_of_input)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(),
exception_message(token_type::end_of_input, "value"), nullptr));
}
}
else
{
// the caller keeps using the input: position it right after
// the value by leaving the character that terminated it
m_lexer.release_lookahead();
}
return !sdp.is_errored();
}
template<typename SAX> template<typename SAX>
JSON_HEDLEY_NON_NULL(2) JSON_HEDLEY_NON_NULL(2)
bool sax_parse_internal(SAX* sax) bool sax_parse_internal(SAX* sax)
@@ -17532,9 +17592,8 @@ class parser
// We are done with this array. Before we can parse a // We are done with this array. Before we can parse a
// new value, we need to evaluate the new state first. // new value, we need to evaluate the new state first.
// By setting skip_to_state_evaluation to true, the next // By setting skip_to_state_evaluation to false, we
// iteration skips parsing a value and evaluates the // are effectively jumping to the beginning of this if.
// enclosing state directly.
JSON_ASSERT(!states.empty()); JSON_ASSERT(!states.empty());
states.pop_back(); states.pop_back();
skip_to_state_evaluation = true; skip_to_state_evaluation = true;
@@ -17594,9 +17653,8 @@ class parser
// We are done with this object. Before we can parse a // We are done with this object. Before we can parse a
// new value, we need to evaluate the new state first. // new value, we need to evaluate the new state first.
// By setting skip_to_state_evaluation to true, the next // By setting skip_to_state_evaluation to false, we
// iteration skips parsing a value and evaluates the // are effectively jumping to the beginning of this if.
// enclosing state directly.
JSON_ASSERT(!states.empty()); JSON_ASSERT(!states.empty());
states.pop_back(); states.pop_back();
skip_to_state_evaluation = true; skip_to_state_evaluation = true;
@@ -22284,9 +22342,21 @@ class binary_writer
case 'd': case 'd':
{ {
const auto dval = el.template get<double>(); const auto dval = el.template get<double>();
in_range = !std::isfinite(dval) || #ifdef __GNUC__
JSON_HEDLEY_DIAGNOSTIC_PUSH
JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal")
#endif
// a value that would be rounded (rather than exactly represented) by the
// narrowing to float is treated like an out-of-range integer element above;
// this is the same criterion write_compact_float() uses for CBOR/MessagePack
in_range = std::isnan(dval) ||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) && (dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
dval <= static_cast<double>((std::numeric_limits<float>::max)())); dval <= static_cast<double>((std::numeric_limits<float>::max)()) &&
static_cast<double>(static_cast<float>(dval)) == dval) ||
std::isinf(dval);
#ifdef __GNUC__
JSON_HEDLEY_DIAGNOSTIC_POP
#endif
break; break;
} }
default: default:
@@ -26193,9 +26263,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
friend class ::nlohmann::detail::json_sax_dom_parser; friend class ::nlohmann::detail::json_sax_dom_parser;
template<typename BasicJsonType, typename InputAdapterType> template<typename BasicJsonType, typename InputAdapterType>
friend class ::nlohmann::detail::json_sax_dom_callback_parser; friend class ::nlohmann::detail::json_sax_dom_callback_parser;
#if JSON_DIAGNOSTIC_POSITIONS
friend struct ::nlohmann::detail::diagnostic_positions;
#endif
friend class ::nlohmann::detail::exception; friend class ::nlohmann::detail::exception;
/// workaround type for MSVC /// workaround type for MSVC
+42 -4
View File
@@ -11,6 +11,7 @@
#define JSON_TESTS_PRIVATE #define JSON_TESTS_PRIVATE
#include <nlohmann/json.hpp> #include <nlohmann/json.hpp>
using nlohmann::json; using nlohmann::json;
using ordered_json = nlohmann::ordered_json;
#include <algorithm> #include <algorithm>
#include <climits> #include <climits>
@@ -2294,29 +2295,33 @@ TEST_CASE("BJData")
SECTION("start_array() in ndarray _ArraySize_") SECTION("start_array() in ndarray _ArraySize_")
{ {
// _ArrayType_ (2 events: key + string) is now emitted before
// _ArraySize_ (see GitHub issue #5661), which shifts the events
// below later by the same 2 events
std::vector<uint8_t> const v = {'[', '$', 'i', '#', '[', '$', 'i', '#', 'i', 2, 2, 1, 1, 2}; std::vector<uint8_t> const v = {'[', '$', 'i', '#', '[', '$', 'i', '#', 'i', 2, 2, 1, 1, 2};
SaxCountdown scp(2); SaxCountdown scp(4);
CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata)); CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata));
} }
SECTION("number_integer() in ndarray _ArraySize_") SECTION("number_integer() in ndarray _ArraySize_")
{ {
std::vector<uint8_t> const v = {'[', '$', 'U', '#', '[', '$', 'i', '#', 'i', 2, 2, 1, 1, 2}; std::vector<uint8_t> const v = {'[', '$', 'U', '#', '[', '$', 'i', '#', 'i', 2, 2, 1, 1, 2};
SaxCountdown scp(3); SaxCountdown scp(5);
CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata)); CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata));
} }
SECTION("key() in ndarray _ArrayType_") SECTION("key() in ndarray _ArrayType_")
{ {
// _ArrayType_ is emitted right after start_object(), before _ArraySize_
std::vector<uint8_t> const v = {'[', '$', 'U', '#', '[', '$', 'U', '#', 'i', 2, 2, 2, 1, 2, 3, 4}; std::vector<uint8_t> const v = {'[', '$', 'U', '#', '[', '$', 'U', '#', 'i', 2, 2, 2, 1, 2, 3, 4};
SaxCountdown scp(6); SaxCountdown scp(1);
CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata)); CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata));
} }
SECTION("string() in ndarray _ArrayType_") SECTION("string() in ndarray _ArrayType_")
{ {
std::vector<uint8_t> const v = {'[', '$', 'U', '#', '[', '$', 'U', '#', 'i', 2, 2, 2, 1, 2, 3, 4}; std::vector<uint8_t> const v = {'[', '$', 'U', '#', '[', '$', 'U', '#', 'i', 2, 2, 2, 1, 2, 3, 4};
SaxCountdown scp(7); SaxCountdown scp(2);
CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata)); CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata));
} }
@@ -2919,6 +2924,22 @@ TEST_CASE("BJData")
CHECK(out_single.at(0) == '{'); CHECK(out_single.at(0) == '{');
CHECK(json::from_bjdata(out_single) == j_single); CHECK(json::from_bjdata(out_single) == j_single);
// a double element that is finite and within the range of "single"
// but is not exactly representable as a float, so narrowing it would
// silently round it (0.1 is read back as 0.10000000149011612); this,
// like the overflow case above, falls back to a plain object (see
// GitHub issue #5661)
json const j_single_rounded = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 0.1}}});
const auto out_single_rounded = json::to_bjdata(j_single_rounded);
CHECK(out_single_rounded.at(0) == '{');
CHECK(json::from_bjdata(out_single_rounded) == j_single_rounded);
// a double element that underflows to 0 when narrowed to "single"
json const j_single_underflow = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 1e-300}}});
const auto out_single_underflow = json::to_bjdata(j_single_underflow);
CHECK(out_single_underflow.at(0) == '{');
CHECK(json::from_bjdata(out_single_underflow) == j_single_underflow);
// in-range boundary values still use the compact ndarray encoding // in-range boundary values still use the compact ndarray encoding
json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {0, 255}}}); json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {0, 255}}});
CHECK(json::to_bjdata(j_uint8_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, 'i', 1, ']', 0, 255})); CHECK(json::to_bjdata(j_uint8_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, 'i', 1, ']', 0, 255}));
@@ -2932,6 +2953,23 @@ TEST_CASE("BJData")
CHECK(json::from_bjdata(out_single_ok) == json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5f, -1.5f}}})); CHECK(json::from_bjdata(out_single_ok) == json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5f, -1.5f}}}));
} }
SECTION("ndarray annotation keys are read back in the documented order")
{
// from_bjdata() must emit the annotation object's keys in the order
// used throughout the documentation, _ArrayType_, _ArraySize_,
// _ArrayData_: the type marker precedes the dimension vector on the
// wire (see get_ubjson_size_type()), so it is known, and emitted,
// before _ArraySize_. For a plain json this key order is invisible
// (its comparison ignores it), but for an ordered_json it is not (see
// GitHub issue #5661).
const ordered_json o = ordered_json::parse(R"({"_ArrayType_":"uint8","_ArraySize_":[2,2],"_ArrayData_":[1,2,3,4]})");
const auto packed = ordered_json::to_bjdata(o);
CHECK(packed.at(0) == '[');
const ordered_json o_back = ordered_json::from_bjdata(packed);
CHECK(o_back == o);
CHECK(o_back.dump() == o.dump());
}
SECTION("ndarray that would not be read back as an annotated object stays as object") SECTION("ndarray that would not be read back as an annotated object stays as object")
{ {
// the reader only restores an annotated object from an ND-array // the reader only restores an annotated object from an ND-array