mirror of
https://github.com/nlohmann/json.git
synced 2026-10-06 06:30:31 +00:00
Merge branch 'develop' into claude/todo-191-plan-508110
- ordered_map.hpp: develop's refactored members, with this branch's @sa links on every public member - byte_container_with_subtype operator==/!= and ordered_map docs: develop's reviewed text (#5638); drop the now unreferenced examples - basic_json.md: keep both version-history additions - add the docset entries that develop's style check now requires for the 25 API pages this branch adds - regenerate tools/api_checker/api_surface.json for develop's API Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
File diff suppressed because it is too large
Load Diff
@@ -8,12 +8,12 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <algorithm> // min
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstring> // strlen
|
||||
#include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next
|
||||
#include <memory> // shared_ptr, make_shared, addressof
|
||||
#include <numeric> // accumulate
|
||||
#include <streambuf> // streambuf
|
||||
#include <string> // string, char_traits
|
||||
#include <type_traits> // enable_if, is_base_of, is_pointer, is_integral, remove_pointer
|
||||
@@ -28,6 +28,7 @@
|
||||
#include <nlohmann/detail/iterators/iterator_traits.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||
#include <nlohmann/detail/string_utils.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -82,8 +83,9 @@ class file_input_adapter
|
||||
};
|
||||
|
||||
/*!
|
||||
Input adapter for a (caching) istream. Ignores a UFT Byte Order Mark at
|
||||
beginning of input. Does not support changing the underlying std::streambuf
|
||||
Input adapter for a (caching) istream. Does not skip a UTF Byte Order Mark
|
||||
itself; that is done by the lexer's skip_bom(). Does not support changing
|
||||
the underlying std::streambuf
|
||||
in mid-input. Maintains underlying std::istream and std::streambuf to support
|
||||
subsequent use of standard std::istream operations to process any input
|
||||
characters following those used in parsing the JSON input. Clears the
|
||||
@@ -451,35 +453,19 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
|
||||
}
|
||||
else
|
||||
{
|
||||
// get the current character
|
||||
const auto wc = input.get_character();
|
||||
// get the current character; converted to an unsigned type so that
|
||||
// a negative unit (wint_t is signed on some platforms) is not
|
||||
// mistaken for an ASCII character or for EOF
|
||||
const auto wc = static_cast<std::uint32_t>(input.get_character());
|
||||
|
||||
// UTF-32 to UTF-8 encoding
|
||||
if (wc < 0x80)
|
||||
if (wc <= 0x10FFFF)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
|
||||
utf8_bytes_filled = 1;
|
||||
}
|
||||
else if (wc <= 0x7FF)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u) & 0x1Fu));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
||||
utf8_bytes_filled = 2;
|
||||
}
|
||||
else if (wc <= 0xFFFF)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u) & 0x0Fu));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
|
||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
||||
utf8_bytes_filled = 3;
|
||||
}
|
||||
else if (wc <= 0x10FFFF)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | ((static_cast<unsigned int>(wc) >> 18u) & 0x07u));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 12u) & 0x3Fu));
|
||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
|
||||
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
||||
utf8_bytes_filled = 4;
|
||||
// UTF-32 to UTF-8 encoding
|
||||
utf8_bytes_filled = 0;
|
||||
encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
|
||||
{
|
||||
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -516,24 +502,15 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
|
||||
// get the current character
|
||||
const auto wc = input.get_character();
|
||||
|
||||
// UTF-16 to UTF-8 encoding
|
||||
if (wc < 0x80)
|
||||
if (0xD800 > wc || wc >= 0xE000)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
|
||||
utf8_bytes_filled = 1;
|
||||
}
|
||||
else if (wc <= 0x7FF)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u)));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
||||
utf8_bytes_filled = 2;
|
||||
}
|
||||
else if (0xD800 > wc || wc >= 0xE000)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u)));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
|
||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
|
||||
utf8_bytes_filled = 3;
|
||||
// a UTF-16 code unit outside the surrogate range is a valid
|
||||
// code point (at most U+FFFF) on its own
|
||||
utf8_bytes_filled = 0;
|
||||
encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
|
||||
{
|
||||
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
|
||||
});
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -547,22 +524,25 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
|
||||
bool valid_pair = false;
|
||||
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
|
||||
{
|
||||
const auto wc2 = static_cast<unsigned int>(input.get_character());
|
||||
// only consume the next unit if it completes the pair
|
||||
const auto wc2 = static_cast<unsigned int>(*input.current);
|
||||
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
|
||||
{
|
||||
input.get_character();
|
||||
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
|
||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu));
|
||||
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (charcode & 0x3Fu));
|
||||
utf8_bytes_filled = 4;
|
||||
utf8_bytes_filled = 0;
|
||||
encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
|
||||
{
|
||||
utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
|
||||
});
|
||||
valid_pair = true;
|
||||
}
|
||||
}
|
||||
|
||||
if (!valid_pair)
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
|
||||
// emit a byte that is never valid UTF-8 (see the UTF-32 case)
|
||||
utf8_bytes[0] = 0xFF;
|
||||
utf8_bytes_filled = 1;
|
||||
}
|
||||
}
|
||||
@@ -769,6 +749,9 @@ struct container_input_adapter_factory< ContainerType,
|
||||
|
||||
static adapter_type create(ContainerType&& container)
|
||||
{
|
||||
// container is forwarded twice on purpose: the resulting begin/end
|
||||
// iterator types must match adapter_type, computed the same way
|
||||
// NOLINTNEXTLINE(bugprone-use-after-move,hicpp-invalid-access-moved)
|
||||
return input_adapter(begin(std::forward<ContainerType>(container)), end(std::forward<ContainerType>(container)));
|
||||
}
|
||||
};
|
||||
@@ -884,9 +867,9 @@ auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) /
|
||||
return input_adapter(array, array + N);
|
||||
}
|
||||
|
||||
// This class only handles inputs of input_buffer_adapter type.
|
||||
// It's required so that expressions like {ptr, len} can be implicitly cast
|
||||
// to the correct adapter.
|
||||
// This class only handles inputs that construct a contiguous_bytes_input_adapter
|
||||
// (e.g. span_input_adapter). It's required so that expressions like {ptr, len}
|
||||
// can be implicitly cast to the correct adapter.
|
||||
class span_input_adapter
|
||||
{
|
||||
public:
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
|
||||
#include <algorithm> // find_if, min
|
||||
#include <cstddef>
|
||||
#include <limits> // numeric_limits
|
||||
#include <string> // string
|
||||
#include <type_traits> // enable_if_t
|
||||
#include <utility> // move, pair
|
||||
@@ -199,6 +200,88 @@ template<typename ArrayType>
|
||||
inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/)
|
||||
{}
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
/*!
|
||||
@brief set the diagnostic positions of a value the DOM SAX parsers just stored
|
||||
|
||||
Shared by json_sax_dom_parser and json_sax_dom_callback_parser. basic_json
|
||||
befriends this struct, as the position members are private.
|
||||
*/
|
||||
struct diagnostic_positions
|
||||
{
|
||||
/*!
|
||||
@param[in,out] v the value that was just parsed
|
||||
@param[in] lexer the lexer that read it, or nullptr to leave @a v alone
|
||||
*/
|
||||
template<typename BasicJsonType, typename LexerType>
|
||||
static void set_from_lexer(BasicJsonType& v, LexerType* lexer)
|
||||
{
|
||||
if (lexer)
|
||||
{
|
||||
// Lexer has read past the current field value, so set the end position to the current position.
|
||||
// The start position will be set below based on the length of the string representation
|
||||
// of the value.
|
||||
v.end_position = lexer->get_position();
|
||||
|
||||
switch (v.type())
|
||||
{
|
||||
case value_t::boolean:
|
||||
{
|
||||
// 4 and 5 are the string length of "true" and "false"
|
||||
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::null:
|
||||
{
|
||||
// 4 is the string length of "null"
|
||||
v.start_position = v.end_position - 4;
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
// escape sequences make the token longer than the value it
|
||||
// parses to, so the start position cannot be derived from
|
||||
// the value; use the offset the lexer recorded instead
|
||||
v.start_position = lexer->get_token_start_position();
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::discarded:
|
||||
{
|
||||
// an object or array the callback of
|
||||
// json_sax_dom_callback_parser rejected has no position
|
||||
v.end_position = std::string::npos;
|
||||
v.start_position = v.end_position;
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::binary:
|
||||
case value_t::number_integer:
|
||||
case value_t::number_unsigned:
|
||||
case value_t::number_float:
|
||||
{
|
||||
v.start_position = v.end_position - lexer->get_string().size();
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::object:
|
||||
case value_t::array:
|
||||
{
|
||||
// object and array are handled in start_object() and start_array() handlers
|
||||
// skip setting the values here.
|
||||
break;
|
||||
}
|
||||
default: // LCOV_EXCL_LINE
|
||||
// Handle all possible types discretely, default handler should never be reached.
|
||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
#endif
|
||||
|
||||
/*!
|
||||
@brief SAX implementation to create a JSON value from SAX events
|
||||
|
||||
@@ -400,76 +483,6 @@ class json_sax_dom_parser
|
||||
|
||||
private:
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
|
||||
{
|
||||
if (m_lexer_ref)
|
||||
{
|
||||
// Lexer has read past the current field value, so set the end position to the current position.
|
||||
// The start position will be set below based on the length of the string representation
|
||||
// of the value.
|
||||
v.end_position = m_lexer_ref->get_position();
|
||||
|
||||
switch (v.type())
|
||||
{
|
||||
case value_t::boolean:
|
||||
{
|
||||
// 4 and 5 are the string length of "true" and "false"
|
||||
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::null:
|
||||
{
|
||||
// 4 is the string length of "null"
|
||||
v.start_position = v.end_position - 4;
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
// escape sequences make the token longer than the value it
|
||||
// parses to, so the start position cannot be derived from
|
||||
// the value; use the offset the lexer recorded instead
|
||||
v.start_position = m_lexer_ref->get_token_start_position();
|
||||
break;
|
||||
}
|
||||
|
||||
// As we handle the start and end positions for values created during parsing,
|
||||
// we do not expect the following value type to be called. Regardless, set the positions
|
||||
// in case this is created manually or through a different constructor. Exclude from lcov
|
||||
// since the exact condition of this switch is esoteric.
|
||||
// LCOV_EXCL_START
|
||||
case value_t::discarded:
|
||||
{
|
||||
v.end_position = std::string::npos;
|
||||
v.start_position = v.end_position;
|
||||
break;
|
||||
}
|
||||
// LCOV_EXCL_STOP
|
||||
case value_t::binary:
|
||||
case value_t::number_integer:
|
||||
case value_t::number_unsigned:
|
||||
case value_t::number_float:
|
||||
{
|
||||
v.start_position = v.end_position - m_lexer_ref->get_string().size();
|
||||
break;
|
||||
}
|
||||
case value_t::object:
|
||||
case value_t::array:
|
||||
{
|
||||
// object and array are handled in start_object() and start_array() handlers
|
||||
// skip setting the values here.
|
||||
break;
|
||||
}
|
||||
default: // LCOV_EXCL_LINE
|
||||
// Handle all possible types discretely, default handler should never be reached.
|
||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
/*!
|
||||
@invariant If the ref stack is empty, then the passed value will be the new
|
||||
root.
|
||||
@@ -485,7 +498,7 @@ class json_sax_dom_parser
|
||||
root = BasicJsonType(std::forward<Value>(v));
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
handle_diagnostic_positions_for_json_value(root);
|
||||
diagnostic_positions::set_from_lexer(root, m_lexer_ref);
|
||||
#endif
|
||||
|
||||
return &root;
|
||||
@@ -498,7 +511,7 @@ class json_sax_dom_parser
|
||||
ref_stack.back()->m_data.m_value.array->emplace_back(std::forward<Value>(v));
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
handle_diagnostic_positions_for_json_value(ref_stack.back()->m_data.m_value.array->back());
|
||||
diagnostic_positions::set_from_lexer(ref_stack.back()->m_data.m_value.array->back(), m_lexer_ref);
|
||||
#endif
|
||||
|
||||
return &(ref_stack.back()->m_data.m_value.array->back());
|
||||
@@ -509,7 +522,7 @@ class json_sax_dom_parser
|
||||
*object_element = BasicJsonType(std::forward<Value>(v));
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
handle_diagnostic_positions_for_json_value(*object_element);
|
||||
diagnostic_positions::set_from_lexer(*object_element, m_lexer_ref);
|
||||
#endif
|
||||
|
||||
return object_element;
|
||||
@@ -698,7 +711,7 @@ class json_sax_dom_callback_parser
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
// Set start/end positions for discarded object.
|
||||
handle_diagnostic_positions_for_json_value(*ref_stack.back());
|
||||
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
@@ -814,7 +827,7 @@ class json_sax_dom_callback_parser
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
// Set start/end positions for discarded array.
|
||||
handle_diagnostic_positions_for_json_value(*ref_stack.back());
|
||||
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
@@ -867,72 +880,6 @@ class json_sax_dom_callback_parser
|
||||
|
||||
private:
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
|
||||
{
|
||||
if (m_lexer_ref)
|
||||
{
|
||||
// Lexer has read past the current field value, so set the end position to the current position.
|
||||
// The start position will be set below based on the length of the string representation
|
||||
// of the value.
|
||||
v.end_position = m_lexer_ref->get_position();
|
||||
|
||||
switch (v.type())
|
||||
{
|
||||
case value_t::boolean:
|
||||
{
|
||||
// 4 and 5 are the string length of "true" and "false"
|
||||
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::null:
|
||||
{
|
||||
// 4 is the string length of "null"
|
||||
v.start_position = v.end_position - 4;
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
// escape sequences make the token longer than the value it
|
||||
// parses to, so the start position cannot be derived from
|
||||
// the value; use the offset the lexer recorded instead
|
||||
v.start_position = m_lexer_ref->get_token_start_position();
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::discarded:
|
||||
{
|
||||
v.end_position = std::string::npos;
|
||||
v.start_position = v.end_position;
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::binary:
|
||||
case value_t::number_integer:
|
||||
case value_t::number_unsigned:
|
||||
case value_t::number_float:
|
||||
{
|
||||
v.start_position = v.end_position - m_lexer_ref->get_string().size();
|
||||
break;
|
||||
}
|
||||
|
||||
case value_t::object:
|
||||
case value_t::array:
|
||||
{
|
||||
// object and array are handled in start_object() and start_array() handlers
|
||||
// skip setting the values here.
|
||||
break;
|
||||
}
|
||||
default: // LCOV_EXCL_LINE
|
||||
// Handle all possible types discretely, default handler should never be reached.
|
||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
/// if there is a pending duplicate-key stash entry for this exact slot,
|
||||
/// remove it from the stash; if restore_value is true, the stashed
|
||||
/// previous value is moved back into the slot first (use this when the
|
||||
@@ -1054,7 +1001,7 @@ class json_sax_dom_callback_parser
|
||||
auto value = BasicJsonType(std::forward<Value>(v));
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
handle_diagnostic_positions_for_json_value(value);
|
||||
diagnostic_positions::set_from_lexer(value, m_lexer_ref);
|
||||
#endif
|
||||
|
||||
// check callback
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <initializer_list> // initializer_list
|
||||
#include <string> // char_traits, string
|
||||
@@ -22,6 +23,7 @@
|
||||
#include <nlohmann/detail/input/string_scan.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||
#include <nlohmann/detail/string_utils.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -220,9 +222,9 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
/////////////////////
|
||||
|
||||
/*!
|
||||
@brief get codepoint from 4 hex characters following `\u`
|
||||
@brief get codepoint from 4 hex characters following `\\u`
|
||||
|
||||
For input "\u c1 c2 c3 c4" the codepoint is:
|
||||
For input "\\u c1 c2 c3 c4" the codepoint is:
|
||||
(c1 * 0x1000) + (c2 * 0x0100) + (c3 * 0x0010) + c4
|
||||
= (c1 << 12) + (c2 << 8) + (c3 << 4) + (c4 << 0)
|
||||
|
||||
@@ -486,32 +488,10 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF);
|
||||
|
||||
// translate codepoint into bytes
|
||||
if (codepoint < 0x80)
|
||||
encode_utf8(static_cast<std::uint32_t>(codepoint), [this](std::uint32_t byte)
|
||||
{
|
||||
// 1-byte characters: 0xxxxxxx (ASCII)
|
||||
add(static_cast<char_int_type>(codepoint));
|
||||
}
|
||||
else if (codepoint <= 0x7FF)
|
||||
{
|
||||
// 2-byte characters: 110xxxxx 10xxxxxx
|
||||
add(static_cast<char_int_type>(0xC0u | (static_cast<unsigned int>(codepoint) >> 6u)));
|
||||
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
|
||||
}
|
||||
else if (codepoint <= 0xFFFF)
|
||||
{
|
||||
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
|
||||
add(static_cast<char_int_type>(0xE0u | (static_cast<unsigned int>(codepoint) >> 12u)));
|
||||
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
|
||||
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
|
||||
}
|
||||
else
|
||||
{
|
||||
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
|
||||
add(static_cast<char_int_type>(0xF0u | (static_cast<unsigned int>(codepoint) >> 18u)));
|
||||
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 12u) & 0x3Fu)));
|
||||
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
|
||||
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
|
||||
}
|
||||
add(static_cast<char_int_type>(byte));
|
||||
});
|
||||
|
||||
break;
|
||||
}
|
||||
@@ -961,10 +941,15 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
case '\n':
|
||||
case '\r':
|
||||
case char_traits<char_type>::eof():
|
||||
return true;
|
||||
|
||||
#if !JSON_STRICT_NUL_HANDLING
|
||||
case '\0':
|
||||
#endif
|
||||
// a NUL byte is the end of the input (see scan()),
|
||||
// so leave it for scan() to see
|
||||
unget();
|
||||
return true;
|
||||
#endif
|
||||
|
||||
default:
|
||||
break;
|
||||
@@ -1409,45 +1394,30 @@ scan_number_done:
|
||||
*/
|
||||
token_type convert_number(token_type number_type, std::size_t mantissa_end)
|
||||
{
|
||||
// If the caller does not need the converted value (only whether the
|
||||
// input is syntactically valid; see json_sax_acceptor/accept()), an
|
||||
// unsigned/integer token can be reported without calling
|
||||
// strtoull()/strtoll() at all, *provided* we can already tell from
|
||||
// the digit count alone that the conversion cannot overflow 64 bits.
|
||||
// Such tokens are always finite and are accepted unconditionally by
|
||||
// the parser regardless of their actual value (parser::sax_parse_internal()
|
||||
// never checks finiteness for value_unsigned/value_integer), so the
|
||||
// classification below is all that is needed.
|
||||
// accept() only needs to know whether the input is valid, so it sets
|
||||
// discard_number_values (see json.hpp), and an integer token whose
|
||||
// digit count shows that it fits is reported without calling
|
||||
// convert_integer(). A number with up to 18 digits always fits into
|
||||
// both std::uint64_t and std::int64_t (18 nines is about 1e18, below
|
||||
// INT64_MAX, which is about 9.2e18). Longer tokens take the exact path
|
||||
// below, including the fallback to floating point when the value does
|
||||
// not fit.
|
||||
//
|
||||
// A decimal number with up to 18 digits is always representable in
|
||||
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
|
||||
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
|
||||
// could not have set errno to ERANGE for it. Numbers with more digits
|
||||
// (rare in practice) fall through to the exact code below, unchanged,
|
||||
// so their handling -- including reclassification to value_float when
|
||||
// the value overflows 64 bits, and rejection when it is not even
|
||||
// finite as a double -- is bit-for-bit identical to before this
|
||||
// optimization.
|
||||
// With a narrower number_unsigned_t/number_integer_t (e.g.
|
||||
// std::uint32_t), the exact path would reclassify some of these tokens
|
||||
// as (finite) floats, while this check reports integers. That does not
|
||||
// change the result of accept(): it always parses through
|
||||
// json_sax_acceptor, whose number callbacks discard their argument and
|
||||
// return true, and the parser rejects neither integers nor finite
|
||||
// floats. value_unsigned/value_integer are left unset here, so a caller
|
||||
// that reads the converted value must not set discard_number_values.
|
||||
//
|
||||
// Note this reasons about std::uint64_t/std::int64_t, not about
|
||||
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
|
||||
// narrower, template parameters -- e.g. std::uint32_t). That is fine
|
||||
// *only* because discard_number_values is exclusively set by
|
||||
// accept() (see json.hpp), and accept() always parses through the
|
||||
// library's own json_sax_acceptor -- never a user-supplied SAX
|
||||
// consumer -- whose number_unsigned()/number_integer()/number_float()
|
||||
// callbacks unconditionally discard their argument and return true.
|
||||
// So for every caller that can reach this branch, neither the token
|
||||
// classification below nor the eventual (possibly narrowed, and on
|
||||
// this fast path left stale/unset) value_unsigned/value_integer is
|
||||
// ever consulted -- an unsigned/integer token is accepted outright,
|
||||
// and even a >18-digit token that this fast path deliberately falls
|
||||
// through for is, once reclassified to value_float, still finite
|
||||
// (and thus accepted) for any digit count that fits in number_unsigned_t
|
||||
// or number_integer_t regardless of that type's width. If this
|
||||
// function is ever taught to run with discard_number_values true for
|
||||
// a caller that *does* read the converted value, this reasoning (and
|
||||
// the fast path below) would need to be revisited.
|
||||
// On contiguous input, scan_number_bulk_contiguous() converts integer
|
||||
// tokens itself and does not pass them to this function, unless
|
||||
// JSON_DIAGNOSTIC_POSITIONS is enabled. This check is therefore only
|
||||
// reached for input without bulk access (e.g. streams), with
|
||||
// JSON_DIAGNOSTIC_POSITIONS, or when scan_number_bulk_contiguous()
|
||||
// falls back to scan_number().
|
||||
if (discard_number_values)
|
||||
{
|
||||
constexpr std::size_t safe_digit_count = 18;
|
||||
@@ -1876,7 +1846,7 @@ scan_number_done:
|
||||
return value_float;
|
||||
}
|
||||
|
||||
/// return current string value (implicitly resets the token; useful only once)
|
||||
/// return current string value
|
||||
string_t& get_string()
|
||||
{
|
||||
// a number token holds '.' regardless of the locale (#4084)
|
||||
@@ -2178,11 +2148,11 @@ scan_number_done:
|
||||
/// the position of the decimal point in token_buffer
|
||||
std::size_t decimal_point_position = std::string::npos;
|
||||
|
||||
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
|
||||
/// token classification and never looks at the converted numeric value;
|
||||
/// when set, scan_number() may skip strtoull()/strtoll() for
|
||||
/// value_unsigned/value_integer tokens whose digit count guarantees they
|
||||
/// fit into 64 bits (see scan_number())
|
||||
/// whether the caller only needs the token types and never looks at the
|
||||
/// converted numeric values; set only by accept(), which parses through
|
||||
/// json_sax_acceptor. When set, convert_number() skips converting integer
|
||||
/// tokens whose digit count guarantees that they fit into 64 bits (see
|
||||
/// there)
|
||||
const bool discard_number_values = false;
|
||||
};
|
||||
|
||||
|
||||
@@ -428,14 +428,14 @@ inline bool parse_float_eisel_lemire(const char* first, const char* last, double
|
||||
}
|
||||
|
||||
std::uint64_t w = 0;
|
||||
int digits = 0; // significant digits in w
|
||||
unsigned int digits = 0; // significant digits in w
|
||||
std::int64_t exponent = 0;
|
||||
bool truncated = false;
|
||||
bool in_fraction = false;
|
||||
for (;;)
|
||||
{
|
||||
// eight digits at a time, as long as they fit into w
|
||||
while (w != 0 && digits <= 19 - 8 && last - p >= 8)
|
||||
while (w != 0 && digits <= 19u - 8u && last - p >= 8)
|
||||
{
|
||||
const std::uint64_t v = read_eight_bytes(p);
|
||||
if (!is_eight_digits(v))
|
||||
@@ -443,7 +443,7 @@ inline bool parse_float_eisel_lemire(const char* first, const char* last, double
|
||||
break;
|
||||
}
|
||||
w = (w * 100000000u) + parse_eight_digits(v);
|
||||
digits += 8;
|
||||
digits += 8u;
|
||||
exponent -= in_fraction ? 8 : 0;
|
||||
p += 8;
|
||||
}
|
||||
@@ -459,7 +459,7 @@ inline bool parse_float_eisel_lemire(const char* first, const char* last, double
|
||||
// leading zeros are not significant, but scale a fraction
|
||||
exponent -= in_fraction ? 1 : 0;
|
||||
}
|
||||
else if (digits < 19)
|
||||
else if (digits < 19u)
|
||||
{
|
||||
w = (w * 10u) + static_cast<std::uint64_t>(c - '0');
|
||||
++digits;
|
||||
|
||||
@@ -54,7 +54,9 @@ using parser_callback_t =
|
||||
/*!
|
||||
@brief syntax analysis
|
||||
|
||||
This class implements a recursive descent parser.
|
||||
This class implements a parser for JSON text. Nested arrays and objects are tracked with an explicit
|
||||
stack instead of recursion, so deeply nested input does not exhaust the call stack, and what is read
|
||||
is reported as SAX events.
|
||||
*/
|
||||
template<typename BasicJsonType, typename InputAdapterType>
|
||||
class parser
|
||||
@@ -98,28 +100,9 @@ class parser
|
||||
if (callback)
|
||||
{
|
||||
json_sax_dom_callback_parser<BasicJsonType, InputAdapterType> sdp(result, callback, allow_exceptions, &m_lexer);
|
||||
sax_parse_internal(&sdp);
|
||||
|
||||
if (strict)
|
||||
{
|
||||
// in strict mode, input must be completely read
|
||||
if (get_token() != token_type::end_of_input)
|
||||
{
|
||||
sdp.parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(),
|
||||
exception_message(token_type::end_of_input, "value"), nullptr));
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// the caller keeps using the input: position it right after
|
||||
// the value by leaving the character that terminated it
|
||||
m_lexer.release_lookahead();
|
||||
}
|
||||
|
||||
// in case of an error, return a discarded value
|
||||
if (sdp.is_errored())
|
||||
if (!parse_dom(sdp, strict))
|
||||
{
|
||||
result = value_t::discarded;
|
||||
return;
|
||||
@@ -135,26 +118,9 @@ class parser
|
||||
else
|
||||
{
|
||||
json_sax_dom_parser<BasicJsonType, InputAdapterType> sdp(result, allow_exceptions, &m_lexer);
|
||||
sax_parse_internal(&sdp);
|
||||
|
||||
if (strict)
|
||||
{
|
||||
// in strict mode, input must be completely read
|
||||
if (get_token() != token_type::end_of_input)
|
||||
{
|
||||
sdp.parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// see above
|
||||
m_lexer.release_lookahead();
|
||||
}
|
||||
|
||||
// in case of an error, return a discarded value
|
||||
if (sdp.is_errored())
|
||||
if (!parse_dom(sdp, strict))
|
||||
{
|
||||
result = value_t::discarded;
|
||||
return;
|
||||
@@ -207,6 +173,46 @@ class parser
|
||||
}
|
||||
|
||||
private:
|
||||
/*!
|
||||
@brief run a DOM SAX parser to completion and position the lexer
|
||||
|
||||
Shared by both branches of @ref parse(): builds no SAX parser itself,
|
||||
but drives an already-constructed @a json_sax_dom_parser or
|
||||
@ref json_sax_dom_callback_parser through @ref sax_parse_internal(),
|
||||
then applies the strict-EOF check (reporting parse_error.101 through
|
||||
@a sdp on failure) or, in non-strict mode, releases the lookahead so
|
||||
the caller can keep reading the input right after the parsed value.
|
||||
|
||||
@param[in,out] sdp the DOM SAX parser to run
|
||||
@param[in] strict whether to expect the last token to be EOF
|
||||
@return whether @a sdp did not report an error
|
||||
*/
|
||||
template<typename DomSax>
|
||||
bool parse_dom(DomSax& sdp, const bool strict)
|
||||
{
|
||||
sax_parse_internal(&sdp);
|
||||
|
||||
if (strict)
|
||||
{
|
||||
// in strict mode, input must be completely read
|
||||
if (get_token() != token_type::end_of_input)
|
||||
{
|
||||
sdp.parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(),
|
||||
exception_message(token_type::end_of_input, "value"), nullptr));
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// the caller keeps using the input: position it right after
|
||||
// the value by leaving the character that terminated it
|
||||
m_lexer.release_lookahead();
|
||||
}
|
||||
|
||||
return !sdp.is_errored();
|
||||
}
|
||||
|
||||
template<typename SAX>
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
bool sax_parse_internal(SAX* sax)
|
||||
@@ -439,8 +445,9 @@ class parser
|
||||
|
||||
// We are done with this array. Before we can parse a
|
||||
// new value, we need to evaluate the new state first.
|
||||
// By setting skip_to_state_evaluation to false, we
|
||||
// are effectively jumping to the beginning of this if.
|
||||
// By setting skip_to_state_evaluation to true, the next
|
||||
// iteration skips parsing a value and evaluates the
|
||||
// enclosing state directly.
|
||||
JSON_ASSERT(!states.empty());
|
||||
states.pop_back();
|
||||
skip_to_state_evaluation = true;
|
||||
@@ -500,8 +507,9 @@ class parser
|
||||
|
||||
// We are done with this object. Before we can parse a
|
||||
// new value, we need to evaluate the new state first.
|
||||
// By setting skip_to_state_evaluation to false, we
|
||||
// are effectively jumping to the beginning of this if.
|
||||
// By setting skip_to_state_evaluation to true, the next
|
||||
// iteration skips parsing a value and evaluates the
|
||||
// enclosing state directly.
|
||||
JSON_ASSERT(!states.empty());
|
||||
states.pop_back();
|
||||
skip_to_state_evaluation = true;
|
||||
|
||||
Reference in New Issue
Block a user