Compare commits

..
Author SHA1 Message Date
Niels Lohmann c5777d5950 Fix from_json() for enums with underlying type bool
get_arithmetic_value() rejects boolean_t, so the default from_json()
for enums failed to compile for an enum whose underlying type is
bool (e.g. enum class Flag : bool { off, on }), even though the
matching to_json() serializes such enums as an unsigned number.

Read the underlying value through number_unsigned_t in that case,
matching what to_json() writes, then cast back to the underlying
type before constructing the enum.

Fixes #5671.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 23:14:35 +02:00
10 changed files with 438 additions and 322 deletions
@@ -167,9 +167,13 @@ template<typename BasicJsonType, typename EnumType,
enable_if_t<std::is_enum<EnumType>::value, int> = 0>
inline void from_json(const BasicJsonType& j, EnumType& e)
{
typename std::underlying_type<EnumType>::type val;
using underlying_type = typename std::underlying_type<EnumType>::type;
// get_arithmetic_value() does not accept boolean_t; read the number that to_json() wrote instead
using value_type = typename std::conditional<std::is_same<underlying_type, typename BasicJsonType::boolean_t>::value,
typename BasicJsonType::number_unsigned_t, underlying_type>::type;
value_type val;
get_arithmetic_value(j, val);
e = static_cast<EnumType>(val);
e = static_cast<EnumType>(static_cast<underlying_type>(val));
}
#endif // JSON_DISABLE_ENUM_SERIALIZATION
@@ -3344,8 +3344,8 @@ class binary_reader
return enter_object(detail::unknown_size());
}
// Note, UBJSON has no binary type of its own; BJData, which shares this
// reader, decodes optimized 'B' arrays as binary in get_ubjson_array().
// Note, no reader for UBJSON binary types is implemented because they do
// not exist
bool get_ubjson_high_precision_number()
{
@@ -4052,7 +4052,7 @@ class binary_reader
#endif
}
/*!
/*
@brief read a number from the input
@tparam NumberType the type of the number
@@ -4062,10 +4062,10 @@ class binary_reader
@return whether conversion completed
@note This function needs to respect the system's endianness, because
bytes in CBOR, MessagePack, UBJSON, and BON8 are stored in network
order (big endian) and therefore need reordering on little endian
systems. On the other hand, BSON and BJData use little endian and
should reorder on big endian systems.
bytes in CBOR, MessagePack, and UBJSON are stored in network order
(big endian) and therefore need reordering on little endian systems.
On the other hand, BSON and BJData use little endian and should reorder
on big endian systems.
*/
template<typename NumberType, bool InputIsLittleEndian = false>
bool get_number(const input_format_t format, NumberType& result)
@@ -8,11 +8,12 @@
#pragma once
#include <algorithm> // min
#include <array> // array
#include <cstddef> // size_t
#include <cstring> // strlen
#include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next
#include <memory> // shared_ptr, make_shared, addressof
#include <numeric> // accumulate
#include <streambuf> // streambuf
#include <string> // string, char_traits
#include <type_traits> // enable_if, is_base_of, is_pointer, is_integral, remove_pointer
@@ -81,9 +82,8 @@ class file_input_adapter
};
/*!
Input adapter for a (caching) istream. Does not skip a UTF Byte Order Mark
itself; that is done by the lexer's skip_bom(). Does not support changing
the underlying std::streambuf
Input adapter for a (caching) istream. Ignores a UFT Byte Order Mark at
beginning of input. Does not support changing the underlying std::streambuf
in mid-input. Maintains underlying std::istream and std::streambuf to support
subsequent use of standard std::istream operations to process any input
characters following those used in parsing the JSON input. Clears the
@@ -862,9 +862,9 @@ auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) /
return input_adapter(array, array + N);
}
// This class only handles inputs that construct a contiguous_bytes_input_adapter
// (e.g. span_input_adapter). It's required so that expressions like {ptr, len}
// can be implicitly cast to the correct adapter.
// This class only handles inputs of input_buffer_adapter type.
// It's required so that expressions like {ptr, len} can be implicitly cast
// to the correct adapter.
class span_input_adapter
{
public:
+142 -89
View File
@@ -10,7 +10,6 @@
#include <algorithm> // find_if, min
#include <cstddef>
#include <limits> // numeric_limits
#include <string> // string
#include <type_traits> // enable_if_t
#include <utility> // move, pair
@@ -176,88 +175,6 @@ template<typename ArrayType>
inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/)
{}
#if JSON_DIAGNOSTIC_POSITIONS
/*!
@brief set the diagnostic positions of a value the DOM SAX parsers just stored
Shared by json_sax_dom_parser and json_sax_dom_callback_parser. basic_json
befriends this struct, as the position members are private.
*/
struct diagnostic_positions
{
/*!
@param[in,out] v the value that was just parsed
@param[in] lexer the lexer that read it, or nullptr to leave @a v alone
*/
template<typename BasicJsonType, typename LexerType>
static void set_from_lexer(BasicJsonType& v, LexerType* lexer)
{
if (lexer)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = lexer->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = lexer->get_token_start_position();
break;
}
case value_t::discarded:
{
// an object or array the callback of
// json_sax_dom_callback_parser rejected has no position
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - lexer->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
}
}
};
#endif
/*!
@brief SAX implementation to create a JSON value from SAX events
@@ -459,6 +376,76 @@ class json_sax_dom_parser
private:
#if JSON_DIAGNOSTIC_POSITIONS
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
{
if (m_lexer_ref)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = m_lexer_ref->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = m_lexer_ref->get_token_start_position();
break;
}
// As we handle the start and end positions for values created during parsing,
// we do not expect the following value type to be called. Regardless, set the positions
// in case this is created manually or through a different constructor. Exclude from lcov
// since the exact condition of this switch is esoteric.
// LCOV_EXCL_START
case value_t::discarded:
{
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
// LCOV_EXCL_STOP
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - m_lexer_ref->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
}
}
}
#endif
/*!
@invariant If the ref stack is empty, then the passed value will be the new
root.
@@ -474,7 +461,7 @@ class json_sax_dom_parser
root = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(root, m_lexer_ref);
handle_diagnostic_positions_for_json_value(root);
#endif
return &root;
@@ -487,7 +474,7 @@ class json_sax_dom_parser
ref_stack.back()->m_data.m_value.array->emplace_back(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(ref_stack.back()->m_data.m_value.array->back(), m_lexer_ref);
handle_diagnostic_positions_for_json_value(ref_stack.back()->m_data.m_value.array->back());
#endif
return &(ref_stack.back()->m_data.m_value.array->back());
@@ -498,7 +485,7 @@ class json_sax_dom_parser
*object_element = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(*object_element, m_lexer_ref);
handle_diagnostic_positions_for_json_value(*object_element);
#endif
return object_element;
@@ -675,7 +662,7 @@ class json_sax_dom_callback_parser
#if JSON_DIAGNOSTIC_POSITIONS
// Set start/end positions for discarded object.
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref);
handle_diagnostic_positions_for_json_value(*ref_stack.back());
#endif
}
}
@@ -791,7 +778,7 @@ class json_sax_dom_callback_parser
#if JSON_DIAGNOSTIC_POSITIONS
// Set start/end positions for discarded array.
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref);
handle_diagnostic_positions_for_json_value(*ref_stack.back());
#endif
}
}
@@ -844,6 +831,72 @@ class json_sax_dom_callback_parser
private:
#if JSON_DIAGNOSTIC_POSITIONS
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
{
if (m_lexer_ref)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = m_lexer_ref->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = m_lexer_ref->get_token_start_position();
break;
}
case value_t::discarded:
{
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - m_lexer_ref->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
}
}
}
#endif
/// if there is a pending duplicate-key stash entry for this exact slot,
/// remove it from the stash; if restore_value is true, the stashed
/// previous value is moved back into the slot first (use this when the
@@ -965,7 +1018,7 @@ class json_sax_dom_callback_parser
auto value = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(value, m_lexer_ref);
handle_diagnostic_positions_for_json_value(value);
#endif
// check callback
+43 -28
View File
@@ -1494,30 +1494,45 @@ scan_number_done:
*/
token_type convert_number(token_type number_type, std::size_t mantissa_end)
{
// accept() only needs to know whether the input is valid, so it sets
// discard_number_values (see json.hpp), and an integer token whose
// digit count shows that it fits is reported without calling
// convert_integer(). A number with up to 18 digits always fits into
// both std::uint64_t and std::int64_t (18 nines is about 1e18, below
// INT64_MAX, which is about 9.2e18). Longer tokens take the exact path
// below, including the fallback to floating point when the value does
// not fit.
// If the caller does not need the converted value (only whether the
// input is syntactically valid; see json_sax_acceptor/accept()), an
// unsigned/integer token can be reported without calling
// strtoull()/strtoll() at all, *provided* we can already tell from
// the digit count alone that the conversion cannot overflow 64 bits.
// Such tokens are always finite and are accepted unconditionally by
// the parser regardless of their actual value (parser::sax_parse_internal()
// never checks finiteness for value_unsigned/value_integer), so the
// classification below is all that is needed.
//
// With a narrower number_unsigned_t/number_integer_t (e.g.
// std::uint32_t), the exact path would reclassify some of these tokens
// as (finite) floats, while this check reports integers. That does not
// change the result of accept(): it always parses through
// json_sax_acceptor, whose number callbacks discard their argument and
// return true, and the parser rejects neither integers nor finite
// floats. value_unsigned/value_integer are left unset here, so a caller
// that reads the converted value must not set discard_number_values.
// A decimal number with up to 18 digits is always representable in
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
// could not have set errno to ERANGE for it. Numbers with more digits
// (rare in practice) fall through to the exact code below, unchanged,
// so their handling -- including reclassification to value_float when
// the value overflows 64 bits, and rejection when it is not even
// finite as a double -- is bit-for-bit identical to before this
// optimization.
//
// On contiguous input, scan_number_bulk_contiguous() converts integer
// tokens itself and does not pass them to this function, unless
// JSON_DIAGNOSTIC_POSITIONS is enabled. This check is therefore only
// reached for input without bulk access (e.g. streams), with
// JSON_DIAGNOSTIC_POSITIONS, or when scan_number_bulk_contiguous()
// falls back to scan_number().
// Note this reasons about std::uint64_t/std::int64_t, not about
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
// narrower, template parameters -- e.g. std::uint32_t). That is fine
// *only* because discard_number_values is exclusively set by
// accept() (see json.hpp), and accept() always parses through the
// library's own json_sax_acceptor -- never a user-supplied SAX
// consumer -- whose number_unsigned()/number_integer()/number_float()
// callbacks unconditionally discard their argument and return true.
// So for every caller that can reach this branch, neither the token
// classification below nor the eventual (possibly narrowed, and on
// this fast path left stale/unset) value_unsigned/value_integer is
// ever consulted -- an unsigned/integer token is accepted outright,
// and even a >18-digit token that this fast path deliberately falls
// through for is, once reclassified to value_float, still finite
// (and thus accepted) for any digit count that fits in number_unsigned_t
// or number_integer_t regardless of that type's width. If this
// function is ever taught to run with discard_number_values true for
// a caller that *does* read the converted value, this reasoning (and
// the fast path below) would need to be revisited.
if (discard_number_values)
{
constexpr std::size_t safe_digit_count = 18;
@@ -2008,7 +2023,7 @@ scan_number_done:
return value_float;
}
/// return current string value
/// return current string value (implicitly resets the token; useful only once)
string_t& get_string()
{
// a number token holds '.' regardless of the locale (#4084)
@@ -2310,11 +2325,11 @@ scan_number_done:
/// the position of the decimal point in token_buffer
std::size_t decimal_point_position = std::string::npos;
/// whether the caller only needs the token types and never looks at the
/// converted numeric values; set only by accept(), which parses through
/// json_sax_acceptor. When set, convert_number() skips converting integer
/// tokens whose digit count guarantees that they fit into 64 bits (see
/// there)
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
/// token classification and never looks at the converted numeric value;
/// when set, scan_number() may skip strtoull()/strtoll() for
/// value_unsigned/value_integer tokens whose digit count guarantees they
/// fit into 64 bits (see scan_number())
const bool discard_number_values = false;
};
+5 -8
View File
@@ -54,8 +54,7 @@ using parser_callback_t =
/*!
@brief syntax analysis
This class implements an iterative parser that keeps the open containers on
an explicit stack and reports what it reads as SAX events.
This class implements a recursive descent parser.
*/
template<typename BasicJsonType, typename InputAdapterType>
class parser
@@ -440,9 +439,8 @@ class parser
// We are done with this array. Before we can parse a
// new value, we need to evaluate the new state first.
// By setting skip_to_state_evaluation to true, the next
// iteration skips parsing a value and evaluates the
// enclosing state directly.
// By setting skip_to_state_evaluation to false, we
// are effectively jumping to the beginning of this if.
JSON_ASSERT(!states.empty());
states.pop_back();
skip_to_state_evaluation = true;
@@ -502,9 +500,8 @@ class parser
// We are done with this object. Before we can parse a
// new value, we need to evaluate the new state first.
// By setting skip_to_state_evaluation to true, the next
// iteration skips parsing a value and evaluates the
// enclosing state directly.
// By setting skip_to_state_evaluation to false, we
// are effectively jumping to the beginning of this if.
JSON_ASSERT(!states.empty());
states.pop_back();
skip_to_state_evaluation = true;
+5 -17
View File
@@ -51,23 +51,11 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally
written by Björn Hoehrmann. See
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
The library checks UTF-8 well-formedness (RFC 3629, section 4) in four
places, which differ in speed, diagnostics, and how they read the input:
- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in
strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR,
MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in
text strings at decode time).
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
diagnostic for each kind of error.
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
bulk string scan, the bulk path of the BON8 reader, and the BON8 writer.
They must accept exactly what the lexer's switch accepts.
- the byte path of binary_reader::get_bon8_string(): BON8 input without bulk
access, and the bytes the bulk path leaves to it.
All four must accept the same set of sequences, so a change to one needs a
matching change to the others.
This decoder is the single source of truth for UTF-8 validation in this
library: it is used both by the serializer (to escape and, in strict mode,
reject ill-formed UTF-8 when dumping a string) and by the binary readers
(to reject ill-formed UTF-8 in CBOR/MessagePack/BSON/UBJSON text strings at
decode time; see @ref is_valid_utf8 below).
@param[in,out] state the current decoder state
@param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT)
-3
View File
@@ -149,9 +149,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
friend class ::nlohmann::detail::json_sax_dom_parser;
template<typename BasicJsonType, typename InputAdapterType>
friend class ::nlohmann::detail::json_sax_dom_callback_parser;
#if JSON_DIAGNOSTIC_POSITIONS
friend struct ::nlohmann::detail::diagnostic_positions;
#endif
friend class ::nlohmann::detail::exception;
/// workaround type for MSVC
+215 -161
View File
@@ -5664,9 +5664,13 @@ template<typename BasicJsonType, typename EnumType,
enable_if_t<std::is_enum<EnumType>::value, int> = 0>
inline void from_json(const BasicJsonType& j, EnumType& e)
{
typename std::underlying_type<EnumType>::type val;
using underlying_type = typename std::underlying_type<EnumType>::type;
// get_arithmetic_value() does not accept boolean_t; read the number that to_json() wrote instead
using value_type = typename std::conditional<std::is_same<underlying_type, typename BasicJsonType::boolean_t>::value,
typename BasicJsonType::number_unsigned_t, underlying_type>::type;
value_type val;
get_arithmetic_value(j, val);
e = static_cast<EnumType>(val);
e = static_cast<EnumType>(static_cast<underlying_type>(val));
}
#endif // JSON_DISABLE_ENUM_SERIALIZATION
@@ -6249,23 +6253,11 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally
written by Björn Hoehrmann. See
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
The library checks UTF-8 well-formedness (RFC 3629, section 4) in four
places, which differ in speed, diagnostics, and how they read the input:
- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in
strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR,
MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in
text strings at decode time).
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
diagnostic for each kind of error.
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
bulk string scan, the bulk path of the BON8 reader, and the BON8 writer.
They must accept exactly what the lexer's switch accepts.
- the byte path of binary_reader::get_bon8_string(): BON8 input without bulk
access, and the bytes the bulk path leaves to it.
All four must accept the same set of sequences, so a change to one needs a
matching change to the others.
This decoder is the single source of truth for UTF-8 validation in this
library: it is used both by the serializer (to escape and, in strict mode,
reject ill-formed UTF-8 when dumping a string) and by the binary readers
(to reject ill-formed UTF-8 in CBOR/MessagePack/BSON/UBJSON text strings at
decode time; see @ref is_valid_utf8 below).
@param[in,out] state the current decoder state
@param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT)
@@ -7560,11 +7552,12 @@ NLOHMANN_JSON_NAMESPACE_END
#include <algorithm> // min
#include <array> // array
#include <cstddef> // size_t
#include <cstring> // strlen
#include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next
#include <memory> // shared_ptr, make_shared, addressof
#include <numeric> // accumulate
#include <streambuf> // streambuf
#include <string> // string, char_traits
#include <type_traits> // enable_if, is_base_of, is_pointer, is_integral, remove_pointer
@@ -7637,9 +7630,8 @@ class file_input_adapter
};
/*!
Input adapter for a (caching) istream. Does not skip a UTF Byte Order Mark
itself; that is done by the lexer's skip_bom(). Does not support changing
the underlying std::streambuf
Input adapter for a (caching) istream. Ignores a UFT Byte Order Mark at
beginning of input. Does not support changing the underlying std::streambuf
in mid-input. Maintains underlying std::istream and std::streambuf to support
subsequent use of standard std::istream operations to process any input
characters following those used in parsing the JSON input. Clears the
@@ -8418,9 +8410,9 @@ auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) /
return input_adapter(array, array + N);
}
// This class only handles inputs that construct a contiguous_bytes_input_adapter
// (e.g. span_input_adapter). It's required so that expressions like {ptr, len}
// can be implicitly cast to the correct adapter.
// This class only handles inputs of input_buffer_adapter type.
// It's required so that expressions like {ptr, len} can be implicitly cast
// to the correct adapter.
class span_input_adapter
{
public:
@@ -8465,7 +8457,6 @@ NLOHMANN_JSON_NAMESPACE_END
#include <algorithm> // find_if, min
#include <cstddef>
#include <limits> // numeric_limits
#include <string> // string
#include <type_traits> // enable_if_t
#include <utility> // move, pair
@@ -10599,30 +10590,45 @@ scan_number_done:
*/
token_type convert_number(token_type number_type, std::size_t mantissa_end)
{
// accept() only needs to know whether the input is valid, so it sets
// discard_number_values (see json.hpp), and an integer token whose
// digit count shows that it fits is reported without calling
// convert_integer(). A number with up to 18 digits always fits into
// both std::uint64_t and std::int64_t (18 nines is about 1e18, below
// INT64_MAX, which is about 9.2e18). Longer tokens take the exact path
// below, including the fallback to floating point when the value does
// not fit.
// If the caller does not need the converted value (only whether the
// input is syntactically valid; see json_sax_acceptor/accept()), an
// unsigned/integer token can be reported without calling
// strtoull()/strtoll() at all, *provided* we can already tell from
// the digit count alone that the conversion cannot overflow 64 bits.
// Such tokens are always finite and are accepted unconditionally by
// the parser regardless of their actual value (parser::sax_parse_internal()
// never checks finiteness for value_unsigned/value_integer), so the
// classification below is all that is needed.
//
// With a narrower number_unsigned_t/number_integer_t (e.g.
// std::uint32_t), the exact path would reclassify some of these tokens
// as (finite) floats, while this check reports integers. That does not
// change the result of accept(): it always parses through
// json_sax_acceptor, whose number callbacks discard their argument and
// return true, and the parser rejects neither integers nor finite
// floats. value_unsigned/value_integer are left unset here, so a caller
// that reads the converted value must not set discard_number_values.
// A decimal number with up to 18 digits is always representable in
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
// could not have set errno to ERANGE for it. Numbers with more digits
// (rare in practice) fall through to the exact code below, unchanged,
// so their handling -- including reclassification to value_float when
// the value overflows 64 bits, and rejection when it is not even
// finite as a double -- is bit-for-bit identical to before this
// optimization.
//
// On contiguous input, scan_number_bulk_contiguous() converts integer
// tokens itself and does not pass them to this function, unless
// JSON_DIAGNOSTIC_POSITIONS is enabled. This check is therefore only
// reached for input without bulk access (e.g. streams), with
// JSON_DIAGNOSTIC_POSITIONS, or when scan_number_bulk_contiguous()
// falls back to scan_number().
// Note this reasons about std::uint64_t/std::int64_t, not about
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
// narrower, template parameters -- e.g. std::uint32_t). That is fine
// *only* because discard_number_values is exclusively set by
// accept() (see json.hpp), and accept() always parses through the
// library's own json_sax_acceptor -- never a user-supplied SAX
// consumer -- whose number_unsigned()/number_integer()/number_float()
// callbacks unconditionally discard their argument and return true.
// So for every caller that can reach this branch, neither the token
// classification below nor the eventual (possibly narrowed, and on
// this fast path left stale/unset) value_unsigned/value_integer is
// ever consulted -- an unsigned/integer token is accepted outright,
// and even a >18-digit token that this fast path deliberately falls
// through for is, once reclassified to value_float, still finite
// (and thus accepted) for any digit count that fits in number_unsigned_t
// or number_integer_t regardless of that type's width. If this
// function is ever taught to run with discard_number_values true for
// a caller that *does* read the converted value, this reasoning (and
// the fast path below) would need to be revisited.
if (discard_number_values)
{
constexpr std::size_t safe_digit_count = 18;
@@ -11113,7 +11119,7 @@ scan_number_done:
return value_float;
}
/// return current string value
/// return current string value (implicitly resets the token; useful only once)
string_t& get_string()
{
// a number token holds '.' regardless of the locale (#4084)
@@ -11415,11 +11421,11 @@ scan_number_done:
/// the position of the decimal point in token_buffer
std::size_t decimal_point_position = std::string::npos;
/// whether the caller only needs the token types and never looks at the
/// converted numeric values; set only by accept(), which parses through
/// json_sax_acceptor. When set, convert_number() skips converting integer
/// tokens whose digit count guarantees that they fit into 64 bits (see
/// there)
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
/// token classification and never looks at the converted numeric value;
/// when set, scan_number() may skip strtoull()/strtoll() for
/// value_unsigned/value_integer tokens whose digit count guarantees they
/// fit into 64 bits (see scan_number())
const bool discard_number_values = false;
};
@@ -11587,88 +11593,6 @@ template<typename ArrayType>
inline void reserve_array(ArrayType& /*arr*/, std::size_t /*len*/, priority_tag<0> /*unused*/)
{}
#if JSON_DIAGNOSTIC_POSITIONS
/*!
@brief set the diagnostic positions of a value the DOM SAX parsers just stored
Shared by json_sax_dom_parser and json_sax_dom_callback_parser. basic_json
befriends this struct, as the position members are private.
*/
struct diagnostic_positions
{
/*!
@param[in,out] v the value that was just parsed
@param[in] lexer the lexer that read it, or nullptr to leave @a v alone
*/
template<typename BasicJsonType, typename LexerType>
static void set_from_lexer(BasicJsonType& v, LexerType* lexer)
{
if (lexer)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = lexer->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = lexer->get_token_start_position();
break;
}
case value_t::discarded:
{
// an object or array the callback of
// json_sax_dom_callback_parser rejected has no position
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - lexer->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
}
}
};
#endif
/*!
@brief SAX implementation to create a JSON value from SAX events
@@ -11870,6 +11794,76 @@ class json_sax_dom_parser
private:
#if JSON_DIAGNOSTIC_POSITIONS
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
{
if (m_lexer_ref)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = m_lexer_ref->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = m_lexer_ref->get_token_start_position();
break;
}
// As we handle the start and end positions for values created during parsing,
// we do not expect the following value type to be called. Regardless, set the positions
// in case this is created manually or through a different constructor. Exclude from lcov
// since the exact condition of this switch is esoteric.
// LCOV_EXCL_START
case value_t::discarded:
{
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
// LCOV_EXCL_STOP
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - m_lexer_ref->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
}
}
}
#endif
/*!
@invariant If the ref stack is empty, then the passed value will be the new
root.
@@ -11885,7 +11879,7 @@ class json_sax_dom_parser
root = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(root, m_lexer_ref);
handle_diagnostic_positions_for_json_value(root);
#endif
return &root;
@@ -11898,7 +11892,7 @@ class json_sax_dom_parser
ref_stack.back()->m_data.m_value.array->emplace_back(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(ref_stack.back()->m_data.m_value.array->back(), m_lexer_ref);
handle_diagnostic_positions_for_json_value(ref_stack.back()->m_data.m_value.array->back());
#endif
return &(ref_stack.back()->m_data.m_value.array->back());
@@ -11909,7 +11903,7 @@ class json_sax_dom_parser
*object_element = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(*object_element, m_lexer_ref);
handle_diagnostic_positions_for_json_value(*object_element);
#endif
return object_element;
@@ -12086,7 +12080,7 @@ class json_sax_dom_callback_parser
#if JSON_DIAGNOSTIC_POSITIONS
// Set start/end positions for discarded object.
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref);
handle_diagnostic_positions_for_json_value(*ref_stack.back());
#endif
}
}
@@ -12202,7 +12196,7 @@ class json_sax_dom_callback_parser
#if JSON_DIAGNOSTIC_POSITIONS
// Set start/end positions for discarded array.
diagnostic_positions::set_from_lexer(*ref_stack.back(), m_lexer_ref);
handle_diagnostic_positions_for_json_value(*ref_stack.back());
#endif
}
}
@@ -12255,6 +12249,72 @@ class json_sax_dom_callback_parser
private:
#if JSON_DIAGNOSTIC_POSITIONS
void handle_diagnostic_positions_for_json_value(BasicJsonType& v)
{
if (m_lexer_ref)
{
// Lexer has read past the current field value, so set the end position to the current position.
// The start position will be set below based on the length of the string representation
// of the value.
v.end_position = m_lexer_ref->get_position();
switch (v.type())
{
case value_t::boolean:
{
// 4 and 5 are the string length of "true" and "false"
v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5);
break;
}
case value_t::null:
{
// 4 is the string length of "null"
v.start_position = v.end_position - 4;
break;
}
case value_t::string:
{
// escape sequences make the token longer than the value it
// parses to, so the start position cannot be derived from
// the value; use the offset the lexer recorded instead
v.start_position = m_lexer_ref->get_token_start_position();
break;
}
case value_t::discarded:
{
v.end_position = std::string::npos;
v.start_position = v.end_position;
break;
}
case value_t::binary:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
{
v.start_position = v.end_position - m_lexer_ref->get_string().size();
break;
}
case value_t::object:
case value_t::array:
{
// object and array are handled in start_object() and start_array() handlers
// skip setting the values here.
break;
}
default: // LCOV_EXCL_LINE
// Handle all possible types discretely, default handler should never be reached.
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE
}
}
}
#endif
/// if there is a pending duplicate-key stash entry for this exact slot,
/// remove it from the stash; if restore_value is true, the stashed
/// previous value is moved back into the slot first (use this when the
@@ -12376,7 +12436,7 @@ class json_sax_dom_callback_parser
auto value = BasicJsonType(std::forward<Value>(v));
#if JSON_DIAGNOSTIC_POSITIONS
diagnostic_positions::set_from_lexer(value, m_lexer_ref);
handle_diagnostic_positions_for_json_value(value);
#endif
// check callback
@@ -16055,8 +16115,8 @@ class binary_reader
return enter_object(detail::unknown_size());
}
// Note, UBJSON has no binary type of its own; BJData, which shares this
// reader, decodes optimized 'B' arrays as binary in get_ubjson_array().
// Note, no reader for UBJSON binary types is implemented because they do
// not exist
bool get_ubjson_high_precision_number()
{
@@ -16763,7 +16823,7 @@ class binary_reader
#endif
}
/*!
/*
@brief read a number from the input
@tparam NumberType the type of the number
@@ -16773,10 +16833,10 @@ class binary_reader
@return whether conversion completed
@note This function needs to respect the system's endianness, because
bytes in CBOR, MessagePack, UBJSON, and BON8 are stored in network
order (big endian) and therefore need reordering on little endian
systems. On the other hand, BSON and BJData use little endian and
should reorder on big endian systems.
bytes in CBOR, MessagePack, and UBJSON are stored in network order
(big endian) and therefore need reordering on little endian systems.
On the other hand, BSON and BJData use little endian and should reorder
on big endian systems.
*/
template<typename NumberType, bool InputIsLittleEndian = false>
bool get_number(const input_format_t format, NumberType& result)
@@ -17130,8 +17190,7 @@ using parser_callback_t =
/*!
@brief syntax analysis
This class implements an iterative parser that keeps the open containers on
an explicit stack and reports what it reads as SAX events.
This class implements a recursive descent parser.
*/
template<typename BasicJsonType, typename InputAdapterType>
class parser
@@ -17516,9 +17575,8 @@ class parser
// We are done with this array. Before we can parse a
// new value, we need to evaluate the new state first.
// By setting skip_to_state_evaluation to true, the next
// iteration skips parsing a value and evaluates the
// enclosing state directly.
// By setting skip_to_state_evaluation to false, we
// are effectively jumping to the beginning of this if.
JSON_ASSERT(!states.empty());
states.pop_back();
skip_to_state_evaluation = true;
@@ -17578,9 +17636,8 @@ class parser
// We are done with this object. Before we can parse a
// new value, we need to evaluate the new state first.
// By setting skip_to_state_evaluation to true, the next
// iteration skips parsing a value and evaluates the
// enclosing state directly.
// By setting skip_to_state_evaluation to false, we
// are effectively jumping to the beginning of this if.
JSON_ASSERT(!states.empty());
states.pop_back();
skip_to_state_evaluation = true;
@@ -26177,9 +26234,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
friend class ::nlohmann::detail::json_sax_dom_parser;
template<typename BasicJsonType, typename InputAdapterType>
friend class ::nlohmann::detail::json_sax_dom_callback_parser;
#if JSON_DIAGNOSTIC_POSITIONS
friend struct ::nlohmann::detail::diagnostic_positions;
#endif
friend class ::nlohmann::detail::exception;
/// workaround type for MSVC
+8
View File
@@ -1358,6 +1358,14 @@ TEST_CASE("value conversion")
CHECK(json(value_1).get<c_enum>() == value_1);
CHECK(json(cpp_enum::value_1).get<cpp_enum>() == cpp_enum::value_1);
}
SECTION("get an enum with underlying type bool (#5671)")
{
enum class bool_enum : bool { off, on };
CHECK(json(bool_enum::off).get<bool_enum>() == bool_enum::off);
CHECK(json(bool_enum::on).get<bool_enum>() == bool_enum::on);
}
#endif
SECTION("more involved conversions")