mirror of
https://github.com/nlohmann/json.git
synced 2026-10-03 05:00:30 +00:00
Add an error_handler parameter for UTF-8 to the binary readers and writers
Adds error_handler_t::keep (invalid UTF-8 sequences are left unchanged) as a fourth error_handler_t value, and threads an error_handler parameter through the binary writers and readers: - to_cbor/to_ubjson/to_bjdata/to_bson gain a trailing error_handler parameter (default strict, matching their existing type_error.316 behavior); keep writes a string value or object key's bytes as is instead of throwing, and replace/ignore sanitize it exactly like dump() would, including for the BSON length prefix. to_msgpack and to_bon8 are unchanged. - from_cbor/from_msgpack/from_ubjson/from_bjdata/from_bson gain a trailing error_handler parameter (default keep, i.e. the lenient behavior every binary reader had in 3.12.0 and still has after #5741); strict checks every string value and object key and raises parse_error.113 for ill-formed UTF-8, honoring allow_exceptions; replace/ignore sanitize it like dump() would. from_bon8 is unchanged, since UTF-8 lead bytes are structural there. dump()'s own keep support writes ill-formed bytes as is, even with ensure_ascii, while still \u-escaping well-formed characters around them as usual. The UTF-8 validity check (is_valid_utf8) and the replace/ignore sanitizing logic (sanitize_utf8) now live in string_utils.hpp, shared by the serializer and the binary reader/writer; error_handler_t itself moved to its own header (detail/output/error_handler.hpp) so that string_utils.hpp does not need to depend on serializer.hpp. See #5529 and #5741. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -31,6 +31,7 @@
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/meta/is_sax.hpp>
|
||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||
#include <nlohmann/detail/output/error_handler.hpp>
|
||||
#include <nlohmann/detail/string_concat.hpp>
|
||||
#include <nlohmann/detail/string_utils.hpp>
|
||||
#include <nlohmann/detail/value_t.hpp>
|
||||
@@ -108,8 +109,16 @@ class binary_reader
|
||||
@brief create a binary reader
|
||||
|
||||
@param[in] adapter input adapter to read from
|
||||
@param[in] format the binary format to parse
|
||||
@param[in] error_handler how to treat text strings and object keys that
|
||||
are not well-formed UTF-8; none of the supported formats
|
||||
requires a decoder to reject those, so the default is to
|
||||
@ref error_handler_t::keep them unchanged, as every binary
|
||||
reader did before this parameter existed
|
||||
*/
|
||||
explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json) noexcept : ia(std::move(adapter)), input_format(format)
|
||||
explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json,
|
||||
const error_handler_t error_handler = error_handler_t::keep) noexcept
|
||||
: ia(std::move(adapter)), input_format(format), error_handler(error_handler)
|
||||
{
|
||||
(void)detail::is_sax_static_asserts<SAX, BasicJsonType> {};
|
||||
}
|
||||
@@ -428,7 +437,7 @@ class binary_reader
|
||||
{
|
||||
if (get_bson_cstr_bulk(result, std::integral_constant<bool, bulk_scan> {}))
|
||||
{
|
||||
return true;
|
||||
return check_string_utf8(result, "key");
|
||||
}
|
||||
|
||||
auto out = std::back_inserter(result);
|
||||
@@ -441,7 +450,7 @@ class binary_reader
|
||||
}
|
||||
if (current == 0x00)
|
||||
{
|
||||
return true;
|
||||
return check_string_utf8(result, "key");
|
||||
}
|
||||
*out++ = static_cast<typename string_t::value_type>(current);
|
||||
}
|
||||
@@ -522,7 +531,7 @@ class binary_reader
|
||||
"string"), nullptr));
|
||||
}
|
||||
|
||||
return true;
|
||||
return check_string_utf8(result, "string");
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -1149,7 +1158,7 @@ class binary_reader
|
||||
|
||||
@return whether string creation completed
|
||||
*/
|
||||
bool get_cbor_string(string_t& result)
|
||||
bool get_cbor_string(string_t& result, const char* context = "string")
|
||||
{
|
||||
// number of indefinite-length strings that have been opened and not
|
||||
// closed yet. RFC 8949, Section 3.2.3 does not permit nesting them,
|
||||
@@ -1179,7 +1188,7 @@ class binary_reader
|
||||
{
|
||||
if (--open == 0)
|
||||
{
|
||||
return true;
|
||||
return check_string_utf8(result, context);
|
||||
}
|
||||
get();
|
||||
continue;
|
||||
@@ -1192,7 +1201,7 @@ class binary_reader
|
||||
|
||||
if (open == 0)
|
||||
{
|
||||
return true;
|
||||
return check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
get();
|
||||
@@ -1216,7 +1225,7 @@ class binary_reader
|
||||
// EOF and major type 3 (text string) are left to get_cbor_string
|
||||
if (current == char_traits<char_type>::eof() || (static_cast<unsigned int>(current) & 0xE0u) == 0x60u)
|
||||
{
|
||||
return get_cbor_string(result);
|
||||
return get_cbor_string(result, "key");
|
||||
}
|
||||
|
||||
const char* found = nullptr;
|
||||
@@ -2004,7 +2013,7 @@ class binary_reader
|
||||
|
||||
@return whether string creation completed
|
||||
*/
|
||||
bool get_msgpack_string(string_t& result)
|
||||
bool get_msgpack_string(string_t& result, const char* context = "string")
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::msgpack, "string")))
|
||||
{
|
||||
@@ -2047,25 +2056,25 @@ class binary_reader
|
||||
case 0xBE:
|
||||
case 0xBF:
|
||||
{
|
||||
return get_string(input_format_t::msgpack, static_cast<unsigned int>(current) & 0x1Fu, result);
|
||||
return get_string(input_format_t::msgpack, static_cast<unsigned int>(current) & 0x1Fu, result) && check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
case 0xD9: // str 8
|
||||
{
|
||||
std::uint8_t len{};
|
||||
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result);
|
||||
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
case 0xDA: // str 16
|
||||
{
|
||||
std::uint16_t len{};
|
||||
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result);
|
||||
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
case 0xDB: // str 32
|
||||
{
|
||||
std::uint32_t len{};
|
||||
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result);
|
||||
return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result) && check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
default:
|
||||
@@ -2143,7 +2152,7 @@ class binary_reader
|
||||
// byte 0xC1 are left to get_msgpack_string
|
||||
if (current == char_traits<char_type>::eof())
|
||||
{
|
||||
return get_msgpack_string(result);
|
||||
return get_msgpack_string(result, "key");
|
||||
}
|
||||
if (current <= 0x7F || current >= 0xE0)
|
||||
{
|
||||
@@ -2159,7 +2168,7 @@ class binary_reader
|
||||
}
|
||||
else
|
||||
{
|
||||
return get_msgpack_string(result);
|
||||
return get_msgpack_string(result, "key");
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -2405,7 +2414,7 @@ class binary_reader
|
||||
if (top.is_object)
|
||||
{
|
||||
key.clear();
|
||||
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key)))
|
||||
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, true, "key") || !sax->key(key)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -2427,7 +2436,7 @@ class binary_reader
|
||||
if (top.is_object)
|
||||
{
|
||||
key.clear();
|
||||
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key)))
|
||||
if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false, "key") || !sax->key(key)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -2495,7 +2504,7 @@ class binary_reader
|
||||
|
||||
@return whether string creation completed
|
||||
*/
|
||||
bool get_ubjson_string(string_t& result, const bool get_char = true)
|
||||
bool get_ubjson_string(string_t& result, const bool get_char = true, const char* context = "string")
|
||||
{
|
||||
if (get_char)
|
||||
{
|
||||
@@ -2516,31 +2525,31 @@ class binary_reader
|
||||
case 'U':
|
||||
{
|
||||
std::uint8_t len{};
|
||||
return get_number(input_format, len) && get_string(input_format, len, result);
|
||||
return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
case 'i':
|
||||
{
|
||||
std::int8_t len{};
|
||||
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result);
|
||||
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
case 'I':
|
||||
{
|
||||
std::int16_t len{};
|
||||
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result);
|
||||
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
case 'l':
|
||||
{
|
||||
std::int32_t len{};
|
||||
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result);
|
||||
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
case 'L':
|
||||
{
|
||||
std::int64_t len{};
|
||||
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result);
|
||||
return get_number(input_format, len) && check_ubjson_string_length(len) && get_string(input_format, len, result) && check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
case 'u':
|
||||
@@ -2550,7 +2559,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint16_t len{};
|
||||
return get_number(input_format, len) && get_string(input_format, len, result);
|
||||
return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
case 'm':
|
||||
@@ -2560,7 +2569,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint32_t len{};
|
||||
return get_number(input_format, len) && get_string(input_format, len, result);
|
||||
return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
case 'M':
|
||||
@@ -2570,7 +2579,7 @@ class binary_reader
|
||||
break;
|
||||
}
|
||||
std::uint64_t len{};
|
||||
return get_number(input_format, len) && get_string(input_format, len, result);
|
||||
return get_number(input_format, len) && get_string(input_format, len, result) && check_string_utf8(result, context);
|
||||
}
|
||||
|
||||
default:
|
||||
@@ -4031,15 +4040,53 @@ class binary_reader
|
||||
const NumberType len,
|
||||
string_t& result)
|
||||
{
|
||||
// Strings are taken as is: none of CBOR (RFC 8949 §3.1 leaves the
|
||||
// choice to the decoder), MessagePack (whose spec explicitly allows
|
||||
// a str object to contain an invalid byte sequence), UBJSON, BJData,
|
||||
// or BSON requires a decoder to reject ill-formed UTF-8. The bytes
|
||||
// are kept unchanged; dump() and the binary writers are the ones
|
||||
// that check them and report type_error.316 if they are not valid.
|
||||
// Strings are taken as is by default: none of CBOR (RFC 8949 §3.1
|
||||
// leaves the choice to the decoder), MessagePack (whose spec
|
||||
// explicitly allows a str object to contain an invalid byte
|
||||
// sequence), UBJSON, BJData, or BSON requires a decoder to reject
|
||||
// ill-formed UTF-8. Checking (and, with @ref error_handler_t::strict,
|
||||
// rejecting, or with `replace`/`ignore`, sanitizing) is opt-in via
|
||||
// @ref error_handler, applied once the whole string (all chunks of
|
||||
// an indefinite-length CBOR string included) has been assembled, by
|
||||
// @ref check_string_utf8 at the call site.
|
||||
return get_bytes(format, len, "string", result);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief validate a decoded text string (value or object key) against @ref error_handler
|
||||
|
||||
None of the binary formats requires a decoder to reject ill-formed UTF-8
|
||||
in a text string (see @ref get_string), so by default
|
||||
(@ref error_handler_t::keep) this does nothing. A stricter
|
||||
@ref error_handler opts into the same well-formedness check @ref
|
||||
serializer::dump_escaped_impl applies when dumping a string:
|
||||
@ref error_handler_t::strict rejects ill-formed input with
|
||||
parse_error.113 (honoring `allow_exceptions` via @a sax), while
|
||||
@ref error_handler_t::replace / @ref error_handler_t::ignore sanitize
|
||||
@a result in place, using the exact same rules.
|
||||
|
||||
@param[in,out] result the already assembled string to check
|
||||
@param[in] context further context information (for diagnostics)
|
||||
@return whether @a result is acceptable (always true for `keep`)
|
||||
*/
|
||||
bool check_string_utf8(string_t& result, const char* context)
|
||||
{
|
||||
if (error_handler == error_handler_t::keep || is_valid_utf8(result))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
if (error_handler == error_handler_t::strict)
|
||||
{
|
||||
auto last_token = get_token_string();
|
||||
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
|
||||
exception_message(input_format, "invalid string: ill-formed UTF-8 byte", context), nullptr));
|
||||
}
|
||||
|
||||
result = sanitize_utf8(result, error_handler);
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief create a byte array by reading bytes from the input
|
||||
|
||||
@@ -4211,6 +4258,9 @@ class binary_reader
|
||||
/// input format
|
||||
const input_format_t input_format = input_format_t::json;
|
||||
|
||||
/// how to treat text strings/object keys that are not well-formed UTF-8
|
||||
const error_handler_t error_handler = error_handler_t::keep;
|
||||
|
||||
/// the SAX parser
|
||||
json_sax_t* sax = nullptr;
|
||||
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
#include <nlohmann/detail/input/binary_reader.hpp>
|
||||
#include <nlohmann/detail/input/string_scan.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/output/error_handler.hpp>
|
||||
#include <nlohmann/detail/output/output_adapters.hpp>
|
||||
#include <nlohmann/detail/string_concat.hpp>
|
||||
#include <nlohmann/detail/string_utils.hpp>
|
||||
@@ -93,8 +94,12 @@ class binary_writer
|
||||
@param[in] sink output sink to write to (a value-type sink such as
|
||||
output_vector_sink, or output_adapter_sink wrapping a
|
||||
type-erased output adapter)
|
||||
@param[in] error_handler_ how to treat a string value or object key that
|
||||
is not valid UTF-8 (CBOR, UBJSON, BJData, and BSON only; never
|
||||
consulted by @ref write_msgpack or @ref write_bon8)
|
||||
*/
|
||||
explicit binary_writer(OutputSinkType sink) : oa(std::move(sink))
|
||||
explicit binary_writer(OutputSinkType sink, const error_handler_t error_handler_ = error_handler_t::strict)
|
||||
: oa(std::move(sink)), error_handler(error_handler_)
|
||||
{}
|
||||
|
||||
/*!
|
||||
@@ -107,10 +112,14 @@ class binary_writer
|
||||
from one.
|
||||
|
||||
@param[in] adapter output adapter to write to
|
||||
@param[in] error_handler_ how to treat a string value or object key that
|
||||
is not valid UTF-8 (CBOR, UBJSON, BJData, and BSON only; never
|
||||
consulted by @ref write_msgpack or @ref write_bon8)
|
||||
*/
|
||||
template < typename SinkType = OutputSinkType,
|
||||
typename std::enable_if < std::is_constructible<SinkType, output_adapter_t<CharType>>::value, int >::type = 0 >
|
||||
explicit binary_writer(output_adapter_t<CharType> adapter) : oa(SinkType(std::move(adapter)))
|
||||
explicit binary_writer(output_adapter_t<CharType> adapter, const error_handler_t error_handler_ = error_handler_t::strict)
|
||||
: oa(SinkType(std::move(adapter))), error_handler(error_handler_)
|
||||
{}
|
||||
|
||||
/*!
|
||||
@@ -215,15 +224,16 @@ class binary_writer
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
check_utf8(*j.m_data.m_value.string, j);
|
||||
string_t storage;
|
||||
const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage);
|
||||
|
||||
// step 1: write control byte and the string length
|
||||
write_cbor_head(0x60, j.m_data.m_value.string->size());
|
||||
write_cbor_head(0x60, value.size());
|
||||
|
||||
// step 2: write the string
|
||||
oa.write_characters(
|
||||
reinterpret_cast<const CharType*>(j.m_data.m_value.string->data()),
|
||||
j.m_data.m_value.string->size());
|
||||
reinterpret_cast<const CharType*>(value.data()),
|
||||
value.size());
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -296,8 +306,14 @@ class binary_writer
|
||||
// el.first is checked here, against the object as
|
||||
// diagnostics context, because write_cbor(el.first)
|
||||
// converts it to a temporary basic_json that would be
|
||||
// used as the context instead
|
||||
check_utf8(el.first, j);
|
||||
// used as the context instead; for error_handler_t::keep
|
||||
// and ::replace/::ignore the recursive write_cbor(el.first)
|
||||
// call below handles the key like any other string, so no
|
||||
// separate check is needed here for those
|
||||
if (error_handler == error_handler_t::strict)
|
||||
{
|
||||
check_utf8(el.first, j);
|
||||
}
|
||||
write_cbor(el.first);
|
||||
write_cbor(el.second);
|
||||
}
|
||||
@@ -691,16 +707,17 @@ class binary_writer
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
check_utf8(*j.m_data.m_value.string, j);
|
||||
string_t storage;
|
||||
const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage);
|
||||
|
||||
if (add_prefix)
|
||||
{
|
||||
oa.write_character(to_char_type('S'));
|
||||
}
|
||||
write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata);
|
||||
write_number_with_ubjson_prefix(value.size(), true, use_bjdata);
|
||||
oa.write_characters(
|
||||
reinterpret_cast<const CharType*>(j.m_data.m_value.string->data()),
|
||||
j.m_data.m_value.string->size());
|
||||
reinterpret_cast<const CharType*>(value.data()),
|
||||
value.size());
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -855,11 +872,12 @@ class binary_writer
|
||||
|
||||
for (const auto& el : *j.m_data.m_value.object)
|
||||
{
|
||||
check_utf8(el.first, j);
|
||||
write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata);
|
||||
string_t storage;
|
||||
const string_t& key = sanitize_utf8_for_write(el.first, j, storage);
|
||||
write_number_with_ubjson_prefix(key.size(), true, use_bjdata);
|
||||
oa.write_characters(
|
||||
reinterpret_cast<const CharType*>(el.first.data()),
|
||||
el.first.size());
|
||||
reinterpret_cast<const CharType*>(key.data()),
|
||||
key.size());
|
||||
write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version);
|
||||
}
|
||||
|
||||
@@ -905,7 +923,7 @@ class binary_writer
|
||||
@throw type_error.316 if @a name is not valid UTF-8, before anything is
|
||||
written
|
||||
*/
|
||||
static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
|
||||
std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
|
||||
{
|
||||
const auto it = name.find(static_cast<typename string_t::value_type>(0));
|
||||
if (JSON_HEDLEY_UNLIKELY(it != BasicJsonType::string_t::npos))
|
||||
@@ -913,9 +931,10 @@ class binary_writer
|
||||
JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j));
|
||||
}
|
||||
|
||||
check_utf8(name, j);
|
||||
string_t storage;
|
||||
const string_t& sanitized = sanitize_utf8_for_write(name, j, storage);
|
||||
|
||||
return /*id*/ 1ul + name.size() + /*zero-terminator*/1u;
|
||||
return /*id*/ 1ul + sanitized.size() + /*zero-terminator*/1u;
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -935,14 +954,28 @@ class binary_writer
|
||||
|
||||
/*!
|
||||
@brief Writes the given @a element_type and @a name to the output adapter
|
||||
|
||||
@a name has already been validated (and, for @ref error_handler_t::strict,
|
||||
found well-formed) by @ref calc_bson_entry_header_size during the earlier
|
||||
size pass, so only @ref error_handler_t::replace / @ref
|
||||
error_handler_t::ignore need to sanitize it again here, to actually write
|
||||
the bytes that size was computed from.
|
||||
*/
|
||||
void write_bson_entry_header(const string_t& name,
|
||||
const std::uint8_t element_type)
|
||||
{
|
||||
oa.write_character(to_char_type(element_type));
|
||||
oa.write_characters(
|
||||
reinterpret_cast<const CharType*>(name.data()),
|
||||
name.size());
|
||||
|
||||
if (error_handler == error_handler_t::keep || error_handler == error_handler_t::strict || is_valid_utf8(name))
|
||||
{
|
||||
oa.write_characters(reinterpret_cast<const CharType*>(name.data()), name.size());
|
||||
}
|
||||
else
|
||||
{
|
||||
const string_t sanitized = sanitize_utf8(name, error_handler);
|
||||
oa.write_characters(reinterpret_cast<const CharType*>(sanitized.data()), sanitized.size());
|
||||
}
|
||||
|
||||
// the terminating null byte is written explicitly rather than taken
|
||||
// from the buffer, so that string_t::data() need not be null-terminated
|
||||
oa.write_character(to_char_type(0x00));
|
||||
@@ -979,27 +1012,41 @@ class binary_writer
|
||||
from reading past a StringType that reports a size larger than what
|
||||
it actually holds.
|
||||
*/
|
||||
static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j)
|
||||
std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<std::int32_t>(value.size())))
|
||||
{
|
||||
check_utf8(value, j);
|
||||
string_t storage;
|
||||
const string_t& sanitized = sanitize_utf8_for_write(value, j, storage);
|
||||
return sizeof(std::int32_t) + sanitized.size() + 1ul;
|
||||
}
|
||||
return sizeof(std::int32_t) + value.size() + 1ul;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief Writes a BSON element with key @a name and string value @a value
|
||||
|
||||
@a value has already been validated (and, for @ref error_handler_t::strict,
|
||||
found well-formed) by @ref calc_bson_string_size during the earlier size
|
||||
pass, so only @ref error_handler_t::replace / @ref error_handler_t::ignore
|
||||
need to sanitize it again here, to actually write the bytes that size was
|
||||
computed from.
|
||||
*/
|
||||
void write_bson_string(const string_t& name,
|
||||
const string_t& value)
|
||||
{
|
||||
write_bson_entry_header(name, 0x02);
|
||||
|
||||
write_number<std::int32_t>(to_bson_length(value.size() + 1ul), true);
|
||||
const bool sanitize = error_handler != error_handler_t::keep
|
||||
&& error_handler != error_handler_t::strict
|
||||
&& !is_valid_utf8(value);
|
||||
const string_t sanitized = sanitize ? sanitize_utf8(value, error_handler) : string_t{};
|
||||
const string_t& written = sanitize ? sanitized : value;
|
||||
|
||||
write_number<std::int32_t>(to_bson_length(written.size() + 1ul), true);
|
||||
oa.write_characters(
|
||||
reinterpret_cast<const CharType*>(value.data()),
|
||||
value.size());
|
||||
reinterpret_cast<const CharType*>(written.data()),
|
||||
written.size());
|
||||
// the terminating null byte is written explicitly rather than taken
|
||||
// from the buffer, so that string_t::data() need not be null-terminated
|
||||
oa.write_character(to_char_type(0x00));
|
||||
@@ -1116,7 +1163,7 @@ class binary_writer
|
||||
@throw type_error.316 if @a j is a string that is not valid UTF-8, before
|
||||
anything is written
|
||||
*/
|
||||
static std::size_t calc_bson_value_size(const BasicJsonType& j)
|
||||
std::size_t calc_bson_value_size(const BasicJsonType& j)
|
||||
{
|
||||
switch (j.type())
|
||||
{
|
||||
@@ -1252,7 +1299,7 @@ class binary_writer
|
||||
@throw type_error.316 if a string value or a key is not valid UTF-8,
|
||||
before anything is written
|
||||
*/
|
||||
static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
|
||||
std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
|
||||
{
|
||||
// the object or array whose entries are being sized, and the ones it
|
||||
// is in; nothing is allocated unless the document nests
|
||||
@@ -2170,6 +2217,59 @@ class binary_writer
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief return @a s as it should be written, honoring @ref error_handler
|
||||
|
||||
Used by @ref write_cbor, @ref write_ubjson (and so @ref write_bjdata), and
|
||||
the BSON writing functions for string values and object keys; never by
|
||||
@ref write_msgpack or @ref write_bon8, which do not take an @ref
|
||||
error_handler (MessagePack's spec allows a str object to contain
|
||||
ill-formed UTF-8, and BON8 always validates, since UTF-8 lead bytes are
|
||||
structural there).
|
||||
|
||||
- @ref error_handler_t::keep: @a s is returned unchanged, without even
|
||||
checking it (the behavior of release 3.12.0 and earlier).
|
||||
- @ref error_handler_t::strict: @ref check_utf8 is called, which throws
|
||||
type_error.316 if @a s is not valid UTF-8.
|
||||
- @ref error_handler_t::replace / @ref error_handler_t::ignore: @a s is
|
||||
sanitized into @a storage with exactly the rules @ref
|
||||
serializer::dump_escaped_impl uses, so that parsing what @ref
|
||||
basic_json::dump produces for the same string and the same handler
|
||||
yields the same result.
|
||||
|
||||
Well-formed input is never copied: this returns a reference to @a s
|
||||
itself in every case but a sanitized `replace`/`ignore` one, so @a
|
||||
storage must outlive the returned reference only then.
|
||||
|
||||
@param[in] s the string (value or object key) to write
|
||||
@param[in] context the value @a s belongs to (for diagnostics)
|
||||
@param[out] storage backing storage for a sanitized copy
|
||||
|
||||
@return a reference to @a s, or to @a storage once it holds a sanitized copy
|
||||
*/
|
||||
const string_t& sanitize_utf8_for_write(const string_t& s, const BasicJsonType& context, string_t& storage) const
|
||||
{
|
||||
switch (error_handler)
|
||||
{
|
||||
case error_handler_t::keep:
|
||||
return s;
|
||||
|
||||
case error_handler_t::strict:
|
||||
check_utf8(s, context);
|
||||
return s;
|
||||
|
||||
case error_handler_t::replace:
|
||||
case error_handler_t::ignore:
|
||||
default:
|
||||
if (is_valid_utf8(s))
|
||||
{
|
||||
return s;
|
||||
}
|
||||
storage = sanitize_utf8(s, error_handler);
|
||||
return storage;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief write an integer in the shortest encoding
|
||||
|
||||
@@ -2498,6 +2598,10 @@ class binary_writer
|
||||
|
||||
/// the output
|
||||
OutputSinkType oa;
|
||||
|
||||
/// how to treat a string value or object key that is not valid UTF-8
|
||||
/// (CBOR, UBJSON, BJData, and BSON only)
|
||||
const error_handler_t error_handler = error_handler_t::strict;
|
||||
};
|
||||
|
||||
} // namespace detail
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <nlohmann/detail/abi_macros.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
|
||||
/// how to treat decoding errors
|
||||
///
|
||||
/// @ref basic_json::dump uses this to decide what to do with ill-formed
|
||||
/// UTF-8 while escaping a string, and the binary writers (@ref
|
||||
/// basic_json::to_cbor, @ref basic_json::to_ubjson, @ref
|
||||
/// basic_json::to_bjdata, @ref basic_json::to_bson) use it the same way for
|
||||
/// string values and object keys. The binary readers (@ref
|
||||
/// basic_json::from_cbor, @ref basic_json::from_msgpack, @ref
|
||||
/// basic_json::from_ubjson, @ref basic_json::from_bjdata, @ref
|
||||
/// basic_json::from_bson) use it to decide whether to check text strings
|
||||
/// and object keys for well-formed UTF-8 at all, since none of those
|
||||
/// formats requires a decoder to do so.
|
||||
enum class error_handler_t
|
||||
{
|
||||
strict, ///< throw a type_error/parse_error exception in case of invalid UTF-8
|
||||
replace, ///< replace invalid UTF-8 sequences with U+FFFD
|
||||
ignore, ///< ignore invalid UTF-8 sequences
|
||||
keep ///< keep invalid UTF-8 sequences unchanged
|
||||
};
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -27,6 +27,7 @@
|
||||
#include <nlohmann/detail/input/string_scan.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/meta/cpp_future.hpp>
|
||||
#include <nlohmann/detail/output/error_handler.hpp>
|
||||
#include <nlohmann/detail/output/output_adapters.hpp>
|
||||
#include <nlohmann/detail/recursion_depth_limit.hpp>
|
||||
#include <nlohmann/detail/string_concat.hpp>
|
||||
@@ -41,14 +42,6 @@ namespace detail
|
||||
// serialization //
|
||||
///////////////////
|
||||
|
||||
/// how to treat decoding errors
|
||||
enum class error_handler_t
|
||||
{
|
||||
strict, ///< throw a type_error exception in case of invalid UTF-8
|
||||
replace, ///< replace invalid UTF-8 sequences with U+FFFD
|
||||
ignore ///< ignore invalid UTF-8 sequences
|
||||
};
|
||||
|
||||
template<typename BasicJsonType>
|
||||
class serializer
|
||||
{
|
||||
@@ -839,6 +832,16 @@ class serializer
|
||||
// EnsureAscii parameter is used, non-ASCII characters
|
||||
if ((codepoint <= 0x1F) || (EnsureAscii && (codepoint >= 0x7F)))
|
||||
{
|
||||
if (EnsureAscii && error_handler == error_handler_t::keep)
|
||||
{
|
||||
// this character was buffered as raw bytes
|
||||
// below in case it turned out to be part of
|
||||
// an ill-formed sequence (which is kept as
|
||||
// is); now that it decoded to a well-formed
|
||||
// code point, undo that and \u-escape it
|
||||
// like any other character instead
|
||||
bytes = bytes_after_last_accept;
|
||||
}
|
||||
if (codepoint <= 0xFFFF)
|
||||
{
|
||||
write_u_escape(bytes, static_cast<std::uint16_t>(codepoint));
|
||||
@@ -937,6 +940,44 @@ class serializer
|
||||
break;
|
||||
}
|
||||
|
||||
case error_handler_t::keep:
|
||||
{
|
||||
// the bytes of this (now abandoned) ill-formed
|
||||
// sequence seen so far are already buffered below
|
||||
// and are kept unchanged in the output
|
||||
if (undumped_chars > 0)
|
||||
{
|
||||
// the byte that ended the sequence may be OK
|
||||
// for itself (e.g., a quote that must still be
|
||||
// escaped, or the lead byte of a well-formed
|
||||
// code point), so read it again
|
||||
--i;
|
||||
}
|
||||
else
|
||||
{
|
||||
// a byte that cannot start a sequence (e.g.,
|
||||
// 0xFF or a stray continuation byte) is kept
|
||||
// as well
|
||||
string_buffer[bytes++] = s[i];
|
||||
}
|
||||
|
||||
// write buffer and reset index; there must be 13 bytes
|
||||
// left, as this is the maximal number of bytes to be
|
||||
// written ("\uxxxx\uxxxx\0") for one code point
|
||||
if (string_buffer.size() - bytes < 13)
|
||||
{
|
||||
put_buffer(string_buffer, bytes);
|
||||
bytes = 0;
|
||||
}
|
||||
|
||||
bytes_after_last_accept = bytes;
|
||||
undumped_chars = 0;
|
||||
|
||||
// continue processing the string
|
||||
state = UTF8_ACCEPT;
|
||||
break;
|
||||
}
|
||||
|
||||
default: // LCOV_EXCL_LINE
|
||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
|
||||
}
|
||||
@@ -945,9 +986,12 @@ class serializer
|
||||
|
||||
default: // decode found yet incomplete multibyte code point
|
||||
{
|
||||
if (!EnsureAscii)
|
||||
if (!EnsureAscii || error_handler == error_handler_t::keep)
|
||||
{
|
||||
// code point will not be escaped - copy byte to buffer
|
||||
// code point will not be escaped (or will be kept as
|
||||
// is if it turns out to be ill-formed) - copy byte to
|
||||
// buffer; dropped again above if it decodes to a
|
||||
// well-formed code point that needs \u-escaping
|
||||
string_buffer[bytes++] = s[i];
|
||||
}
|
||||
++undumped_chars;
|
||||
@@ -998,6 +1042,14 @@ class serializer
|
||||
break;
|
||||
}
|
||||
|
||||
case error_handler_t::keep:
|
||||
{
|
||||
// write the ill-formed trailing bytes as is; they were
|
||||
// buffered above regardless of EnsureAscii
|
||||
put_buffer(string_buffer, bytes);
|
||||
break;
|
||||
}
|
||||
|
||||
default: // LCOV_EXCL_LINE
|
||||
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
|
||||
}
|
||||
|
||||
@@ -16,6 +16,7 @@
|
||||
|
||||
#include <nlohmann/detail/abi_macros.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/output/error_handler.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -179,5 +180,134 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std:
|
||||
return state;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check a string for well-formed UTF-8 (RFC 3629, section 4)
|
||||
|
||||
Used by the binary readers (CBOR, MessagePack, UBJSON, BJData, BSON) when an
|
||||
@ref error_handler_t other than `keep` is requested for a text string value
|
||||
or object key: none of those formats requires a decoder to reject ill-formed
|
||||
UTF-8 on its own, so the check is opt-in there, unlike the JSON lexer and the
|
||||
serializer's @ref decode -based escaping, which always run it.
|
||||
|
||||
@param[in] s the string to check
|
||||
@param[in] first the index to start checking at
|
||||
@return whether `s.substr(first)` is well-formed UTF-8
|
||||
|
||||
@sa @ref decode
|
||||
*/
|
||||
template<typename StringType>
|
||||
inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept
|
||||
{
|
||||
std::uint8_t state = UTF8_ACCEPT;
|
||||
std::uint32_t codepoint = 0;
|
||||
|
||||
for (std::size_t i = first; i < s.size(); ++i)
|
||||
{
|
||||
decode(state, codepoint, static_cast<std::uint8_t>(s[i]));
|
||||
if (state == UTF8_REJECT)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return state == UTF8_ACCEPT;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief sanitize a string with ill-formed UTF-8 for @ref error_handler_t::replace or @ref error_handler_t::ignore
|
||||
|
||||
Replaces every maximal ill-formed subsequence with U+FFFD (`replace`) or
|
||||
drops it (`ignore`), using exactly the same boundaries @ref
|
||||
serializer::dump_escaped_impl uses while escaping a string: a byte that does
|
||||
not extend the sequence started by the previous byte(s) is reread as the
|
||||
start of a new one, instead of being swallowed along with them.
|
||||
|
||||
@pre @a error_handler is @ref error_handler_t::replace or @ref error_handler_t::ignore
|
||||
@note Well-formed input is copied through unchanged, including bytes (e.g.
|
||||
control characters or quotes) that @ref serializer::dump_escaped_impl
|
||||
would itself escape; this function only concerns itself with
|
||||
well-formedness, not with producing valid JSON text.
|
||||
|
||||
@param[in] s the string to sanitize
|
||||
@param[in] error_handler @ref error_handler_t::replace or @ref error_handler_t::ignore
|
||||
|
||||
@return @a s with every ill-formed subsequence replaced or removed
|
||||
|
||||
@sa @ref decode
|
||||
*/
|
||||
template<typename StringType>
|
||||
inline StringType sanitize_utf8(const StringType& s, const error_handler_t error_handler)
|
||||
{
|
||||
JSON_ASSERT(error_handler == error_handler_t::replace || error_handler == error_handler_t::ignore);
|
||||
|
||||
StringType result;
|
||||
result.reserve(s.size());
|
||||
|
||||
std::uint32_t codepoint = 0;
|
||||
std::uint8_t state = UTF8_ACCEPT;
|
||||
// length of result after the last accepted code point
|
||||
std::size_t result_len_after_last_accept = 0;
|
||||
// whether bytes of an as yet unresolved sequence were already appended
|
||||
bool pending = false;
|
||||
|
||||
for (std::size_t i = 0; i < s.size(); ++i)
|
||||
{
|
||||
switch (decode(state, codepoint, static_cast<std::uint8_t>(s[i])))
|
||||
{
|
||||
case UTF8_ACCEPT: // decode found a well-formed code point
|
||||
{
|
||||
result.push_back(s[i]);
|
||||
result_len_after_last_accept = result.size();
|
||||
pending = false;
|
||||
break;
|
||||
}
|
||||
|
||||
case UTF8_REJECT: // decode found an ill-formed byte
|
||||
{
|
||||
// in case we saw this byte for the first time, read it again,
|
||||
// because it may be fine for itself, just not for the
|
||||
// sequence that came before it
|
||||
if (pending)
|
||||
{
|
||||
--i;
|
||||
}
|
||||
|
||||
// drop the bytes of the ill-formed sequence buffered below
|
||||
result.resize(result_len_after_last_accept);
|
||||
|
||||
if (error_handler == error_handler_t::replace)
|
||||
{
|
||||
result.append("\xEF\xBF\xBD");
|
||||
result_len_after_last_accept = result.size();
|
||||
}
|
||||
|
||||
pending = false;
|
||||
state = UTF8_ACCEPT;
|
||||
break;
|
||||
}
|
||||
|
||||
default: // decode found yet incomplete multibyte code point
|
||||
{
|
||||
result.push_back(s[i]);
|
||||
pending = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// the string ended with an incomplete sequence
|
||||
if (state != UTF8_ACCEPT)
|
||||
{
|
||||
result.resize(result_len_after_last_accept);
|
||||
|
||||
if (error_handler == error_handler_t::replace)
|
||||
{
|
||||
result.append("\xEF\xBF\xBD");
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
+69
-46
@@ -201,9 +201,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
// used by the vector-returning to_* overloads
|
||||
template<typename CharType> using vector_binary_writer =
|
||||
::nlohmann::detail::binary_writer<basic_json, CharType, ::nlohmann::detail::output_vector_sink<CharType>>;
|
||||
template<typename CharType> static vector_binary_writer<CharType> vector_writer(std::vector<CharType>& v)
|
||||
template<typename CharType> static vector_binary_writer<CharType> vector_writer(
|
||||
std::vector<CharType>& v, const ::nlohmann::detail::error_handler_t error_handler = ::nlohmann::detail::error_handler_t::strict)
|
||||
{
|
||||
return vector_binary_writer<CharType>(::nlohmann::detail::output_vector_sink<CharType>(v));
|
||||
return vector_binary_writer<CharType>(::nlohmann::detail::output_vector_sink<CharType>(v), error_handler);
|
||||
}
|
||||
|
||||
JSON_PRIVATE_UNLESS_TESTED:
|
||||
@@ -5445,26 +5446,29 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
public:
|
||||
/// @brief create a CBOR serialization of a given JSON value
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/to_cbor/
|
||||
static std::vector<std::uint8_t> to_cbor(const basic_json& j)
|
||||
static std::vector<std::uint8_t> to_cbor(const basic_json& j,
|
||||
const error_handler_t error_handler = error_handler_t::strict)
|
||||
{
|
||||
std::vector<std::uint8_t> result;
|
||||
result.reserve(detail::binary_reserve_hint(j));
|
||||
vector_writer(result).write_cbor(j);
|
||||
vector_writer(result, error_handler).write_cbor(j);
|
||||
return result;
|
||||
}
|
||||
|
||||
/// @brief create a CBOR serialization of a given JSON value
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/to_cbor/
|
||||
static void to_cbor(const basic_json& j, detail::output_adapter<std::uint8_t> o)
|
||||
static void to_cbor(const basic_json& j, detail::output_adapter<std::uint8_t> o,
|
||||
const error_handler_t error_handler = error_handler_t::strict)
|
||||
{
|
||||
binary_writer<std::uint8_t>(o).write_cbor(j);
|
||||
binary_writer<std::uint8_t>(o, error_handler).write_cbor(j);
|
||||
}
|
||||
|
||||
/// @brief create a CBOR serialization of a given JSON value
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/to_cbor/
|
||||
static void to_cbor(const basic_json& j, detail::output_adapter<char> o)
|
||||
static void to_cbor(const basic_json& j, detail::output_adapter<char> o,
|
||||
const error_handler_t error_handler = error_handler_t::strict)
|
||||
{
|
||||
binary_writer<char>(o).write_cbor(j);
|
||||
binary_writer<char>(o, error_handler).write_cbor(j);
|
||||
}
|
||||
|
||||
/// @brief create a MessagePack serialization of a given JSON value
|
||||
@@ -5495,28 +5499,31 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/
|
||||
static std::vector<std::uint8_t> to_ubjson(const basic_json& j,
|
||||
const bool use_size = false,
|
||||
const bool use_type = false)
|
||||
const bool use_type = false,
|
||||
const error_handler_t error_handler = error_handler_t::strict)
|
||||
{
|
||||
std::vector<std::uint8_t> result;
|
||||
result.reserve(detail::binary_reserve_hint(j));
|
||||
vector_writer(result).write_ubjson(j, use_size, use_type);
|
||||
vector_writer(result, error_handler).write_ubjson(j, use_size, use_type);
|
||||
return result;
|
||||
}
|
||||
|
||||
/// @brief create a UBJSON serialization of a given JSON value
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/
|
||||
static void to_ubjson(const basic_json& j, detail::output_adapter<std::uint8_t> o,
|
||||
const bool use_size = false, const bool use_type = false)
|
||||
const bool use_size = false, const bool use_type = false,
|
||||
const error_handler_t error_handler = error_handler_t::strict)
|
||||
{
|
||||
binary_writer<std::uint8_t>(o).write_ubjson(j, use_size, use_type);
|
||||
binary_writer<std::uint8_t>(o, error_handler).write_ubjson(j, use_size, use_type);
|
||||
}
|
||||
|
||||
/// @brief create a UBJSON serialization of a given JSON value
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/to_ubjson/
|
||||
static void to_ubjson(const basic_json& j, detail::output_adapter<char> o,
|
||||
const bool use_size = false, const bool use_type = false)
|
||||
const bool use_size = false, const bool use_type = false,
|
||||
const error_handler_t error_handler = error_handler_t::strict)
|
||||
{
|
||||
binary_writer<char>(o).write_ubjson(j, use_size, use_type);
|
||||
binary_writer<char>(o, error_handler).write_ubjson(j, use_size, use_type);
|
||||
}
|
||||
|
||||
/// @brief create a BJData serialization of a given JSON value
|
||||
@@ -5524,11 +5531,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
static std::vector<std::uint8_t> to_bjdata(const basic_json& j,
|
||||
const bool use_size = false,
|
||||
const bool use_type = false,
|
||||
const bjdata_version_t version = bjdata_version_t::draft2)
|
||||
const bjdata_version_t version = bjdata_version_t::draft2,
|
||||
const error_handler_t error_handler = error_handler_t::strict)
|
||||
{
|
||||
std::vector<std::uint8_t> result;
|
||||
result.reserve(detail::binary_reserve_hint(j));
|
||||
vector_writer(result).write_ubjson(j, use_size, use_type, true, true, version);
|
||||
vector_writer(result, error_handler).write_ubjson(j, use_size, use_type, true, true, version);
|
||||
return result;
|
||||
}
|
||||
|
||||
@@ -5536,42 +5544,47 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/
|
||||
static void to_bjdata(const basic_json& j, detail::output_adapter<std::uint8_t> o,
|
||||
const bool use_size = false, const bool use_type = false,
|
||||
const bjdata_version_t version = bjdata_version_t::draft2)
|
||||
const bjdata_version_t version = bjdata_version_t::draft2,
|
||||
const error_handler_t error_handler = error_handler_t::strict)
|
||||
{
|
||||
binary_writer<std::uint8_t>(o).write_ubjson(j, use_size, use_type, true, true, version);
|
||||
binary_writer<std::uint8_t>(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version);
|
||||
}
|
||||
|
||||
/// @brief create a BJData serialization of a given JSON value
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/to_bjdata/
|
||||
static void to_bjdata(const basic_json& j, detail::output_adapter<char> o,
|
||||
const bool use_size = false, const bool use_type = false,
|
||||
const bjdata_version_t version = bjdata_version_t::draft2)
|
||||
const bjdata_version_t version = bjdata_version_t::draft2,
|
||||
const error_handler_t error_handler = error_handler_t::strict)
|
||||
{
|
||||
binary_writer<char>(o).write_ubjson(j, use_size, use_type, true, true, version);
|
||||
binary_writer<char>(o, error_handler).write_ubjson(j, use_size, use_type, true, true, version);
|
||||
}
|
||||
|
||||
/// @brief create a BSON serialization of a given JSON value
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/to_bson/
|
||||
static std::vector<std::uint8_t> to_bson(const basic_json& j)
|
||||
static std::vector<std::uint8_t> to_bson(const basic_json& j,
|
||||
const error_handler_t error_handler = error_handler_t::strict)
|
||||
{
|
||||
std::vector<std::uint8_t> result;
|
||||
result.reserve(detail::binary_reserve_hint(j));
|
||||
vector_writer(result).write_bson(j);
|
||||
vector_writer(result, error_handler).write_bson(j);
|
||||
return result;
|
||||
}
|
||||
|
||||
/// @brief create a BSON serialization of a given JSON value
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/to_bson/
|
||||
static void to_bson(const basic_json& j, detail::output_adapter<std::uint8_t> o)
|
||||
static void to_bson(const basic_json& j, detail::output_adapter<std::uint8_t> o,
|
||||
const error_handler_t error_handler = error_handler_t::strict)
|
||||
{
|
||||
binary_writer<std::uint8_t>(o).write_bson(j);
|
||||
binary_writer<std::uint8_t>(o, error_handler).write_bson(j);
|
||||
}
|
||||
|
||||
/// @brief create a BSON serialization of a given JSON value
|
||||
/// @sa https://json.nlohmann.me/api/basic_json/to_bson/
|
||||
static void to_bson(const basic_json& j, detail::output_adapter<char> o)
|
||||
static void to_bson(const basic_json& j, detail::output_adapter<char> o,
|
||||
const error_handler_t error_handler = error_handler_t::strict)
|
||||
{
|
||||
binary_writer<char>(o).write_bson(j);
|
||||
binary_writer<char>(o, error_handler).write_bson(j);
|
||||
}
|
||||
|
||||
/// @brief create a BON8 serialization of a given JSON value
|
||||
@@ -5605,12 +5618,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
static basic_json from_cbor(InputType&& i,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true,
|
||||
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error)
|
||||
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error,
|
||||
const error_handler_t error_handler = error_handler_t::keep)
|
||||
{
|
||||
basic_json result;
|
||||
auto ia = detail::input_adapter(std::forward<InputType>(i));
|
||||
detail::json_sax_dom_parser<basic_json, decltype(ia)> sdp(result, allow_exceptions);
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved]
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::cbor, error_handler).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved]
|
||||
{
|
||||
result = value_t::discarded;
|
||||
}
|
||||
@@ -5625,12 +5639,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
static basic_json from_cbor(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true,
|
||||
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error)
|
||||
const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error,
|
||||
const error_handler_t error_handler = error_handler_t::keep)
|
||||
{
|
||||
basic_json result;
|
||||
auto ia = detail::input_adapter(std::move(first), std::move(last));
|
||||
detail::json_sax_dom_parser<basic_json, decltype(ia)> sdp(result, allow_exceptions);
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved]
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::cbor, error_handler).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved]
|
||||
{
|
||||
result = value_t::discarded;
|
||||
}
|
||||
@@ -5672,12 +5687,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_msgpack(InputType&& i,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true)
|
||||
const bool allow_exceptions = true,
|
||||
const error_handler_t error_handler = error_handler_t::keep)
|
||||
{
|
||||
basic_json result;
|
||||
auto ia = detail::input_adapter(std::forward<InputType>(i));
|
||||
detail::json_sax_dom_parser<basic_json, decltype(ia)> sdp(result, allow_exceptions);
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::msgpack, error_handler).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
{
|
||||
result = value_t::discarded;
|
||||
}
|
||||
@@ -5691,12 +5707,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_msgpack(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true)
|
||||
const bool allow_exceptions = true,
|
||||
const error_handler_t error_handler = error_handler_t::keep)
|
||||
{
|
||||
basic_json result;
|
||||
auto ia = detail::input_adapter(std::move(first), std::move(last));
|
||||
detail::json_sax_dom_parser<basic_json, decltype(ia)> sdp(result, allow_exceptions);
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::msgpack, error_handler).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
{
|
||||
result = value_t::discarded;
|
||||
}
|
||||
@@ -5736,12 +5753,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_ubjson(InputType&& i,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true)
|
||||
const bool allow_exceptions = true,
|
||||
const error_handler_t error_handler = error_handler_t::keep)
|
||||
{
|
||||
basic_json result;
|
||||
auto ia = detail::input_adapter(std::forward<InputType>(i));
|
||||
detail::json_sax_dom_parser<basic_json, decltype(ia)> sdp(result, allow_exceptions);
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::ubjson, error_handler).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
{
|
||||
result = value_t::discarded;
|
||||
}
|
||||
@@ -5755,12 +5773,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_ubjson(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true)
|
||||
const bool allow_exceptions = true,
|
||||
const error_handler_t error_handler = error_handler_t::keep)
|
||||
{
|
||||
basic_json result;
|
||||
auto ia = detail::input_adapter(std::move(first), std::move(last));
|
||||
detail::json_sax_dom_parser<basic_json, decltype(ia)> sdp(result, allow_exceptions);
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::ubjson, error_handler).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
{
|
||||
result = value_t::discarded;
|
||||
}
|
||||
@@ -5800,12 +5819,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_bjdata(InputType&& i,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true)
|
||||
const bool allow_exceptions = true,
|
||||
const error_handler_t error_handler = error_handler_t::keep)
|
||||
{
|
||||
basic_json result;
|
||||
auto ia = detail::input_adapter(std::forward<InputType>(i));
|
||||
detail::json_sax_dom_parser<basic_json, decltype(ia)> sdp(result, allow_exceptions);
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::bjdata, error_handler).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
{
|
||||
result = value_t::discarded;
|
||||
}
|
||||
@@ -5819,12 +5839,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_bjdata(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true)
|
||||
const bool allow_exceptions = true,
|
||||
const error_handler_t error_handler = error_handler_t::keep)
|
||||
{
|
||||
basic_json result;
|
||||
auto ia = detail::input_adapter(std::move(first), std::move(last));
|
||||
detail::json_sax_dom_parser<basic_json, decltype(ia)> sdp(result, allow_exceptions);
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::bjdata, error_handler).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
{
|
||||
result = value_t::discarded;
|
||||
}
|
||||
@@ -5874,12 +5895,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_bson(InputType&& i,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true)
|
||||
const bool allow_exceptions = true,
|
||||
const error_handler_t error_handler = error_handler_t::keep)
|
||||
{
|
||||
basic_json result;
|
||||
auto ia = detail::input_adapter(std::forward<InputType>(i));
|
||||
detail::json_sax_dom_parser<basic_json, decltype(ia)> sdp(result, allow_exceptions);
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::bson, error_handler).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
{
|
||||
result = value_t::discarded;
|
||||
}
|
||||
@@ -5893,12 +5915,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
JSON_HEDLEY_WARN_UNUSED_RESULT
|
||||
static basic_json from_bson(IteratorType first, SentinelType last,
|
||||
const bool strict = true,
|
||||
const bool allow_exceptions = true)
|
||||
const bool allow_exceptions = true,
|
||||
const error_handler_t error_handler = error_handler_t::keep)
|
||||
{
|
||||
basic_json result;
|
||||
auto ia = detail::input_adapter(std::move(first), std::move(last));
|
||||
detail::json_sax_dom_parser<basic_json, decltype(ia)> sdp(result, allow_exceptions);
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
if (!binary_reader<decltype(ia)>(std::move(ia), input_format_t::bson, error_handler).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved]
|
||||
{
|
||||
result = value_t::discarded;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user