mirror of
https://github.com/nlohmann/json.git
synced 2026-10-05 22:20:30 +00:00
Add an error_handler parameter for UTF-8 to the binary readers and writers
Adds error_handler_t::keep (invalid UTF-8 sequences are left unchanged) as a fourth error_handler_t value, and threads an error_handler parameter through the binary writers and readers: - to_cbor/to_ubjson/to_bjdata/to_bson gain a trailing error_handler parameter (default strict, matching their existing type_error.316 behavior); keep writes a string value or object key's bytes as is instead of throwing, and replace/ignore sanitize it exactly like dump() would, including for the BSON length prefix. to_msgpack and to_bon8 are unchanged. - from_cbor/from_msgpack/from_ubjson/from_bjdata/from_bson gain a trailing error_handler parameter (default keep, i.e. the lenient behavior every binary reader had in 3.12.0 and still has after #5741); strict checks every string value and object key and raises parse_error.113 for ill-formed UTF-8, honoring allow_exceptions; replace/ignore sanitize it like dump() would. from_bon8 is unchanged, since UTF-8 lead bytes are structural there. dump()'s own keep support writes ill-formed bytes as is, even with ensure_ascii, while still \u-escaping well-formed characters around them as usual. The UTF-8 validity check (is_valid_utf8) and the replace/ignore sanitizing logic (sanitize_utf8) now live in string_utils.hpp, shared by the serializer and the binary reader/writer; error_handler_t itself moved to its own header (detail/output/error_handler.hpp) so that string_utils.hpp does not need to depend on serializer.hpp. See #5529 and #5741. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -26,6 +26,7 @@
|
||||
#include <nlohmann/detail/input/binary_reader.hpp>
|
||||
#include <nlohmann/detail/input/string_scan.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/output/error_handler.hpp>
|
||||
#include <nlohmann/detail/output/output_adapters.hpp>
|
||||
#include <nlohmann/detail/string_concat.hpp>
|
||||
#include <nlohmann/detail/string_utils.hpp>
|
||||
@@ -93,8 +94,12 @@ class binary_writer
|
||||
@param[in] sink output sink to write to (a value-type sink such as
|
||||
output_vector_sink, or output_adapter_sink wrapping a
|
||||
type-erased output adapter)
|
||||
@param[in] error_handler_ how to treat a string value or object key that
|
||||
is not valid UTF-8 (CBOR, UBJSON, BJData, and BSON only; never
|
||||
consulted by @ref write_msgpack or @ref write_bon8)
|
||||
*/
|
||||
explicit binary_writer(OutputSinkType sink) : oa(std::move(sink))
|
||||
explicit binary_writer(OutputSinkType sink, const error_handler_t error_handler_ = error_handler_t::strict)
|
||||
: oa(std::move(sink)), error_handler(error_handler_)
|
||||
{}
|
||||
|
||||
/*!
|
||||
@@ -107,10 +112,14 @@ class binary_writer
|
||||
from one.
|
||||
|
||||
@param[in] adapter output adapter to write to
|
||||
@param[in] error_handler_ how to treat a string value or object key that
|
||||
is not valid UTF-8 (CBOR, UBJSON, BJData, and BSON only; never
|
||||
consulted by @ref write_msgpack or @ref write_bon8)
|
||||
*/
|
||||
template < typename SinkType = OutputSinkType,
|
||||
typename std::enable_if < std::is_constructible<SinkType, output_adapter_t<CharType>>::value, int >::type = 0 >
|
||||
explicit binary_writer(output_adapter_t<CharType> adapter) : oa(SinkType(std::move(adapter)))
|
||||
explicit binary_writer(output_adapter_t<CharType> adapter, const error_handler_t error_handler_ = error_handler_t::strict)
|
||||
: oa(SinkType(std::move(adapter))), error_handler(error_handler_)
|
||||
{}
|
||||
|
||||
/*!
|
||||
@@ -215,15 +224,16 @@ class binary_writer
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
check_utf8(*j.m_data.m_value.string, j);
|
||||
string_t storage;
|
||||
const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage);
|
||||
|
||||
// step 1: write control byte and the string length
|
||||
write_cbor_head(0x60, j.m_data.m_value.string->size());
|
||||
write_cbor_head(0x60, value.size());
|
||||
|
||||
// step 2: write the string
|
||||
oa.write_characters(
|
||||
reinterpret_cast<const CharType*>(j.m_data.m_value.string->data()),
|
||||
j.m_data.m_value.string->size());
|
||||
reinterpret_cast<const CharType*>(value.data()),
|
||||
value.size());
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -296,8 +306,14 @@ class binary_writer
|
||||
// el.first is checked here, against the object as
|
||||
// diagnostics context, because write_cbor(el.first)
|
||||
// converts it to a temporary basic_json that would be
|
||||
// used as the context instead
|
||||
check_utf8(el.first, j);
|
||||
// used as the context instead; for error_handler_t::keep
|
||||
// and ::replace/::ignore the recursive write_cbor(el.first)
|
||||
// call below handles the key like any other string, so no
|
||||
// separate check is needed here for those
|
||||
if (error_handler == error_handler_t::strict)
|
||||
{
|
||||
check_utf8(el.first, j);
|
||||
}
|
||||
write_cbor(el.first);
|
||||
write_cbor(el.second);
|
||||
}
|
||||
@@ -691,16 +707,17 @@ class binary_writer
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
check_utf8(*j.m_data.m_value.string, j);
|
||||
string_t storage;
|
||||
const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage);
|
||||
|
||||
if (add_prefix)
|
||||
{
|
||||
oa.write_character(to_char_type('S'));
|
||||
}
|
||||
write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata);
|
||||
write_number_with_ubjson_prefix(value.size(), true, use_bjdata);
|
||||
oa.write_characters(
|
||||
reinterpret_cast<const CharType*>(j.m_data.m_value.string->data()),
|
||||
j.m_data.m_value.string->size());
|
||||
reinterpret_cast<const CharType*>(value.data()),
|
||||
value.size());
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -855,11 +872,12 @@ class binary_writer
|
||||
|
||||
for (const auto& el : *j.m_data.m_value.object)
|
||||
{
|
||||
check_utf8(el.first, j);
|
||||
write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata);
|
||||
string_t storage;
|
||||
const string_t& key = sanitize_utf8_for_write(el.first, j, storage);
|
||||
write_number_with_ubjson_prefix(key.size(), true, use_bjdata);
|
||||
oa.write_characters(
|
||||
reinterpret_cast<const CharType*>(el.first.data()),
|
||||
el.first.size());
|
||||
reinterpret_cast<const CharType*>(key.data()),
|
||||
key.size());
|
||||
write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version);
|
||||
}
|
||||
|
||||
@@ -905,7 +923,7 @@ class binary_writer
|
||||
@throw type_error.316 if @a name is not valid UTF-8, before anything is
|
||||
written
|
||||
*/
|
||||
static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
|
||||
std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
|
||||
{
|
||||
const auto it = name.find(static_cast<typename string_t::value_type>(0));
|
||||
if (JSON_HEDLEY_UNLIKELY(it != BasicJsonType::string_t::npos))
|
||||
@@ -913,9 +931,10 @@ class binary_writer
|
||||
JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j));
|
||||
}
|
||||
|
||||
check_utf8(name, j);
|
||||
string_t storage;
|
||||
const string_t& sanitized = sanitize_utf8_for_write(name, j, storage);
|
||||
|
||||
return /*id*/ 1ul + name.size() + /*zero-terminator*/1u;
|
||||
return /*id*/ 1ul + sanitized.size() + /*zero-terminator*/1u;
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -935,14 +954,28 @@ class binary_writer
|
||||
|
||||
/*!
|
||||
@brief Writes the given @a element_type and @a name to the output adapter
|
||||
|
||||
@a name has already been validated (and, for @ref error_handler_t::strict,
|
||||
found well-formed) by @ref calc_bson_entry_header_size during the earlier
|
||||
size pass, so only @ref error_handler_t::replace / @ref
|
||||
error_handler_t::ignore need to sanitize it again here, to actually write
|
||||
the bytes that size was computed from.
|
||||
*/
|
||||
void write_bson_entry_header(const string_t& name,
|
||||
const std::uint8_t element_type)
|
||||
{
|
||||
oa.write_character(to_char_type(element_type));
|
||||
oa.write_characters(
|
||||
reinterpret_cast<const CharType*>(name.data()),
|
||||
name.size());
|
||||
|
||||
if (error_handler == error_handler_t::keep || error_handler == error_handler_t::strict || is_valid_utf8(name))
|
||||
{
|
||||
oa.write_characters(reinterpret_cast<const CharType*>(name.data()), name.size());
|
||||
}
|
||||
else
|
||||
{
|
||||
const string_t sanitized = sanitize_utf8(name, error_handler);
|
||||
oa.write_characters(reinterpret_cast<const CharType*>(sanitized.data()), sanitized.size());
|
||||
}
|
||||
|
||||
// the terminating null byte is written explicitly rather than taken
|
||||
// from the buffer, so that string_t::data() need not be null-terminated
|
||||
oa.write_character(to_char_type(0x00));
|
||||
@@ -979,27 +1012,41 @@ class binary_writer
|
||||
from reading past a StringType that reports a size larger than what
|
||||
it actually holds.
|
||||
*/
|
||||
static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j)
|
||||
std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j)
|
||||
{
|
||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<std::int32_t>(value.size())))
|
||||
{
|
||||
check_utf8(value, j);
|
||||
string_t storage;
|
||||
const string_t& sanitized = sanitize_utf8_for_write(value, j, storage);
|
||||
return sizeof(std::int32_t) + sanitized.size() + 1ul;
|
||||
}
|
||||
return sizeof(std::int32_t) + value.size() + 1ul;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief Writes a BSON element with key @a name and string value @a value
|
||||
|
||||
@a value has already been validated (and, for @ref error_handler_t::strict,
|
||||
found well-formed) by @ref calc_bson_string_size during the earlier size
|
||||
pass, so only @ref error_handler_t::replace / @ref error_handler_t::ignore
|
||||
need to sanitize it again here, to actually write the bytes that size was
|
||||
computed from.
|
||||
*/
|
||||
void write_bson_string(const string_t& name,
|
||||
const string_t& value)
|
||||
{
|
||||
write_bson_entry_header(name, 0x02);
|
||||
|
||||
write_number<std::int32_t>(to_bson_length(value.size() + 1ul), true);
|
||||
const bool sanitize = error_handler != error_handler_t::keep
|
||||
&& error_handler != error_handler_t::strict
|
||||
&& !is_valid_utf8(value);
|
||||
const string_t sanitized = sanitize ? sanitize_utf8(value, error_handler) : string_t{};
|
||||
const string_t& written = sanitize ? sanitized : value;
|
||||
|
||||
write_number<std::int32_t>(to_bson_length(written.size() + 1ul), true);
|
||||
oa.write_characters(
|
||||
reinterpret_cast<const CharType*>(value.data()),
|
||||
value.size());
|
||||
reinterpret_cast<const CharType*>(written.data()),
|
||||
written.size());
|
||||
// the terminating null byte is written explicitly rather than taken
|
||||
// from the buffer, so that string_t::data() need not be null-terminated
|
||||
oa.write_character(to_char_type(0x00));
|
||||
@@ -1116,7 +1163,7 @@ class binary_writer
|
||||
@throw type_error.316 if @a j is a string that is not valid UTF-8, before
|
||||
anything is written
|
||||
*/
|
||||
static std::size_t calc_bson_value_size(const BasicJsonType& j)
|
||||
std::size_t calc_bson_value_size(const BasicJsonType& j)
|
||||
{
|
||||
switch (j.type())
|
||||
{
|
||||
@@ -1252,7 +1299,7 @@ class binary_writer
|
||||
@throw type_error.316 if a string value or a key is not valid UTF-8,
|
||||
before anything is written
|
||||
*/
|
||||
static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
|
||||
std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
|
||||
{
|
||||
// the object or array whose entries are being sized, and the ones it
|
||||
// is in; nothing is allocated unless the document nests
|
||||
@@ -2170,6 +2217,59 @@ class binary_writer
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief return @a s as it should be written, honoring @ref error_handler
|
||||
|
||||
Used by @ref write_cbor, @ref write_ubjson (and so @ref write_bjdata), and
|
||||
the BSON writing functions for string values and object keys; never by
|
||||
@ref write_msgpack or @ref write_bon8, which do not take an @ref
|
||||
error_handler (MessagePack's spec allows a str object to contain
|
||||
ill-formed UTF-8, and BON8 always validates, since UTF-8 lead bytes are
|
||||
structural there).
|
||||
|
||||
- @ref error_handler_t::keep: @a s is returned unchanged, without even
|
||||
checking it (the behavior of release 3.12.0 and earlier).
|
||||
- @ref error_handler_t::strict: @ref check_utf8 is called, which throws
|
||||
type_error.316 if @a s is not valid UTF-8.
|
||||
- @ref error_handler_t::replace / @ref error_handler_t::ignore: @a s is
|
||||
sanitized into @a storage with exactly the rules @ref
|
||||
serializer::dump_escaped_impl uses, so that parsing what @ref
|
||||
basic_json::dump produces for the same string and the same handler
|
||||
yields the same result.
|
||||
|
||||
Well-formed input is never copied: this returns a reference to @a s
|
||||
itself in every case but a sanitized `replace`/`ignore` one, so @a
|
||||
storage must outlive the returned reference only then.
|
||||
|
||||
@param[in] s the string (value or object key) to write
|
||||
@param[in] context the value @a s belongs to (for diagnostics)
|
||||
@param[out] storage backing storage for a sanitized copy
|
||||
|
||||
@return a reference to @a s, or to @a storage once it holds a sanitized copy
|
||||
*/
|
||||
const string_t& sanitize_utf8_for_write(const string_t& s, const BasicJsonType& context, string_t& storage) const
|
||||
{
|
||||
switch (error_handler)
|
||||
{
|
||||
case error_handler_t::keep:
|
||||
return s;
|
||||
|
||||
case error_handler_t::strict:
|
||||
check_utf8(s, context);
|
||||
return s;
|
||||
|
||||
case error_handler_t::replace:
|
||||
case error_handler_t::ignore:
|
||||
default:
|
||||
if (is_valid_utf8(s))
|
||||
{
|
||||
return s;
|
||||
}
|
||||
storage = sanitize_utf8(s, error_handler);
|
||||
return storage;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief write an integer in the shortest encoding
|
||||
|
||||
@@ -2498,6 +2598,10 @@ class binary_writer
|
||||
|
||||
/// the output
|
||||
OutputSinkType oa;
|
||||
|
||||
/// how to treat a string value or object key that is not valid UTF-8
|
||||
/// (CBOR, UBJSON, BJData, and BSON only)
|
||||
const error_handler_t error_handler = error_handler_t::strict;
|
||||
};
|
||||
|
||||
} // namespace detail
|
||||
|
||||
Reference in New Issue
Block a user