diff --git a/CMakeLists.txt b/CMakeLists.txt index f0a6771dc..b5092ae49 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -61,6 +61,7 @@ option(JSON_Install "Install CMake targets during install option(JSON_MultipleHeaders "Use non-amalgamated version of the library." ON) option(JSON_SystemInclude "Include as system headers (skip for clang-tidy)." OFF) option(JSON_StrictNulHandling "Build with strict NUL-byte handling enabled." OFF) +option(JSON_StrictBinaryUTF8 "Build with UTF-8 checks in the CBOR, UBJSON, BJData, and BSON writers enabled." OFF) if (JSON_CI) include(ci) @@ -118,6 +119,10 @@ if (JSON_StrictNulHandling) message(STATUS "Strict NUL-byte handling enabled (JSON_STRICT_NUL_HANDLING=1)") endif() +if (JSON_StrictBinaryUTF8) + message(STATUS "Strict UTF-8 checks in binary writers enabled (JSON_STRICT_BINARY_UTF8=1)") +endif() + if (JSON_Diagnostic_Positions) message(STATUS "Diagnostic positions enabled (JSON_DIAGNOSTIC_POSITIONS=1)") endif() @@ -153,6 +158,7 @@ target_compile_definitions( $<$:JSON_DIAGNOSTIC_POSITIONS=1> $<$:JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1> $<$:JSON_STRICT_NUL_HANDLING=1> + $<$:JSON_STRICT_BINARY_UTF8=1> ) target_include_directories( diff --git a/cmake/ci.cmake b/cmake/ci.cmake index 353975b8e..59a1b1d4e 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -691,7 +691,7 @@ ci_get_cmake(4.0.0 CMAKE_4_0_0_BINARY) # the tests require CMake 3.13 or later, so they are excluded for CMake 3.5.0 set(JSON_CMAKE_FLAGS_3_5_0 JSON_Diagnostics JSON_Diagnostic_Positions JSON_GlobalUDLs JSON_ImplicitConversions JSON_DisableEnumSerialization JSON_LegacyDiscardedValueComparison JSON_Install JSON_MultipleHeaders JSON_SystemInclude JSON_Valgrind - JSON_StrictNulHandling) + JSON_StrictNulHandling JSON_StrictBinaryUTF8) set(JSON_CMAKE_FLAGS_3_31_6 JSON_BuildTests ${JSON_CMAKE_FLAGS_3_5_0}) set(JSON_CMAKE_FLAGS_4_0_0 JSON_BuildTests ${JSON_CMAKE_FLAGS_3_5_0}) diff --git a/docs/mkdocs/docs/api/basic_json/to_bjdata.md b/docs/mkdocs/docs/api/basic_json/to_bjdata.md index 1473e7b70..139abed53 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/to_bjdata.md @@ -56,8 +56,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false. -- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is - not valid UTF-8 +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not + valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are + written unchanged ## Complexity @@ -92,4 +93,5 @@ Linear in the size of the JSON value `j`. - Added in version 3.11.0. - BJData version parameter (for draft3 binary encoding) added in version 3.12.0. -- Throwing `type_error.316` for a string or object key that is not valid UTF-8 added in version 3.13.0. \ No newline at end of file +- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. \ No newline at end of file diff --git a/docs/mkdocs/docs/api/basic_json/to_bson.md b/docs/mkdocs/docs/api/basic_json/to_bson.md index 16b19119a..5426b2b07 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bson.md +++ b/docs/mkdocs/docs/api/basic_json/to_bson.md @@ -46,8 +46,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value exceeds 255, the maximum of the BSON binary subtype; example: `"subtype 70000 is too large for the BSON binary subtype (max 255)"` -- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key is - not valid UTF-8 +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key is not valid + UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are + written unchanged ## Complexity @@ -84,5 +85,6 @@ pass before anything is written. - Added in version 3.4.0. - Linear in the size of `j`, and no longer limited by the call stack for deeply nested values, since version 3.13.0. - `out_of_range.415` is now detected before anything is written, like the other exceptions above, since version 3.13.0. -- Throwing `type_error.316` for a string value or object key that is not valid UTF-8, detected before anything is - written, added in version 3.13.0. +- Throwing `type_error.316` for a string value or object key that is not valid UTF-8 if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled, detected before anything is written, + added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/to_cbor.md b/docs/mkdocs/docs/api/basic_json/to_cbor.md index 10449a0c1..cac8a0917 100644 --- a/docs/mkdocs/docs/api/basic_json/to_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/to_cbor.md @@ -37,8 +37,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va ## Exceptions -- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is - not valid UTF-8 +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not + valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are + written unchanged ## Complexity @@ -73,4 +74,5 @@ Linear in the size of the JSON value `j`. - Added in version 2.0.9. - Compact representation of floating-point numbers added in version 3.8.0. -- Throwing `type_error.316` for a string or object key that is not valid UTF-8 added in version 3.13.0. +- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/to_ubjson.md b/docs/mkdocs/docs/api/basic_json/to_ubjson.md index 4fc87dcf9..1d48528ee 100644 --- a/docs/mkdocs/docs/api/basic_json/to_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/to_ubjson.md @@ -49,8 +49,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va - Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size` is false. -- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is - not valid UTF-8 +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is not + valid UTF-8 and [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled; otherwise, the bytes are + written unchanged ## Complexity @@ -84,4 +85,5 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 3.1.0. -- Throwing `type_error.316` for a string or object key that is not valid UTF-8 added in version 3.13.0. +- Throwing `type_error.316` for a string or object key that is not valid UTF-8 if + [`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index e917e0ae2..d88500efa 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -18,6 +18,8 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_PRECISE_STREAM_POSITION**](json_precise_stream_position.md) - opt in to leaving an input stream positioned right after a parsed number +- [**JSON_STRICT_BINARY_UTF8**](json_strict_binary_utf8.md) - opt in to checking strings for valid UTF-8 in the CBOR, + UBJSON, BJData, and BSON writers - [**JSON_STRICT_NUL_HANDLING**](json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of treating it as end of input diff --git a/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md new file mode 100644 index 000000000..7fe6d9a58 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_strict_binary_utf8.md @@ -0,0 +1,97 @@ +# JSON_STRICT_BINARY_UTF8 + +```cpp +#define JSON_STRICT_BINARY_UTF8 /* value */ +``` + +When defined to `1`, the binary writers [`to_cbor`](../basic_json/to_cbor.md), [`to_ubjson`](../basic_json/to_ubjson.md), +[`to_bjdata`](../basic_json/to_bjdata.md), and [`to_bson`](../basic_json/to_bson.md) check every string value and +object key for valid UTF-8 and throw [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for +ill-formed UTF-8, like [`dump`](../basic_json/dump.md) does. Without it, they write the bytes unchanged. + +The macro does not affect: + +- [`to_msgpack`](../basic_json/to_msgpack.md): the MessagePack specification allows a `str` value to contain bytes that + are not valid UTF-8, so it always writes them unchanged. +- [`to_bon8`](../basic_json/to_bon8.md): BON8 always checks, because the UTF-8 lead bytes mark where a string ends. +- The binary readers ([`from_cbor`](../basic_json/from_cbor.md), [`from_msgpack`](../basic_json/from_msgpack.md), + [`from_ubjson`](../basic_json/from_ubjson.md), [`from_bjdata`](../basic_json/from_bjdata.md), + [`from_bson`](../basic_json/from_bson.md)): none of these formats requires a decoder to reject ill-formed UTF-8, so + they always return the bytes unchanged. + +## Default definition + +The default value is `0` (disabled, the behavior of version 3.12.0 and earlier is preserved). + +```cpp +#define JSON_STRICT_BINARY_UTF8 0 +``` + +## Notes + +!!! note "Background" + + CBOR, UBJSON, BJData, and BSON all require strings to be UTF-8. Up to version 3.12.0, the writers did not check + this, so they could produce output that other decoders reject. Checking by default would break code that stores + other encodings (for instance ISO 8859-1) in a string and only ever writes it to a binary format, so this macro + offers the check as an opt-in ahead of version 4.0.0, where it is planned to become the default (see + [#5529](https://github.com/nlohmann/json/issues/5529) and [#5651](https://github.com/nlohmann/json/issues/5651)). + +!!! warning "Opt-in only" + + This macro must be defined **before** including ``. Defining it after the include has no + effect. + +!!! note "ABI compatibility" + + The value of this macro is encoded in the [namespace](../../features/namespace.md) (tag `_sbu8`), resulting in + distinct symbol names. Translation units compiled with and without it can therefore be linked into the same program + without One Definition Rule (ODR) violations, but they cannot exchange instances of library types. + +## Examples + +??? example "Default behavior (macro not defined)" + + Without the macro, the bytes are written unchanged: + + ```cpp + #include + + using json = nlohmann::json; + + int main() + { + auto v = json::to_cbor(json("\xFF")); + // v is {0x61, 0xFF} + } + ``` + +??? example "Opt-in check (macro defined to 1)" + + With the macro, ill-formed UTF-8 is rejected: + + ```cpp + #define JSON_STRICT_BINARY_UTF8 1 + #include + + using json = nlohmann::json; + + int main() + { + auto v = json::to_cbor(json("\xFF")); + // throws type_error.316: invalid UTF-8 byte at index 0: 0xFF + } + ``` + +## See also + +- [**to_cbor**](../basic_json/to_cbor.md) - create a CBOR serialization of a JSON value +- [**to_ubjson**](../basic_json/to_ubjson.md) - create a UBJSON serialization of a JSON value +- [**to_bjdata**](../basic_json/to_bjdata.md) - create a BJData serialization of a JSON value +- [**to_bson**](../basic_json/to_bson.md) - create a BSON serialization of a JSON value +- [**error_handler_t**](../basic_json/error_handler_t.md) - how [`dump`](../basic_json/dump.md) treats ill-formed UTF-8 + +## Version history + +- Added in version 3.13.0. +- Planned to become the default (with the macro removed) in version 4.0.0. diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index 5220c819b..f52deb540 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -65,9 +65,10 @@ The library uses the following mapping from JSON values types to BJData types ac !!! warning "UTF-8 validation of string values and object keys" - BJData strings must use UTF-8 encoding. `to_bjdata()` validates the bytes of every string value and object key - and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8, so a - value with such a string cannot be serialized in the first place. + BJData strings must use UTF-8 encoding. By default, `to_bjdata()` writes the bytes of string values and object keys + unchanged, even if they are not valid UTF-8. If + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead. !!! info "Unused BJData markers" @@ -220,8 +221,8 @@ The library maps BJData types to JSON value types as follows: value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error - handler is passed that replaces or ignores the ill-formed bytes. `to_bjdata()` is strict as well (see above), so - a value read this way cannot be written back to BJData. + handler is passed that replaces or ignores the ill-formed bytes. By default, `to_bjdata()` writes such a value + back unchanged (see above). !!! info "Round trips" diff --git a/docs/mkdocs/docs/features/binary_formats/bson.md b/docs/mkdocs/docs/features/binary_formats/bson.md index acc7ca6d9..b60f90c3c 100644 --- a/docs/mkdocs/docs/features/binary_formats/bson.md +++ b/docs/mkdocs/docs/features/binary_formats/bson.md @@ -114,12 +114,11 @@ The library maps BSON record types to JSON value types as follows: The BSON specification requires `string` values (type `0x02`) to be valid UTF-8, but this is not required of a decoder. `from_bson()` accepts a `string` value whose bytes are not valid UTF-8 and hands them back unchanged. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error - handler is passed that replaces or ignores the ill-formed bytes. `to_bson()` is strict as well and throws the - same exception for a string value or element (key) name that is not valid UTF-8, so an object with such a key - or value cannot be produced in the first place, even though `from_bson()` would accept it from another source. - Element (key) names are never validated on read, since they are read byte-by-byte as a C string. `binary` - values (type `0x05`) are unaffected, since they are not required to hold text. + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error handler is + passed that replaces or ignores the ill-formed bytes. By default, `to_bson()` writes such a string value or element + (key) name unchanged; if [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it + throws the same exception instead. Element (key) names are never validated on read, since they are read byte-by-byte + as a C string. `binary` values (type `0x05`) are unaffected, since they are not required to hold text. ??? example diff --git a/docs/mkdocs/docs/features/binary_formats/cbor.md b/docs/mkdocs/docs/features/binary_formats/cbor.md index de594d837..5ce05765c 100644 --- a/docs/mkdocs/docs/features/binary_formats/cbor.md +++ b/docs/mkdocs/docs/features/binary_formats/cbor.md @@ -191,14 +191,14 @@ The library maps CBOR types to JSON value types as follows: !!! warning "Ill-formed UTF-8 in text strings" - [RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings - (major type 3) to be valid UTF-8, but leaves it up to the decoder whether to enforce this. This library does - not: `from_cbor()` accepts a text string (object keys included) whose bytes are not valid UTF-8 and hands them - back unchanged. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws - [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error - handler is passed that replaces or ignores the ill-formed bytes. `to_cbor()` is strict as well and throws the - same exception for a string value or object key that is not valid UTF-8, so such a value cannot be written back - to CBOR. Byte strings (major type 2) are unaffected, since they are not required to hold text. + [RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings (major + type 3) to be valid UTF-8, but leaves it up to the decoder whether to enforce this. This library does not: + `from_cbor()` accepts a text string (object keys included) whose bytes are not valid UTF-8 and hands them back + unchanged. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error handler is + passed that replaces or ignores the ill-formed bytes. By default, `to_cbor()` writes such a value back unchanged; if + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws the same exception + instead. Byte strings (major type 2) are unaffected, since they are not required to hold text. !!! warning "Tagged items" diff --git a/docs/mkdocs/docs/features/binary_formats/ubjson.md b/docs/mkdocs/docs/features/binary_formats/ubjson.md index 39da9628e..6903d066c 100644 --- a/docs/mkdocs/docs/features/binary_formats/ubjson.md +++ b/docs/mkdocs/docs/features/binary_formats/ubjson.md @@ -49,9 +49,10 @@ The library uses the following mapping from JSON values types to UBJSON types ac !!! warning "UTF-8 validation of string values and object keys" - UBJSON's required string encoding is UTF-8. `to_ubjson()` validates the bytes of every string value and object - key and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8, so - a value with such a string cannot be serialized in the first place. + UBJSON's required string encoding is UTF-8. By default, `to_ubjson()` writes the bytes of string values and object + keys unchanged, even if they are not valid UTF-8. If + [`JSON_STRICT_BINARY_UTF8`](../../api/macros/json_strict_binary_utf8.md) is enabled, it throws + [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8 instead. !!! info "Unused UBJSON markers" @@ -132,8 +133,8 @@ The library maps UBJSON types to JSON value types as follows: value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error - handler is passed that replaces or ignores the ill-formed bytes. `to_ubjson()` is strict as well (see above), so - a value read this way cannot be written back to UBJSON. + handler is passed that replaces or ignores the ill-formed bytes. By default, `to_ubjson()` writes such a value + back unchanged (see above). ??? example diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index a4479712f..ff24c55ca 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -138,6 +138,20 @@ using the library with compilers that do not fully support C++11 and may only wo See [full documentation of `JSON_SKIP_UNSUPPORTED_COMPILER_CHECK`](../api/macros/json_skip_unsupported_compiler_check.md). +## `JSON_STRICT_BINARY_UTF8` + +When defined to `1`, [`to_cbor`](../api/basic_json/to_cbor.md), [`to_ubjson`](../api/basic_json/to_ubjson.md), +[`to_bjdata`](../api/basic_json/to_bjdata.md), and [`to_bson`](../api/basic_json/to_bson.md) throw +[`type_error.316`](../home/exceptions.md#jsonexceptiontype_error316) for a string value or object key that is not +valid UTF-8. The default value is `0`, which writes the bytes unchanged as before version 3.13.0; this is planned to +become the default in version 4.0.0. + +The check can also be enabled with the CMake option +[`JSON_StrictBinaryUTF8`](../integration/cmake.md#json_strictbinaryutf8) (`OFF` by default) which sets +`JSON_STRICT_BINARY_UTF8` accordingly. + +See [full documentation of `JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md). + ## `JSON_STRICT_NUL_HANDLING` When defined to `1`, a `'\0'` (NUL) byte anywhere in the input is rejected with `parse_error.101`, like any other diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index 577f5e221..dbddac13d 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -20,6 +20,7 @@ The complete default namespace name is derived as follows: `_bics`. - [`JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md) defined non-zero appends `_psp`. - [`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md) defined non-zero appends `_snul`. + - [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) defined non-zero appends `_sbu8`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index 1d1bf33d5..84c0c9d41 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -747,6 +747,11 @@ The `unflatten()` function only works for an object whose keys are JSON Pointers The `dump()` function only works with UTF-8 encoded strings; that is, if you assign a `std::string` to a JSON value, make sure it is UTF-8 encoded. +If [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md) is enabled, the binary writers +[`to_cbor()`](../api/basic_json/to_cbor.md), [`to_ubjson()`](../api/basic_json/to_ubjson.md), +[`to_bjdata()`](../api/basic_json/to_bjdata.md), and [`to_bson()`](../api/basic_json/to_bson.md) throw this exception +for a string value or object key that is not valid UTF-8 as well. + !!! failure "Example message" Calling `dump()` on a JSON value containing an ISO 8859-1 encoded string: diff --git a/docs/mkdocs/docs/integration/cmake.md b/docs/mkdocs/docs/integration/cmake.md index 4160584d5..5b78df133 100644 --- a/docs/mkdocs/docs/integration/cmake.md +++ b/docs/mkdocs/docs/integration/cmake.md @@ -204,6 +204,11 @@ Use the non-amalgamated version of the library. This option is `ON` by default. Treat the library headers like system headers (i.e., adding `SYSTEM` to the [`target_include_directories`](https://cmake.org/cmake/help/latest/command/target_include_directories.html) call) to check for this library by tools like Clang-Tidy. This option is `OFF` by default. +### `JSON_StrictBinaryUTF8` + +Check string values and object keys for valid UTF-8 in the CBOR, UBJSON, BJData, and BSON writers, by defining the +macro [`JSON_STRICT_BINARY_UTF8`](../api/macros/json_strict_binary_utf8.md). This option is `OFF` by default. + ### `JSON_StrictNulHandling` Reject a `'\0'` (NUL) byte in the input instead of treating it as end of input, by defining the macro diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 4bd207e60..09aa165b0 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -301,6 +301,7 @@ nav: - 'JSON_PRECISE_STREAM_POSITION': api/macros/json_precise_stream_position.md - 'JSON_SKIP_LIBRARY_VERSION_CHECK': api/macros/json_skip_library_version_check.md - 'JSON_SKIP_UNSUPPORTED_COMPILER_CHECK': api/macros/json_skip_unsupported_compiler_check.md + - 'JSON_STRICT_BINARY_UTF8': api/macros/json_strict_binary_utf8.md - 'JSON_STRICT_NUL_HANDLING': api/macros/json_strict_nul_handling.md - 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md - 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md diff --git a/include/nlohmann/detail/abi_macros.hpp b/include/nlohmann/detail/abi_macros.hpp index 0bace616a..0153c8706 100644 --- a/include/nlohmann/detail/abi_macros.hpp +++ b/include/nlohmann/detail/abi_macros.hpp @@ -46,6 +46,10 @@ #define JSON_STRICT_NUL_HANDLING 0 #endif +#ifndef JSON_STRICT_BINARY_UTF8 + #define JSON_STRICT_BINARY_UTF8 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -82,14 +86,20 @@ #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING #endif +#if JSON_STRICT_BINARY_UTF8 + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8 +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -98,7 +108,8 @@ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ - NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \ + NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index 1e6e6cce6..f8be43829 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -41,6 +41,7 @@ #undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_PRECISE_STREAM_POSITION #undef JSON_STRICT_NUL_HANDLING + #undef JSON_STRICT_BINARY_UTF8 #endif #include diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index d777b2b43..da5aa0cf4 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -115,8 +115,8 @@ class binary_writer /*! @param[in] j JSON value to serialize - @throw type_error.316 if a string value or an object key is not valid - UTF-8 + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or an object key is not valid UTF-8 @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) @@ -147,8 +147,8 @@ class binary_writer /*! @param[in] j JSON value to serialize - @throw type_error.316 if a string value or an object key is not valid - UTF-8 + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or an object key is not valid UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -215,7 +215,7 @@ class binary_writer case value_t::string: { - check_utf8(*j.m_data.m_value.string, j); + check_text_utf8(*j.m_data.m_value.string, j); // step 1: write control byte and the string length write_cbor_head(0x60, j.m_data.m_value.string->size()); @@ -297,7 +297,7 @@ class binary_writer // diagnostics context, because write_cbor(el.first) // converts it to a temporary basic_json that would be // used as the context instead - check_utf8(el.first, j); + check_text_utf8(el.first, j); write_cbor(el.first); write_cbor(el.second); } @@ -640,8 +640,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 - @throw type_error.316 if a string value or an object key is not valid - UTF-8 + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or an object key is not valid UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -691,7 +691,7 @@ class binary_writer case value_t::string: { - check_utf8(*j.m_data.m_value.string, j); + check_text_utf8(*j.m_data.m_value.string, j); if (add_prefix) { @@ -855,7 +855,7 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { - check_utf8(el.first, j); + check_text_utf8(el.first, j); write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(el.first.data()), @@ -902,8 +902,8 @@ class binary_writer and the entry name size (and its null-terminator). @throw out_of_range.409 if @a name contains U+0000, before anything is written - @throw type_error.316 if @a name is not valid UTF-8, before anything is - written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a name is + not valid UTF-8, before anything is written */ static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { @@ -913,7 +913,7 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - check_utf8(name, j); + check_text_utf8(name, j); return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; } @@ -970,8 +970,8 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value - @throw type_error.316 if @a value is not valid UTF-8, before anything is - written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a value + is not valid UTF-8, before anything is written @note The UTF-8 check is skipped if @a value is already too long for the 32-bit BSON length field (@ref to_bson_length rejects it later, once @@ -983,7 +983,7 @@ class binary_writer { if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) { - check_utf8(value, j); + check_text_utf8(value, j); } return sizeof(std::int32_t) + value.size() + 1ul; } @@ -1113,8 +1113,8 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written - @throw type_error.316 if @a j is a string that is not valid UTF-8, before - anything is written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a j is a + string that is not valid UTF-8, before anything is written */ static std::size_t calc_bson_value_size(const BasicJsonType& j) { @@ -1249,8 +1249,8 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written - @throw type_error.316 if a string value or a key is not valid UTF-8, - before anything is written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or a key is not valid UTF-8, before anything is written */ static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { @@ -2170,6 +2170,29 @@ class binary_writer } } + /*! + @brief check a CBOR, UBJSON, BJData, or BSON text string for valid UTF-8 + + The check only happens if JSON_STRICT_BINARY_UTF8 is enabled. Otherwise, + the bytes are written unchanged, as before version 3.13.0. MessagePack + always writes the bytes as is, and BON8 always checks them (see + @ref check_utf8). + + @param[in] s the string to check + @param[in] context the value that holds @a s (for diagnostics) + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a s is + not valid UTF-8 + */ + static void check_text_utf8(const string_t& s, const BasicJsonType& context) + { +#if JSON_STRICT_BINARY_UTF8 + check_utf8(s, context); +#else + static_cast(s); + static_cast(context); +#endif + } + /*! @brief write an integer in the shortest encoding diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 19bd2da1e..65d8424fe 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -104,6 +104,10 @@ #define JSON_STRICT_NUL_HANDLING 0 #endif +#ifndef JSON_STRICT_BINARY_UTF8 + #define JSON_STRICT_BINARY_UTF8 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -140,14 +144,20 @@ #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING #endif +#if JSON_STRICT_BINARY_UTF8 + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8 +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -156,7 +166,8 @@ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ - NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \ + NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ @@ -21193,8 +21204,8 @@ class binary_writer /*! @param[in] j JSON value to serialize - @throw type_error.316 if a string value or an object key is not valid - UTF-8 + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or an object key is not valid UTF-8 @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) @@ -21225,8 +21236,8 @@ class binary_writer /*! @param[in] j JSON value to serialize - @throw type_error.316 if a string value or an object key is not valid - UTF-8 + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or an object key is not valid UTF-8 */ void write_cbor(const BasicJsonType& j) { @@ -21293,7 +21304,7 @@ class binary_writer case value_t::string: { - check_utf8(*j.m_data.m_value.string, j); + check_text_utf8(*j.m_data.m_value.string, j); // step 1: write control byte and the string length write_cbor_head(0x60, j.m_data.m_value.string->size()); @@ -21375,7 +21386,7 @@ class binary_writer // diagnostics context, because write_cbor(el.first) // converts it to a temporary basic_json that would be // used as the context instead - check_utf8(el.first, j); + check_text_utf8(el.first, j); write_cbor(el.first); write_cbor(el.second); } @@ -21718,8 +21729,8 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 - @throw type_error.316 if a string value or an object key is not valid - UTF-8 + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or an object key is not valid UTF-8 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, @@ -21769,7 +21780,7 @@ class binary_writer case value_t::string: { - check_utf8(*j.m_data.m_value.string, j); + check_text_utf8(*j.m_data.m_value.string, j); if (add_prefix) { @@ -21933,7 +21944,7 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { - check_utf8(el.first, j); + check_text_utf8(el.first, j); write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(el.first.data()), @@ -21980,8 +21991,8 @@ class binary_writer and the entry name size (and its null-terminator). @throw out_of_range.409 if @a name contains U+0000, before anything is written - @throw type_error.316 if @a name is not valid UTF-8, before anything is - written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a name is + not valid UTF-8, before anything is written */ static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j) { @@ -21991,7 +22002,7 @@ class binary_writer JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j)); } - check_utf8(name, j); + check_text_utf8(name, j); return /*id*/ 1ul + name.size() + /*zero-terminator*/1u; } @@ -22048,8 +22059,8 @@ class binary_writer /*! @return The size of the BSON-encoded string in @a value - @throw type_error.316 if @a value is not valid UTF-8, before anything is - written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a value + is not valid UTF-8, before anything is written @note The UTF-8 check is skipped if @a value is already too long for the 32-bit BSON length field (@ref to_bson_length rejects it later, once @@ -22061,7 +22072,7 @@ class binary_writer { if (JSON_HEDLEY_LIKELY(value_in_range_of(value.size()))) { - check_utf8(value, j); + check_text_utf8(value, j); } return sizeof(std::int32_t) + value.size() + 1ul; } @@ -22191,8 +22202,8 @@ class binary_writer is neither an object nor an array @throw out_of_range.415 if @a j is binary with a subtype that does not fit into a byte, before anything is written - @throw type_error.316 if @a j is a string that is not valid UTF-8, before - anything is written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a j is a + string that is not valid UTF-8, before anything is written */ static std::size_t calc_bson_value_size(const BasicJsonType& j) { @@ -22327,8 +22338,8 @@ class binary_writer written @throw out_of_range.415 if a binary value's subtype does not fit into a byte, before anything is written - @throw type_error.316 if a string value or a key is not valid UTF-8, - before anything is written + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and a string + value or a key is not valid UTF-8, before anything is written */ static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { @@ -23248,6 +23259,29 @@ class binary_writer } } + /*! + @brief check a CBOR, UBJSON, BJData, or BSON text string for valid UTF-8 + + The check only happens if JSON_STRICT_BINARY_UTF8 is enabled. Otherwise, + the bytes are written unchanged, as before version 3.13.0. MessagePack + always writes the bytes as is, and BON8 always checks them (see + @ref check_utf8). + + @param[in] s the string to check + @param[in] context the value that holds @a s (for diagnostics) + @throw type_error.316 if JSON_STRICT_BINARY_UTF8 is enabled and @a s is + not valid UTF-8 + */ + static void check_text_utf8(const string_t& s, const BasicJsonType& context) + { +#if JSON_STRICT_BINARY_UTF8 + check_utf8(s, context); +#else + static_cast(s); + static_cast(context); +#endif + } + /*! @brief write an integer in the shortest encoding @@ -33813,6 +33847,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_PRECISE_STREAM_POSITION #undef JSON_STRICT_NUL_HANDLING + #undef JSON_STRICT_BINARY_UTF8 #endif // #include diff --git a/single_include/nlohmann/json_fwd.hpp b/single_include/nlohmann/json_fwd.hpp index 3ee7afa73..7cadc9b1c 100644 --- a/single_include/nlohmann/json_fwd.hpp +++ b/single_include/nlohmann/json_fwd.hpp @@ -63,6 +63,10 @@ #define JSON_STRICT_NUL_HANDLING 0 #endif +#ifndef JSON_STRICT_BINARY_UTF8 + #define JSON_STRICT_BINARY_UTF8 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -99,14 +103,20 @@ #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING #endif +#if JSON_STRICT_BINARY_UTF8 + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 _sbu8 +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8 +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) json_abi ## a ## b ## c ## d ## e ## f ## g +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f, g) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f, g) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -115,7 +125,8 @@ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ - NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING, \ + NLOHMANN_JSON_ABI_TAG_STRICT_BINARY_UTF8) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index 879322dd0..e4c627060 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -44,6 +44,10 @@ TEST_CASE("default namespace") expected += "_snul"; #endif +#if JSON_STRICT_BINARY_UTF8 + expected += "_sbu8"; +#endif + expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json"; diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index 4b1eb6ee4..858964695 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -45,6 +45,10 @@ TEST_CASE("default namespace without version component") expected += "_snul"; #endif +#if JSON_STRICT_BINARY_UTF8 + expected += "_sbu8"; +#endif + expected += "::basic_json"; // fallback for Clang diff --git a/tests/src/unit-binary_utf8_strict.cpp b/tests/src/unit-binary_utf8_strict.cpp new file mode 100644 index 000000000..c83b5938c --- /dev/null +++ b/tests/src/unit-binary_utf8_strict.cpp @@ -0,0 +1,110 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// The binary writers check strings and object keys for valid UTF-8 only if +// JSON_STRICT_BINARY_UTF8 is enabled (planned to be the default in 4.0.0). +// Without it, they write the bytes unchanged, as before version 3.13.0; the +// tests for that are next to the other tests of each format. +#ifdef JSON_STRICT_BINARY_UTF8 + #undef JSON_STRICT_BINARY_UTF8 +#endif + +#define JSON_STRICT_BINARY_UTF8 1 + +#include +using nlohmann::json; + +#include +#include + +TEST_CASE("JSON_STRICT_BINARY_UTF8 (see #5529, #5651)") +{ + SECTION("CBOR") + { + // a string value with ill-formed UTF-8 is rejected + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_cbor(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + + // binary values are not text and are unaffected + CHECK_NOTHROW(json::to_cbor(json::binary(std::vector({0xFF})))); + + // a value read back from CBOR with ill-formed bytes cannot be written + // back either (the reader is lenient regardless of the macro) + const json j = json::from_cbor(std::vector({0x62, 0xc0, 0xae})); + CHECK_THROWS_WITH_AS(json::to_cbor(j), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + } + + SECTION("UBJSON") + { + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_ubjson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + + SECTION("BJData") + { + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected the same way + CHECK_THROWS_WITH_AS(json::to_bjdata(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + + SECTION("BSON") + { + // to_bson() rejects the same kind of ill-formed string value, before + // any bytes reach the output adapter (the BSON document length + // prefix must be known up front, so nothing is written incrementally) + std::vector out{0x42}; // a sentinel byte the writer must not touch + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}, nlohmann::detail::output_adapter(out)), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + CHECK(out == std::vector {0x42}); + + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // a truncated multi-byte sequence + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC3"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + // an encoded surrogate half (U+D800) + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xED\xA0\x80"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + // an overlong encoding of '.' + CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC0\xAF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + + // an object key with ill-formed UTF-8 is rejected as well; unlike + // the reader (which never validates element names), the writer + // checks both string values and object keys + CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + } + + SECTION("MessagePack and BON8 are unaffected") + { + // MessagePack allows any bytes in a str, so to_msgpack() writes them as + // is; BON8 always checks, because the lead bytes mark where strings end + CHECK(json::to_msgpack(json("\xFF")) == std::vector({0xa1, 0xff})); + CHECK_THROWS_AS(json::to_bon8(json("\xFF")), json::type_error&); + } +} diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 09a46ceb2..ef73c5e82 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -3913,15 +3913,17 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") // none of the binary format specs requires a decoder to reject // ill-formed UTF-8 in a text string, so a value whose bytes are // not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') - // round-trips byte for byte as a string value; to_bjdata() is - // strict, so such a value cannot be written back + // round-trips byte for byte as a string value; to_bjdata() writes + // the bytes unchanged, as before 3.13.0, unless + // JSON_STRICT_BINARY_UTF8 is enabled (see + // unit-binary_utf8_strict.cpp) const std::vector v = {'S', 'i', 2, 0xc0, 0xae}; json j; CHECK_NOTHROW(j = json::from_bjdata(v)); REQUIRE(j.is_string()); CHECK(j.get_ref() == std::string("\xc0\xae")); CHECK_THROWS_AS(j.dump(), json::type_error&); - CHECK_THROWS_AS(json::to_bjdata(j), json::type_error&); + CHECK(json::from_bjdata(json::to_bjdata(j)) == j); // the same bytes as an object key round-trip as well const std::vector v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'}; @@ -3929,18 +3931,18 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") CHECK_NOTHROW(j_key = json::from_bjdata(v_key)); REQUIRE(j_key.is_object()); CHECK(j_key.contains(std::string("\xc0\xae"))); - CHECK_THROWS_AS(json::to_bjdata(j_key), json::type_error&); + CHECK(json::from_bjdata(json::to_bjdata(j_key)) == j_key); - CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + CHECK(json::from_bjdata(json::to_bjdata(json("\xFF"))) == json("\xFF")); // a truncated multi-byte sequence - CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + CHECK(json::from_bjdata(json::to_bjdata(json("\xC3"))) == json("\xC3")); // an encoded surrogate half (U+D800) - CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + CHECK(json::from_bjdata(json::to_bjdata(json("\xED\xA0\x80"))) == json("\xED\xA0\x80")); // an overlong encoding of '.' - CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + CHECK(json::from_bjdata(json::to_bjdata(json("\xC0\xAF"))) == json("\xC0\xAF")); - // an object key with ill-formed UTF-8 is rejected the same way - CHECK_THROWS_WITH_AS(json::to_bjdata(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // an object key with ill-formed UTF-8 is kept the same way + CHECK(json::from_bjdata(json::to_bjdata(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); } } diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index 97a437d6b..92a14e6fe 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -175,28 +175,20 @@ TEST_CASE("BSON") CHECK(j["s"].get_ref() == std::string("\xc0\xae")); // dump() still requires valid UTF-8 and throws for such a value CHECK_THROWS_AS(j.dump(), json::type_error&); - // to_bson() is strict as well, so the value cannot be written back - CHECK_THROWS_AS(json::to_bson(j), json::type_error&); + // to_bson() writes the bytes back unchanged, as before 3.13.0, + // unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp) + CHECK(json::from_bson(json::to_bson(j)) == j); - // to_bson() rejects the same kind of ill-formed string value, before - // any bytes reach the output adapter (the BSON document length - // prefix must be known up front, so nothing is written incrementally) - std::vector out{0x42}; // a sentinel byte the writer must not touch - CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}, nlohmann::detail::output_adapter(out)), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); - CHECK(out == std::vector {0x42}); - - CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}})) == json{{"s", "\xFF"}}); // a truncated multi-byte sequence - CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC3"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + CHECK(json::from_bson(json::to_bson(json{{"s", "\xC3"}})) == json{{"s", "\xC3"}}); // an encoded surrogate half (U+D800) - CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xED\xA0\x80"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + CHECK(json::from_bson(json::to_bson(json{{"s", "\xED\xA0\x80"}})) == json{{"s", "\xED\xA0\x80"}}); // an overlong encoding of '.' - CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC0\xAF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + CHECK(json::from_bson(json::to_bson(json{{"s", "\xC0\xAF"}})) == json{{"s", "\xC0\xAF"}}); - // an object key with ill-formed UTF-8 is rejected as well; unlike - // the reader (which never validates element names), the writer - // checks both string values and object keys - CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // an object key with ill-formed UTF-8 is kept as well + CHECK(json::from_bson(json::to_bson(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); } SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON") diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index ed2ba98ac..633f50fea 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -1821,8 +1821,9 @@ TEST_CASE("CBOR") // unless an error handler that replaces or ignores the bytes is // passed CHECK_THROWS_AS(j_value.dump(), json::type_error&); - // to_cbor() is strict as well, so the value cannot be written back - CHECK_THROWS_AS(json::to_cbor(j_value), json::type_error&); + // to_cbor() writes the bytes back unchanged, as before 3.13.0, + // unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp) + CHECK(json::from_cbor(json::to_cbor(j_value)) == j_value); // the same bytes as an object key round-trip as well const std::vector ill_formed_key = {0xa1, 0x62, 0xc0, 0xae, 0x01}; @@ -1830,7 +1831,7 @@ TEST_CASE("CBOR") CHECK_NOTHROW(j_key = json::from_cbor(ill_formed_key)); REQUIRE(j_key.is_object()); CHECK(j_key.contains(std::string("\xc0\xae"))); - CHECK_THROWS_AS(json::to_cbor(j_key), json::type_error&); + CHECK(json::from_cbor(json::to_cbor(j_key)) == j_key); // a CBOR byte string (major type 2) with the very same bytes is // NOT text and must still be accepted as-is @@ -1843,20 +1844,21 @@ TEST_CASE("CBOR") CHECK(json::from_cbor(json::to_cbor(j)) == j); } - SECTION("to_cbor rejects ill-formed UTF-8 (see #5651)") + SECTION("to_cbor keeps ill-formed UTF-8 (see #5651)") { - // to_cbor() must reject the same ill-formed strings from_cbor() - // rejects, so a value it accepts can always be read back - CHECK_THROWS_WITH_AS(json::to_cbor(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // to_cbor() writes the bytes unchanged, as before 3.13.0, unless + // JSON_STRICT_BINARY_UTF8 is enabled (see + // unit-binary_utf8_strict.cpp); from_cbor() reads them back as is + CHECK(json::from_cbor(json::to_cbor(json("\xFF"))) == json("\xFF")); // a truncated multi-byte sequence - CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + CHECK(json::from_cbor(json::to_cbor(json("\xC3"))) == json("\xC3")); // an encoded surrogate half (U+D800) - CHECK_THROWS_WITH_AS(json::to_cbor(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + CHECK(json::from_cbor(json::to_cbor(json("\xED\xA0\x80"))) == json("\xED\xA0\x80")); // an overlong encoding of '.' - CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + CHECK(json::from_cbor(json::to_cbor(json("\xC0\xAF"))) == json("\xC0\xAF")); - // an object key with ill-formed UTF-8 is rejected the same way - CHECK_THROWS_WITH_AS(json::to_cbor(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // an object key with ill-formed UTF-8 is kept the same way + CHECK(json::from_cbor(json::to_cbor(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); // binary values are not text and are unaffected CHECK_NOTHROW(json::to_cbor(json::binary(std::vector({0xFF})))); @@ -1877,7 +1879,7 @@ TEST_CASE("CBOR") CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0xff}))); CHECK(_ == "\xc3"); CHECK_THROWS_AS(_.dump(), json::type_error&); - CHECK_THROWS_AS(json::to_cbor(_), json::type_error&); + CHECK(json::from_cbor(json::to_cbor(_)) == _); // an ill-formed later chunk is kept after valid ones CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff}))); diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index 0ca68f384..450882a12 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -2511,15 +2511,17 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") // none of the binary format specs requires a decoder to reject // ill-formed UTF-8 in a text string, so a value whose bytes are // not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') - // round-trips byte for byte as a string value; to_ubjson() is - // strict, so such a value cannot be written back + // round-trips byte for byte as a string value; to_ubjson() writes + // the bytes unchanged, as before 3.13.0, unless + // JSON_STRICT_BINARY_UTF8 is enabled (see + // unit-binary_utf8_strict.cpp) const std::vector v = {'S', 'i', 2, 0xc0, 0xae}; json j; CHECK_NOTHROW(j = json::from_ubjson(v)); REQUIRE(j.is_string()); CHECK(j.get_ref() == std::string("\xc0\xae")); CHECK_THROWS_AS(j.dump(), json::type_error&); - CHECK_THROWS_AS(json::to_ubjson(j), json::type_error&); + CHECK(json::from_ubjson(json::to_ubjson(j)) == j); // the same bytes as an object key round-trip as well const std::vector v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'}; @@ -2527,18 +2529,18 @@ TEST_CASE("Universal Binary JSON Specification Examples 1") CHECK_NOTHROW(j_key = json::from_ubjson(v_key)); REQUIRE(j_key.is_object()); CHECK(j_key.contains(std::string("\xc0\xae"))); - CHECK_THROWS_AS(json::to_ubjson(j_key), json::type_error&); + CHECK(json::from_ubjson(json::to_ubjson(j_key)) == j_key); - CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + CHECK(json::from_ubjson(json::to_ubjson(json("\xFF"))) == json("\xFF")); // a truncated multi-byte sequence - CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&); + CHECK(json::from_ubjson(json::to_ubjson(json("\xC3"))) == json("\xC3")); // an encoded surrogate half (U+D800) - CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + CHECK(json::from_ubjson(json::to_ubjson(json("\xED\xA0\x80"))) == json("\xED\xA0\x80")); // an overlong encoding of '.' - CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&); + CHECK(json::from_ubjson(json::to_ubjson(json("\xC0\xAF"))) == json("\xC0\xAF")); - // an object key with ill-formed UTF-8 is rejected the same way - CHECK_THROWS_WITH_AS(json::to_ubjson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // an object key with ill-formed UTF-8 is kept the same way + CHECK(json::from_ubjson(json::to_ubjson(json{{"\xFF", 1}})) == json{{"\xFF", 1}}); } }