Compare commits

..
Author SHA1 Message Date
Niels Lohmann dc194c94d9 Stop excluding the Unicode tests in CI
The compiler matrix, icpc/icpx/nvhpc, the default-compiler job, the
Windows MinGW and clang-cl jobs, and AppVeyor Debug builds excluded
test-unicode because it took several minutes. It now takes ~20 s, so
run it everywhere except under Valgrind.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-27 22:07:47 +02:00
Niels Lohmann 9ffd033af7 Merge the Unicode tests into unit-unicode.cpp
The Unicode tests were split into five files (#2889) so they could run
in parallel. After #5418 they take ~20 s together (unoptimized), so one
file is enough. This removes four copies of the check helpers and the
progress output, whose hard-coded totals were stale, and drops
doctest::skip() so JSON_FastTests configurations run them too. The
raised timeout for test-unicode4 is no longer needed.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-27 22:07:46 +02:00
Niels Lohmann d304501fa7 Cover every byte class in the ill-formed UTF-8 sweeps
Pinning the bytes around the invalid one to 0x80 lost coverage: with
error_handler_t::ignore/replace, the serializer re-reads the invalid
byte and decodes the following bytes, and its decoder distinguishes
the continuation classes 0x80-0x8F, 0x90-0x9F, and 0xA0-0xBF, as do
the lexer's range checks.

Iterate those positions over the first and last byte of each class
within the valid range instead (utils::utf8_continuation_bytes). The
invalid byte still takes all 256 values. Defining
JSON_TEST_UTF8_EXHAUSTIVE restores the full Cartesian product, with
the same assertion counts as before #5418.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-27 22:00:35 +02:00
Niels Lohmann 21c8935269 Check the JSON Pointer roundtrip for every code point again
Sampling every 64th element of all_unicode.json gave up the claim that
every code point is tested for a small saving; the stringstream
removal keeps most of the speedup.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-27 22:00:34 +02:00
Niels Lohmann 96cedfec81 Compile unit-msgpack.cpp only once
unit-msgpack.cpp mentioned JSON_HAS_CPP_17 for a single std::byte test
case, so the whole file was also compiled and run as test-msgpack_cpp17.
Move that test case into unit-msgpack-cpp17.cpp, as was done for
unit-items-cpp17.cpp.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-27 21:48:35 +02:00
Niels Lohmann d27698ee55 Run the cheap binary format size tests unconditionally
Only jeopardy.json (52 MB, ~20 s unoptimized) needs doctest::skip();
move it into its own test case so canada.json, twitter.json,
citm_catalog.json and sample.json (~1.5 s) also run in JSON_FastTests
configurations.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-27 21:47:17 +02:00
Niels Lohmann 295ff0f778 Speed up unit-unicode1
Format \uxxxx escapes by hand instead of constructing a stringstream
for each of the ~1.1M code points, and check the JSON Pointer
escape/unescape roundtrip on every 64th element of all_unicode.json
plus '~' and '/', the only characters escaping treats specially.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-27 21:46:39 +02:00
Niels Lohmann e802c98da8 Pin unrelated bytes in the remaining ill-formed UTF-8 sweeps
The wrong-4th-byte sections of unit-unicode3/4/5 became live with #5499
and swept every valid 2nd/3rd byte again (2.4M iterations in unicode4
alone). Pin those bytes like the 2nd/3rd-byte sections, and do the
same for the 3-byte sequences in unit-unicode2.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-27 21:46:08 +02:00
elix3r c6bb0844f3 Cut Unicode ill-formed byte sweeps to one representative prefix
Wrong-2nd and wrong-3rd-byte sections iterated every valid trailing
byte, which dominated Linux --no-skip runtime without testing extra
properties. Pin those bytes to a single valid continuation.

The fourth-byte typo is left to #5416 so this change stands alone.

Signed-off-by: elix3r <157088510+22elix3r@users.noreply.github.com>
Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-27 21:44:47 +02:00
27 changed files with 1913 additions and 3215 deletions
+1 -6
View File
@@ -83,9 +83,4 @@ build_script:
- cmake --build . --config "%configuration%" --parallel 2
test_script:
- if "%configuration%"=="Release" ctest -C "%configuration%" --parallel 2 --output-on-failure
# On Debug builds, skip test-unicode_all
# as it is extremely slow to run and cause
# occasional timeouts on AppVeyor.
# More info: https://github.com/nlohmann/json/pull/1570
- if "%configuration%"=="Debug" ctest --exclude-regex "test-unicode" -C "%configuration%" --parallel 2 --output-on-failure
- ctest -C "%configuration%" --parallel 2 --output-on-failure
+2 -2
View File
@@ -174,7 +174,7 @@ jobs:
- name: Build
run: cmake --build build --parallel 10
- name: Test
run: cd build ; ctest -j 10 -C Debug --exclude-regex "test-unicode" --output-on-failure
run: cd build ; ctest -j 10 -C Debug --output-on-failure
clang-cl-12:
runs-on: windows-2022
@@ -191,7 +191,7 @@ jobs:
- name: Build
run: cmake --build build --config Debug --parallel 10
- name: Test
run: cd build ; ctest -j 10 -C Debug --exclude-regex "test-unicode" --output-on-failure
run: cd build ; ctest -j 10 -C Debug --output-on-failure
ci_module_cpp20:
runs-on: windows-2022
+7 -6
View File
@@ -413,13 +413,14 @@ add_custom_target(ci_test_single_header
# Valgrind.
###############################################################################
# The Unicode test (~17M assertions) is too slow under Valgrind.
add_custom_target(ci_test_valgrind
COMMAND CXX=${GCC_TOOL} ${CMAKE_COMMAND}
-DCMAKE_BUILD_TYPE=Debug -GNinja
-DJSON_BuildTests=ON -DJSON_Valgrind=ON
-S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_valgrind
COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_valgrind
COMMAND cd ${PROJECT_BINARY_DIR}/build_valgrind && ${CMAKE_CTEST_COMMAND} -L valgrind --parallel ${N} --output-on-failure
COMMAND cd ${PROJECT_BINARY_DIR}/build_valgrind && ${CMAKE_CTEST_COMMAND} -L valgrind --exclude-regex "test-unicode" --parallel ${N} --output-on-failure
COMMENT "Compile and test with Valgrind"
)
@@ -717,7 +718,7 @@ foreach(COMPILER g++-4.8 g++-4.9 g++-5 g++-6 g++-7 g++-8 g++-9 g++-10 g++-11 cla
-S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_compiler_${COMPILER}
${ADDITIONAL_FLAGS}
COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_compiler_${COMPILER}
COMMAND cd ${PROJECT_BINARY_DIR}/build_compiler_${COMPILER} && ${CMAKE_CTEST_COMMAND} --parallel ${N} --exclude-regex "test-unicode" --output-on-failure
COMMAND cd ${PROJECT_BINARY_DIR}/build_compiler_${COMPILER} && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure
COMMENT "Compile and test with ${COMPILER}"
)
endif()
@@ -731,7 +732,7 @@ add_custom_target(ci_test_compiler_default
-S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_compiler_default
${ADDITIONAL_FLAGS}
COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_compiler_default --parallel ${N}
COMMAND cd ${PROJECT_BINARY_DIR}/build_compiler_default && ${CMAKE_CTEST_COMMAND} --parallel ${N} --exclude-regex "test-unicode" -LE git_required --output-on-failure
COMMAND cd ${PROJECT_BINARY_DIR}/build_compiler_default && ${CMAKE_CTEST_COMMAND} --parallel ${N} -LE git_required --output-on-failure
COMMENT "Compile and test with default C++ compiler"
)
@@ -769,7 +770,7 @@ add_custom_target(ci_icpc
-DJSON_BuildTests=ON -DJSON_FastTests=ON
-S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_icpc
COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_icpc
COMMAND cd ${PROJECT_BINARY_DIR}/build_icpc && ${CMAKE_CTEST_COMMAND} --parallel ${N} --exclude-regex "test-unicode" --output-on-failure
COMMAND cd ${PROJECT_BINARY_DIR}/build_icpc && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure
COMMENT "Compile and test with ICPC"
)
@@ -780,7 +781,7 @@ add_custom_target(ci_icpx
-DJSON_BuildTests=ON -DJSON_FastTests=ON
-S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_icpx
COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_icpx
COMMAND cd ${PROJECT_BINARY_DIR}/build_icpx && ${CMAKE_CTEST_COMMAND} --parallel ${N} --exclude-regex "test-unicode" --output-on-failure
COMMAND cd ${PROJECT_BINARY_DIR}/build_icpx && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure
COMMENT "Compile and test with ICPX (Intel oneAPI DPC++/C++)"
)
@@ -816,7 +817,7 @@ add_custom_target(ci_nvhpc
COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_nvhpc
# the pipes are escaped so the surrounding shell passes them to ctest verbatim
# instead of treating them as shell pipe operators
COMMAND cd ${PROJECT_BINARY_DIR}/build_nvhpc && ${CMAKE_CTEST_COMMAND} --parallel ${N} --exclude-regex "test-unicode\\|test-comparison_cpp20\\|test-comparison_legacy_cpp20\\|test-constructor1_cpp11\\|test-deserialization_cpp20" --output-on-failure
COMMAND cd ${PROJECT_BINARY_DIR}/build_nvhpc && ${CMAKE_CTEST_COMMAND} --parallel ${N} --exclude-regex "test-comparison_cpp20\\|test-comparison_legacy_cpp20\\|test-constructor1_cpp11\\|test-deserialization_cpp20" --output-on-failure
COMMENT "Compile and test with NVIDIA HPC SDK (nvc++)"
)
+2 -2
View File
@@ -80,8 +80,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
the end of the file was not reached when `strict` was set to true
- Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from CBOR were
used in the given input or if the input is not valid CBOR
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other
types are not supported, as JSON object keys are always strings) or a string is malformed
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string was expected as a map key,
but not found
## Complexity
@@ -73,8 +73,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
the end of the file was not reached when `strict` was set to true
- Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if unsupported features from
MessagePack were used in the given input or if the input is not valid MessagePack
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a map key is not a string (keys of other
types are not supported, as JSON object keys are always strings) or a string is malformed
- Throws [parse_error.113](../../home/exceptions.md#jsonexceptionparse_error113) if a string was expected as a map key,
but not found
## Complexity
@@ -174,20 +174,7 @@ The library maps CBOR types to JSON value types as follows:
!!! warning "Object keys"
CBOR allows map keys of any type, whereas JSON only allows strings as keys in object values. Therefore, CBOR maps
with keys other than text strings (major type 3) are rejected with a
[`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with `allow_exceptions` set
to `false`, a discarded value) naming the type of the key that was found, for instance:
```
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found an unsigned integer; last byte: 0x01
```
This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed
on. This is a deliberate restriction of the library's JSON value model, not an oversight: formats built on CBOR
maps with integer keys, such as COSE ([RFC 9052](https://www.rfc-editor.org/rfc/rfc9052.html)) or CWT
([RFC 8392](https://www.rfc-editor.org/rfc/rfc8392.html)), cannot be read with this library and need a
general-purpose CBOR library instead.
CBOR allows map keys of any type, whereas JSON only allows strings as keys in object values. Therefore, CBOR maps with keys other than UTF-8 strings are rejected.
!!! warning "UTF-8 validation of text strings"
@@ -138,21 +138,6 @@ The library maps MessagePack types to JSON value types as follows:
Any MessagePack output created by `to_msgpack` can be successfully parsed by `from_msgpack`.
!!! warning "Object keys"
MessagePack allows map keys of any type, whereas JSON only allows strings as keys in object values. Like the
JSON-compatible [profile](https://github.com/msgpack/msgpack/blob/master/spec.md#profile) sketched in the
MessagePack specification, this library restricts map keys to `str` values. Maps with keys of any other type are
rejected with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with
`allow_exceptions` set to `false`, a discarded value) naming the type of the key that was found, for instance:
```
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack object key: only string keys are supported, but found nil; last byte: 0xC0
```
This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed
on. Such input needs a general-purpose MessagePack library instead.
!!! warning "UTF-8 validation of string values"
The MessagePack specification requires `str` values (`fixstr`, `str 8`, `str 16`, `str 32`) to be valid UTF-8.
+2 -9
View File
@@ -343,20 +343,13 @@ A string could not be read from a [binary format](../features/binary_formats/ind
string was read where one was required (for instance as a map key), the string's length specification is invalid, or
the string's bytes are not valid UTF-8.
CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings. Maps with keys of any other
type (for instance integers or `null`) are therefore not supported; see the notes on
[CBOR](../features/binary_formats/cbor.md) and [MessagePack](../features/binary_formats/messagepack.md).
!!! failure "Example messages"
```
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found an unsigned integer; last byte: 0x01
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xFF
```
```
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack object key: only string keys are supported, but found nil; last byte: 0xC0
```
```
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack string: expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0xFF
```
```
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing UBJSON char: byte after 'C' must be in range 0x00..0x7F; last byte: 0x82
+2 -168
View File
@@ -1324,80 +1324,6 @@ class binary_reader
}
}
/*!
@brief reads a CBOR object key
RFC 8949 allows any data item as a map key, but only strings have a
counterpart in JSON. A key of any other type is rejected with a message
naming that type, rather than the one @ref get_cbor_string gives for a
malformed string.
@param[out] result created key
@return whether key creation completed
*/
bool get_cbor_object_key(string_t& result)
{
// EOF and major type 3 (text string) are left to get_cbor_string
if (current == char_traits<char_type>::eof() || (static_cast<unsigned int>(current) & 0xE0u) == 0x60u)
{
return get_cbor_string(result);
}
const char* found = nullptr;
switch (static_cast<unsigned int>(current) >> 5u)
{
case 0:
found = "an unsigned integer";
break;
case 1:
found = "a negative integer";
break;
case 2:
found = "a byte string";
break;
case 4:
found = "an array";
break;
case 5:
found = "a map";
break;
case 6:
found = "a tag";
break;
default: // major type 7
switch (current)
{
case 0xF4:
case 0xF5:
found = "a boolean";
break;
case 0xF6:
found = "null";
break;
case 0xF7:
found = "undefined";
break;
case 0xF9:
case 0xFA:
case 0xFB:
found = "a floating-point number";
break;
case 0xFF:
found = "a break stop code";
break;
default:
found = "a simple value";
break;
}
break;
}
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(input_format_t::cbor, concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr));
}
/*!
@brief reads a definite-length CBOR byte array
@@ -1642,7 +1568,7 @@ class binary_reader
if (top.is_object)
{
key.clear();
if (JSON_HEDLEY_UNLIKELY(!get_cbor_object_key(key) || !sax->key(key)))
if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key)))
{
return false;
}
@@ -2143,98 +2069,6 @@ class binary_reader
}
}
/*!
@brief reads a MessagePack object key
The MessagePack specification allows any type as a map key, but only
strings have a counterpart in JSON. A key of any other type is rejected
with a message naming that type, rather than the one @ref
get_msgpack_string gives for a malformed string.
@param[out] result created key
@return whether key creation completed
*/
bool get_msgpack_object_key(string_t& result)
{
const char* found = nullptr;
switch (current)
{
case 0xC0:
found = "nil";
break;
case 0xC2:
case 0xC3:
found = "a boolean";
break;
case 0xCA:
case 0xCB:
found = "a float";
break;
case 0xC4:
case 0xC5:
case 0xC6:
found = "a bin";
break;
case 0xC7:
case 0xC8:
case 0xC9:
case 0xD4:
case 0xD5:
case 0xD6:
case 0xD7:
case 0xD8:
found = "an ext";
break;
case 0xCC:
case 0xCD:
case 0xCE:
case 0xCF:
case 0xD0:
case 0xD1:
case 0xD2:
case 0xD3:
found = "an integer";
break;
case 0xDC:
case 0xDD:
found = "an array";
break;
case 0xDE:
case 0xDF:
found = "a map";
break;
default:
// fixint, fixmap, and fixarray; strings, EOF, and the unused
// byte 0xC1 are left to get_msgpack_string
if (current == char_traits<char_type>::eof())
{
return get_msgpack_string(result);
}
if (current <= 0x7F || current >= 0xE0)
{
found = "an integer";
}
else if (current <= 0x8F)
{
found = "a map";
}
else if (current <= 0x9F)
{
found = "an array";
}
else
{
return get_msgpack_string(result);
}
break;
}
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(input_format_t::msgpack, concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr));
}
/*!
@brief reads a MessagePack byte array
@@ -2397,7 +2231,7 @@ class binary_reader
{
get();
key.clear();
if (JSON_HEDLEY_UNLIKELY(!get_msgpack_object_key(key) || !sax->key(key)))
if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key)))
{
return false;
}
+39 -76
View File
@@ -206,6 +206,7 @@ class lexer : public lexer_base<BasicJsonType>
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept
: ia(std::move(adapter))
, ignore_comments(ignore_comments_)
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
, discard_number_values(discard_number_values_)
{}
@@ -221,7 +222,8 @@ class lexer : public lexer_base<BasicJsonType>
// locales
/////////////////////
/// return the decimal point of the current locale
/// return the locale-dependent decimal point
JSON_HEDLEY_PURE
static char get_decimal_point() noexcept
{
const auto* loc = localeconv();
@@ -1090,10 +1092,9 @@ class lexer : public lexer_base<BasicJsonType>
token_type::value_float if number could be successfully scanned,
token_type::parse_error otherwise
@note The scanner is independent of the current locale: token_buffer
always holds `.`. Only the std::strtod fallback of convert_number()
depends on the locale, and it looks up the decimal point right
before converting (see convert_float_locale_aware()).
@note The scanner is independent of the current locale. Internally, the
locale's decimal point is used instead of `.` to work with the
locale-dependent converters.
*/
token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated.
{
@@ -1182,7 +1183,7 @@ scan_number_zero:
{
case '.':
{
add(current);
add(decimal_point_char);
decimal_point_position = token_buffer.size() - 1;
goto scan_number_decimal1;
}
@@ -1219,7 +1220,7 @@ scan_number_any1:
case '.':
{
add(current);
add(decimal_point_char);
decimal_point_position = token_buffer.size() - 1;
goto scan_number_decimal1;
}
@@ -1461,9 +1462,9 @@ scan_number_done:
// Only a number below 1 can carry further insignificant zeros, and only
// while the count stays at the limit does removing them change the
// answer - so this loop is skipped for all but a few tokens. The
// fraction is located through decimal_point_position rather than by
// searching '.'.
// answer - so this loop is skipped for all but a few tokens. Note
// token_buffer holds the locale's decimal point, so the fraction is
// located through decimal_point_position rather than by searching '.'.
if (lead_zero != 0)
{
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
@@ -1481,8 +1482,8 @@ scan_number_done:
@brief convert the number text in token_buffer to its value and token type
The digit sequence in token_buffer has already been validated (by the
scan_number() state machine or by the contiguous fast path) and holds '.'
as decimal point, independent of the locale. Integers are parsed first and fall
scan_number() state machine or by the contiguous fast path) and holds the
locale decimal point in place of '.'. Integers are parsed first and fall
back to floating point on overflow. This is shared so both scanners produce
identical results.
@@ -1562,7 +1563,7 @@ scan_number_done:
// integer conversion above overflowed. Prefer std::from_chars
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
// otherwise the exact Clinger fast path (double only); otherwise the
// locale-aware strtof/strtod/strtold.
// locale-aware strtof/strtod.
if (parse_float_from_chars(num_begin, num_end, value_float))
{
return token_type::value_float;
@@ -1571,75 +1572,26 @@ scan_number_done:
// extra pass over the token's bytes, which otherwise shows up on
// high-precision inputs such as canada.json
if (mantissa_fits_clinger(mantissa_end)
&& parse_float_fast(num_begin, num_end, value_float))
&& parse_float_fast(num_begin, num_end, decimal_point_char, value_float))
{
return token_type::value_float;
}
convert_float_locale_aware();
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
strtof(value_float, token_buffer.data(), &endptr);
// we checked the number format before
JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size());
return token_type::value_float;
}
/*!
@brief convert the float in token_buffer with strtof/strtod/strtold
These functions expect the decimal point of the *current* locale, so it is
looked up right before the conversion instead of once when the lexer is
constructed: a locale change in between (by a parser callback, a SAX
handler, or another thread) must not truncate the value (#5198). The
token has been validated before, so if the conversion stops early and the
decimal point changed in the meantime, the locale changed between the
lookup and the call, and the conversion is repeated with the new decimal
point. If the decimal point did not change, a retry cannot succeed: the
locale's decimal point is not a single character (e.g., the two-byte
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
The value strtod parsed up to that point is kept, as before this change.
Note that changing the locale in another thread *while* strtod runs is
undefined behavior of the C library, which this function cannot prevent.
*/
void convert_float_locale_aware()
{
const bool has_dot = decimal_point_position != std::string::npos;
char decimal_point = get_decimal_point();
for (;;)
{
const bool substitute = has_dot && decimal_point != '.';
if (substitute)
{
token_buffer[decimal_point_position] = static_cast<typename string_t::value_type>(decimal_point);
}
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
strtof(value_float, token_buffer.data(), &endptr);
if (substitute)
{
// get_string() hands the token to the SAX interface with '.'
token_buffer[decimal_point_position] = '.';
}
if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size()))
{
return;
}
// retry only if the locale changed; otherwise, this would loop forever
const char current_decimal_point = get_decimal_point();
if (current_decimal_point == decimal_point)
{
return;
}
decimal_point = current_decimal_point;
}
}
/*!
@brief contiguous fast path for scanning a number
Parses the whole number token straight from the input buffer, avoiding the
per-character get()/add() of scan_number(). On success it fills token_buffer
(as scan_number() does) and
(with the locale decimal point substituted, as scan_number() does) and
returns the token type. On anything it does not fully recognize as a
well-formed number it makes no state change and returns
token_type::uninitialized, so the caller falls back to scan_number(), which
@@ -1755,11 +1707,16 @@ scan_number_done:
}
#endif
// materialize the token exactly as scan_number() would. reset() already
// cleared token_buffer, so append() fills it (assign() is avoided
// because custom string_t types need not provide it)
// materialize the token exactly as scan_number() would, substituting the
// locale decimal point so convert_number()'s strtof fallback stays valid.
// reset() already cleared token_buffer, so append() fills it (assign() is
// avoided because custom string_t types need not provide it)
token_buffer.append(reinterpret_cast<const typename string_t::value_type*>(data), len);
decimal_point_position = dot_index;
if (dot_index != std::string::npos)
{
token_buffer[dot_index] = static_cast<typename string_t::value_type>(decimal_point_char);
decimal_point_position = dot_index;
}
ia.bulk_skip(len - 1);
position.chars_read_total += (len - 1);
@@ -2026,7 +1983,11 @@ scan_number_done:
/// return current string value (implicitly resets the token; useful only once)
string_t& get_string()
{
// a number token holds '.' regardless of the locale (#4084)
// translate decimal points from locale back to '.' (#4084)
if (decimal_point_char != '.' && decimal_point_position != std::string::npos)
{
token_buffer[decimal_point_position] = '.';
}
return token_buffer;
}
@@ -2322,7 +2283,9 @@ scan_number_done:
number_unsigned_t value_unsigned = 0;
number_float_t value_float = 0;
/// the position of the decimal point in token_buffer
/// the decimal point
const char_int_type decimal_point_char = '.';
/// the position of the decimal point in the input
std::size_t decimal_point_position = std::string::npos;
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
+13 -8
View File
@@ -118,12 +118,14 @@ std::strtod. The parser only activates for number_float_t == double; float and
long double keep the std::strtof/std::strtold paths (see the templated overload
below).
@param[in] first pointer to the first character of the number
@param[in] last pointer past the last character
@param[out] out the parsed value on success
@param[in] first pointer to the first character of the number
@param[in] last pointer past the last character
@param[in] decimal_point the (locale-dependent) decimal point character
@param[out] out the parsed value on success
@return true if the value was parsed exactly; false to fall back to strtod
*/
inline bool parse_float_fast(const char* first, const char* last, double& out) noexcept
template<typename DecimalPointType>
bool parse_float_fast(const char* first, const char* last, DecimalPointType decimal_point, double& out) noexcept
{
#if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0
// Clinger's fast path is only exact when double operations are evaluated in
@@ -134,6 +136,7 @@ inline bool parse_float_fast(const char* first, const char* last, double& out) n
// std::from_chars / std::strtod path.
static_cast<void>(first);
static_cast<void>(last);
static_cast<void>(decimal_point);
static_cast<void>(out);
return false;
#else
@@ -172,7 +175,7 @@ inline bool parse_float_fast(const char* first, const char* last, double& out) n
++num_digits;
fractional_digits += static_cast<int>(seen_dot);
}
else if (c == '.')
else if (static_cast<DecimalPointType>(c) == decimal_point)
{
if (JSON_HEDLEY_UNLIKELY(seen_dot))
{
@@ -257,8 +260,8 @@ inline bool parse_float_fast(const char* first, const char* last, double& out) n
}
/// fast float path is only exact for `double`; decline for float/long double
template<typename FloatType>
bool parse_float_fast(const char* /*first*/, const char* /*last*/, FloatType& /*out*/) noexcept
template<typename DecimalPointType, typename FloatType>
bool parse_float_fast(const char* /*first*/, const char* /*last*/, DecimalPointType /*decimal_point*/, FloatType& /*out*/) noexcept
{
return false;
}
@@ -270,7 +273,9 @@ std::from_chars is locale-independent, correctly rounded, and - via the
Eisel-Lemire algorithm in modern standard libraries - much faster than strtod
over the whole value range (not just the Clinger subset). It is used only when
__cpp_lib_to_chars indicates full floating-point support and only when it
consumes the entire token ([first, last)). An under-/overflow (result_out_of_range) also declines, so
consumes the entire token ([first, last)); a partial parse means the buffer
uses a non-'.' locale decimal point, in which case the caller falls back to the
locale-aware path. An under-/overflow (result_out_of_range) also declines, so
the caller's strtod fallback supplies the well-defined ±inf/0 result the parser
expects (side-stepping the P4168 divergence between implementations).
+54 -252
View File
@@ -8605,12 +8605,14 @@ std::strtod. The parser only activates for number_float_t == double; float and
long double keep the std::strtof/std::strtold paths (see the templated overload
below).
@param[in] first pointer to the first character of the number
@param[in] last pointer past the last character
@param[out] out the parsed value on success
@param[in] first pointer to the first character of the number
@param[in] last pointer past the last character
@param[in] decimal_point the (locale-dependent) decimal point character
@param[out] out the parsed value on success
@return true if the value was parsed exactly; false to fall back to strtod
*/
inline bool parse_float_fast(const char* first, const char* last, double& out) noexcept
template<typename DecimalPointType>
bool parse_float_fast(const char* first, const char* last, DecimalPointType decimal_point, double& out) noexcept
{
#if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0
// Clinger's fast path is only exact when double operations are evaluated in
@@ -8621,6 +8623,7 @@ inline bool parse_float_fast(const char* first, const char* last, double& out) n
// std::from_chars / std::strtod path.
static_cast<void>(first);
static_cast<void>(last);
static_cast<void>(decimal_point);
static_cast<void>(out);
return false;
#else
@@ -8659,7 +8662,7 @@ inline bool parse_float_fast(const char* first, const char* last, double& out) n
++num_digits;
fractional_digits += static_cast<int>(seen_dot);
}
else if (c == '.')
else if (static_cast<DecimalPointType>(c) == decimal_point)
{
if (JSON_HEDLEY_UNLIKELY(seen_dot))
{
@@ -8744,8 +8747,8 @@ inline bool parse_float_fast(const char* first, const char* last, double& out) n
}
/// fast float path is only exact for `double`; decline for float/long double
template<typename FloatType>
bool parse_float_fast(const char* /*first*/, const char* /*last*/, FloatType& /*out*/) noexcept
template<typename DecimalPointType, typename FloatType>
bool parse_float_fast(const char* /*first*/, const char* /*last*/, DecimalPointType /*decimal_point*/, FloatType& /*out*/) noexcept
{
return false;
}
@@ -8757,7 +8760,9 @@ std::from_chars is locale-independent, correctly rounded, and - via the
Eisel-Lemire algorithm in modern standard libraries - much faster than strtod
over the whole value range (not just the Clinger subset). It is used only when
__cpp_lib_to_chars indicates full floating-point support and only when it
consumes the entire token ([first, last)). An under-/overflow (result_out_of_range) also declines, so
consumes the entire token ([first, last)); a partial parse means the buffer
uses a non-'.' locale decimal point, in which case the caller falls back to the
locale-aware path. An under-/overflow (result_out_of_range) also declines, so
the caller's strtod fallback supplies the well-defined ±inf/0 result the parser
expects (side-stepping the P4168 divergence between implementations).
@@ -9298,6 +9303,7 @@ class lexer : public lexer_base<BasicJsonType>
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept
: ia(std::move(adapter))
, ignore_comments(ignore_comments_)
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
, discard_number_values(discard_number_values_)
{}
@@ -9313,7 +9319,8 @@ class lexer : public lexer_base<BasicJsonType>
// locales
/////////////////////
/// return the decimal point of the current locale
/// return the locale-dependent decimal point
JSON_HEDLEY_PURE
static char get_decimal_point() noexcept
{
const auto* loc = localeconv();
@@ -10182,10 +10189,9 @@ class lexer : public lexer_base<BasicJsonType>
token_type::value_float if number could be successfully scanned,
token_type::parse_error otherwise
@note The scanner is independent of the current locale: token_buffer
always holds `.`. Only the std::strtod fallback of convert_number()
depends on the locale, and it looks up the decimal point right
before converting (see convert_float_locale_aware()).
@note The scanner is independent of the current locale. Internally, the
locale's decimal point is used instead of `.` to work with the
locale-dependent converters.
*/
token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated.
{
@@ -10274,7 +10280,7 @@ scan_number_zero:
{
case '.':
{
add(current);
add(decimal_point_char);
decimal_point_position = token_buffer.size() - 1;
goto scan_number_decimal1;
}
@@ -10311,7 +10317,7 @@ scan_number_any1:
case '.':
{
add(current);
add(decimal_point_char);
decimal_point_position = token_buffer.size() - 1;
goto scan_number_decimal1;
}
@@ -10553,9 +10559,9 @@ scan_number_done:
// Only a number below 1 can carry further insignificant zeros, and only
// while the count stays at the limit does removing them change the
// answer - so this loop is skipped for all but a few tokens. The
// fraction is located through decimal_point_position rather than by
// searching '.'.
// answer - so this loop is skipped for all but a few tokens. Note
// token_buffer holds the locale's decimal point, so the fraction is
// located through decimal_point_position rather than by searching '.'.
if (lead_zero != 0)
{
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
@@ -10573,8 +10579,8 @@ scan_number_done:
@brief convert the number text in token_buffer to its value and token type
The digit sequence in token_buffer has already been validated (by the
scan_number() state machine or by the contiguous fast path) and holds '.'
as decimal point, independent of the locale. Integers are parsed first and fall
scan_number() state machine or by the contiguous fast path) and holds the
locale decimal point in place of '.'. Integers are parsed first and fall
back to floating point on overflow. This is shared so both scanners produce
identical results.
@@ -10654,7 +10660,7 @@ scan_number_done:
// integer conversion above overflowed. Prefer std::from_chars
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
// otherwise the exact Clinger fast path (double only); otherwise the
// locale-aware strtof/strtod/strtold.
// locale-aware strtof/strtod.
if (parse_float_from_chars(num_begin, num_end, value_float))
{
return token_type::value_float;
@@ -10663,75 +10669,26 @@ scan_number_done:
// extra pass over the token's bytes, which otherwise shows up on
// high-precision inputs such as canada.json
if (mantissa_fits_clinger(mantissa_end)
&& parse_float_fast(num_begin, num_end, value_float))
&& parse_float_fast(num_begin, num_end, decimal_point_char, value_float))
{
return token_type::value_float;
}
convert_float_locale_aware();
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
strtof(value_float, token_buffer.data(), &endptr);
// we checked the number format before
JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size());
return token_type::value_float;
}
/*!
@brief convert the float in token_buffer with strtof/strtod/strtold
These functions expect the decimal point of the *current* locale, so it is
looked up right before the conversion instead of once when the lexer is
constructed: a locale change in between (by a parser callback, a SAX
handler, or another thread) must not truncate the value (#5198). The
token has been validated before, so if the conversion stops early and the
decimal point changed in the meantime, the locale changed between the
lookup and the call, and the conversion is repeated with the new decimal
point. If the decimal point did not change, a retry cannot succeed: the
locale's decimal point is not a single character (e.g., the two-byte
U+066B of ar_EG.UTF-8 or fa_IR.UTF-8) and cannot be substituted in place.
The value strtod parsed up to that point is kept, as before this change.
Note that changing the locale in another thread *while* strtod runs is
undefined behavior of the C library, which this function cannot prevent.
*/
void convert_float_locale_aware()
{
const bool has_dot = decimal_point_position != std::string::npos;
char decimal_point = get_decimal_point();
for (;;)
{
const bool substitute = has_dot && decimal_point != '.';
if (substitute)
{
token_buffer[decimal_point_position] = static_cast<typename string_t::value_type>(decimal_point);
}
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
strtof(value_float, token_buffer.data(), &endptr);
if (substitute)
{
// get_string() hands the token to the SAX interface with '.'
token_buffer[decimal_point_position] = '.';
}
if (JSON_HEDLEY_LIKELY(endptr == token_buffer.data() + token_buffer.size()))
{
return;
}
// retry only if the locale changed; otherwise, this would loop forever
const char current_decimal_point = get_decimal_point();
if (current_decimal_point == decimal_point)
{
return;
}
decimal_point = current_decimal_point;
}
}
/*!
@brief contiguous fast path for scanning a number
Parses the whole number token straight from the input buffer, avoiding the
per-character get()/add() of scan_number(). On success it fills token_buffer
(as scan_number() does) and
(with the locale decimal point substituted, as scan_number() does) and
returns the token type. On anything it does not fully recognize as a
well-formed number it makes no state change and returns
token_type::uninitialized, so the caller falls back to scan_number(), which
@@ -10847,11 +10804,16 @@ scan_number_done:
}
#endif
// materialize the token exactly as scan_number() would. reset() already
// cleared token_buffer, so append() fills it (assign() is avoided
// because custom string_t types need not provide it)
// materialize the token exactly as scan_number() would, substituting the
// locale decimal point so convert_number()'s strtof fallback stays valid.
// reset() already cleared token_buffer, so append() fills it (assign() is
// avoided because custom string_t types need not provide it)
token_buffer.append(reinterpret_cast<const typename string_t::value_type*>(data), len);
decimal_point_position = dot_index;
if (dot_index != std::string::npos)
{
token_buffer[dot_index] = static_cast<typename string_t::value_type>(decimal_point_char);
decimal_point_position = dot_index;
}
ia.bulk_skip(len - 1);
position.chars_read_total += (len - 1);
@@ -11118,7 +11080,11 @@ scan_number_done:
/// return current string value (implicitly resets the token; useful only once)
string_t& get_string()
{
// a number token holds '.' regardless of the locale (#4084)
// translate decimal points from locale back to '.' (#4084)
if (decimal_point_char != '.' && decimal_point_position != std::string::npos)
{
token_buffer[decimal_point_position] = '.';
}
return token_buffer;
}
@@ -11414,7 +11380,9 @@ scan_number_done:
number_unsigned_t value_unsigned = 0;
number_float_t value_float = 0;
/// the position of the decimal point in token_buffer
/// the decimal point
const char_int_type decimal_point_char = '.';
/// the position of the decimal point in the input
std::size_t decimal_point_position = std::string::npos;
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
@@ -14091,80 +14059,6 @@ class binary_reader
}
}
/*!
@brief reads a CBOR object key
RFC 8949 allows any data item as a map key, but only strings have a
counterpart in JSON. A key of any other type is rejected with a message
naming that type, rather than the one @ref get_cbor_string gives for a
malformed string.
@param[out] result created key
@return whether key creation completed
*/
bool get_cbor_object_key(string_t& result)
{
// EOF and major type 3 (text string) are left to get_cbor_string
if (current == char_traits<char_type>::eof() || (static_cast<unsigned int>(current) & 0xE0u) == 0x60u)
{
return get_cbor_string(result);
}
const char* found = nullptr;
switch (static_cast<unsigned int>(current) >> 5u)
{
case 0:
found = "an unsigned integer";
break;
case 1:
found = "a negative integer";
break;
case 2:
found = "a byte string";
break;
case 4:
found = "an array";
break;
case 5:
found = "a map";
break;
case 6:
found = "a tag";
break;
default: // major type 7
switch (current)
{
case 0xF4:
case 0xF5:
found = "a boolean";
break;
case 0xF6:
found = "null";
break;
case 0xF7:
found = "undefined";
break;
case 0xF9:
case 0xFA:
case 0xFB:
found = "a floating-point number";
break;
case 0xFF:
found = "a break stop code";
break;
default:
found = "a simple value";
break;
}
break;
}
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(input_format_t::cbor, concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr));
}
/*!
@brief reads a definite-length CBOR byte array
@@ -14409,7 +14303,7 @@ class binary_reader
if (top.is_object)
{
key.clear();
if (JSON_HEDLEY_UNLIKELY(!get_cbor_object_key(key) || !sax->key(key)))
if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key)))
{
return false;
}
@@ -14910,98 +14804,6 @@ class binary_reader
}
}
/*!
@brief reads a MessagePack object key
The MessagePack specification allows any type as a map key, but only
strings have a counterpart in JSON. A key of any other type is rejected
with a message naming that type, rather than the one @ref
get_msgpack_string gives for a malformed string.
@param[out] result created key
@return whether key creation completed
*/
bool get_msgpack_object_key(string_t& result)
{
const char* found = nullptr;
switch (current)
{
case 0xC0:
found = "nil";
break;
case 0xC2:
case 0xC3:
found = "a boolean";
break;
case 0xCA:
case 0xCB:
found = "a float";
break;
case 0xC4:
case 0xC5:
case 0xC6:
found = "a bin";
break;
case 0xC7:
case 0xC8:
case 0xC9:
case 0xD4:
case 0xD5:
case 0xD6:
case 0xD7:
case 0xD8:
found = "an ext";
break;
case 0xCC:
case 0xCD:
case 0xCE:
case 0xCF:
case 0xD0:
case 0xD1:
case 0xD2:
case 0xD3:
found = "an integer";
break;
case 0xDC:
case 0xDD:
found = "an array";
break;
case 0xDE:
case 0xDF:
found = "a map";
break;
default:
// fixint, fixmap, and fixarray; strings, EOF, and the unused
// byte 0xC1 are left to get_msgpack_string
if (current == char_traits<char_type>::eof())
{
return get_msgpack_string(result);
}
if (current <= 0x7F || current >= 0xE0)
{
found = "an integer";
}
else if (current <= 0x8F)
{
found = "a map";
}
else if (current <= 0x9F)
{
found = "an array";
}
else
{
return get_msgpack_string(result);
}
break;
}
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read,
exception_message(input_format_t::msgpack, concat("only string keys are supported, but found ", found, "; last byte: 0x", last_token), "object key"), nullptr));
}
/*!
@brief reads a MessagePack byte array
@@ -15164,7 +14966,7 @@ class binary_reader
{
get();
key.clear();
if (JSON_HEDLEY_UNLIKELY(!get_msgpack_object_key(key) || !sax->key(key)))
if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key)))
{
return false;
}
-3
View File
@@ -129,9 +129,6 @@ json_test_set_test_options(test-disabled_exceptions
#$<$<CXX_COMPILER_ID:MSVC>:/EH>
)
# raise timeout of expensive Unicode test
json_test_set_test_options(test-unicode4 TEST_PROPERTIES TIMEOUT 3000)
#############################################################################
# add unit tests
#############################################################################
+28
View File
@@ -8,6 +8,7 @@
#pragma once
#include <array> // array
#include <cstdint> // uint8_t
#include <cstddef> // size_t
#include <fstream> // ifstream, istreambuf_iterator, ios
@@ -42,6 +43,33 @@ T next_integer_sample(T i, T last, T stride)
return n < last ? n : last;
}
// UTF-8 continuation bytes in [lo, hi] that stand in for all of them in the
// ill-formed UTF-8 tests. Both the lexer's range checks and the serializer's
// decoder (detail::decode) only distinguish the classes 0x80..0x8F, 0x90..0x9F,
// and 0xA0..0xBF, so the first and last byte of each class within [lo, hi]
// exercise every behavior while a test sweeps another byte position through
// all 256 values (#5418). Define JSON_TEST_UTF8_EXHAUSTIVE to get every byte.
inline std::vector<int> utf8_continuation_bytes(int lo, int hi)
{
std::vector<int> result;
#ifdef JSON_TEST_UTF8_EXHAUSTIVE
for (int byte = lo; byte <= hi; ++byte)
{
result.push_back(byte);
}
#else
static const std::array<int, 6> class_ends = {{0x80, 0x8F, 0x90, 0x9F, 0xA0, 0xBF}};
for (const int byte : class_ends)
{
if (lo <= byte && byte <= hi)
{
result.push_back(byte);
}
}
#endif
return result;
}
inline std::vector<std::uint8_t> read_binary_file(const std::string& filename)
{
std::ifstream file(filename, std::ios::binary);
+45 -43
View File
@@ -14,7 +14,7 @@ using nlohmann::json;
#include <fstream>
#include "make_test_data_available.hpp"
TEST_CASE("Binary Formats" * doctest::skip())
TEST_CASE("Binary Formats")
{
SECTION("canada.json")
{
@@ -142,48 +142,6 @@ TEST_CASE("Binary Formats" * doctest::skip())
CHECK((100.0 * double(ubjson_3_size) / double(json_size)) == Approx(84.963));
}
SECTION("jeopardy.json")
{
const auto* filename = TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json";
json j = json::parse(std::ifstream(filename));
const auto json_size = j.dump().size();
const auto bjdata_1_size = json::to_bjdata(j).size();
const auto bjdata_2_size = json::to_bjdata(j, true).size();
const auto bjdata_3_size = json::to_bjdata(j, true, true).size();
const auto bon8_size = json::to_bon8(j).size();
const auto bson_size = json::to_bson({{"", j}}).size(); // wrap array in object for BSON
const auto cbor_size = json::to_cbor(j).size();
const auto msgpack_size = json::to_msgpack(j).size();
const auto ubjson_1_size = json::to_ubjson(j).size();
const auto ubjson_2_size = json::to_ubjson(j, true).size();
const auto ubjson_3_size = json::to_ubjson(j, true, true).size();
CHECK(json_size == 52508728);
CHECK(bjdata_1_size == 50710965);
CHECK(bjdata_2_size == 51144830);
CHECK(bjdata_3_size == 51144830);
CHECK(bon8_size == 45942080);
CHECK(bson_size == 56008520);
CHECK(cbor_size == 46187320);
CHECK(msgpack_size == 46158575);
CHECK(ubjson_1_size == 50710965);
CHECK(ubjson_2_size == 51144830);
CHECK(ubjson_3_size == 49861422);
CHECK((100.0 * double(json_size) / double(json_size)) == Approx(100.0));
CHECK((100.0 * double(bjdata_1_size) / double(json_size)) == Approx(96.576));
CHECK((100.0 * double(bjdata_2_size) / double(json_size)) == Approx(97.402));
CHECK((100.0 * double(bjdata_3_size) / double(json_size)) == Approx(97.402));
CHECK((100.0 * double(bon8_size) / double(json_size)) == Approx(87.494));
CHECK((100.0 * double(bson_size) / double(json_size)) == Approx(106.665));
CHECK((100.0 * double(cbor_size) / double(json_size)) == Approx(87.961));
CHECK((100.0 * double(msgpack_size) / double(json_size)) == Approx(87.906));
CHECK((100.0 * double(ubjson_1_size) / double(json_size)) == Approx(96.576));
CHECK((100.0 * double(ubjson_2_size) / double(json_size)) == Approx(97.402));
CHECK((100.0 * double(ubjson_3_size) / double(json_size)) == Approx(94.958));
}
SECTION("sample.json")
{
const auto* filename = TEST_DATA_DIRECTORY "/json_testsuite/sample.json";
@@ -224,3 +182,47 @@ TEST_CASE("Binary Formats" * doctest::skip())
CHECK((100.0 * double(ubjson_3_size) / double(json_size)) == Approx(89.450));
}
}
// jeopardy.json is 52 MB and produces ~500 MB of serialization output, so it
// is kept apart from the cheap corpus files above (#5418)
TEST_CASE("Binary Formats (jeopardy.json)" * doctest::skip())
{
const auto* filename = TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json";
json j = json::parse(std::ifstream(filename));
const auto json_size = j.dump().size();
const auto bjdata_1_size = json::to_bjdata(j).size();
const auto bjdata_2_size = json::to_bjdata(j, true).size();
const auto bjdata_3_size = json::to_bjdata(j, true, true).size();
const auto bon8_size = json::to_bon8(j).size();
const auto bson_size = json::to_bson({{"", j}}).size(); // wrap array in object for BSON
const auto cbor_size = json::to_cbor(j).size();
const auto msgpack_size = json::to_msgpack(j).size();
const auto ubjson_1_size = json::to_ubjson(j).size();
const auto ubjson_2_size = json::to_ubjson(j, true).size();
const auto ubjson_3_size = json::to_ubjson(j, true, true).size();
CHECK(json_size == 52508728);
CHECK(bjdata_1_size == 50710965);
CHECK(bjdata_2_size == 51144830);
CHECK(bjdata_3_size == 51144830);
CHECK(bon8_size == 45942080);
CHECK(bson_size == 56008520);
CHECK(cbor_size == 46187320);
CHECK(msgpack_size == 46158575);
CHECK(ubjson_1_size == 50710965);
CHECK(ubjson_2_size == 51144830);
CHECK(ubjson_3_size == 49861422);
CHECK((100.0 * double(json_size) / double(json_size)) == Approx(100.0));
CHECK((100.0 * double(bjdata_1_size) / double(json_size)) == Approx(96.576));
CHECK((100.0 * double(bjdata_2_size) / double(json_size)) == Approx(97.402));
CHECK((100.0 * double(bjdata_3_size) / double(json_size)) == Approx(97.402));
CHECK((100.0 * double(bon8_size) / double(json_size)) == Approx(87.494));
CHECK((100.0 * double(bson_size) / double(json_size)) == Approx(106.665));
CHECK((100.0 * double(cbor_size) / double(json_size)) == Approx(87.961));
CHECK((100.0 * double(msgpack_size) / double(json_size)) == Approx(87.906));
CHECK((100.0 * double(ubjson_1_size) / double(json_size)) == Approx(96.576));
CHECK((100.0 * double(ubjson_2_size) / double(json_size)) == Approx(97.402));
CHECK((100.0 * double(ubjson_3_size) / double(json_size)) == Approx(94.958));
}
+2 -43
View File
@@ -1830,51 +1830,10 @@ TEST_CASE("CBOR")
SECTION("invalid string in map")
{
json _;
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xa1, 0xff, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found a break stop code; last byte: 0xFF", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xa1, 0xff, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xFF", json::parse_error&);
CHECK(json::from_cbor(std::vector<uint8_t>({0xa1, 0xff, 0x01}), true, false).is_discarded());
}
SECTION("non-string key (see #2766 and #3381)")
{
// only text strings map to JSON object keys; any other key is
// rejected with a message naming its type
const std::vector<std::pair<std::vector<std::uint8_t>, std::string>> cases =
{
{{0xA1, 0x01, 0x01}, "an unsigned integer; last byte: 0x01"},
{{0xA1, 0x20, 0x01}, "a negative integer; last byte: 0x20"},
{{0xA1, 0x41, 0x61, 0x01}, "a byte string; last byte: 0x41"},
{{0xA1, 0x80, 0x01}, "an array; last byte: 0x80"},
{{0xA1, 0xA0, 0x01}, "a map; last byte: 0xA0"},
{{0xA1, 0xC0, 0x61, 0x61, 0x01}, "a tag; last byte: 0xC0"},
{{0xA1, 0xF4, 0x01}, "a boolean; last byte: 0xF4"},
{{0xA1, 0xF5, 0x01}, "a boolean; last byte: 0xF5"},
{{0xA1, 0xF6, 0x01}, "null; last byte: 0xF6"},
{{0xA1, 0xF7, 0x01}, "undefined; last byte: 0xF7"},
{{0xA1, 0xF9, 0x3C, 0x00, 0x01}, "a floating-point number; last byte: 0xF9"},
{{0xA1, 0xFA, 0x3F, 0x80, 0x00, 0x00, 0x01}, "a floating-point number; last byte: 0xFA"},
{{0xA1, 0xFB, 0x3F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01}, "a floating-point number; last byte: 0xFB"},
{{0xA1, 0xE0, 0x01}, "a simple value; last byte: 0xE0"},
{{0xA1, 0xF8, 0x20, 0x01}, "a simple value; last byte: 0xF8"},
// indefinite-length map
{{0xBF, 0x01, 0x01, 0xFF}, "an unsigned integer; last byte: 0x01"},
};
for (const auto& c : cases)
{
CAPTURE(c.first)
const std::string expected = "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found " + c.second;
json _;
CHECK_THROWS_WITH_AS(_ = json::from_cbor(c.first), expected.c_str(), json::parse_error&);
CHECK(json::from_cbor(c.first, true, false).is_discarded());
}
// a key of major type 3 with a reserved length is still reported as
// a malformed string, and a missing key as the end of input
json _;
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing CBOR string: unexpected end of input", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&);
}
SECTION("invalid UTF-8 in string (see #5529)")
{
// a two-character text string (major type 3) whose bytes are not
@@ -2325,7 +2284,7 @@ TEST_CASE("CBOR indefinite-length strings do not recurse per chunk")
SECTION("a break marker outside an indefinite-length string is not a string")
{
// 0xFF only closes a string that was opened; on its own it is not one
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0xFF, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found a break stop code; last byte: 0xFF", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0xFF, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xFF", json::parse_error&);
}
}
+1 -1
View File
@@ -666,7 +666,7 @@ TEST_CASE("parse_float_fast declines what it cannot convert exactly")
// always safe: the caller then falls back to a slower, exact conversion.
const auto fast = [](const std::string & s, double & out)
{
return nlohmann::detail::parse_float_fast(s.data(), s.data() + s.size(), out);
return nlohmann::detail::parse_float_fast(s.data(), s.data() + s.size(), '.', out);
};
double out = 0;
-210
View File
@@ -12,12 +12,7 @@
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <array>
#include <clocale>
#include <map>
#include <string>
#include <utility>
#include <vector>
struct ParserImpl final: public nlohmann::json_sax<json>
{
@@ -180,208 +175,3 @@ TEST_CASE("locale-dependent test (LC_NUMERIC=de_DE)")
MESSAGE("locale de_DE is not usable");
}
}
namespace
{
// records the numbers of a flat array and switches LC_NUMERIC to the given
// locale once the array opens - after the lexer was constructed, but before
// any number in the array is lexed
struct LocaleSwitchingSax final: public nlohmann::json_sax<json>
{
explicit LocaleSwitchingSax(const char* switch_to)
: locale_after_open(switch_to)
{}
bool null() override
{
return true;
}
bool boolean(bool /*val*/) override
{
return true;
}
bool number_integer(json::number_integer_t /*val*/) override
{
return true;
}
bool number_unsigned(json::number_unsigned_t /*val*/) override
{
return true;
}
bool number_float(json::number_float_t val, const json::string_t& s) override
{
values.push_back(val);
strings.push_back(s);
return true;
}
bool string(json::string_t& /*val*/) override
{
return true;
}
bool binary(json::binary_t& /*val*/) override
{
return true;
}
bool start_object(std::size_t /*val*/) override
{
return true;
}
bool key(json::string_t& /*val*/) override
{
return true;
}
bool end_object() override
{
return true;
}
bool start_array(std::size_t /*val*/) override
{
switched = std::setlocale(LC_NUMERIC, locale_after_open.c_str()) != nullptr;
return true;
}
bool end_array() override
{
return true;
}
bool parse_error(std::size_t /*val*/, const std::string& /*val*/, const nlohmann::detail::exception& /*val*/) override
{
return false;
}
std::string locale_after_open;
bool switched = false;
std::vector<json::number_float_t> values {}; // NOLINT(readability-redundant-member-init)
std::vector<json::string_t> strings {}; // NOLINT(readability-redundant-member-init)
};
} // namespace
TEST_CASE("locale changes between lexer construction and number conversion (#5198)")
{
// The numbers are chosen so that the conversion also takes the strtod
// fallback, which honors the locale that is current at conversion time:
// too many significant digits for Clinger's fast path, an underflow that
// std::from_chars rejects, and a plain value.
const std::vector<std::string> numbers = {"3.14159265358979323846", "1.5e-400", "12.34", "-0.000123456789012345678"};
std::string text = "[";
for (const auto& n : numbers)
{
text += (text.size() == 1 ? "" : ",") + n;
}
text += "]";
using long_double_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, long double>;
// reference values, parsed without a locale switch
REQUIRE(std::setlocale(LC_NUMERIC, "C") != nullptr);
const json expected = json::parse(text);
const long_double_json expected_ld = long_double_json::parse(text);
const std::array<std::pair<const char*, const char*>, 2> transitions =
{
{
{"C", "de_DE"},
{"de_DE", "C"}
}
};
for (const auto& transition : transitions)
{
CAPTURE(transition.first);
CAPTURE(transition.second);
if (std::setlocale(LC_NUMERIC, transition.first) == nullptr)
{
MESSAGE("locale is not usable");
continue;
}
// SAX parsing
{
LocaleSwitchingSax sax(transition.second);
CHECK(json::sax_parse(text, &sax));
if (sax.switched)
{
CHECK(sax.values == expected.get<std::vector<json::number_float_t>>());
CHECK(sax.strings == numbers);
}
}
// DOM parsing with a callback
{
bool switched = false;
const auto cb = [&](int /*depth*/, json::parse_event_t event, json& /*parsed*/)
{
if (event == json::parse_event_t::array_start)
{
switched = std::setlocale(LC_NUMERIC, transition.second) != nullptr;
}
return true;
};
const json j = json::parse(text, cb);
if (switched)
{
CHECK(j == expected);
}
}
// a long double goes through std::strtold unless std::from_chars supports it
{
bool switched = false;
const auto cb = [&](int /*depth*/, long_double_json::parse_event_t event, long_double_json& /*parsed*/)
{
if (event == long_double_json::parse_event_t::array_start)
{
switched = std::setlocale(LC_NUMERIC, transition.second) != nullptr;
}
return true;
};
const long_double_json j = long_double_json::parse(text, cb);
if (switched)
{
CHECK(j == expected_ld);
}
}
}
std::setlocale(LC_NUMERIC, "C");
}
TEST_CASE("locale with a multi-byte decimal point")
{
// Some locales use a decimal point that is not a single character, e.g.
// U+066B ARABIC DECIMAL SEPARATOR (two bytes in UTF-8). It cannot be
// substituted in place for '.', so the strtod fallback stops early. The
// conversion must still terminate rather than retry forever.
const std::array<const char*, 6> names = {{"ar_EG.UTF-8", "ar_SA.UTF-8", "fa_IR.UTF-8", "ps_AF.UTF-8", "ar_EG", "fa_IR"}};
bool tested = false;
for (const char* name : names)
{
if (std::setlocale(LC_NUMERIC, name) == nullptr)
{
continue;
}
const std::string decimal_point = std::localeconv()->decimal_point;
if (decimal_point.size() < 2)
{
continue;
}
CAPTURE(name);
tested = true;
// too many significant digits for Clinger's fast path, and an underflow
// that std::from_chars rejects: both reach the strtod fallback
json j;
CHECK_NOTHROW(j = json::parse("[3.14159265358979323846, 1.5e-400, -0.000123456789012345678]"));
CHECK(j.is_array());
CHECK(json::accept("3.14159265358979323846"));
// a value the locale-independent paths convert is not affected
CHECK(json::parse("12.5") == 12.5);
}
if (!tested)
{
MESSAGE("no locale with a multi-byte decimal point is usable");
}
std::setlocale(LC_NUMERIC, "C");
}
+99
View File
@@ -0,0 +1,99 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
// This file contains the C++17-only part of unit-msgpack.cpp (std::byte
// input). It is kept in a separate translation unit so the (much larger)
// unit-msgpack.cpp does not need to be compiled and run a second time just
// for this one test case (#5418).
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
using nlohmann::json;
#ifdef JSON_HAS_CPP_17
#include <cstddef>
#include <vector>
// Test suite for verifying MessagePack handling with std::byte input
TEST_CASE("MessagePack with std::byte")
{
SECTION("std::byte compatibility")
{
SECTION("vector roundtrip")
{
json original =
{
{"name", "test"},
{"value", 42},
{"array", {1, 2, 3}}
};
std::vector<uint8_t> temp = json::to_msgpack(original);
// Convert the uint8_t vector to std::byte vector
std::vector<std::byte> msgpack_data(temp.size());
for (size_t i = 0; i < temp.size(); ++i)
{
msgpack_data[i] = std::byte(temp[i]);
}
// Deserialize from std::byte vector back to JSON
json from_bytes;
CHECK_NOTHROW(from_bytes = json::from_msgpack(msgpack_data));
CHECK(from_bytes == original);
}
SECTION("empty vector")
{
const std::vector<std::byte> empty_data;
CHECK_THROWS_WITH_AS([&]()
{
[[maybe_unused]] auto result = json::from_msgpack(empty_data);
return true;
}
(),
"[json.exception.parse_error.110] parse error at byte 1: syntax error while parsing MessagePack value: unexpected end of input",
json::parse_error&);
}
SECTION("comparison with workaround")
{
json original =
{
{"string", "hello"},
{"integer", 42},
{"float", 3.14},
{"boolean", true},
{"null", nullptr},
{"array", {1, 2, 3}},
{"object", {{"key", "value"}}}
};
std::vector<uint8_t> temp = json::to_msgpack(original);
std::vector<std::byte> msgpack_data(temp.size());
for (size_t i = 0; i < temp.size(); ++i)
{
msgpack_data[i] = std::byte(temp[i]);
}
// Attempt direct deserialization using std::byte input
const json direct_result = json::from_msgpack(msgpack_data);
// Test the workaround approach: reinterpret as unsigned char* and use iterator range
const auto* const char_start = reinterpret_cast<unsigned char const*>(msgpack_data.data());
const auto* const char_end = char_start + msgpack_data.size();
json workaround_result = json::from_msgpack(char_start, char_end);
// Verify that the final deserialized JSON matches the original JSON
CHECK(direct_result == workaround_result);
CHECK(direct_result == original);
}
}
}
#endif
+1 -139
View File
@@ -1551,69 +1551,10 @@ TEST_CASE("MessagePack")
SECTION("invalid string in map")
{
json _;
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81, 0xff, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack object key: only string keys are supported, but found an integer; last byte: 0xFF", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81, 0xff, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack string: expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0xFF", json::parse_error&);
CHECK(json::from_msgpack(std::vector<uint8_t>({0x81, 0xff, 0x01}), true, false).is_discarded());
}
SECTION("non-string key (see #3381)")
{
// only strings map to JSON object keys; any other key is rejected
// with a message naming its type
const std::vector<std::pair<std::vector<std::uint8_t>, std::string>> cases =
{
{{0x81, 0xC0, 0x01}, "nil; last byte: 0xC0"},
{{0x81, 0xC2, 0x01}, "a boolean; last byte: 0xC2"},
{{0x81, 0xC3, 0x01}, "a boolean; last byte: 0xC3"},
{{0x81, 0xCA, 0x3F, 0x80, 0x00, 0x00, 0x01}, "a float; last byte: 0xCA"},
{{0x81, 0xCB, 0x3F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01}, "a float; last byte: 0xCB"},
{{0x81, 0xC4, 0x00, 0x01}, "a bin; last byte: 0xC4"},
{{0x81, 0xC5, 0x00, 0x00, 0x01}, "a bin; last byte: 0xC5"},
{{0x81, 0xC6, 0x00, 0x00, 0x00, 0x00, 0x01}, "a bin; last byte: 0xC6"},
{{0x81, 0xC7, 0x00, 0x01, 0x01}, "an ext; last byte: 0xC7"},
{{0x81, 0xC8, 0x00, 0x00, 0x01, 0x01}, "an ext; last byte: 0xC8"},
{{0x81, 0xC9, 0x00, 0x00, 0x00, 0x00, 0x01, 0x01}, "an ext; last byte: 0xC9"},
{{0x81, 0xD4, 0x01, 0x00, 0x01}, "an ext; last byte: 0xD4"},
{{0x81, 0xD5, 0x01, 0x00, 0x00, 0x01}, "an ext; last byte: 0xD5"},
{{0x81, 0xD6, 0x01, 0x00, 0x00, 0x00, 0x00, 0x01}, "an ext; last byte: 0xD6"},
{{0x81, 0xD7, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01}, "an ext; last byte: 0xD7"},
{{0x81, 0xD8, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01}, "an ext; last byte: 0xD8"},
{{0x81, 0xCC, 0x01, 0x01}, "an integer; last byte: 0xCC"},
{{0x81, 0xCD, 0x00, 0x01, 0x01}, "an integer; last byte: 0xCD"},
{{0x81, 0xCE, 0x00, 0x00, 0x00, 0x01, 0x01}, "an integer; last byte: 0xCE"},
{{0x81, 0xCF, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x01}, "an integer; last byte: 0xCF"},
{{0x81, 0xD0, 0x01, 0x01}, "an integer; last byte: 0xD0"},
{{0x81, 0xD1, 0x00, 0x01, 0x01}, "an integer; last byte: 0xD1"},
{{0x81, 0xD2, 0x00, 0x00, 0x00, 0x01, 0x01}, "an integer; last byte: 0xD2"},
{{0x81, 0xD3, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x01}, "an integer; last byte: 0xD3"},
{{0x81, 0x00, 0x01}, "an integer; last byte: 0x00"},
{{0x81, 0x7F, 0x01}, "an integer; last byte: 0x7F"},
{{0x81, 0xE0, 0x01}, "an integer; last byte: 0xE0"},
{{0x81, 0x80, 0x01}, "a map; last byte: 0x80"},
{{0x81, 0x8F, 0x01}, "a map; last byte: 0x8F"},
{{0x81, 0xDE, 0x00, 0x00, 0x01}, "a map; last byte: 0xDE"},
{{0x81, 0xDF, 0x00, 0x00, 0x00, 0x00, 0x01}, "a map; last byte: 0xDF"},
{{0x81, 0x90, 0x01}, "an array; last byte: 0x90"},
{{0x81, 0x9F, 0x01}, "an array; last byte: 0x9F"},
{{0x81, 0xDC, 0x00, 0x00, 0x01}, "an array; last byte: 0xDC"},
{{0x81, 0xDD, 0x00, 0x00, 0x00, 0x00, 0x01}, "an array; last byte: 0xDD"},
};
for (const auto& c : cases)
{
CAPTURE(c.first)
const std::string expected = "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack object key: only string keys are supported, but found " + c.second;
json _;
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(c.first), expected.c_str(), json::parse_error&);
CHECK(json::from_msgpack(c.first, true, false).is_discarded());
}
json _;
// the unused byte 0xC1 is still reported as a malformed string
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81, 0xC1, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing MessagePack string: expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0xC1", json::parse_error&);
// a missing key is still reported as the end of input
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing MessagePack string: unexpected end of input", json::parse_error&);
}
SECTION("invalid UTF-8 in string (see #5529)")
{
// a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8
@@ -2144,85 +2085,6 @@ TEST_CASE("MessagePack roundtrips" * doctest::skip())
}
}
#ifdef JSON_HAS_CPP_17
// Test suite for verifying MessagePack handling with std::byte input
TEST_CASE("MessagePack with std::byte")
{
SECTION("std::byte compatibility")
{
SECTION("vector roundtrip")
{
json original =
{
{"name", "test"},
{"value", 42},
{"array", {1, 2, 3}}
};
std::vector<uint8_t> temp = json::to_msgpack(original);
// Convert the uint8_t vector to std::byte vector
std::vector<std::byte> msgpack_data(temp.size());
for (size_t i = 0; i < temp.size(); ++i)
{
msgpack_data[i] = std::byte(temp[i]);
}
// Deserialize from std::byte vector back to JSON
json from_bytes;
CHECK_NOTHROW(from_bytes = json::from_msgpack(msgpack_data));
CHECK(from_bytes == original);
}
SECTION("empty vector")
{
const std::vector<std::byte> empty_data;
CHECK_THROWS_WITH_AS([&]()
{
[[maybe_unused]] auto result = json::from_msgpack(empty_data);
return true;
}
(),
"[json.exception.parse_error.110] parse error at byte 1: syntax error while parsing MessagePack value: unexpected end of input",
json::parse_error&);
}
SECTION("comparison with workaround")
{
json original =
{
{"string", "hello"},
{"integer", 42},
{"float", 3.14},
{"boolean", true},
{"null", nullptr},
{"array", {1, 2, 3}},
{"object", {{"key", "value"}}}
};
std::vector<uint8_t> temp = json::to_msgpack(original);
std::vector<std::byte> msgpack_data(temp.size());
for (size_t i = 0; i < temp.size(); ++i)
{
msgpack_data[i] = std::byte(temp[i]);
}
// Attempt direct deserialization using std::byte input
const json direct_result = json::from_msgpack(msgpack_data);
// Test the workaround approach: reinterpret as unsigned char* and use iterator range
const auto* const char_start = reinterpret_cast<unsigned char const*>(msgpack_data.data());
const auto* const char_end = char_start + msgpack_data.size();
json workaround_result = json::from_msgpack(char_start, char_end);
// Verify that the final deserialized JSON matches the original JSON
CHECK(direct_result == workaround_result);
CHECK(direct_result == original);
}
}
}
#endif
// the fake sizes below do not fit into a 32-bit std::size_t
// with clang and libstdc++ 10, the std::filesystem::path conversion that
// C++17 builds consider for every string type is ambiguous for a class
+3 -3
View File
@@ -1018,7 +1018,7 @@ TEST_CASE("regression tests 1")
};
json _;
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR object key: only string keys are supported, but found an array; last byte: 0x98", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x98", json::parse_error&);
// related test case: nonempty UTF-8 string (indefinite length)
std::vector<uint8_t> const vec1 {0x7f, 0x61, 0x61};
@@ -1065,7 +1065,7 @@ TEST_CASE("regression tests 1")
};
json _;
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec1), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing CBOR object key: only string keys are supported, but found a map; last byte: 0xB4", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec1), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xB4", json::parse_error&);
// related test case: double-precision
std::vector<uint8_t> const vec2
@@ -1077,7 +1077,7 @@ TEST_CASE("regression tests 1")
0x96, 0x96, 0xb4, 0xb4, 0xfa, 0x94, 0x94, 0x61,
0x61, 0x61, 0x61, 0x61, 0x61, 0x61, 0x61, 0xfb
};
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec2), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing CBOR object key: only string keys are supported, but found a map; last byte: 0xB4", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vec2), "[json.exception.parse_error.113] parse error at byte 13: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0xB4", json::parse_error&);
}
SECTION("issue #452 - Heap-buffer-overflow (OSS-Fuzz issue 585)")
File diff suppressed because it is too large Load Diff
-623
View File
@@ -1,623 +0,0 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
// for some reason including this after the json header leads to linker errors with VS 2017...
#include <locale>
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <fstream>
#include <sstream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
TEST_CASE("Unicode (1/5)" * doctest::skip())
{
SECTION("\\uxxxx sequences")
{
// create an escaped string from a code point
const auto codepoint_to_unicode = [](std::size_t cp)
{
// code points are represented as a six-character sequence: a
// reverse solidus, followed by the lowercase letter u, followed
// by four hexadecimal digits that encode the character's code
// point
std::stringstream ss;
ss << "\\u" << std::setw(4) << std::setfill('0') << std::hex << cp;
return ss.str();
};
SECTION("correct sequences")
{
// generate all UTF-8 code points; in total, 1112064 code points are
// generated: 0x1FFFFF code points - 2048 invalid values between
// 0xD800 and 0xDFFF.
for (std::size_t cp = 0; cp <= 0x10FFFFu; ++cp)
{
// string to store the code point as in \uxxxx format
std::string json_text = "\"";
// decide whether to use one or two \uxxxx sequences
if (cp < 0x10000u)
{
// The Unicode standard permanently reserves these code point
// values for UTF-16 encoding of the high and low surrogates, and
// they will never be assigned a character, so there should be no
// reason to encode them. The official Unicode standard says that
// no UTF forms, including UTF-16, can encode these code points.
if (cp >= 0xD800u && cp <= 0xDFFFu)
{
// if we would not skip these code points, we would get a
// "missing low surrogate" exception
continue;
}
// code points in the Basic Multilingual Plane can be
// represented with one \uxxxx sequence
json_text += codepoint_to_unicode(cp);
}
else
{
// To escape an extended character that is not in the Basic
// Multilingual Plane, the character is represented as a
// 12-character sequence, encoding the UTF-16 surrogate pair
const auto codepoint1 = 0xd800u + (((cp - 0x10000u) >> 10) & 0x3ffu);
const auto codepoint2 = 0xdc00u + ((cp - 0x10000u) & 0x3ffu);
json_text += codepoint_to_unicode(codepoint1) + codepoint_to_unicode(codepoint2);
}
json_text += "\"";
CAPTURE(json_text)
json _;
CHECK_NOTHROW(_ = json::parse(json_text));
}
}
SECTION("incorrect sequences")
{
SECTION("incorrect surrogate values")
{
json _;
CHECK_THROWS_WITH_AS(_ = json::parse("\"\\uDC00\\uDC00\""), "[json.exception.parse_error.101] parse error at line 1, column 7: syntax error while parsing value - invalid string: surrogate U+DC00..U+DFFF must follow U+D800..U+DBFF; last read: '\"\\uDC00'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse("\"\\uD7FF\\uDC00\""), "[json.exception.parse_error.101] parse error at line 1, column 13: syntax error while parsing value - invalid string: surrogate U+DC00..U+DFFF must follow U+D800..U+DBFF; last read: '\"\\uD7FF\\uDC00'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse("\"\\uD800]\""), "[json.exception.parse_error.101] parse error at line 1, column 8: syntax error while parsing value - invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF; last read: '\"\\uD800]'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse("\"\\uD800\\v\""), "[json.exception.parse_error.101] parse error at line 1, column 9: syntax error while parsing value - invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF; last read: '\"\\uD800\\v'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse("\"\\uD800\\u123\""), "[json.exception.parse_error.101] parse error at line 1, column 13: syntax error while parsing value - invalid string: '\\u' must be followed by 4 hex digits; last read: '\"\\uD800\\u123\"'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse("\"\\uD800\\uDBFF\""), "[json.exception.parse_error.101] parse error at line 1, column 13: syntax error while parsing value - invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF; last read: '\"\\uD800\\uDBFF'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse("\"\\uD800\\uE000\""), "[json.exception.parse_error.101] parse error at line 1, column 13: syntax error while parsing value - invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF; last read: '\"\\uD800\\uE000'", json::parse_error&);
}
}
#if 0 // NOLINT(readability-avoid-unconditional-preprocessor-if)
SECTION("incorrect sequences")
{
SECTION("high surrogate without low surrogate")
{
// D800..DBFF are high surrogates and must be followed by low
// surrogates DC00..DFFF; here, nothing follows
for (std::size_t cp = 0xD800u; cp <= 0xDBFFu; ++cp)
{
std::string json_text = "\"" + codepoint_to_unicode(cp) + "\"";
CAPTURE(json_text)
CHECK_THROWS_AS(json::parse(json_text), json::parse_error&);
}
}
SECTION("high surrogate with wrong low surrogate")
{
// D800..DBFF are high surrogates and must be followed by low
// surrogates DC00..DFFF; here a different sequence follows
for (std::size_t cp1 = 0xD800u; cp1 <= 0xDBFFu; ++cp1)
{
for (std::size_t cp2 = 0x0000u; cp2 <= 0xFFFFu; ++cp2)
{
if (0xDC00u <= cp2 && cp2 <= 0xDFFFu)
{
continue;
}
std::string json_text = "\"" + codepoint_to_unicode(cp1) + codepoint_to_unicode(cp2) + "\"";
CAPTURE(json_text)
CHECK_THROWS_AS(json::parse(json_text), json::parse_error&);
}
}
}
SECTION("low surrogate without high surrogate")
{
// low surrogates DC00..DFFF must follow high surrogates; here,
// they occur alone
for (std::size_t cp = 0xDC00u; cp <= 0xDFFFu; ++cp)
{
std::string json_text = "\"" + codepoint_to_unicode(cp) + "\"";
CAPTURE(json_text)
CHECK_THROWS_AS(json::parse(json_text), json::parse_error&);
}
}
}
#endif
}
SECTION("read all unicode characters")
{
// read a file with all Unicode characters stored as single-character
// strings in a JSON array
std::ifstream f(TEST_DATA_DIRECTORY "/json_nlohmann_tests/all_unicode.json");
json j;
CHECK_NOTHROW(f >> j);
// the array has 1112064 + 1 elements (a terminating "null" value)
// Note: 1112064 = 0x1FFFFF code points - 2048 invalid values between
// 0xD800 and 0xDFFF.
CHECK(j.size() == 1112065);
SECTION("check JSON Pointers")
{
for (const auto& s : j)
{
// skip non-string JSON values
if (!s.is_string())
{
continue;
}
auto ptr = s.get<std::string>();
// tilde must be followed by 0 or 1
if (ptr == "~")
{
ptr += "0";
}
// JSON Pointers must begin with "/"
ptr.insert(0, "/");
CHECK_NOTHROW(json::json_pointer("/" + ptr));
// check escape/unescape roundtrip
auto escaped = nlohmann::detail::escape(ptr);
nlohmann::detail::unescape(escaped);
CHECK(escaped == ptr);
}
}
}
SECTION("ignore byte-order-mark")
{
SECTION("in a stream")
{
// read a file with a UTF-8 BOM
std::ifstream f(TEST_DATA_DIRECTORY "/json_nlohmann_tests/bom.json");
json j;
CHECK_NOTHROW(f >> j);
}
SECTION("with an iterator")
{
std::string i = "\xef\xbb\xbf{\n \"foo\": true\n}";
json _;
CHECK_NOTHROW(_ = json::parse(i.begin(), i.end()));
}
}
SECTION("error for incomplete/wrong BOM")
{
json _;
CHECK_THROWS_AS(_ = json::parse("\xef\xbb"), json::parse_error&);
CHECK_THROWS_AS(_ = json::parse("\xef\xbb\xbb"), json::parse_error&);
}
}
namespace
{
void roundtrip(bool success_expected, const std::string& s);
void roundtrip(bool success_expected, const std::string& s)
{
CAPTURE(s)
json _;
// create JSON string value
const json j = s;
// create JSON text
const std::string ps = std::string("\"") + s + "\"";
if (success_expected)
{
// serialization succeeds
// dump() is nodiscard; this only checks that dumping does not throw
CHECK_NOTHROW(utils::ignore_return_value(j.dump()));
// exclude parse test for U+0000
if (s[0] != '\0')
{
// parsing JSON text succeeds
CHECK_NOTHROW(_ = json::parse(ps));
}
// roundtrip succeeds
CHECK_NOTHROW(_ = json::parse(j.dump()));
// after roundtrip, the same string is stored
const json jr = json::parse(j.dump());
CHECK(jr.get<std::string>() == s);
}
else
{
// serialization fails
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// parsing JSON text fails
CHECK_THROWS_AS(_ = json::parse(ps), json::parse_error&);
}
}
} // namespace
TEST_CASE("Markus Kuhn's UTF-8 decoder capability and stress test")
{
// Markus Kuhn <http://www.cl.cam.ac.uk/~mgk25/> - 2015-08-28 - CC BY 4.0
// http://www.cl.cam.ac.uk/~mgk25/ucs/examples/UTF-8-test.txt
SECTION("1 Some correct UTF-8 text")
{
roundtrip(true, "κόσμε");
}
SECTION("2 Boundary condition test cases")
{
SECTION("2.1 First possible sequence of a certain length")
{
// 2.1.1 1 byte (U-00000000)
roundtrip(true, std::string("\0", 1));
// 2.1.2 2 bytes (U-00000080)
roundtrip(true, "\xc2\x80");
// 2.1.3 3 bytes (U-00000800)
roundtrip(true, "\xe0\xa0\x80");
// 2.1.4 4 bytes (U-00010000)
roundtrip(true, "\xf0\x90\x80\x80");
// 2.1.5 5 bytes (U-00200000)
roundtrip(false, "\xF8\x88\x80\x80\x80");
// 2.1.6 6 bytes (U-04000000)
roundtrip(false, "\xFC\x84\x80\x80\x80\x80");
}
SECTION("2.2 Last possible sequence of a certain length")
{
// 2.2.1 1 byte (U-0000007F)
roundtrip(true, "\x7f");
// 2.2.2 2 bytes (U-000007FF)
roundtrip(true, "\xdf\xbf");
// 2.2.3 3 bytes (U-0000FFFF)
roundtrip(true, "\xef\xbf\xbf");
// 2.2.4 4 bytes (U-001FFFFF)
roundtrip(false, "\xF7\xBF\xBF\xBF");
// 2.2.5 5 bytes (U-03FFFFFF)
roundtrip(false, "\xFB\xBF\xBF\xBF\xBF");
// 2.2.6 6 bytes (U-7FFFFFFF)
roundtrip(false, "\xFD\xBF\xBF\xBF\xBF\xBF");
}
SECTION("2.3 Other boundary conditions")
{
// 2.3.1 U-0000D7FF = ed 9f bf
roundtrip(true, "\xed\x9f\xbf");
// 2.3.2 U-0000E000 = ee 80 80
roundtrip(true, "\xee\x80\x80");
// 2.3.3 U-0000FFFD = ef bf bd
roundtrip(true, "\xef\xbf\xbd");
// 2.3.4 U-0010FFFF = f4 8f bf bf
roundtrip(true, "\xf4\x8f\xbf\xbf");
// 2.3.5 U-00110000 = f4 90 80 80
roundtrip(false, "\xf4\x90\x80\x80");
}
}
SECTION("3 Malformed sequences")
{
SECTION("3.1 Unexpected continuation bytes")
{
// Each unexpected continuation byte should be separately signalled as a
// malformed sequence of its own.
// 3.1.1 First continuation byte 0x80
roundtrip(false, "\x80");
// 3.1.2 Last continuation byte 0xbf
roundtrip(false, "\xbf");
// 3.1.3 2 continuation bytes
roundtrip(false, "\x80\xbf");
// 3.1.4 3 continuation bytes
roundtrip(false, "\x80\xbf\x80");
// 3.1.5 4 continuation bytes
roundtrip(false, "\x80\xbf\x80\xbf");
// 3.1.6 5 continuation bytes
roundtrip(false, "\x80\xbf\x80\xbf\x80");
// 3.1.7 6 continuation bytes
roundtrip(false, "\x80\xbf\x80\xbf\x80\xbf");
// 3.1.8 7 continuation bytes
roundtrip(false, "\x80\xbf\x80\xbf\x80\xbf\x80");
// 3.1.9 Sequence of all 64 possible continuation bytes (0x80-0xbf)
roundtrip(false, "\x80\x81\x82\x83\x84\x85\x86\x87\x88\x89\x8a\x8b\x8c\x8d\x8e\x8f\x90\x91\x92\x93\x94\x95\x96\x97\x98\x99\x9a\x9b\x9c\x9d\x9e\x9f\xa0\xa1\xa2\xa3\xa4\xa5\xa6\xa7\xa8\xa9\xaa\xab\xac\xad\xae\xaf\xb0\xb1\xb2\xb3\xb4\xb5\xb6\xb7\xb8\xb9\xba\xbb\xbc\xbd\xbe\xbf");
}
SECTION("3.2 Lonely start characters")
{
// 3.2.1 All 32 first bytes of 2-byte sequences (0xc0-0xdf)
roundtrip(false, "\xc0 \xc1 \xc2 \xc3 \xc4 \xc5 \xc6 \xc7 \xc8 \xc9 \xca \xcb \xcc \xcd \xce \xcf \xd0 \xd1 \xd2 \xd3 \xd4 \xd5 \xd6 \xd7 \xd8 \xd9 \xda \xdb \xdc \xdd \xde \xdf");
// 3.2.2 All 16 first bytes of 3-byte sequences (0xe0-0xef)
roundtrip(false, "\xe0 \xe1 \xe2 \xe3 \xe4 \xe5 \xe6 \xe7 \xe8 \xe9 \xea \xeb \xec \xed \xee \xef");
// 3.2.3 All 8 first bytes of 4-byte sequences (0xf0-0xf7)
roundtrip(false, "\xf0 \xf1 \xf2 \xf3 \xf4 \xf5 \xf6 \xf7");
// 3.2.4 All 4 first bytes of 5-byte sequences (0xf8-0xfb)
roundtrip(false, "\xf8 \xf9 \xfa \xfb");
// 3.2.5 All 2 first bytes of 6-byte sequences (0xfc-0xfd)
roundtrip(false, "\xfc \xfd");
}
SECTION("3.3 Sequences with last continuation byte missing")
{
// All bytes of an incomplete sequence should be signalled as a single
// malformed sequence, i.e., you should see only a single replacement
// character in each of the next 10 tests. (Characters as in section 2)
// 3.3.1 2-byte sequence with last byte missing (U+0000)
roundtrip(false, "\xc0");
// 3.3.2 3-byte sequence with last byte missing (U+0000)
roundtrip(false, "\xe0\x80");
// 3.3.3 4-byte sequence with last byte missing (U+0000)
roundtrip(false, "\xf0\x80\x80");
// 3.3.4 5-byte sequence with last byte missing (U+0000)
roundtrip(false, "\xf8\x80\x80\x80");
// 3.3.5 6-byte sequence with last byte missing (U+0000)
roundtrip(false, "\xfc\x80\x80\x80\x80");
// 3.3.6 2-byte sequence with last byte missing (U-000007FF)
roundtrip(false, "\xdf");
// 3.3.7 3-byte sequence with last byte missing (U-0000FFFF)
roundtrip(false, "\xef\xbf");
// 3.3.8 4-byte sequence with last byte missing (U-001FFFFF)
roundtrip(false, "\xf7\xbf\xbf");
// 3.3.9 5-byte sequence with last byte missing (U-03FFFFFF)
roundtrip(false, "\xfb\xbf\xbf\xbf");
// 3.3.10 6-byte sequence with last byte missing (U-7FFFFFFF)
roundtrip(false, "\xfd\xbf\xbf\xbf\xbf");
}
SECTION("3.4 Concatenation of incomplete sequences")
{
// All the 10 sequences of 3.3 concatenated, you should see 10 malformed
// sequences being signalled:
roundtrip(false, "\xc0\xe0\x80\xf0\x80\x80\xf8\x80\x80\x80\xfc\x80\x80\x80\x80\xdf\xef\xbf\xf7\xbf\xbf\xfb\xbf\xbf\xbf\xfd\xbf\xbf\xbf\xbf");
}
SECTION("3.5 Impossible bytes")
{
// The following two bytes cannot appear in a correct UTF-8 string
// 3.5.1 fe
roundtrip(false, "\xfe");
// 3.5.2 ff
roundtrip(false, "\xff");
// 3.5.3 fe fe ff ff
roundtrip(false, "\xfe\xfe\xff\xff");
}
}
SECTION("4 Overlong sequences")
{
// The following sequences are not malformed according to the letter of
// the Unicode 2.0 standard. However, they are longer then necessary and
// a correct UTF-8 encoder is not allowed to produce them. A "safe UTF-8
// decoder" should reject them just like malformed sequences for two
// reasons: (1) It helps to debug applications if overlong sequences are
// not treated as valid representations of characters, because this helps
// to spot problems more quickly. (2) Overlong sequences provide
// alternative representations of characters, that could maliciously be
// used to bypass filters that check only for ASCII characters. For
// instance, a 2-byte encoded line feed (LF) would not be caught by a
// line counter that counts only 0x0a bytes, but it would still be
// processed as a line feed by an unsafe UTF-8 decoder later in the
// pipeline. From a security point of view, ASCII compatibility of UTF-8
// sequences means also, that ASCII characters are *only* allowed to be
// represented by ASCII bytes in the range 0x00-0x7f. To ensure this
// aspect of ASCII compatibility, use only "safe UTF-8 decoders" that
// reject overlong UTF-8 sequences for which a shorter encoding exists.
SECTION("4.1 Examples of an overlong ASCII character")
{
// With a safe UTF-8 decoder, all the following five overlong
// representations of the ASCII character slash ("/") should be rejected
// like a malformed UTF-8 sequence, for instance by substituting it with
// a replacement character. If you see a slash below, you do not have a
// safe UTF-8 decoder!
// 4.1.1 U+002F = c0 af
roundtrip(false, "\xc0\xaf");
// 4.1.2 U+002F = e0 80 af
roundtrip(false, "\xe0\x80\xaf");
// 4.1.3 U+002F = f0 80 80 af
roundtrip(false, "\xf0\x80\x80\xaf");
// 4.1.4 U+002F = f8 80 80 80 af
roundtrip(false, "\xf8\x80\x80\x80\xaf");
// 4.1.5 U+002F = fc 80 80 80 80 af
roundtrip(false, "\xfc\x80\x80\x80\x80\xaf");
}
SECTION("4.2 Maximum overlong sequences")
{
// Below you see the highest Unicode value that is still resulting in an
// overlong sequence if represented with the given number of bytes. This
// is a boundary test for safe UTF-8 decoders. All five characters should
// be rejected like malformed UTF-8 sequences.
// 4.2.1 U-0000007F = c1 bf
roundtrip(false, "\xc1\xbf");
// 4.2.2 U-000007FF = e0 9f bf
roundtrip(false, "\xe0\x9f\xbf");
// 4.2.3 U-0000FFFF = f0 8f bf bf
roundtrip(false, "\xf0\x8f\xbf\xbf");
// 4.2.4 U-001FFFFF = f8 87 bf bf bf
roundtrip(false, "\xf8\x87\xbf\xbf\xbf");
// 4.2.5 U-03FFFFFF = fc 83 bf bf bf bf
roundtrip(false, "\xfc\x83\xbf\xbf\xbf\xbf");
}
SECTION("4.3 Overlong representation of the NUL character")
{
// The following five sequences should also be rejected like malformed
// UTF-8 sequences and should not be treated like the ASCII NUL
// character.
// 4.3.1 U+0000 = c0 80
roundtrip(false, "\xc0\x80");
// 4.3.2 U+0000 = e0 80 80
roundtrip(false, "\xe0\x80\x80");
// 4.3.3 U+0000 = f0 80 80 80
roundtrip(false, "\xf0\x80\x80\x80");
// 4.3.4 U+0000 = f8 80 80 80 80
roundtrip(false, "\xf8\x80\x80\x80\x80");
// 4.3.5 U+0000 = fc 80 80 80 80 80
roundtrip(false, "\xfc\x80\x80\x80\x80\x80");
}
}
SECTION("5 Illegal code positions")
{
// The following UTF-8 sequences should be rejected like malformed
// sequences, because they never represent valid ISO 10646 characters and
// a UTF-8 decoder that accepts them might introduce security problems
// comparable to overlong UTF-8 sequences.
SECTION("5.1 Single UTF-16 surrogates")
{
// 5.1.1 U+D800 = ed a0 80
roundtrip(false, "\xed\xa0\x80");
// 5.1.2 U+DB7F = ed ad bf
roundtrip(false, "\xed\xad\xbf");
// 5.1.3 U+DB80 = ed ae 80
roundtrip(false, "\xed\xae\x80");
// 5.1.4 U+DBFF = ed af bf
roundtrip(false, "\xed\xaf\xbf");
// 5.1.5 U+DC00 = ed b0 80
roundtrip(false, "\xed\xb0\x80");
// 5.1.6 U+DF80 = ed be 80
roundtrip(false, "\xed\xbe\x80");
// 5.1.7 U+DFFF = ed bf bf
roundtrip(false, "\xed\xbf\xbf");
}
SECTION("5.2 Paired UTF-16 surrogates")
{
// 5.2.1 U+D800 U+DC00 = ed a0 80 ed b0 80
roundtrip(false, "\xed\xa0\x80\xed\xb0\x80");
// 5.2.2 U+D800 U+DFFF = ed a0 80 ed bf bf
roundtrip(false, "\xed\xa0\x80\xed\xbf\xbf");
// 5.2.3 U+DB7F U+DC00 = ed ad bf ed b0 80
roundtrip(false, "\xed\xad\xbf\xed\xb0\x80");
// 5.2.4 U+DB7F U+DFFF = ed ad bf ed bf bf
roundtrip(false, "\xed\xad\xbf\xed\xbf\xbf");
// 5.2.5 U+DB80 U+DC00 = ed ae 80 ed b0 80
roundtrip(false, "\xed\xae\x80\xed\xb0\x80");
// 5.2.6 U+DB80 U+DFFF = ed ae 80 ed bf bf
roundtrip(false, "\xed\xae\x80\xed\xbf\xbf");
// 5.2.7 U+DBFF U+DC00 = ed af bf ed b0 80
roundtrip(false, "\xed\xaf\xbf\xed\xb0\x80");
// 5.2.8 U+DBFF U+DFFF = ed af bf ed bf bf
roundtrip(false, "\xed\xaf\xbf\xed\xbf\xbf");
}
SECTION("5.3 Noncharacter code positions")
{
// The following "noncharacters" are "reserved for internal use" by
// applications, and according to older versions of the Unicode Standard
// "should never be interchanged". Unicode Corrigendum #9 dropped the
// latter restriction. Nevertheless, their presence in incoming UTF-8 data
// can remain a potential security risk, depending on what use is made of
// these codes subsequently. Examples of such internal use:
//
// - Some file APIs with 16-bit characters may use the integer value -1
// = U+FFFF to signal an end-of-file (EOF) or error condition.
//
// - In some UTF-16 receivers, code point U+FFFE might trigger a
// byte-swap operation (to convert between UTF-16LE and UTF-16BE).
//
// With such internal use of noncharacters, it may be desirable and safer
// to block those code points in UTF-8 decoders, as they should never
// occur legitimately in incoming UTF-8 data, and could trigger unsafe
// behaviour in subsequent processing.
// Particularly problematic noncharacters in 16-bit applications:
// 5.3.1 U+FFFE = ef bf be
roundtrip(true, "\xef\xbf\xbe");
// 5.3.2 U+FFFF = ef bf bf
roundtrip(true, "\xef\xbf\xbf");
// 5.3.3 U+FDD0 .. U+FDEF
roundtrip(true, "\xEF\xB7\x90");
roundtrip(true, "\xEF\xB7\x91");
roundtrip(true, "\xEF\xB7\x92");
roundtrip(true, "\xEF\xB7\x93");
roundtrip(true, "\xEF\xB7\x94");
roundtrip(true, "\xEF\xB7\x95");
roundtrip(true, "\xEF\xB7\x96");
roundtrip(true, "\xEF\xB7\x97");
roundtrip(true, "\xEF\xB7\x98");
roundtrip(true, "\xEF\xB7\x99");
roundtrip(true, "\xEF\xB7\x9A");
roundtrip(true, "\xEF\xB7\x9B");
roundtrip(true, "\xEF\xB7\x9C");
roundtrip(true, "\xEF\xB7\x9D");
roundtrip(true, "\xEF\xB7\x9E");
roundtrip(true, "\xEF\xB7\x9F");
roundtrip(true, "\xEF\xB7\xA0");
roundtrip(true, "\xEF\xB7\xA1");
roundtrip(true, "\xEF\xB7\xA2");
roundtrip(true, "\xEF\xB7\xA3");
roundtrip(true, "\xEF\xB7\xA4");
roundtrip(true, "\xEF\xB7\xA5");
roundtrip(true, "\xEF\xB7\xA6");
roundtrip(true, "\xEF\xB7\xA7");
roundtrip(true, "\xEF\xB7\xA8");
roundtrip(true, "\xEF\xB7\xA9");
roundtrip(true, "\xEF\xB7\xAA");
roundtrip(true, "\xEF\xB7\xAB");
roundtrip(true, "\xEF\xB7\xAC");
roundtrip(true, "\xEF\xB7\xAD");
roundtrip(true, "\xEF\xB7\xAE");
roundtrip(true, "\xEF\xB7\xAF");
// 5.3.4 U+nFFFE U+nFFFF (for n = 1..10)
roundtrip(true, "\xF0\x9F\xBF\xBF");
roundtrip(true, "\xF0\xAF\xBF\xBF");
roundtrip(true, "\xF0\xBF\xBF\xBF");
roundtrip(true, "\xF1\x8F\xBF\xBF");
roundtrip(true, "\xF1\x9F\xBF\xBF");
roundtrip(true, "\xF1\xAF\xBF\xBF");
roundtrip(true, "\xF1\xBF\xBF\xBF");
roundtrip(true, "\xF2\x8F\xBF\xBF");
roundtrip(true, "\xF2\x9F\xBF\xBF");
roundtrip(true, "\xF2\xAF\xBF\xBF");
}
}
}
-612
View File
@@ -1,612 +0,0 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
// for some reason including this after the json header leads to linker errors with VS 2017...
#include <locale>
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <fstream>
#include <sstream>
#include <iostream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
// this test suite uses static variables with non-trivial destructors
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
DOCTEST_CLANG_SUPPRESS_WARNING("-Wexit-time-destructors")
namespace
{
extern size_t calls;
size_t calls = 0;
void check_utf8dump(bool success_expected, int byte1, int byte2, int byte3, int byte4);
void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 = -1, int byte4 = -1)
{
static std::string json_string;
json_string.clear();
CAPTURE(byte1)
CAPTURE(byte2)
CAPTURE(byte3)
CAPTURE(byte4)
json_string += std::string(1, static_cast<char>(byte1));
if (byte2 != -1)
{
json_string += std::string(1, static_cast<char>(byte2));
}
if (byte3 != -1)
{
json_string += std::string(1, static_cast<char>(byte3));
}
if (byte4 != -1)
{
json_string += std::string(1, static_cast<char>(byte4));
}
CAPTURE(json_string)
// store the string in a JSON value
static json j;
static json j2;
j = json_string;
j2 = "abc" + json_string + "xyz";
static std::string s_ignored;
static std::string s_ignored2;
static std::string s_ignored_ascii;
static std::string s_ignored2_ascii;
static std::string s_replaced;
static std::string s_replaced2;
static std::string s_replaced_ascii;
static std::string s_replaced2_ascii;
// dumping with ignore/replace must not throw in any case
s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore);
s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore);
s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore);
s_ignored2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::ignore);
s_replaced = j.dump(-1, ' ', false, json::error_handler_t::replace);
s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace);
s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace);
s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace);
if (success_expected)
{
static std::string s_strict;
// strict mode must not throw if success is expected
s_strict = j.dump();
// all dumps should agree on the string
CHECK(s_strict == s_ignored);
CHECK(s_strict == s_replaced);
}
else
{
// strict mode must throw if success is not expected
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// ignore and replace must create different dumps
CHECK(s_ignored != s_replaced);
// check that replace string contains a replacement character
CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos);
}
// check that prefix and suffix are preserved
CHECK(s_ignored2.substr(1, 3) == "abc");
CHECK(s_ignored2.substr(s_ignored2.size() - 4, 3) == "xyz");
CHECK(s_ignored2_ascii.substr(1, 3) == "abc");
CHECK(s_ignored2_ascii.substr(s_ignored2_ascii.size() - 4, 3) == "xyz");
CHECK(s_replaced2.substr(1, 3) == "abc");
CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz");
CHECK(s_replaced2_ascii.substr(1, 3) == "abc");
CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz");
}
void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);
// create and check a JSON string with up to four UTF-8 bytes
void check_utf8string(bool success_expected, int byte1, int byte2 = -1, int byte3 = -1, int byte4 = -1)
{
if (++calls % 100000 == 0)
{
std::cout << calls << " of 455355 UTF-8 strings checked" << std::endl; // NOLINT(performance-avoid-endl)
}
static std::string json_string;
json_string = "\"";
CAPTURE(byte1)
json_string += std::string(1, static_cast<char>(byte1));
if (byte2 != -1)
{
CAPTURE(byte2)
json_string += std::string(1, static_cast<char>(byte2));
}
if (byte3 != -1)
{
CAPTURE(byte3)
json_string += std::string(1, static_cast<char>(byte3));
}
if (byte4 != -1)
{
CAPTURE(byte4)
json_string += std::string(1, static_cast<char>(byte4));
}
json_string += "\"";
CAPTURE(json_string)
json _;
if (success_expected)
{
CHECK_NOTHROW(_ = json::parse(json_string));
}
else
{
CHECK_THROWS_AS(_ = json::parse(json_string), json::parse_error&);
}
}
} // namespace
TEST_CASE("Unicode (2/5)" * doctest::skip())
{
SECTION("RFC 3629")
{
/*
RFC 3629 describes in Sect. 4 the syntax of UTF-8 byte sequences as
follows:
A UTF-8 string is a sequence of octets representing a sequence of UCS
characters. An octet sequence is valid UTF-8 only if it matches the
following syntax, which is derived from the rules for encoding UTF-8
and is expressed in the ABNF of [RFC2234].
UTF8-octets = *( UTF8-char )
UTF8-char = UTF8-1 / UTF8-2 / UTF8-3 / UTF8-4
UTF8-1 = %x00-7F
UTF8-2 = %xC2-DF UTF8-tail
UTF8-3 = %xE0 %xA0-BF UTF8-tail / %xE1-EC 2( UTF8-tail ) /
%xED %x80-9F UTF8-tail / %xEE-EF 2( UTF8-tail )
UTF8-4 = %xF0 %x90-BF 2( UTF8-tail ) / %xF1-F3 3( UTF8-tail ) /
%xF4 %x80-8F 2( UTF8-tail )
UTF8-tail = %x80-BF
*/
SECTION("ill-formed first byte")
{
for (int byte1 = 0x80; byte1 <= 0xC1; ++byte1)
{
check_utf8string(false, byte1);
check_utf8dump(false, byte1);
}
for (int byte1 = 0xF5; byte1 <= 0xFF; ++byte1)
{
check_utf8string(false, byte1);
check_utf8dump(false, byte1);
}
}
SECTION("UTF8-1 (x00-x7F)")
{
SECTION("well-formed")
{
for (int byte1 = 0x00; byte1 <= 0x7F; ++byte1)
{
// unescaped control characters are parse errors in JSON
if (0x00 <= byte1 && byte1 <= 0x1F)
{
check_utf8string(false, byte1);
continue;
}
// a single quote is a parse error in JSON
if (byte1 == 0x22)
{
check_utf8string(false, byte1);
continue;
}
// a single backslash is a parse error in JSON
if (byte1 == 0x5C)
{
check_utf8string(false, byte1);
continue;
}
// all other characters are OK
check_utf8string(true, byte1);
check_utf8dump(true, byte1);
}
}
}
SECTION("UTF8-2 (xC2-xDF UTF8-tail)")
{
SECTION("well-formed")
{
for (int byte1 = 0xC2; byte1 <= 0xDF; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
{
check_utf8string(true, byte1, byte2);
check_utf8dump(true, byte1, byte2);
}
}
}
SECTION("ill-formed: missing second byte")
{
for (int byte1 = 0xC2; byte1 <= 0xDF; ++byte1)
{
check_utf8string(false, byte1);
check_utf8dump(false, byte1);
}
}
SECTION("ill-formed: wrong second byte")
{
for (int byte1 = 0xC2; byte1 <= 0xDF; ++byte1)
{
for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2)
{
// skip correct second byte
if (0x80 <= byte2 && byte2 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2);
check_utf8dump(false, byte1, byte2);
}
}
}
}
SECTION("UTF8-3 (xE0 xA0-BF UTF8-tail)")
{
SECTION("well-formed")
{
for (int byte1 = 0xE0; byte1 <= 0xE0; ++byte1)
{
for (int byte2 = 0xA0; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(true, byte1, byte2, byte3);
check_utf8dump(true, byte1, byte2, byte3);
}
}
}
}
SECTION("ill-formed: missing second byte")
{
for (int byte1 = 0xE0; byte1 <= 0xE0; ++byte1)
{
check_utf8string(false, byte1);
check_utf8dump(false, byte1);
}
}
SECTION("ill-formed: missing third byte")
{
for (int byte1 = 0xE0; byte1 <= 0xE0; ++byte1)
{
for (int byte2 = 0xA0; byte2 <= 0xBF; ++byte2)
{
check_utf8string(false, byte1, byte2);
check_utf8dump(false, byte1, byte2);
}
}
}
SECTION("ill-formed: wrong second byte")
{
for (int byte1 = 0xE0; byte1 <= 0xE0; ++byte1)
{
for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2)
{
// skip correct second byte
if (0xA0 <= byte2 && byte2 <= 0xBF)
{
continue;
}
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
}
SECTION("ill-formed: wrong third byte")
{
for (int byte1 = 0xE0; byte1 <= 0xE0; ++byte1)
{
for (int byte2 = 0xA0; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
{
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
}
}
SECTION("UTF8-3 (xE1-xEC UTF8-tail UTF8-tail)")
{
SECTION("well-formed")
{
for (int byte1 = 0xE1; byte1 <= 0xEC; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(true, byte1, byte2, byte3);
check_utf8dump(true, byte1, byte2, byte3);
}
}
}
}
SECTION("ill-formed: missing second byte")
{
for (int byte1 = 0xE1; byte1 <= 0xEC; ++byte1)
{
check_utf8string(false, byte1);
check_utf8dump(false, byte1);
}
}
SECTION("ill-formed: missing third byte")
{
for (int byte1 = 0xE1; byte1 <= 0xEC; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
{
check_utf8string(false, byte1, byte2);
check_utf8dump(false, byte1, byte2);
}
}
}
SECTION("ill-formed: wrong second byte")
{
for (int byte1 = 0xE1; byte1 <= 0xEC; ++byte1)
{
for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2)
{
// skip correct second byte
if (0x80 <= byte2 && byte2 <= 0xBF)
{
continue;
}
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
}
SECTION("ill-formed: wrong third byte")
{
for (int byte1 = 0xE1; byte1 <= 0xEC; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
{
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
}
}
SECTION("UTF8-3 (xED x80-9F UTF8-tail)")
{
SECTION("well-formed")
{
for (int byte1 = 0xED; byte1 <= 0xED; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0x9F; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(true, byte1, byte2, byte3);
check_utf8dump(true, byte1, byte2, byte3);
}
}
}
}
SECTION("ill-formed: missing second byte")
{
for (int byte1 = 0xED; byte1 <= 0xED; ++byte1)
{
check_utf8string(false, byte1);
check_utf8dump(false, byte1);
}
}
SECTION("ill-formed: missing third byte")
{
for (int byte1 = 0xED; byte1 <= 0xED; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0x9F; ++byte2)
{
check_utf8string(false, byte1, byte2);
check_utf8dump(false, byte1, byte2);
}
}
}
SECTION("ill-formed: wrong second byte")
{
for (int byte1 = 0xED; byte1 <= 0xED; ++byte1)
{
for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2)
{
// skip correct second byte
if (0x80 <= byte2 && byte2 <= 0x9F)
{
continue;
}
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
}
SECTION("ill-formed: wrong third byte")
{
for (int byte1 = 0xED; byte1 <= 0xED; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0x9F; ++byte2)
{
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
{
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
}
}
SECTION("UTF8-3 (xEE-xEF UTF8-tail UTF8-tail)")
{
SECTION("well-formed")
{
for (int byte1 = 0xEE; byte1 <= 0xEF; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(true, byte1, byte2, byte3);
check_utf8dump(true, byte1, byte2, byte3);
}
}
}
}
SECTION("ill-formed: missing second byte")
{
for (int byte1 = 0xEE; byte1 <= 0xEF; ++byte1)
{
check_utf8string(false, byte1);
check_utf8dump(false, byte1);
}
}
SECTION("ill-formed: missing third byte")
{
for (int byte1 = 0xEE; byte1 <= 0xEF; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
{
check_utf8string(false, byte1, byte2);
check_utf8dump(false, byte1, byte2);
}
}
}
SECTION("ill-formed: wrong second byte")
{
for (int byte1 = 0xEE; byte1 <= 0xEF; ++byte1)
{
for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2)
{
// skip correct second byte
if (0x80 <= byte2 && byte2 <= 0xBF)
{
continue;
}
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
}
SECTION("ill-formed: wrong third byte")
{
for (int byte1 = 0xEE; byte1 <= 0xEF; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
{
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
}
}
}
}
DOCTEST_CLANG_SUPPRESS_WARNING_POP
-326
View File
@@ -1,326 +0,0 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
// for some reason including this after the json header leads to linker errors with VS 2017...
#include <locale>
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <fstream>
#include <sstream>
#include <iostream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
// this test suite uses static variables with non-trivial destructors
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
DOCTEST_CLANG_SUPPRESS_WARNING("-Wexit-time-destructors")
namespace
{
extern size_t calls;
size_t calls = 0;
void check_utf8dump(bool success_expected, int byte1, int byte2, int byte3, int byte4);
void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 = -1, int byte4 = -1)
{
static std::string json_string;
json_string.clear();
CAPTURE(byte1)
CAPTURE(byte2)
CAPTURE(byte3)
CAPTURE(byte4)
json_string += std::string(1, static_cast<char>(byte1));
if (byte2 != -1)
{
json_string += std::string(1, static_cast<char>(byte2));
}
if (byte3 != -1)
{
json_string += std::string(1, static_cast<char>(byte3));
}
if (byte4 != -1)
{
json_string += std::string(1, static_cast<char>(byte4));
}
CAPTURE(json_string)
// store the string in a JSON value
static json j;
static json j2;
j = json_string;
j2 = "abc" + json_string + "xyz";
static std::string s_ignored;
static std::string s_ignored2;
static std::string s_ignored_ascii;
static std::string s_ignored2_ascii;
static std::string s_replaced;
static std::string s_replaced2;
static std::string s_replaced_ascii;
static std::string s_replaced2_ascii;
// dumping with ignore/replace must not throw in any case
s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore);
s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore);
s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore);
s_ignored2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::ignore);
s_replaced = j.dump(-1, ' ', false, json::error_handler_t::replace);
s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace);
s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace);
s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace);
if (success_expected)
{
static std::string s_strict;
// strict mode must not throw if success is expected
s_strict = j.dump();
// all dumps should agree on the string
CHECK(s_strict == s_ignored);
CHECK(s_strict == s_replaced);
}
else
{
// strict mode must throw if success is not expected
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// ignore and replace must create different dumps
CHECK(s_ignored != s_replaced);
// check that replace string contains a replacement character
CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos);
}
// check that prefix and suffix are preserved
CHECK(s_ignored2.substr(1, 3) == "abc");
CHECK(s_ignored2.substr(s_ignored2.size() - 4, 3) == "xyz");
CHECK(s_ignored2_ascii.substr(1, 3) == "abc");
CHECK(s_ignored2_ascii.substr(s_ignored2_ascii.size() - 4, 3) == "xyz");
CHECK(s_replaced2.substr(1, 3) == "abc");
CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz");
CHECK(s_replaced2_ascii.substr(1, 3) == "abc");
CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz");
}
void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);
// create and check a JSON string with up to four UTF-8 bytes
void check_utf8string(bool success_expected, int byte1, int byte2 = -1, int byte3 = -1, int byte4 = -1)
{
if (++calls % 100000 == 0)
{
std::cout << calls << " of 1641521 UTF-8 strings checked" << std::endl; // NOLINT(performance-avoid-endl)
}
static std::string json_string;
json_string = "\"";
CAPTURE(byte1)
json_string += std::string(1, static_cast<char>(byte1));
if (byte2 != -1)
{
CAPTURE(byte2)
json_string += std::string(1, static_cast<char>(byte2));
}
if (byte3 != -1)
{
CAPTURE(byte3)
json_string += std::string(1, static_cast<char>(byte3));
}
if (byte4 != -1)
{
CAPTURE(byte4)
json_string += std::string(1, static_cast<char>(byte4));
}
json_string += "\"";
CAPTURE(json_string)
json _;
if (success_expected)
{
CHECK_NOTHROW(_ = json::parse(json_string));
}
else
{
CHECK_THROWS_AS(_ = json::parse(json_string), json::parse_error&);
}
}
} // namespace
TEST_CASE("Unicode (3/5)" * doctest::skip())
{
SECTION("RFC 3629")
{
/*
RFC 3629 describes in Sect. 4 the syntax of UTF-8 byte sequences as
follows:
A UTF-8 string is a sequence of octets representing a sequence of UCS
characters. An octet sequence is valid UTF-8 only if it matches the
following syntax, which is derived from the rules for encoding UTF-8
and is expressed in the ABNF of [RFC2234].
UTF8-octets = *( UTF8-char )
UTF8-char = UTF8-1 / UTF8-2 / UTF8-3 / UTF8-4
UTF8-1 = %x00-7F
UTF8-2 = %xC2-DF UTF8-tail
UTF8-3 = %xE0 %xA0-BF UTF8-tail / %xE1-EC 2( UTF8-tail ) /
%xED %x80-9F UTF8-tail / %xEE-EF 2( UTF8-tail )
UTF8-4 = %xF0 %x90-BF 2( UTF8-tail ) / %xF1-F3 3( UTF8-tail ) /
%xF4 %x80-8F 2( UTF8-tail )
UTF8-tail = %x80-BF
*/
SECTION("UTF8-4 (xF0 x90-BF UTF8-tail UTF8-tail)")
{
SECTION("well-formed")
{
for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1)
{
for (int byte2 = 0x90; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4)
{
check_utf8string(true, byte1, byte2, byte3, byte4);
check_utf8dump(true, byte1, byte2, byte3, byte4);
}
}
}
}
}
SECTION("ill-formed: missing second byte")
{
for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1)
{
check_utf8string(false, byte1);
check_utf8dump(false, byte1);
}
}
SECTION("ill-formed: missing third byte")
{
for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1)
{
for (int byte2 = 0x90; byte2 <= 0xBF; ++byte2)
{
check_utf8string(false, byte1, byte2);
check_utf8dump(false, byte1, byte2);
}
}
}
SECTION("ill-formed: missing fourth byte")
{
for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1)
{
for (int byte2 = 0x90; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
}
SECTION("ill-formed: wrong second byte")
{
for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1)
{
for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2)
{
// skip correct second byte
if (0x90 <= byte2 && byte2 <= 0xBF)
{
continue;
}
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4)
{
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
}
}
}
SECTION("ill-formed: wrong third byte")
{
for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1)
{
for (int byte2 = 0x90; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
{
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
continue;
}
for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4)
{
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
}
}
}
SECTION("ill-formed: wrong fourth byte")
{
for (int byte1 = 0xF0; byte1 <= 0xF0; ++byte1)
{
for (int byte2 = 0x90; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip correct fourth byte
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
}
}
}
}
}
}
DOCTEST_CLANG_SUPPRESS_WARNING_POP
-326
View File
@@ -1,326 +0,0 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
// for some reason including this after the json header leads to linker errors with VS 2017...
#include <locale>
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <fstream>
#include <sstream>
#include <iostream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
// this test suite uses static variables with non-trivial destructors
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
DOCTEST_CLANG_SUPPRESS_WARNING("-Wexit-time-destructors")
namespace
{
extern size_t calls;
size_t calls = 0;
void check_utf8dump(bool success_expected, int byte1, int byte2, int byte3, int byte4);
void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 = -1, int byte4 = -1)
{
static std::string json_string;
json_string.clear();
CAPTURE(byte1)
CAPTURE(byte2)
CAPTURE(byte3)
CAPTURE(byte4)
json_string += std::string(1, static_cast<char>(byte1));
if (byte2 != -1)
{
json_string += std::string(1, static_cast<char>(byte2));
}
if (byte3 != -1)
{
json_string += std::string(1, static_cast<char>(byte3));
}
if (byte4 != -1)
{
json_string += std::string(1, static_cast<char>(byte4));
}
CAPTURE(json_string)
// store the string in a JSON value
static json j;
static json j2;
j = json_string;
j2 = "abc" + json_string + "xyz";
static std::string s_ignored;
static std::string s_ignored2;
static std::string s_ignored_ascii;
static std::string s_ignored2_ascii;
static std::string s_replaced;
static std::string s_replaced2;
static std::string s_replaced_ascii;
static std::string s_replaced2_ascii;
// dumping with ignore/replace must not throw in any case
s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore);
s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore);
s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore);
s_ignored2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::ignore);
s_replaced = j.dump(-1, ' ', false, json::error_handler_t::replace);
s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace);
s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace);
s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace);
if (success_expected)
{
static std::string s_strict;
// strict mode must not throw if success is expected
s_strict = j.dump();
// all dumps should agree on the string
CHECK(s_strict == s_ignored);
CHECK(s_strict == s_replaced);
}
else
{
// strict mode must throw if success is not expected
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// ignore and replace must create different dumps
CHECK(s_ignored != s_replaced);
// check that replace string contains a replacement character
CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos);
}
// check that prefix and suffix are preserved
CHECK(s_ignored2.substr(1, 3) == "abc");
CHECK(s_ignored2.substr(s_ignored2.size() - 4, 3) == "xyz");
CHECK(s_ignored2_ascii.substr(1, 3) == "abc");
CHECK(s_ignored2_ascii.substr(s_ignored2_ascii.size() - 4, 3) == "xyz");
CHECK(s_replaced2.substr(1, 3) == "abc");
CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz");
CHECK(s_replaced2_ascii.substr(1, 3) == "abc");
CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz");
}
void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);
// create and check a JSON string with up to four UTF-8 bytes
void check_utf8string(bool success_expected, int byte1, int byte2 = -1, int byte3 = -1, int byte4 = -1)
{
if (++calls % 100000 == 0)
{
std::cout << calls << " of 5517507 UTF-8 strings checked" << std::endl; // NOLINT(performance-avoid-endl)
}
static std::string json_string;
json_string = "\"";
CAPTURE(byte1)
json_string += std::string(1, static_cast<char>(byte1));
if (byte2 != -1)
{
CAPTURE(byte2)
json_string += std::string(1, static_cast<char>(byte2));
}
if (byte3 != -1)
{
CAPTURE(byte3)
json_string += std::string(1, static_cast<char>(byte3));
}
if (byte4 != -1)
{
CAPTURE(byte4)
json_string += std::string(1, static_cast<char>(byte4));
}
json_string += "\"";
CAPTURE(json_string)
json _;
if (success_expected)
{
CHECK_NOTHROW(_ = json::parse(json_string));
}
else
{
CHECK_THROWS_AS(_ = json::parse(json_string), json::parse_error&);
}
}
} // namespace
TEST_CASE("Unicode (4/5)" * doctest::skip())
{
SECTION("RFC 3629")
{
/*
RFC 3629 describes in Sect. 4 the syntax of UTF-8 byte sequences as
follows:
A UTF-8 string is a sequence of octets representing a sequence of UCS
characters. An octet sequence is valid UTF-8 only if it matches the
following syntax, which is derived from the rules for encoding UTF-8
and is expressed in the ABNF of [RFC2234].
UTF8-octets = *( UTF8-char )
UTF8-char = UTF8-1 / UTF8-2 / UTF8-3 / UTF8-4
UTF8-1 = %x00-7F
UTF8-2 = %xC2-DF UTF8-tail
UTF8-3 = %xE0 %xA0-BF UTF8-tail / %xE1-EC 2( UTF8-tail ) /
%xED %x80-9F UTF8-tail / %xEE-EF 2( UTF8-tail )
UTF8-4 = %xF0 %x90-BF 2( UTF8-tail ) / %xF1-F3 3( UTF8-tail ) /
%xF4 %x80-8F 2( UTF8-tail )
UTF8-tail = %x80-BF
*/
SECTION("UTF8-4 (xF1-F3 UTF8-tail UTF8-tail UTF8-tail)")
{
SECTION("well-formed")
{
for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4)
{
check_utf8string(true, byte1, byte2, byte3, byte4);
check_utf8dump(true, byte1, byte2, byte3, byte4);
}
}
}
}
}
SECTION("ill-formed: missing second byte")
{
for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1)
{
check_utf8string(false, byte1);
check_utf8dump(false, byte1);
}
}
SECTION("ill-formed: missing third byte")
{
for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
{
check_utf8string(false, byte1, byte2);
check_utf8dump(false, byte1, byte2);
}
}
}
SECTION("ill-formed: missing fourth byte")
{
for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
}
SECTION("ill-formed: wrong second byte")
{
for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1)
{
for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2)
{
// skip correct second byte
if (0x80 <= byte2 && byte2 <= 0xBF)
{
continue;
}
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4)
{
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
}
}
}
SECTION("ill-formed: wrong third byte")
{
for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
{
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
continue;
}
for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4)
{
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
}
}
}
SECTION("ill-formed: wrong fourth byte")
{
for (int byte1 = 0xF1; byte1 <= 0xF3; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0xBF; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip correct fourth byte
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
}
}
}
}
}
}
DOCTEST_CLANG_SUPPRESS_WARNING_POP
-326
View File
@@ -1,326 +0,0 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
// for some reason including this after the json header leads to linker errors with VS 2017...
#include <locale>
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <fstream>
#include <sstream>
#include <iostream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
// this test suite uses static variables with non-trivial destructors
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
DOCTEST_CLANG_SUPPRESS_WARNING("-Wexit-time-destructors")
namespace
{
extern size_t calls;
size_t calls = 0;
void check_utf8dump(bool success_expected, int byte1, int byte2, int byte3, int byte4);
void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 = -1, int byte4 = -1)
{
static std::string json_string;
json_string.clear();
CAPTURE(byte1)
CAPTURE(byte2)
CAPTURE(byte3)
CAPTURE(byte4)
json_string += std::string(1, static_cast<char>(byte1));
if (byte2 != -1)
{
json_string += std::string(1, static_cast<char>(byte2));
}
if (byte3 != -1)
{
json_string += std::string(1, static_cast<char>(byte3));
}
if (byte4 != -1)
{
json_string += std::string(1, static_cast<char>(byte4));
}
CAPTURE(json_string)
// store the string in a JSON value
static json j;
static json j2;
j = json_string;
j2 = "abc" + json_string + "xyz";
static std::string s_ignored;
static std::string s_ignored2;
static std::string s_ignored_ascii;
static std::string s_ignored2_ascii;
static std::string s_replaced;
static std::string s_replaced2;
static std::string s_replaced_ascii;
static std::string s_replaced2_ascii;
// dumping with ignore/replace must not throw in any case
s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore);
s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore);
s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore);
s_ignored2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::ignore);
s_replaced = j.dump(-1, ' ', false, json::error_handler_t::replace);
s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace);
s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace);
s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace);
if (success_expected)
{
static std::string s_strict;
// strict mode must not throw if success is expected
s_strict = j.dump();
// all dumps should agree on the string
CHECK(s_strict == s_ignored);
CHECK(s_strict == s_replaced);
}
else
{
// strict mode must throw if success is not expected
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// ignore and replace must create different dumps
CHECK(s_ignored != s_replaced);
// check that replace string contains a replacement character
CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos);
}
// check that prefix and suffix are preserved
CHECK(s_ignored2.substr(1, 3) == "abc");
CHECK(s_ignored2.substr(s_ignored2.size() - 4, 3) == "xyz");
CHECK(s_ignored2_ascii.substr(1, 3) == "abc");
CHECK(s_ignored2_ascii.substr(s_ignored2_ascii.size() - 4, 3) == "xyz");
CHECK(s_replaced2.substr(1, 3) == "abc");
CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz");
CHECK(s_replaced2_ascii.substr(1, 3) == "abc");
CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz");
}
void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);
// create and check a JSON string with up to four UTF-8 bytes
void check_utf8string(bool success_expected, int byte1, int byte2 = -1, int byte3 = -1, int byte4 = -1)
{
if (++calls % 100000 == 0)
{
std::cout << calls << " of 1246225 UTF-8 strings checked" << std::endl; // NOLINT(performance-avoid-endl)
}
static std::string json_string;
json_string = "\"";
CAPTURE(byte1)
json_string += std::string(1, static_cast<char>(byte1));
if (byte2 != -1)
{
CAPTURE(byte2)
json_string += std::string(1, static_cast<char>(byte2));
}
if (byte3 != -1)
{
CAPTURE(byte3)
json_string += std::string(1, static_cast<char>(byte3));
}
if (byte4 != -1)
{
CAPTURE(byte4)
json_string += std::string(1, static_cast<char>(byte4));
}
json_string += "\"";
CAPTURE(json_string)
json _;
if (success_expected)
{
CHECK_NOTHROW(_ = json::parse(json_string));
}
else
{
CHECK_THROWS_AS(_ = json::parse(json_string), json::parse_error&);
}
}
} // namespace
TEST_CASE("Unicode (5/5)" * doctest::skip())
{
SECTION("RFC 3629")
{
/*
RFC 3629 describes in Sect. 4 the syntax of UTF-8 byte sequences as
follows:
A UTF-8 string is a sequence of octets representing a sequence of UCS
characters. An octet sequence is valid UTF-8 only if it matches the
following syntax, which is derived from the rules for encoding UTF-8
and is expressed in the ABNF of [RFC2234].
UTF8-octets = *( UTF8-char )
UTF8-char = UTF8-1 / UTF8-2 / UTF8-3 / UTF8-4
UTF8-1 = %x00-7F
UTF8-2 = %xC2-DF UTF8-tail
UTF8-3 = %xE0 %xA0-BF UTF8-tail / %xE1-EC 2( UTF8-tail ) /
%xED %x80-9F UTF8-tail / %xEE-EF 2( UTF8-tail )
UTF8-4 = %xF0 %x90-BF 2( UTF8-tail ) / %xF1-F3 3( UTF8-tail ) /
%xF4 %x80-8F 2( UTF8-tail )
UTF8-tail = %x80-BF
*/
SECTION("UTF8-4 (xF4 x80-8F UTF8-tail UTF8-tail)")
{
SECTION("well-formed")
{
for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0x8F; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4)
{
check_utf8string(true, byte1, byte2, byte3, byte4);
check_utf8dump(true, byte1, byte2, byte3, byte4);
}
}
}
}
}
SECTION("ill-formed: missing second byte")
{
for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1)
{
check_utf8string(false, byte1);
check_utf8dump(false, byte1);
}
}
SECTION("ill-formed: missing third byte")
{
for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0x8F; ++byte2)
{
check_utf8string(false, byte1, byte2);
check_utf8dump(false, byte1, byte2);
}
}
}
SECTION("ill-formed: missing fourth byte")
{
for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0x8F; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
check_utf8string(false, byte1, byte2, byte3);
check_utf8dump(false, byte1, byte2, byte3);
}
}
}
}
SECTION("ill-formed: wrong second byte")
{
for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1)
{
for (int byte2 = 0x00; byte2 <= 0xFF; ++byte2)
{
// skip correct second byte
if (0x80 <= byte2 && byte2 <= 0x8F)
{
continue;
}
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4)
{
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
}
}
}
SECTION("ill-formed: wrong third byte")
{
for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0x8F; ++byte2)
{
for (int byte3 = 0x00; byte3 <= 0xFF; ++byte3)
{
// skip correct third byte
if (0x80 <= byte3 && byte3 <= 0xBF)
{
continue;
}
for (int byte4 = 0x80; byte4 <= 0xBF; ++byte4)
{
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
}
}
}
SECTION("ill-formed: wrong fourth byte")
{
for (int byte1 = 0xF4; byte1 <= 0xF4; ++byte1)
{
for (int byte2 = 0x80; byte2 <= 0x8F; ++byte2)
{
for (int byte3 = 0x80; byte3 <= 0xBF; ++byte3)
{
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip correct fourth byte
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
check_utf8string(false, byte1, byte2, byte3, byte4);
check_utf8dump(false, byte1, byte2, byte3, byte4);
}
}
}
}
}
}
}
}
DOCTEST_CLANG_SUPPRESS_WARNING_POP