diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index 49ee9d3d6..69a3cbc45 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -100,7 +100,7 @@ jobs: container: ubuntu:focal strategy: matrix: - target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls] + target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_simdutf] steps: - name: Install build-essential run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev diff --git a/cmake/ci.cmake b/cmake/ci.cmake index 6b1d325d8..18fef2075 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -212,6 +212,24 @@ add_custom_target(ci_test_legacycomparison COMMENT "Compile and test with legacy discarded value comparison enabled" ) +############################################################################### +# Validate UTF-8 with simdutf. +############################################################################### + +add_custom_target(ci_test_simdutf + COMMAND ${CMAKE_COMMAND} + -DCMAKE_BUILD_TYPE=Debug -GNinja + -DJSON_BuildTests=ON -DJSON_TestSimdutf=ON + # simdutf needs C++17, so the library falls back to its scalar validator + # below that: build the suite at C++11 to cover the fallback with the macro + # defined, and at C++17 to run every test against simdutf itself + "-DJSON_TestStandards=11\;17" + -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_simdutf + COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_simdutf + COMMAND cd ${PROJECT_BINARY_DIR}/build_simdutf && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure + COMMENT "Compile and test with simdutf UTF-8 validation enabled" +) + ############################################################################### # Enable brace-init copy semantics. ############################################################################### diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index 507c04932..e818f032a 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -24,6 +24,7 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_NO_IO**](json_no_io.md) - switch off functions relying on certain C++ I/O headers - [**JSON_SKIP_UNSUPPORTED_COMPILER_CHECK**](json_skip_unsupported_compiler_check.md) - do not warn about unsupported compilers - [**JSON_USE_GLOBAL_UDLS**](json_use_global_udls.md) - place user-defined string literals (UDLs) into the global namespace +- [**JSON_USE_SIMDUTF**](json_use_simdutf.md) - use the simdutf library to accelerate UTF-8 validation ## Library version diff --git a/docs/mkdocs/docs/api/macros/json_use_simdutf.md b/docs/mkdocs/docs/api/macros/json_use_simdutf.md new file mode 100644 index 000000000..611c1e9f1 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_use_simdutf.md @@ -0,0 +1,71 @@ +# JSON_USE_SIMDUTF + +```cpp +#define JSON_USE_SIMDUTF +``` + +When defined, the parser validates the UTF-8 content of JSON strings that come from a **contiguous byte input** +(`std::string`, `std::vector`/``, string literals, `const char*` ranges, …) using the +[simdutf](https://github.com/simdutf/simdutf) library instead of the built-in scalar validator. On text with many +non-ASCII characters (e.g. CJK or emoji) this can validate several times faster. + +This is an **opt-in external dependency**. The library itself remains header-only and its behavior is unchanged: the +same input is accepted or rejected either way, and every parse error is reported at the same position with the same +message (simdutf is only used to fast-path *valid* runs; anything it flags falls back to the scalar path so the exact +diagnostic is preserved). Streaming inputs (files, `std::istream`, wide strings, user-defined adapters) always use the +scalar path. + +When `JSON_USE_SIMDUTF` is defined you must make the `simdutf.h` header available on the include path and link the +simdutf library. When it is not defined, no simdutf header is included and there is no dependency. + +!!! note "Requires C++17" + + simdutf requires C++17 and its header rejects older standards with an `#!cpp #error`. The backend is therefore only + compiled in from C++17 on. In C++11 and C++14 the macro has no effect and the scalar validator is used, which + accepts and rejects exactly the same input -- only throughput differs. Setting the macro project-wide is therefore + safe even when some translation units are built with an older standard. + +!!! warning "Define consistently" + + The macro selects between two definitions of the same inline validation function. It must therefore be defined + identically for **every** translation unit that includes the library; mixing translation units that define it with + ones that do not is an ODR violation. Prefer setting it as a compile definition on the target rather than with + `#!cpp #define` in individual source files. + +## Default definition + +By default, `#!cpp JSON_USE_SIMDUTF` is not defined and the portable C++11 scalar validator is used. + +```cpp +#undef JSON_USE_SIMDUTF +``` + +## Examples + +??? example + + The code below enables the simdutf backend for UTF-8 validation. + + ```cpp + #define JSON_USE_SIMDUTF 1 + #include + + ... + ``` + + The project must also link against simdutf, e.g. with CMake: + + ```cmake + target_compile_definitions(your_target PRIVATE JSON_USE_SIMDUTF) + target_link_libraries(your_target PRIVATE simdutf::simdutf) + ``` + +!!! hint "Testing this configuration" + + The unit tests can be built against the simdutf backend with the CMake option `JSON_TestSimdutf` (`OFF` by + default), which fetches simdutf and defines `JSON_USE_SIMDUTF` for every test target. The `ci_test_simdutf` target + runs the whole test suite in that configuration. + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index 1d169fdeb..c4602fa5a 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -137,6 +137,14 @@ behavior is deprecated and switched off (`0`) by default. See [full documentation of `JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md). +## `JSON_USE_SIMDUTF` + +When defined, UTF-8 validation of JSON strings read from contiguous byte input is delegated to the +[simdutf](https://github.com/simdutf/simdutf) library instead of the built-in scalar validator. This is an opt-in +external dependency and is not defined by default. + +See [full documentation of `JSON_USE_SIMDUTF`](../api/macros/json_use_simdutf.md). + ## `NLOHMANN_DEFINE_TYPE_*(...)`, `NLOHMANN_DEFINE_DERIVED_TYPE_*(...)` The library defines 12 macros to simplify the serialization/deserialization of types. See the page on diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 2e1337f47..ec3e462c1 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -296,6 +296,7 @@ nav: - 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md - 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md - 'JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON': api/macros/json_use_legacy_discarded_value_comparison.md + - 'JSON_USE_SIMDUTF': api/macros/json_use_simdutf.md - 'NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_derived_type.md - 'NLOHMANN_DEFINE_TYPE_INTRUSIVE, NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_intrusive.md - 'NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_non_intrusive.md diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index ba8df07a6..bd19d32a8 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -155,11 +155,31 @@ class input_stream_adapter // General-purpose iterator-based adapter. It might not be as fast as // theoretically possible for some containers, but it is extremely versatile. -// SentinelType defaults to IteratorType for backward compatibility, but may -// be a different type (e.g., a C++20 sentinel or counted_iterator). +// SentinelType defaults to IteratorType for backward compatibility, but may be +// a different type, e.g. a C++20 sentinel such as std::default_sentinel_t when +// IteratorType is a std::counted_iterator. template class iterator_input_adapter { + // Whether the number of elements between two positions can be computed in + // O(1): either the iterator and the sentinel have the same type (plain + // std::distance) or, in C++20, the sentinel is a sized sentinel for the + // iterator (std::ranges::distance), e.g. std::default_sentinel_t paired + // with std::counted_iterator. + // + // JSON_HAS_RANGES gates the C++20 branch: on standard libraries with an + // incomplete (libstdc++ < 11, see #4440) evaluating + // std::contiguous_iterator on a std::counted_iterator is a hard error + // instead of yielding false, and these traits are instantiated for every + // adapter. Such toolchains fall back to the pointer-only test and simply + // use the byte-at-a-time scanner. + static constexpr bool sentinel_is_sized = +#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) + std::is_same::value || std::sized_sentinel_for; +#else + std::is_same::value; +#endif + public: using char_type = typename std::iterator_traits::value_type; @@ -171,7 +191,7 @@ class iterator_input_adapter // in wide_string_input_adapter, which does not expose this). static constexpr bool supports_seek = std::is_same::iterator_category, std::random_access_iterator_tag>::value - && std::is_same::value + && sentinel_is_sized && sizeof(char_type) == 1; iterator_input_adapter(IteratorType first, SentinelType last) @@ -219,30 +239,60 @@ class iterator_input_adapter private: // whether IteratorType refers to a contiguous range and therefore supports // a std::memcpy fast path (pointers always do; in C++20 we can also detect - // library iterators such as those of std::vector and std::string). - // Computing the available element count needs either same-type iterators - // (plain std::distance) or, in C++20, a sized sentinel (std::ranges::distance), - // e.g. std::counted_iterator paired with std::default_sentinel_t. - static constexpr bool iterator_is_contiguous = -#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) - (std::is_same::value || std::sized_sentinel_for) - && (std::contiguous_iterator || std::is_pointer::value); + // library iterators such as those of std::vector and std::string). The + // available element count must also be computable in O(1), hence + // sentinel_is_sized. + static constexpr bool iterator_is_contiguous = sentinel_is_sized && +#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) + (std::contiguous_iterator || std::is_pointer::value); #else - std::is_same::value && std::is_pointer::value; + std::is_pointer::value; #endif + // number of unread elements in [current, end) + std::size_t remaining_count() const + { +#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) + // std::ranges::distance also supports sized sentinels of a different + // type (e.g. std::counted_iterator + std::default_sentinel_t) + return static_cast(std::ranges::distance(current, end)); +#else + return static_cast(std::distance(current, end)); +#endif + } + + public: + // Whether the remaining input is a single contiguous block of 1-byte + // elements that the lexer can inspect directly (used for the SWAR string + // fast path). + static constexpr bool supports_bulk_scan = + iterator_is_contiguous && sizeof(char_type) == 1; + + // Pointer to the next unread element; only valid when bulk_remaining() > 0. + const char_type* bulk_data() const + { + return &*current; + } + + // Number of unread elements available as one contiguous block. + std::size_t bulk_remaining() const + { + return remaining_count(); + } + + // Consume @a n elements previously inspected via bulk_data(). + void bulk_skip(std::size_t n) + { + std::advance(current, static_cast::difference_type>(n)); + } + + private: // contiguous fast path: bulk copy the remaining range with std::memcpy template std::size_t get_elements_impl(T* dest, std::size_t count, std::true_type /*contiguous*/) { const std::size_t wanted = count * sizeof(T); -#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) - // std::ranges::distance also supports sized sentinels of a different - // type (e.g. std::counted_iterator + std::default_sentinel_t) - const std::size_t available = static_cast(std::ranges::distance(current, end)) * sizeof(char_type); -#else - const std::size_t available = static_cast(std::distance(current, end)) * sizeof(char_type); -#endif + const std::size_t available = remaining_count() * sizeof(char_type); const std::size_t copied = (std::min)(wanted, available); if (JSON_HEDLEY_LIKELY(copied != 0)) { @@ -570,6 +620,46 @@ typename iterator_input_adapter_factory::adapter_typ return factory_type::create(first, last); } +// The element type a container's data() points at, cv-qualifiers removed. +// Ill-formed - and therefore SFINAE-friendly - for types without data(). +template +using container_data_t = typename std::remove_cv().data()) >::type >::type; + +// The container's own element type, cv-qualifiers removed. It is looked up on +// the bare type so it is also found when ContainerType is deduced as a +// reference by the forwarding-reference overload below. +template +using container_value_t = typename std::remove_cv < + typename std::remove_cv::type>::type::value_type >::type; + +// Detect a container that stores its elements contiguously as single bytes +// (std::string, std::vector, std::array, +// std::string_view, ...). Such inputs are wrapped in a pointer-based adapter so +// they benefit from the contiguous fast paths (bulk string scanning, memcpy for +// binary formats) in every C++ standard - not only in C++20, where the standard +// library iterators model std::contiguous_iterator and are detected directly. +// +// data() and size() on their own would be duck typing: they say nothing about +// size() counting the units data() points at, and reading [data(), data() + +// size()) as bytes would be wrong for a type where it does not. Requiring the +// container's own value_type to be that same single-byte element ties the two +// together; every contiguous standard container satisfies it. Anything else +// keeps the iterator-based adapter, which is always correct - only slower. +template +struct is_contiguous_byte_container : std::false_type {}; + +template +struct is_contiguous_byte_container < ContainerType, void_t < + container_data_t, + container_value_t, +decltype(std::declval().size()) >> + : std::integral_constant < bool, + std::is_pointer().data())>::value&& + std::is_integral>::value&& + sizeof(container_data_t) == 1 && + std::is_same, container_value_t>::value > {}; + // Convenience shorthand from container to iterator // Enables ADL on begin(container) and end(container) // Encloses the using declarations in namespace for not to leak them to outside scope @@ -597,12 +687,32 @@ struct container_input_adapter_factory< ContainerType, } // namespace container_input_adapter_factory_impl -template -typename container_input_adapter_factory_impl::container_input_adapter_factory::adapter_type input_adapter(ContainerType&& container) +// General container path (iterator-based). Contiguous single-byte containers +// are excluded here and routed through the pointer-based overload below. +template < typename ContainerType, + enable_if_t < !is_contiguous_byte_container::value, int > = 0 > +typename container_input_adapter_factory_impl::container_input_adapter_factory::adapter_type input_adapter(ContainerType && container) { return container_input_adapter_factory_impl::container_input_adapter_factory::create(std::forward(container)); } +// Contiguous single-byte containers (std::string, std::vector, ...) are +// wrapped in a pointer-based adapter so the contiguous fast paths apply in every +// standard. The pointer keeps the container's own element type (const char* for +// std::string, const std::uint8_t* for std::vector, ...), so the +// resulting char_type - and therefore the parsing behavior - is byte-for-byte +// identical to the iterator-based path; only the raw pointer additionally +// enables the bulk fast paths. The container outlives the adapter for the whole +// parse (temporaries live until the end of the full expression), exactly as the +// iterators it replaces did. +template < typename ContainerType, + enable_if_t < is_contiguous_byte_container::value, int > = 0 > +auto input_adapter(const ContainerType& container) +-> decltype(input_adapter(container.data(), container.data() + container.size())) +{ + return input_adapter(container.data(), container.data() + container.size()); +} + // specialization for std::string using string_input_adapter_type = decltype(input_adapter(std::declval())); diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index c241e793b..bc31337f9 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -19,7 +19,9 @@ #include // vector #include +#include #include +#include #include #include @@ -125,6 +127,25 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/) return false; } +// Detect whether an input adapter exposes a contiguous byte block that the +// lexer can scan directly (see iterator_input_adapter::supports_bulk_scan). +// Adapters without the flag - file, stream, wide-string, user-defined - fall +// back to the character-at-a-time string scanner. +template +using detect_supports_bulk_scan = decltype(InputAdapterType::supports_bulk_scan); + +template +constexpr bool input_adapter_supports_bulk_scan(std::true_type /*detected*/) +{ + return InputAdapterType::supports_bulk_scan; +} + +template +constexpr bool input_adapter_supports_bulk_scan(std::false_type /*detected*/) +{ + return false; +} + /*! @brief lexical analysis @@ -146,6 +167,14 @@ class lexer : public lexer_base static constexpr bool lazy_token_string = input_adapter_supports_seek(is_detected {}); + /// whether string scanning may bulk-consume runs of ordinary characters + /// directly from a contiguous input buffer (SWAR fast path). This requires + /// the token to be reconstructible lazily (lazy_token_string), so bypassing + /// the per-character capture in get() cannot lose error diagnostics. + static constexpr bool bulk_scan = + lazy_token_string + && input_adapter_supports_bulk_scan(is_detected {}); + public: using token_type = typename lexer_base::token_type; @@ -266,6 +295,40 @@ class lexer : public lexer_base return true; } + /// contiguous input: bulk-append the run of ordinary characters and complete + /// well-formed UTF-8 sequences starting at the current read position, leaving + /// the first byte that needs individual handling (the closing quote, an + /// escape, a control character, or an ill-formed UTF-8 byte) for get() + void scan_string_bulk(std::true_type /*bulk*/) + { + // a pending unget must be consumed through the normal path first + if (next_unget) + { + return; + } + const std::size_t remaining = ia.bulk_remaining(); + if (remaining == 0) + { + return; + } + const auto* const data = reinterpret_cast(ia.bulk_data()); + + const std::size_t pos = string_bulk_run(data, remaining); + if (pos == 0) + { + return; + } + token_buffer.append(reinterpret_cast(data), pos); + ia.bulk_skip(pos); + // the run contains no newline (all bytes < 0x20 are treated as special), + // so only the flat character counters advance + position.chars_read_total += pos; + position.chars_read_current_line += pos; + } + + /// streaming input: no bulk fast path + void scan_string_bulk(std::false_type /*bulk*/) const noexcept {} + /*! @brief scan a string literal @@ -291,6 +354,10 @@ class lexer : public lexer_base while (true) { + // bulk-consume ordinary characters from contiguous input, then + // handle the next special byte through the switch below + scan_string_bulk(std::integral_constant {}); + // get the next character switch (get()) { @@ -1009,6 +1076,12 @@ class lexer : public lexer_base // changed if minus sign, decimal point, or exponent is read token_type number_type = token_type::value_unsigned; + // offset just past the last mantissa byte in token_buffer (i.e. the + // index of 'e'/'E', or the whole token when there is no exponent). + // convert_number() uses it to count significant digits; npos means + // "not seen an exponent yet" and is resolved at scan_number_done + std::size_t mantissa_end = std::string::npos; + // state (init): we just found out we need to scan a number switch (current) { @@ -1194,6 +1267,9 @@ scan_number_decimal2: scan_number_exponent: // we just parsed an exponent number_type = token_type::value_float; + // this label is reached only right after the 'e'/'E' was appended (from + // the zero, any1, and decimal2 states), so the mantissa ends before it + mantissa_end = token_buffer.size() - 1; switch (get()) { case '+': @@ -1280,6 +1356,116 @@ scan_number_done: // we are done scanning a number) unget(); + // no exponent was scanned: the mantissa spans the whole token + if (mantissa_end == std::string::npos) + { + mantissa_end = token_buffer.size(); + } + + return convert_number(number_type, mantissa_end); + } + + /*! + @brief convert an already-validated integer token to its value + + The digit sequence in [first, last) has been validated by the caller, so a + dedicated parser can avoid the locale/errno overhead of std::strtoull. + + @return the token type on success; token_type::uninitialized if @a + number_type is not an integer type or the value does not fit, in + which case the caller falls back to the floating-point conversion + (matching the previous std::strtoull/std::strtoll behavior) + */ + token_type convert_integer(token_type number_type, const char* first, const char* last) + { + if (number_type == token_type::value_unsigned) + { + if (parse_integer_unsigned(first, last, value_unsigned)) + { + return token_type::value_unsigned; + } + } + else if (number_type == token_type::value_integer) + { + if (parse_integer_signed(first, last, value_integer)) + { + return token_type::value_integer; + } + } + + return token_type::uninitialized; + } + + /*! + @brief check whether Clinger's fast path can still succeed for this token + + parse_float_fast() needs a significand below 2^53. A mantissa with 17 or + more significant digits is at least 10^16 and therefore always exceeds it, + so calling the fast path would walk the token one extra time only to + decline before strtod has to run anyway. + + Significant digits are the mantissa's digits from the first nonzero one on; + the sign, the decimal point, leading zeros, and the exponent do not count. + The answer is derived from indices - the digits are not scanned again - so + this stays off the hot path of the number scanners. + + @param[in] mantissa_end offset just past the last mantissa byte in + token_buffer + @return false if parse_float_fast() is guaranteed to decline + */ + bool mantissa_fits_clinger(std::size_t mantissa_end) const + { + // 10^16 already exceeds 2^53, so 17 digits can never fit + constexpr std::size_t limit = 17; + + const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u; + const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u; + // the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so + // a leading zero can only be a lone "0", which is not significant + const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u; + JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero); + std::size_t digits = mantissa_end - neg - has_dot - lead_zero; + + if (JSON_HEDLEY_LIKELY(digits < limit)) + { + return true; + } + + // Only a number below 1 can carry further insignificant zeros, and only + // while the count stays at the limit does removing them change the + // answer - so this loop is skipped for all but a few tokens. Note + // token_buffer holds the locale's decimal point, so the fraction is + // located through decimal_point_position rather than by searching '.'. + if (lead_zero != 0) + { + JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit + for (std::size_t i = decimal_point_position + 1; + digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i) + { + --digits; + } + } + + return digits < limit; + } + + /*! + @brief convert the number text in token_buffer to its value and token type + + The digit sequence in token_buffer has already been validated (by the + scan_number() state machine or by the contiguous fast path) and holds the + locale decimal point in place of '.'. Integers are parsed first and fall + back to floating point on overflow. This is shared so both scanners produce + identical results. + + @param[in] mantissa_end offset just past the last mantissa byte in + token_buffer (the index of 'e'/'E', or + token_buffer.size() when there is no exponent); + used to skip Clinger's fast path when it cannot + possibly succeed - see mantissa_fits_clinger() + */ + token_type convert_number(token_type number_type, std::size_t mantissa_end) + { // If the caller does not need the converted value (only whether the // input is syntactically valid; see json_sax_acceptor/accept()), an // unsigned/integer token can be reported without calling @@ -1332,45 +1518,37 @@ scan_number_done: } } - char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) - errno = 0; + const char* const num_begin = token_buffer.data(); + const char* const num_end = num_begin + token_buffer.size(); - // try to parse integers first and fall back to floats - if (number_type == token_type::value_unsigned) + if (number_type != token_type::value_float) { - const auto x = std::strtoull(token_buffer.data(), &endptr, 10); - - // we checked the number format before - JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); - - if (errno != ERANGE) + const token_type integer_result = convert_integer(number_type, num_begin, num_end); + if (integer_result != token_type::uninitialized) { - value_unsigned = static_cast(x); - if (value_unsigned == x) - { - return token_type::value_unsigned; - } - } - } - else if (number_type == token_type::value_integer) - { - const auto x = std::strtoll(token_buffer.data(), &endptr, 10); - - // we checked the number format before - JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); - - if (errno != ERANGE) - { - value_integer = static_cast(x); - if (value_integer == x) - { - return token_type::value_integer; - } + return integer_result; } } // this code is reached if we parse a floating-point number or if an - // integer conversion above failed + // integer conversion above overflowed. Prefer std::from_chars + // (Eisel-Lemire, locale-independent, correctly rounded) when available; + // otherwise the exact Clinger fast path (double only); otherwise the + // locale-aware strtof/strtod. + if (parse_float_from_chars(num_begin, num_end, value_float)) + { + return token_type::value_float; + } + // Skipping a fast path that cannot succeed is lossless and saves a full + // extra pass over the token's bytes, which otherwise shows up on + // high-precision inputs such as canada.json + if (mantissa_fits_clinger(mantissa_end) + && parse_float_fast(num_begin, num_end, decimal_point_char, value_float)) + { + return token_type::value_float; + } + + char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) strtof(value_float, token_buffer.data(), &endptr); // we checked the number format before @@ -1379,6 +1557,158 @@ scan_number_done: return token_type::value_float; } + /*! + @brief contiguous fast path for scanning a number + + Parses the whole number token straight from the input buffer, avoiding the + per-character get()/add() of scan_number(). On success it fills token_buffer + (with the locale decimal point substituted, as scan_number() does) and + returns the token type. On anything it does not fully recognize as a + well-formed number it makes no state change and returns + token_type::uninitialized, so the caller falls back to scan_number(), which + then produces the exact diagnostic. @a current is the first digit or the + leading minus (already read); the remaining bytes are taken from the adapter. + */ + token_type scan_number_bulk_contiguous() + { + // a pending unget offsets the buffer position from current; fall back + if (next_unget) + { + return token_type::uninitialized; + } + const std::size_t rem = ia.bulk_remaining(); + if (rem == 0) + { + // the first digit is the last input byte; let scan_number() finish + return token_type::uninitialized; + } + // the byte before the next unread one is current (contiguous input) + const char* const data = reinterpret_cast(ia.bulk_data()) - 1; + const std::size_t avail = rem + 1; + + // validate + classify the number extent (mirrors scan_number()'s grammar) + std::size_t i = 0; + std::size_t dot_index = std::string::npos; + token_type number_type = token_type::value_unsigned; + if (data[0] == '-') + { + number_type = token_type::value_integer; + i = 1; + if (i >= avail) + { + return token_type::uninitialized; + } + } + if (data[i] == '0') + { + ++i; + } + else if (data[i] >= '1' && data[i] <= '9') + { + ++i; + while (i < avail && data[i] >= '0' && data[i] <= '9') + { + ++i; + } + } + else + { + return token_type::uninitialized; + } + if (i < avail && data[i] == '.') + { + number_type = token_type::value_float; + dot_index = i; + ++i; + if (i >= avail || !(data[i] >= '0' && data[i] <= '9')) + { + return token_type::uninitialized; + } + while (i < avail && data[i] >= '0' && data[i] <= '9') + { + ++i; + } + } + // the mantissa ends here, whether or not an exponent part follows + const std::size_t mantissa_end = i; + if (i < avail && (data[i] == 'e' || data[i] == 'E')) + { + number_type = token_type::value_float; + ++i; + if (i < avail && (data[i] == '+' || data[i] == '-')) + { + ++i; + } + if (i >= avail || !(data[i] >= '0' && data[i] <= '9')) + { + return token_type::uninitialized; + } + while (i < avail && data[i] >= '0' && data[i] <= '9') + { + ++i; + } + } + const std::size_t len = i; + + // reset() records where this token starts (for diagnostics), so it has + // to run before the input position advances below + reset(); + + // An integer token needs no token_buffer: the SAX callbacks for + // number_integer/number_unsigned take only the value, and the overflow + // diagnostic rebuilds the text from the input. Convert straight from the + // input buffer and leave token_buffer empty. (JSON_DIAGNOSTIC_POSITIONS + // derives a number's start position from get_string().size(), so there + // the token still has to be materialized.) +#if !JSON_DIAGNOSTIC_POSITIONS + if (number_type != token_type::value_float) + { + const token_type integer_result = convert_integer(number_type, data, data + len); + if (JSON_HEDLEY_LIKELY(integer_result != token_type::uninitialized)) + { + ia.bulk_skip(len - 1); + position.chars_read_total += (len - 1); + position.chars_read_current_line += (len - 1); + return integer_result; + } + // The value does not fit an integer, so this token converts as a + // float. Recording that here keeps convert_number() below from + // repeating the integer attempt that just failed. + number_type = token_type::value_float; + } +#endif + + // materialize the token exactly as scan_number() would, substituting the + // locale decimal point so convert_number()'s strtof fallback stays valid. + // reset() already cleared token_buffer, so append() fills it (assign() is + // avoided because custom string_t types need not provide it) + token_buffer.append(reinterpret_cast(data), len); + if (dot_index != std::string::npos) + { + token_buffer[dot_index] = static_cast(decimal_point_char); + decimal_point_position = dot_index; + } + + ia.bulk_skip(len - 1); + position.chars_read_total += (len - 1); + position.chars_read_current_line += (len - 1); + + return convert_number(number_type, mantissa_end); + } + + /// contiguous input: try the number fast path, else the byte-path scanner + token_type scan_number_dispatch(std::true_type /*bulk*/) + { + const token_type t = scan_number_bulk_contiguous(); + return (t != token_type::uninitialized) ? t : scan_number(); + } + + /// streaming input: always use the byte-path scanner + token_type scan_number_dispatch(std::false_type /*bulk*/) + { + return scan_number(); + } + /*! @param[in] literal_text the literal text to expect @param[in] length the length of the passed literal text @@ -1482,6 +1812,9 @@ scan_number_done: if (current == '\n') { ++position.lines_read; + // remember the column the newline was read at: chars_read_current_line + // is about to be cleared, and a matching unget() cannot reconstruct it + chars_read_before_newline = position.chars_read_current_line; position.chars_read_current_line = 0; } @@ -1538,12 +1871,20 @@ scan_number_done: --position.chars_read_total; // in case we "unget" a newline, we have to also decrement the lines_read + // and restore the column that get() cleared when it saw the newline; + // chars_read_current_line == 0 can only mean the last get() read one if (position.chars_read_current_line == 0) { if (position.lines_read > 0) { --position.lines_read; } + + // chars_read_before_newline counts the newline itself, which is the + // character being ungotten, hence the -1 + position.chars_read_current_line = (chars_read_before_newline > 0) + ? chars_read_before_newline - 1 + : 0; } else { @@ -1810,7 +2151,7 @@ scan_number_done: case '7': case '8': case '9': - return scan_number(); + return scan_number_dispatch(std::integral_constant {}); // end of input (the null byte is needed when parsing from // string literals) @@ -1841,6 +2182,10 @@ scan_number_done: /// the start position of the current token position_t position {}; + /// the value chars_read_current_line had when the last newline was read, so + /// that unget() can restore the column instead of leaving it at 0 + std::size_t chars_read_before_newline = 0; + /// raw input token string for error messages; only populated for streaming /// adapters (seekable adapters reconstruct it lazily via token_string_start) std::vector token_string {}; diff --git a/include/nlohmann/detail/input/number_parse.hpp b/include/nlohmann/detail/input/number_parse.hpp new file mode 100644 index 000000000..e50c3f67f --- /dev/null +++ b/include/nlohmann/detail/input/number_parse.hpp @@ -0,0 +1,302 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // array +#include // FLT_EVAL_METHOD +#include // size_t +#include // int64_t, uint64_t +#include // numeric_limits + +#include + +// std::from_chars lives in , but being in C++17 mode does not +// guarantee the header exists: GCC 7 sets __cplusplus to C++17 yet ships no +// (added in GCC 8; floating-point support in GCC 11). Guard the +// include with __has_include so such toolchains fall back to the scalar path. +#if defined(JSON_HAS_CPP_17) && defined(__has_include) + #if __has_include() + #include // from_chars (only used when __cpp_lib_to_chars is defined) + #include // errc + #endif +#endif + +// This file contains the value-conversion helpers used by the lexer to turn an +// already-validated number token into a value, without the locale/errno +// overhead of std::strtoull/std::strtod. They are free functions so the lexer +// stays focused on scanning; see lexer::convert_number(). + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief fast integer parser for an already-validated unsigned integer + +The number scanner has already checked that [first, last) is a valid JSON +integer, so this only needs to accumulate the digits and detect overflow. This +avoids the locale/errno machinery of std::strtoull, which dominates +integer-heavy inputs. + +@param[in] first pointer to the first character (a digit) +@param[in] last pointer past the last character +@param[out] value the parsed value on success +@return true if the value fit into @a NumberUnsignedType; false on overflow, in + which case the caller falls back to floating-point parsing (matching the + previous std::strtoull behavior) +*/ +template +bool parse_integer_unsigned(const char* first, const char* last, NumberUnsignedType& value) noexcept +{ + // accumulate in the widest unsigned type used by the previous strtoull + // path so the overflow behavior is unchanged for custom number types + std::uint64_t x = 0; + constexpr std::uint64_t cutoff = (std::numeric_limits::max)() / 10u; + constexpr std::uint64_t cutlim = (std::numeric_limits::max)() % 10u; + for (const char* p = first; p != last; ++p) + { + const auto digit = static_cast(static_cast(*p) - static_cast('0')); + if (JSON_HEDLEY_UNLIKELY(x > cutoff || (x == cutoff && digit > cutlim))) + { + return false; + } + x = (x * 10u) + digit; + } + value = static_cast(x); + // reject values that do not round-trip into a narrower NumberUnsignedType + return static_cast(value) == x; +} + +/*! +@brief fast integer parser for an already-validated negative integer + +@param[in] first pointer to the leading '-' +@param[in] last pointer past the last character +@param[out] value the parsed (negative) value on success +@return true on success; false on overflow (caller falls back to float) +*/ +template +bool parse_integer_signed(const char* first, const char* last, NumberIntegerType& value) noexcept +{ + // the state machine only reaches the signed path via a leading '-' + JSON_ASSERT(first != last && *first == '-'); + std::uint64_t magnitude = 0; + // |INT64_MIN| == INT64_MAX + 1; this is the largest admissible magnitude + constexpr std::uint64_t limit = static_cast((std::numeric_limits::max)()) + 1u; + for (const char* p = first + 1; p != last; ++p) + { + const auto digit = static_cast(static_cast(*p) - static_cast('0')); + if (JSON_HEDLEY_UNLIKELY(magnitude > (limit - digit) / 10u)) + { + return false; + } + magnitude = (magnitude * 10u) + digit; + } + const std::int64_t x = (magnitude == limit) + ? (std::numeric_limits::min)() + : -static_cast(magnitude); + value = static_cast(x); + // reject values that do not round-trip into a narrower NumberIntegerType + return static_cast(value) == x; +} + +/*! +@brief exact fast path for parsing a `double` (Clinger's algorithm) + +For the common case - at most 19 significant digits, a decimal exponent in +[-22, 22], and a significand below 2^53 - the value equals significand * +10^exp computed in IEEE-754 double arithmetic, which is exact under +round-to-nearest because both operands are exactly representable. This is the +same fast path used by fast_float/simdjson; the general cases are left to +std::strtod. The parser only activates for number_float_t == double; float and +long double keep the std::strtof/std::strtold paths (see the templated overload +below). + +@param[in] first pointer to the first character of the number +@param[in] last pointer past the last character +@param[in] decimal_point the (locale-dependent) decimal point character +@param[out] out the parsed value on success +@return true if the value was parsed exactly; false to fall back to strtod +*/ +template +bool parse_float_fast(const char* first, const char* last, DecimalPointType decimal_point, double& out) noexcept +{ +#if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0 + // Clinger's fast path is only exact when double operations are evaluated in + // true double precision. On platforms that keep intermediates in extended + // precision (e.g. the x87 FPU on 32-bit x86, where FLT_EVAL_METHOD == 2) the + // single significand * 10^scale step is double-rounded and can be 1 ULP off, + // so decline and let the caller fall back to the correctly-rounded + // std::from_chars / std::strtod path. + static_cast(first); + static_cast(last); + static_cast(decimal_point); + static_cast(out); + return false; +#else + static const std::array powers_of_ten = + { + { + 1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11, + 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22 + } + }; + + const char* p = first; + bool negative = false; + if (p != last && (*p == '-' || *p == '+')) + { + negative = (*p == '-'); + ++p; + } + + std::uint64_t significand = 0; + int num_digits = 0; + int fractional_digits = 0; + bool seen_dot = false; + bool any_digit = false; + for (; p != last; ++p) + { + const char c = *p; + if (c >= '0' && c <= '9') + { + any_digit = true; + if (JSON_HEDLEY_UNLIKELY(num_digits >= 19)) + { + return false; // significand may not fit into uint64_t + } + significand = (significand * 10u) + static_cast(c - '0'); + ++num_digits; + fractional_digits += static_cast(seen_dot); + } + else if (static_cast(c) == decimal_point) + { + if (JSON_HEDLEY_UNLIKELY(seen_dot)) + { + return false; + } + seen_dot = true; + } + else if (c == 'e' || c == 'E') + { + ++p; + break; + } + else + { + return false; + } + } + if (JSON_HEDLEY_UNLIKELY(!any_digit)) + { + return false; + } + + int exponent = 0; + if (p != last) // an exponent part remains + { + bool exp_negative = false; + if (p != last && (*p == '-' || *p == '+')) + { + exp_negative = (*p == '-'); + ++p; + } + bool any_exp_digit = false; + for (; p != last; ++p) + { + if (JSON_HEDLEY_UNLIKELY(*p < '0' || *p > '9')) + { + return false; + } + exponent = (exponent * 10) + (*p - '0'); + any_exp_digit = true; + if (JSON_HEDLEY_UNLIKELY(exponent > 9999)) + { + return false; + } + } + if (JSON_HEDLEY_UNLIKELY(!any_exp_digit)) + { + return false; + } + if (exp_negative) + { + exponent = -exponent; + } + } + + const int scale = exponent - fractional_digits; + if (JSON_HEDLEY_UNLIKELY(significand >= (static_cast(1) << 53))) + { + return false; // significand not exactly representable as double + } + + auto result = static_cast(significand); + if (scale >= 0) + { + if (JSON_HEDLEY_UNLIKELY(scale > 22)) + { + return false; + } + result *= powers_of_ten[static_cast(scale)]; + } + else + { + if (JSON_HEDLEY_UNLIKELY(-scale > 22)) + { + return false; + } + result /= powers_of_ten[static_cast(-scale)]; + } + out = negative ? -result : result; + return true; +#endif +} + +/// fast float path is only exact for `double`; decline for float/long double +template +bool parse_float_fast(const char* /*first*/, const char* /*last*/, DecimalPointType /*decimal_point*/, FloatType& /*out*/) noexcept +{ + return false; +} + +/*! +@brief parse a float with std::from_chars (Eisel-Lemire) when available + +std::from_chars is locale-independent, correctly rounded, and - via the +Eisel-Lemire algorithm in modern standard libraries - much faster than strtod +over the whole value range (not just the Clinger subset). It is used only when +__cpp_lib_to_chars indicates full floating-point support and only when it +consumes the entire token ([first, last)); a partial parse means the buffer +uses a non-'.' locale decimal point, in which case the caller falls back to the +locale-aware path. An under-/overflow (result_out_of_range) also declines, so +the caller's strtod fallback supplies the well-defined ±inf/0 result the parser +expects (side-stepping the P4168 divergence between implementations). + +@return true if the value was parsed exactly and fully; false to fall back +*/ +template +bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept +{ + // JSON_HAS_CPP_17 must gate the use as well as the include above: + // some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even + // in C++14 mode, where is not included. +#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars) + const auto result = std::from_chars(first, last, out); + return result.ec == std::errc() && result.ptr == last; +#else + static_cast(first); + static_cast(last); + static_cast(out); + return false; +#endif +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/input/string_scan.hpp b/include/nlohmann/detail/input/string_scan.hpp new file mode 100644 index 000000000..dc5b07a54 --- /dev/null +++ b/include/nlohmann/detail/input/string_scan.hpp @@ -0,0 +1,241 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // size_t +#include // uint64_t +#include // memcpy + +#include + +// Optional SIMD backend for bulk UTF-8 validation. This is an opt-in external +// dependency: nlohmann/json itself stays header-only and the C++11 scalar +// validator below is always available; defining JSON_USE_SIMDUTF additionally +// requires the simdutf headers on the include path and linking the simdutf +// library. See string_bulk_run(). +// +// simdutf.h itself requires C++17 - it rejects older standards with an #error - +// so the backend is only compiled in from C++17 on. Below that the macro has no +// effect and the scalar validator is used; it accepts and rejects exactly the +// same input, so only throughput differs. macro_scope.hpp is included above to +// have JSON_HAS_CPP_17 available for this test. +#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17) + #include +#endif + +// This file contains the byte-level string-scanning helpers used by the lexer's +// contiguous fast path. They operate purely on raw bytes (no dependency on the +// lexer's template parameters) so they are free functions, keeping the lexer +// itself focused on the state machine; see lexer::scan_string_bulk(). + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +// classify a single byte as needing individual string handling: the closing +// quote, an escape, a control character, or a non-ASCII (UTF-8) +// lead/continuation byte. Ordinary bytes (0x20..0x7F except '"' and '\\') are +// copied verbatim, which the bulk scanner does 8 bytes at a time. +inline bool is_string_special(unsigned char c) noexcept +{ + return c == '\"' || c == '\\' || c < 0x20u || c >= 0x80u; +} + +// SWAR helper: return a word whose high bit is set in every byte of @a v that +// is_string_special(); zero if the 8 bytes are all ordinary. +inline std::uint64_t swar_string_special(std::uint64_t v) noexcept +{ + constexpr std::uint64_t ones = 0x0101010101010101ull; + constexpr std::uint64_t high = 0x8080808080808080ull; + const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22) + const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C) + const std::uint64_t has_quote = (q - ones) & ~q & high; + const std::uint64_t has_backslash = (b - ones) & ~b & high; + const std::uint64_t has_control = (v - 0x2020202020202020ull) & ~v & high; // < 0x20 + const std::uint64_t has_non_ascii = v & high; // >= 0x80 + return has_quote | has_backslash | has_control | has_non_ascii; +} + +// return the index of the first is_string_special() byte in [data, data+n), or +// n if every byte is ordinary; scans 8 bytes at a time +inline std::size_t find_string_special(const unsigned char* data, std::size_t n) noexcept +{ + std::size_t i = 0; + for (; i + 8 <= n; i += 8) + { + std::uint64_t word = 0; + std::memcpy(&word, data + i, sizeof(word)); + if (swar_string_special(word) != 0) + { + // a special byte is in this word; locate it (endian-agnostic) + for (std::size_t j = 0; j < 8; ++j) + { + if (is_string_special(data[i + j])) + { + return i + j; + } + } + } + } + for (; i < n; ++i) + { + if (is_string_special(data[i])) + { + return i; + } + } + return n; +} + +// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its +// length (2..4) only when the bytes form a *well-formed* sequence using exactly +// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts +// precisely what the byte path accepts. Returns 0 for anything that is invalid, +// incomplete, or that the byte path must diagnose (the caller then defers to +// that path, keeping error messages unchanged). Lead bytes < 0x80 are handled +// by the caller and never passed here. +inline std::size_t validate_one_utf8(const unsigned char* data, std::size_t avail) noexcept +{ + const unsigned char c0 = data[0]; + if (c0 >= 0xC2 && c0 <= 0xDF) // U+0080..U+07FF + { + if (avail >= 2 && data[1] >= 0x80 && data[1] <= 0xBF) + { + return 2; + } + } + else if (c0 == 0xE0) // U+0800..U+0FFF + { + if (avail >= 3 && data[1] >= 0xA0 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF) + { + return 3; + } + } + else if ((c0 >= 0xE1 && c0 <= 0xEC) || c0 == 0xEE || c0 == 0xEF) // U+1000..U+CFFF, U+E000..U+FFFF + { + if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF) + { + return 3; + } + } + else if (c0 == 0xED) // U+D000..U+D7FF (excludes surrogates) + { + if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0x9F && data[2] >= 0x80 && data[2] <= 0xBF) + { + return 3; + } + } + else if (c0 == 0xF0) // U+10000..U+3FFFF + { + if (avail >= 4 && data[1] >= 0x90 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF) + { + return 4; + } + } + else if (c0 >= 0xF1 && c0 <= 0xF3) // U+40000..U+FFFFF + { + if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF) + { + return 4; + } + } + else if (c0 == 0xF4) // U+100000..U+10FFFF + { + if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0x8F && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF) + { + return 4; + } + } + return 0; // invalid, incomplete, or must be diagnosed by the byte path +} + +// Scalar (C++11) computation of the bulk run length: the number of leading +// bytes in [data, data+n) that are ordinary ASCII or complete well-formed UTF-8 +// sequences, stopping before the first byte that needs individual handling (the +// closing quote, an escape, a control character, or an ill-formed/truncated +// sequence). ASCII is skipped 8 bytes at a time. +inline std::size_t scalar_string_bulk_run(const unsigned char* data, std::size_t n) noexcept +{ + std::size_t pos = 0; + while (pos < n) + { + pos += find_string_special(data + pos, n - pos); + if (pos >= n || data[pos] < 0x80u) + { + break; // end of buffer, or a quote/escape/control byte + } + const std::size_t seq = validate_one_utf8(data + pos, n - pos); + if (seq == 0) + { + break; // ill-formed or truncated: let the byte path diagnose it + } + pos += seq; + } + return pos; +} + +#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17) +// Index of the first quote/escape/control byte in [data, data+n) (non-ASCII +// bytes are *not* stops here - the whole run is handed to simdutf), or n. +inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t n) noexcept +{ + constexpr std::uint64_t ones = 0x0101010101010101ull; + constexpr std::uint64_t high = 0x8080808080808080ull; + std::size_t i = 0; + for (; i + 8 <= n; i += 8) + { + std::uint64_t v = 0; + std::memcpy(&v, data + i, sizeof(v)); + const std::uint64_t q = v ^ 0x2222222222222222ull; + const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; + const std::uint64_t hit = ((q - ones) & ~q & high) + | ((b - ones) & ~b & high) + | ((v - 0x2020202020202020ull) & ~v & high); + if (hit != 0) + { + for (std::size_t j = 0; j < 8; ++j) + { + const unsigned char c = data[i + j]; + if (c == '\"' || c == '\\' || c < 0x20u) + { + return i + j; + } + } + } + } + for (; i < n; ++i) + { + const unsigned char c = data[i]; + if (c == '\"' || c == '\\' || c < 0x20u) + { + return i; + } + } + return n; +} +#endif + +// Backend-dispatched bulk run length. With JSON_USE_SIMDUTF the run up to the +// next delimiter is validated in one shot by simdutf; on the rare failure the +// scalar helper recomputes the exact valid prefix so the byte path still +// produces the precise diagnostic. Without it, the pure scalar path is used. +inline std::size_t string_bulk_run(const unsigned char* data, std::size_t n) noexcept +{ +#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17) + const std::size_t run = find_string_delimiter(data, n); + if (run != 0 && simdutf::validate_utf8(reinterpret_cast(data), run)) + { + return run; + } +#endif + return scalar_string_bulk_run(data, n); +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 1d1f290bc..10e7fff6f 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7237,11 +7237,31 @@ class input_stream_adapter // General-purpose iterator-based adapter. It might not be as fast as // theoretically possible for some containers, but it is extremely versatile. -// SentinelType defaults to IteratorType for backward compatibility, but may -// be a different type (e.g., a C++20 sentinel or counted_iterator). +// SentinelType defaults to IteratorType for backward compatibility, but may be +// a different type, e.g. a C++20 sentinel such as std::default_sentinel_t when +// IteratorType is a std::counted_iterator. template class iterator_input_adapter { + // Whether the number of elements between two positions can be computed in + // O(1): either the iterator and the sentinel have the same type (plain + // std::distance) or, in C++20, the sentinel is a sized sentinel for the + // iterator (std::ranges::distance), e.g. std::default_sentinel_t paired + // with std::counted_iterator. + // + // JSON_HAS_RANGES gates the C++20 branch: on standard libraries with an + // incomplete (libstdc++ < 11, see #4440) evaluating + // std::contiguous_iterator on a std::counted_iterator is a hard error + // instead of yielding false, and these traits are instantiated for every + // adapter. Such toolchains fall back to the pointer-only test and simply + // use the byte-at-a-time scanner. + static constexpr bool sentinel_is_sized = +#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) + std::is_same::value || std::sized_sentinel_for; +#else + std::is_same::value; +#endif + public: using char_type = typename std::iterator_traits::value_type; @@ -7253,7 +7273,7 @@ class iterator_input_adapter // in wide_string_input_adapter, which does not expose this). static constexpr bool supports_seek = std::is_same::iterator_category, std::random_access_iterator_tag>::value - && std::is_same::value + && sentinel_is_sized && sizeof(char_type) == 1; iterator_input_adapter(IteratorType first, SentinelType last) @@ -7301,30 +7321,60 @@ class iterator_input_adapter private: // whether IteratorType refers to a contiguous range and therefore supports // a std::memcpy fast path (pointers always do; in C++20 we can also detect - // library iterators such as those of std::vector and std::string). - // Computing the available element count needs either same-type iterators - // (plain std::distance) or, in C++20, a sized sentinel (std::ranges::distance), - // e.g. std::counted_iterator paired with std::default_sentinel_t. - static constexpr bool iterator_is_contiguous = -#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) - (std::is_same::value || std::sized_sentinel_for) - && (std::contiguous_iterator || std::is_pointer::value); + // library iterators such as those of std::vector and std::string). The + // available element count must also be computable in O(1), hence + // sentinel_is_sized. + static constexpr bool iterator_is_contiguous = sentinel_is_sized && +#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) + (std::contiguous_iterator || std::is_pointer::value); #else - std::is_same::value && std::is_pointer::value; + std::is_pointer::value; #endif + // number of unread elements in [current, end) + std::size_t remaining_count() const + { +#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) + // std::ranges::distance also supports sized sentinels of a different + // type (e.g. std::counted_iterator + std::default_sentinel_t) + return static_cast(std::ranges::distance(current, end)); +#else + return static_cast(std::distance(current, end)); +#endif + } + + public: + // Whether the remaining input is a single contiguous block of 1-byte + // elements that the lexer can inspect directly (used for the SWAR string + // fast path). + static constexpr bool supports_bulk_scan = + iterator_is_contiguous && sizeof(char_type) == 1; + + // Pointer to the next unread element; only valid when bulk_remaining() > 0. + const char_type* bulk_data() const + { + return &*current; + } + + // Number of unread elements available as one contiguous block. + std::size_t bulk_remaining() const + { + return remaining_count(); + } + + // Consume @a n elements previously inspected via bulk_data(). + void bulk_skip(std::size_t n) + { + std::advance(current, static_cast::difference_type>(n)); + } + + private: // contiguous fast path: bulk copy the remaining range with std::memcpy template std::size_t get_elements_impl(T* dest, std::size_t count, std::true_type /*contiguous*/) { const std::size_t wanted = count * sizeof(T); -#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) - // std::ranges::distance also supports sized sentinels of a different - // type (e.g. std::counted_iterator + std::default_sentinel_t) - const std::size_t available = static_cast(std::ranges::distance(current, end)) * sizeof(char_type); -#else - const std::size_t available = static_cast(std::distance(current, end)) * sizeof(char_type); -#endif + const std::size_t available = remaining_count() * sizeof(char_type); const std::size_t copied = (std::min)(wanted, available); if (JSON_HEDLEY_LIKELY(copied != 0)) { @@ -7652,6 +7702,46 @@ typename iterator_input_adapter_factory::adapter_typ return factory_type::create(first, last); } +// The element type a container's data() points at, cv-qualifiers removed. +// Ill-formed - and therefore SFINAE-friendly - for types without data(). +template +using container_data_t = typename std::remove_cv().data()) >::type >::type; + +// The container's own element type, cv-qualifiers removed. It is looked up on +// the bare type so it is also found when ContainerType is deduced as a +// reference by the forwarding-reference overload below. +template +using container_value_t = typename std::remove_cv < + typename std::remove_cv::type>::type::value_type >::type; + +// Detect a container that stores its elements contiguously as single bytes +// (std::string, std::vector, std::array, +// std::string_view, ...). Such inputs are wrapped in a pointer-based adapter so +// they benefit from the contiguous fast paths (bulk string scanning, memcpy for +// binary formats) in every C++ standard - not only in C++20, where the standard +// library iterators model std::contiguous_iterator and are detected directly. +// +// data() and size() on their own would be duck typing: they say nothing about +// size() counting the units data() points at, and reading [data(), data() + +// size()) as bytes would be wrong for a type where it does not. Requiring the +// container's own value_type to be that same single-byte element ties the two +// together; every contiguous standard container satisfies it. Anything else +// keeps the iterator-based adapter, which is always correct - only slower. +template +struct is_contiguous_byte_container : std::false_type {}; + +template +struct is_contiguous_byte_container < ContainerType, void_t < + container_data_t, + container_value_t, +decltype(std::declval().size()) >> + : std::integral_constant < bool, + std::is_pointer().data())>::value&& + std::is_integral>::value&& + sizeof(container_data_t) == 1 && + std::is_same, container_value_t>::value > {}; + // Convenience shorthand from container to iterator // Enables ADL on begin(container) and end(container) // Encloses the using declarations in namespace for not to leak them to outside scope @@ -7679,12 +7769,32 @@ struct container_input_adapter_factory< ContainerType, } // namespace container_input_adapter_factory_impl -template -typename container_input_adapter_factory_impl::container_input_adapter_factory::adapter_type input_adapter(ContainerType&& container) +// General container path (iterator-based). Contiguous single-byte containers +// are excluded here and routed through the pointer-based overload below. +template < typename ContainerType, + enable_if_t < !is_contiguous_byte_container::value, int > = 0 > +typename container_input_adapter_factory_impl::container_input_adapter_factory::adapter_type input_adapter(ContainerType && container) { return container_input_adapter_factory_impl::container_input_adapter_factory::create(std::forward(container)); } +// Contiguous single-byte containers (std::string, std::vector, ...) are +// wrapped in a pointer-based adapter so the contiguous fast paths apply in every +// standard. The pointer keeps the container's own element type (const char* for +// std::string, const std::uint8_t* for std::vector, ...), so the +// resulting char_type - and therefore the parsing behavior - is byte-for-byte +// identical to the iterator-based path; only the raw pointer additionally +// enables the bulk fast paths. The container outlives the adapter for the whole +// parse (temporaries live until the end of the full expression), exactly as the +// iterators it replaces did. +template < typename ContainerType, + enable_if_t < is_contiguous_byte_container::value, int > = 0 > +auto input_adapter(const ContainerType& container) +-> decltype(input_adapter(container.data(), container.data() + container.size())) +{ + return input_adapter(container.data(), container.data() + container.size()); +} + // specialization for std::string using string_input_adapter_type = decltype(input_adapter(std::declval())); @@ -7813,8 +7923,557 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // array +#include // FLT_EVAL_METHOD +#include // size_t +#include // int64_t, uint64_t +#include // numeric_limits + +// #include + + +// std::from_chars lives in , but being in C++17 mode does not +// guarantee the header exists: GCC 7 sets __cplusplus to C++17 yet ships no +// (added in GCC 8; floating-point support in GCC 11). Guard the +// include with __has_include so such toolchains fall back to the scalar path. +#if defined(JSON_HAS_CPP_17) && defined(__has_include) + #if __has_include() + #include // from_chars (only used when __cpp_lib_to_chars is defined) + #include // errc + #endif +#endif + +// This file contains the value-conversion helpers used by the lexer to turn an +// already-validated number token into a value, without the locale/errno +// overhead of std::strtoull/std::strtod. They are free functions so the lexer +// stays focused on scanning; see lexer::convert_number(). + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief fast integer parser for an already-validated unsigned integer + +The number scanner has already checked that [first, last) is a valid JSON +integer, so this only needs to accumulate the digits and detect overflow. This +avoids the locale/errno machinery of std::strtoull, which dominates +integer-heavy inputs. + +@param[in] first pointer to the first character (a digit) +@param[in] last pointer past the last character +@param[out] value the parsed value on success +@return true if the value fit into @a NumberUnsignedType; false on overflow, in + which case the caller falls back to floating-point parsing (matching the + previous std::strtoull behavior) +*/ +template +bool parse_integer_unsigned(const char* first, const char* last, NumberUnsignedType& value) noexcept +{ + // accumulate in the widest unsigned type used by the previous strtoull + // path so the overflow behavior is unchanged for custom number types + std::uint64_t x = 0; + constexpr std::uint64_t cutoff = (std::numeric_limits::max)() / 10u; + constexpr std::uint64_t cutlim = (std::numeric_limits::max)() % 10u; + for (const char* p = first; p != last; ++p) + { + const auto digit = static_cast(static_cast(*p) - static_cast('0')); + if (JSON_HEDLEY_UNLIKELY(x > cutoff || (x == cutoff && digit > cutlim))) + { + return false; + } + x = (x * 10u) + digit; + } + value = static_cast(x); + // reject values that do not round-trip into a narrower NumberUnsignedType + return static_cast(value) == x; +} + +/*! +@brief fast integer parser for an already-validated negative integer + +@param[in] first pointer to the leading '-' +@param[in] last pointer past the last character +@param[out] value the parsed (negative) value on success +@return true on success; false on overflow (caller falls back to float) +*/ +template +bool parse_integer_signed(const char* first, const char* last, NumberIntegerType& value) noexcept +{ + // the state machine only reaches the signed path via a leading '-' + JSON_ASSERT(first != last && *first == '-'); + std::uint64_t magnitude = 0; + // |INT64_MIN| == INT64_MAX + 1; this is the largest admissible magnitude + constexpr std::uint64_t limit = static_cast((std::numeric_limits::max)()) + 1u; + for (const char* p = first + 1; p != last; ++p) + { + const auto digit = static_cast(static_cast(*p) - static_cast('0')); + if (JSON_HEDLEY_UNLIKELY(magnitude > (limit - digit) / 10u)) + { + return false; + } + magnitude = (magnitude * 10u) + digit; + } + const std::int64_t x = (magnitude == limit) + ? (std::numeric_limits::min)() + : -static_cast(magnitude); + value = static_cast(x); + // reject values that do not round-trip into a narrower NumberIntegerType + return static_cast(value) == x; +} + +/*! +@brief exact fast path for parsing a `double` (Clinger's algorithm) + +For the common case - at most 19 significant digits, a decimal exponent in +[-22, 22], and a significand below 2^53 - the value equals significand * +10^exp computed in IEEE-754 double arithmetic, which is exact under +round-to-nearest because both operands are exactly representable. This is the +same fast path used by fast_float/simdjson; the general cases are left to +std::strtod. The parser only activates for number_float_t == double; float and +long double keep the std::strtof/std::strtold paths (see the templated overload +below). + +@param[in] first pointer to the first character of the number +@param[in] last pointer past the last character +@param[in] decimal_point the (locale-dependent) decimal point character +@param[out] out the parsed value on success +@return true if the value was parsed exactly; false to fall back to strtod +*/ +template +bool parse_float_fast(const char* first, const char* last, DecimalPointType decimal_point, double& out) noexcept +{ +#if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0 + // Clinger's fast path is only exact when double operations are evaluated in + // true double precision. On platforms that keep intermediates in extended + // precision (e.g. the x87 FPU on 32-bit x86, where FLT_EVAL_METHOD == 2) the + // single significand * 10^scale step is double-rounded and can be 1 ULP off, + // so decline and let the caller fall back to the correctly-rounded + // std::from_chars / std::strtod path. + static_cast(first); + static_cast(last); + static_cast(decimal_point); + static_cast(out); + return false; +#else + static const std::array powers_of_ten = + { + { + 1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11, + 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22 + } + }; + + const char* p = first; + bool negative = false; + if (p != last && (*p == '-' || *p == '+')) + { + negative = (*p == '-'); + ++p; + } + + std::uint64_t significand = 0; + int num_digits = 0; + int fractional_digits = 0; + bool seen_dot = false; + bool any_digit = false; + for (; p != last; ++p) + { + const char c = *p; + if (c >= '0' && c <= '9') + { + any_digit = true; + if (JSON_HEDLEY_UNLIKELY(num_digits >= 19)) + { + return false; // significand may not fit into uint64_t + } + significand = (significand * 10u) + static_cast(c - '0'); + ++num_digits; + fractional_digits += static_cast(seen_dot); + } + else if (static_cast(c) == decimal_point) + { + if (JSON_HEDLEY_UNLIKELY(seen_dot)) + { + return false; + } + seen_dot = true; + } + else if (c == 'e' || c == 'E') + { + ++p; + break; + } + else + { + return false; + } + } + if (JSON_HEDLEY_UNLIKELY(!any_digit)) + { + return false; + } + + int exponent = 0; + if (p != last) // an exponent part remains + { + bool exp_negative = false; + if (p != last && (*p == '-' || *p == '+')) + { + exp_negative = (*p == '-'); + ++p; + } + bool any_exp_digit = false; + for (; p != last; ++p) + { + if (JSON_HEDLEY_UNLIKELY(*p < '0' || *p > '9')) + { + return false; + } + exponent = (exponent * 10) + (*p - '0'); + any_exp_digit = true; + if (JSON_HEDLEY_UNLIKELY(exponent > 9999)) + { + return false; + } + } + if (JSON_HEDLEY_UNLIKELY(!any_exp_digit)) + { + return false; + } + if (exp_negative) + { + exponent = -exponent; + } + } + + const int scale = exponent - fractional_digits; + if (JSON_HEDLEY_UNLIKELY(significand >= (static_cast(1) << 53))) + { + return false; // significand not exactly representable as double + } + + auto result = static_cast(significand); + if (scale >= 0) + { + if (JSON_HEDLEY_UNLIKELY(scale > 22)) + { + return false; + } + result *= powers_of_ten[static_cast(scale)]; + } + else + { + if (JSON_HEDLEY_UNLIKELY(-scale > 22)) + { + return false; + } + result /= powers_of_ten[static_cast(-scale)]; + } + out = negative ? -result : result; + return true; +#endif +} + +/// fast float path is only exact for `double`; decline for float/long double +template +bool parse_float_fast(const char* /*first*/, const char* /*last*/, DecimalPointType /*decimal_point*/, FloatType& /*out*/) noexcept +{ + return false; +} + +/*! +@brief parse a float with std::from_chars (Eisel-Lemire) when available + +std::from_chars is locale-independent, correctly rounded, and - via the +Eisel-Lemire algorithm in modern standard libraries - much faster than strtod +over the whole value range (not just the Clinger subset). It is used only when +__cpp_lib_to_chars indicates full floating-point support and only when it +consumes the entire token ([first, last)); a partial parse means the buffer +uses a non-'.' locale decimal point, in which case the caller falls back to the +locale-aware path. An under-/overflow (result_out_of_range) also declines, so +the caller's strtod fallback supplies the well-defined ±inf/0 result the parser +expects (side-stepping the P4168 divergence between implementations). + +@return true if the value was parsed exactly and fully; false to fall back +*/ +template +bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept +{ + // JSON_HAS_CPP_17 must gate the use as well as the include above: + // some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even + // in C++14 mode, where is not included. +#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars) + const auto result = std::from_chars(first, last, out); + return result.ec == std::errc() && result.ptr == last; +#else + static_cast(first); + static_cast(last); + static_cast(out); + return false; +#endif +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // size_t +#include // uint64_t +#include // memcpy + +// #include + + +// Optional SIMD backend for bulk UTF-8 validation. This is an opt-in external +// dependency: nlohmann/json itself stays header-only and the C++11 scalar +// validator below is always available; defining JSON_USE_SIMDUTF additionally +// requires the simdutf headers on the include path and linking the simdutf +// library. See string_bulk_run(). +// +// simdutf.h itself requires C++17 - it rejects older standards with an #error - +// so the backend is only compiled in from C++17 on. Below that the macro has no +// effect and the scalar validator is used; it accepts and rejects exactly the +// same input, so only throughput differs. macro_scope.hpp is included above to +// have JSON_HAS_CPP_17 available for this test. +#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17) + #include +#endif + +// This file contains the byte-level string-scanning helpers used by the lexer's +// contiguous fast path. They operate purely on raw bytes (no dependency on the +// lexer's template parameters) so they are free functions, keeping the lexer +// itself focused on the state machine; see lexer::scan_string_bulk(). + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +// classify a single byte as needing individual string handling: the closing +// quote, an escape, a control character, or a non-ASCII (UTF-8) +// lead/continuation byte. Ordinary bytes (0x20..0x7F except '"' and '\\') are +// copied verbatim, which the bulk scanner does 8 bytes at a time. +inline bool is_string_special(unsigned char c) noexcept +{ + return c == '\"' || c == '\\' || c < 0x20u || c >= 0x80u; +} + +// SWAR helper: return a word whose high bit is set in every byte of @a v that +// is_string_special(); zero if the 8 bytes are all ordinary. +inline std::uint64_t swar_string_special(std::uint64_t v) noexcept +{ + constexpr std::uint64_t ones = 0x0101010101010101ull; + constexpr std::uint64_t high = 0x8080808080808080ull; + const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22) + const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C) + const std::uint64_t has_quote = (q - ones) & ~q & high; + const std::uint64_t has_backslash = (b - ones) & ~b & high; + const std::uint64_t has_control = (v - 0x2020202020202020ull) & ~v & high; // < 0x20 + const std::uint64_t has_non_ascii = v & high; // >= 0x80 + return has_quote | has_backslash | has_control | has_non_ascii; +} + +// return the index of the first is_string_special() byte in [data, data+n), or +// n if every byte is ordinary; scans 8 bytes at a time +inline std::size_t find_string_special(const unsigned char* data, std::size_t n) noexcept +{ + std::size_t i = 0; + for (; i + 8 <= n; i += 8) + { + std::uint64_t word = 0; + std::memcpy(&word, data + i, sizeof(word)); + if (swar_string_special(word) != 0) + { + // a special byte is in this word; locate it (endian-agnostic) + for (std::size_t j = 0; j < 8; ++j) + { + if (is_string_special(data[i + j])) + { + return i + j; + } + } + } + } + for (; i < n; ++i) + { + if (is_string_special(data[i])) + { + return i; + } + } + return n; +} + +// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its +// length (2..4) only when the bytes form a *well-formed* sequence using exactly +// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts +// precisely what the byte path accepts. Returns 0 for anything that is invalid, +// incomplete, or that the byte path must diagnose (the caller then defers to +// that path, keeping error messages unchanged). Lead bytes < 0x80 are handled +// by the caller and never passed here. +inline std::size_t validate_one_utf8(const unsigned char* data, std::size_t avail) noexcept +{ + const unsigned char c0 = data[0]; + if (c0 >= 0xC2 && c0 <= 0xDF) // U+0080..U+07FF + { + if (avail >= 2 && data[1] >= 0x80 && data[1] <= 0xBF) + { + return 2; + } + } + else if (c0 == 0xE0) // U+0800..U+0FFF + { + if (avail >= 3 && data[1] >= 0xA0 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF) + { + return 3; + } + } + else if ((c0 >= 0xE1 && c0 <= 0xEC) || c0 == 0xEE || c0 == 0xEF) // U+1000..U+CFFF, U+E000..U+FFFF + { + if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF) + { + return 3; + } + } + else if (c0 == 0xED) // U+D000..U+D7FF (excludes surrogates) + { + if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0x9F && data[2] >= 0x80 && data[2] <= 0xBF) + { + return 3; + } + } + else if (c0 == 0xF0) // U+10000..U+3FFFF + { + if (avail >= 4 && data[1] >= 0x90 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF) + { + return 4; + } + } + else if (c0 >= 0xF1 && c0 <= 0xF3) // U+40000..U+FFFFF + { + if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF) + { + return 4; + } + } + else if (c0 == 0xF4) // U+100000..U+10FFFF + { + if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0x8F && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF) + { + return 4; + } + } + return 0; // invalid, incomplete, or must be diagnosed by the byte path +} + +// Scalar (C++11) computation of the bulk run length: the number of leading +// bytes in [data, data+n) that are ordinary ASCII or complete well-formed UTF-8 +// sequences, stopping before the first byte that needs individual handling (the +// closing quote, an escape, a control character, or an ill-formed/truncated +// sequence). ASCII is skipped 8 bytes at a time. +inline std::size_t scalar_string_bulk_run(const unsigned char* data, std::size_t n) noexcept +{ + std::size_t pos = 0; + while (pos < n) + { + pos += find_string_special(data + pos, n - pos); + if (pos >= n || data[pos] < 0x80u) + { + break; // end of buffer, or a quote/escape/control byte + } + const std::size_t seq = validate_one_utf8(data + pos, n - pos); + if (seq == 0) + { + break; // ill-formed or truncated: let the byte path diagnose it + } + pos += seq; + } + return pos; +} + +#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17) +// Index of the first quote/escape/control byte in [data, data+n) (non-ASCII +// bytes are *not* stops here - the whole run is handed to simdutf), or n. +inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t n) noexcept +{ + constexpr std::uint64_t ones = 0x0101010101010101ull; + constexpr std::uint64_t high = 0x8080808080808080ull; + std::size_t i = 0; + for (; i + 8 <= n; i += 8) + { + std::uint64_t v = 0; + std::memcpy(&v, data + i, sizeof(v)); + const std::uint64_t q = v ^ 0x2222222222222222ull; + const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; + const std::uint64_t hit = ((q - ones) & ~q & high) + | ((b - ones) & ~b & high) + | ((v - 0x2020202020202020ull) & ~v & high); + if (hit != 0) + { + for (std::size_t j = 0; j < 8; ++j) + { + const unsigned char c = data[i + j]; + if (c == '\"' || c == '\\' || c < 0x20u) + { + return i + j; + } + } + } + } + for (; i < n; ++i) + { + const unsigned char c = data[i]; + if (c == '\"' || c == '\\' || c < 0x20u) + { + return i; + } + } + return n; +} +#endif + +// Backend-dispatched bulk run length. With JSON_USE_SIMDUTF the run up to the +// next delimiter is validated in one shot by simdutf; on the rare failure the +// scalar helper recomputes the exact valid prefix so the byte path still +// produces the precise diagnostic. Without it, the pure scalar path is used. +inline std::size_t string_bulk_run(const unsigned char* data, std::size_t n) noexcept +{ +#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17) + const std::size_t run = find_string_delimiter(data, n); + if (run != 0 && simdutf::validate_utf8(reinterpret_cast(data), run)) + { + return run; + } +#endif + return scalar_string_bulk_run(data, n); +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // #include // #include @@ -7922,6 +8581,25 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/) return false; } +// Detect whether an input adapter exposes a contiguous byte block that the +// lexer can scan directly (see iterator_input_adapter::supports_bulk_scan). +// Adapters without the flag - file, stream, wide-string, user-defined - fall +// back to the character-at-a-time string scanner. +template +using detect_supports_bulk_scan = decltype(InputAdapterType::supports_bulk_scan); + +template +constexpr bool input_adapter_supports_bulk_scan(std::true_type /*detected*/) +{ + return InputAdapterType::supports_bulk_scan; +} + +template +constexpr bool input_adapter_supports_bulk_scan(std::false_type /*detected*/) +{ + return false; +} + /*! @brief lexical analysis @@ -7943,6 +8621,14 @@ class lexer : public lexer_base static constexpr bool lazy_token_string = input_adapter_supports_seek(is_detected {}); + /// whether string scanning may bulk-consume runs of ordinary characters + /// directly from a contiguous input buffer (SWAR fast path). This requires + /// the token to be reconstructible lazily (lazy_token_string), so bypassing + /// the per-character capture in get() cannot lose error diagnostics. + static constexpr bool bulk_scan = + lazy_token_string + && input_adapter_supports_bulk_scan(is_detected {}); + public: using token_type = typename lexer_base::token_type; @@ -8063,6 +8749,40 @@ class lexer : public lexer_base return true; } + /// contiguous input: bulk-append the run of ordinary characters and complete + /// well-formed UTF-8 sequences starting at the current read position, leaving + /// the first byte that needs individual handling (the closing quote, an + /// escape, a control character, or an ill-formed UTF-8 byte) for get() + void scan_string_bulk(std::true_type /*bulk*/) + { + // a pending unget must be consumed through the normal path first + if (next_unget) + { + return; + } + const std::size_t remaining = ia.bulk_remaining(); + if (remaining == 0) + { + return; + } + const auto* const data = reinterpret_cast(ia.bulk_data()); + + const std::size_t pos = string_bulk_run(data, remaining); + if (pos == 0) + { + return; + } + token_buffer.append(reinterpret_cast(data), pos); + ia.bulk_skip(pos); + // the run contains no newline (all bytes < 0x20 are treated as special), + // so only the flat character counters advance + position.chars_read_total += pos; + position.chars_read_current_line += pos; + } + + /// streaming input: no bulk fast path + void scan_string_bulk(std::false_type /*bulk*/) const noexcept {} + /*! @brief scan a string literal @@ -8088,6 +8808,10 @@ class lexer : public lexer_base while (true) { + // bulk-consume ordinary characters from contiguous input, then + // handle the next special byte through the switch below + scan_string_bulk(std::integral_constant {}); + // get the next character switch (get()) { @@ -8806,6 +9530,12 @@ class lexer : public lexer_base // changed if minus sign, decimal point, or exponent is read token_type number_type = token_type::value_unsigned; + // offset just past the last mantissa byte in token_buffer (i.e. the + // index of 'e'/'E', or the whole token when there is no exponent). + // convert_number() uses it to count significant digits; npos means + // "not seen an exponent yet" and is resolved at scan_number_done + std::size_t mantissa_end = std::string::npos; + // state (init): we just found out we need to scan a number switch (current) { @@ -8991,6 +9721,9 @@ scan_number_decimal2: scan_number_exponent: // we just parsed an exponent number_type = token_type::value_float; + // this label is reached only right after the 'e'/'E' was appended (from + // the zero, any1, and decimal2 states), so the mantissa ends before it + mantissa_end = token_buffer.size() - 1; switch (get()) { case '+': @@ -9077,6 +9810,116 @@ scan_number_done: // we are done scanning a number) unget(); + // no exponent was scanned: the mantissa spans the whole token + if (mantissa_end == std::string::npos) + { + mantissa_end = token_buffer.size(); + } + + return convert_number(number_type, mantissa_end); + } + + /*! + @brief convert an already-validated integer token to its value + + The digit sequence in [first, last) has been validated by the caller, so a + dedicated parser can avoid the locale/errno overhead of std::strtoull. + + @return the token type on success; token_type::uninitialized if @a + number_type is not an integer type or the value does not fit, in + which case the caller falls back to the floating-point conversion + (matching the previous std::strtoull/std::strtoll behavior) + */ + token_type convert_integer(token_type number_type, const char* first, const char* last) + { + if (number_type == token_type::value_unsigned) + { + if (parse_integer_unsigned(first, last, value_unsigned)) + { + return token_type::value_unsigned; + } + } + else if (number_type == token_type::value_integer) + { + if (parse_integer_signed(first, last, value_integer)) + { + return token_type::value_integer; + } + } + + return token_type::uninitialized; + } + + /*! + @brief check whether Clinger's fast path can still succeed for this token + + parse_float_fast() needs a significand below 2^53. A mantissa with 17 or + more significant digits is at least 10^16 and therefore always exceeds it, + so calling the fast path would walk the token one extra time only to + decline before strtod has to run anyway. + + Significant digits are the mantissa's digits from the first nonzero one on; + the sign, the decimal point, leading zeros, and the exponent do not count. + The answer is derived from indices - the digits are not scanned again - so + this stays off the hot path of the number scanners. + + @param[in] mantissa_end offset just past the last mantissa byte in + token_buffer + @return false if parse_float_fast() is guaranteed to decline + */ + bool mantissa_fits_clinger(std::size_t mantissa_end) const + { + // 10^16 already exceeds 2^53, so 17 digits can never fit + constexpr std::size_t limit = 17; + + const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u; + const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u; + // the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so + // a leading zero can only be a lone "0", which is not significant + const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u; + JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero); + std::size_t digits = mantissa_end - neg - has_dot - lead_zero; + + if (JSON_HEDLEY_LIKELY(digits < limit)) + { + return true; + } + + // Only a number below 1 can carry further insignificant zeros, and only + // while the count stays at the limit does removing them change the + // answer - so this loop is skipped for all but a few tokens. Note + // token_buffer holds the locale's decimal point, so the fraction is + // located through decimal_point_position rather than by searching '.'. + if (lead_zero != 0) + { + JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit + for (std::size_t i = decimal_point_position + 1; + digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i) + { + --digits; + } + } + + return digits < limit; + } + + /*! + @brief convert the number text in token_buffer to its value and token type + + The digit sequence in token_buffer has already been validated (by the + scan_number() state machine or by the contiguous fast path) and holds the + locale decimal point in place of '.'. Integers are parsed first and fall + back to floating point on overflow. This is shared so both scanners produce + identical results. + + @param[in] mantissa_end offset just past the last mantissa byte in + token_buffer (the index of 'e'/'E', or + token_buffer.size() when there is no exponent); + used to skip Clinger's fast path when it cannot + possibly succeed - see mantissa_fits_clinger() + */ + token_type convert_number(token_type number_type, std::size_t mantissa_end) + { // If the caller does not need the converted value (only whether the // input is syntactically valid; see json_sax_acceptor/accept()), an // unsigned/integer token can be reported without calling @@ -9129,45 +9972,37 @@ scan_number_done: } } - char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) - errno = 0; + const char* const num_begin = token_buffer.data(); + const char* const num_end = num_begin + token_buffer.size(); - // try to parse integers first and fall back to floats - if (number_type == token_type::value_unsigned) + if (number_type != token_type::value_float) { - const auto x = std::strtoull(token_buffer.data(), &endptr, 10); - - // we checked the number format before - JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); - - if (errno != ERANGE) + const token_type integer_result = convert_integer(number_type, num_begin, num_end); + if (integer_result != token_type::uninitialized) { - value_unsigned = static_cast(x); - if (value_unsigned == x) - { - return token_type::value_unsigned; - } - } - } - else if (number_type == token_type::value_integer) - { - const auto x = std::strtoll(token_buffer.data(), &endptr, 10); - - // we checked the number format before - JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); - - if (errno != ERANGE) - { - value_integer = static_cast(x); - if (value_integer == x) - { - return token_type::value_integer; - } + return integer_result; } } // this code is reached if we parse a floating-point number or if an - // integer conversion above failed + // integer conversion above overflowed. Prefer std::from_chars + // (Eisel-Lemire, locale-independent, correctly rounded) when available; + // otherwise the exact Clinger fast path (double only); otherwise the + // locale-aware strtof/strtod. + if (parse_float_from_chars(num_begin, num_end, value_float)) + { + return token_type::value_float; + } + // Skipping a fast path that cannot succeed is lossless and saves a full + // extra pass over the token's bytes, which otherwise shows up on + // high-precision inputs such as canada.json + if (mantissa_fits_clinger(mantissa_end) + && parse_float_fast(num_begin, num_end, decimal_point_char, value_float)) + { + return token_type::value_float; + } + + char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) strtof(value_float, token_buffer.data(), &endptr); // we checked the number format before @@ -9176,6 +10011,158 @@ scan_number_done: return token_type::value_float; } + /*! + @brief contiguous fast path for scanning a number + + Parses the whole number token straight from the input buffer, avoiding the + per-character get()/add() of scan_number(). On success it fills token_buffer + (with the locale decimal point substituted, as scan_number() does) and + returns the token type. On anything it does not fully recognize as a + well-formed number it makes no state change and returns + token_type::uninitialized, so the caller falls back to scan_number(), which + then produces the exact diagnostic. @a current is the first digit or the + leading minus (already read); the remaining bytes are taken from the adapter. + */ + token_type scan_number_bulk_contiguous() + { + // a pending unget offsets the buffer position from current; fall back + if (next_unget) + { + return token_type::uninitialized; + } + const std::size_t rem = ia.bulk_remaining(); + if (rem == 0) + { + // the first digit is the last input byte; let scan_number() finish + return token_type::uninitialized; + } + // the byte before the next unread one is current (contiguous input) + const char* const data = reinterpret_cast(ia.bulk_data()) - 1; + const std::size_t avail = rem + 1; + + // validate + classify the number extent (mirrors scan_number()'s grammar) + std::size_t i = 0; + std::size_t dot_index = std::string::npos; + token_type number_type = token_type::value_unsigned; + if (data[0] == '-') + { + number_type = token_type::value_integer; + i = 1; + if (i >= avail) + { + return token_type::uninitialized; + } + } + if (data[i] == '0') + { + ++i; + } + else if (data[i] >= '1' && data[i] <= '9') + { + ++i; + while (i < avail && data[i] >= '0' && data[i] <= '9') + { + ++i; + } + } + else + { + return token_type::uninitialized; + } + if (i < avail && data[i] == '.') + { + number_type = token_type::value_float; + dot_index = i; + ++i; + if (i >= avail || !(data[i] >= '0' && data[i] <= '9')) + { + return token_type::uninitialized; + } + while (i < avail && data[i] >= '0' && data[i] <= '9') + { + ++i; + } + } + // the mantissa ends here, whether or not an exponent part follows + const std::size_t mantissa_end = i; + if (i < avail && (data[i] == 'e' || data[i] == 'E')) + { + number_type = token_type::value_float; + ++i; + if (i < avail && (data[i] == '+' || data[i] == '-')) + { + ++i; + } + if (i >= avail || !(data[i] >= '0' && data[i] <= '9')) + { + return token_type::uninitialized; + } + while (i < avail && data[i] >= '0' && data[i] <= '9') + { + ++i; + } + } + const std::size_t len = i; + + // reset() records where this token starts (for diagnostics), so it has + // to run before the input position advances below + reset(); + + // An integer token needs no token_buffer: the SAX callbacks for + // number_integer/number_unsigned take only the value, and the overflow + // diagnostic rebuilds the text from the input. Convert straight from the + // input buffer and leave token_buffer empty. (JSON_DIAGNOSTIC_POSITIONS + // derives a number's start position from get_string().size(), so there + // the token still has to be materialized.) +#if !JSON_DIAGNOSTIC_POSITIONS + if (number_type != token_type::value_float) + { + const token_type integer_result = convert_integer(number_type, data, data + len); + if (JSON_HEDLEY_LIKELY(integer_result != token_type::uninitialized)) + { + ia.bulk_skip(len - 1); + position.chars_read_total += (len - 1); + position.chars_read_current_line += (len - 1); + return integer_result; + } + // The value does not fit an integer, so this token converts as a + // float. Recording that here keeps convert_number() below from + // repeating the integer attempt that just failed. + number_type = token_type::value_float; + } +#endif + + // materialize the token exactly as scan_number() would, substituting the + // locale decimal point so convert_number()'s strtof fallback stays valid. + // reset() already cleared token_buffer, so append() fills it (assign() is + // avoided because custom string_t types need not provide it) + token_buffer.append(reinterpret_cast(data), len); + if (dot_index != std::string::npos) + { + token_buffer[dot_index] = static_cast(decimal_point_char); + decimal_point_position = dot_index; + } + + ia.bulk_skip(len - 1); + position.chars_read_total += (len - 1); + position.chars_read_current_line += (len - 1); + + return convert_number(number_type, mantissa_end); + } + + /// contiguous input: try the number fast path, else the byte-path scanner + token_type scan_number_dispatch(std::true_type /*bulk*/) + { + const token_type t = scan_number_bulk_contiguous(); + return (t != token_type::uninitialized) ? t : scan_number(); + } + + /// streaming input: always use the byte-path scanner + token_type scan_number_dispatch(std::false_type /*bulk*/) + { + return scan_number(); + } + /*! @param[in] literal_text the literal text to expect @param[in] length the length of the passed literal text @@ -9279,6 +10266,9 @@ scan_number_done: if (current == '\n') { ++position.lines_read; + // remember the column the newline was read at: chars_read_current_line + // is about to be cleared, and a matching unget() cannot reconstruct it + chars_read_before_newline = position.chars_read_current_line; position.chars_read_current_line = 0; } @@ -9335,12 +10325,20 @@ scan_number_done: --position.chars_read_total; // in case we "unget" a newline, we have to also decrement the lines_read + // and restore the column that get() cleared when it saw the newline; + // chars_read_current_line == 0 can only mean the last get() read one if (position.chars_read_current_line == 0) { if (position.lines_read > 0) { --position.lines_read; } + + // chars_read_before_newline counts the newline itself, which is the + // character being ungotten, hence the -1 + position.chars_read_current_line = (chars_read_before_newline > 0) + ? chars_read_before_newline - 1 + : 0; } else { @@ -9607,7 +10605,7 @@ scan_number_done: case '7': case '8': case '9': - return scan_number(); + return scan_number_dispatch(std::integral_constant {}); // end of input (the null byte is needed when parsing from // string literals) @@ -9638,6 +10636,10 @@ scan_number_done: /// the start position of the current token position_t position {}; + /// the value chars_read_current_line had when the last newline was read, so + /// that unget() can restore the column instead of leaving it at 0 + std::size_t chars_read_before_newline = 0; + /// raw input token string for error messages; only populated for streaming /// adapters (seekable adapters reconstruct it lazily via token_string_start) std::vector token_string {}; diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 2d0aaaf70..322e40d80 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -2,6 +2,9 @@ cmake_minimum_required(VERSION 3.13...4.0) option(JSON_Valgrind "Execute test suite with Valgrind." OFF) option(JSON_FastTests "Skip expensive/slow tests." OFF) +option(JSON_TestSimdutf "Build the unit tests against the simdutf UTF-8 validation backend." OFF) + +set(JSON_SIMDUTF_VERSION 9.1.0 CACHE STRING "The simdutf version used by JSON_TestSimdutf.") set(JSON_32bitTest AUTO CACHE STRING "Enable the 32bit unit test (ON/OFF/AUTO/ONLY).") set(JSON_TestStandards "" CACHE STRING "The list of standards to test explicitly.") @@ -194,6 +197,71 @@ if(test_force) endif() message(STATUS "${msg}") +############################################################################# +# optionally validate UTF-8 with simdutf (JSON_USE_SIMDUTF) +############################################################################# + +# The simdutf backend is opt-in and not vendored, so it is fetched here rather +# than being a checked-in dependency. Everything below hangs off test_main, +# whose usage requirements every test target inherits; the library target and +# the installed CMake package are deliberately left untouched. +if (JSON_TestSimdutf) + # simdutf requires C++17, both to compile itself and to be reachable from + # the library, which keeps its scalar validator below that. Find a tested + # standard that satisfies it. + set(simdutf_standard "") + foreach(cxx_standard ${test_cxx_standards}) + if(NOT cxx_standard LESS 17 AND compiler_supports_cpp_${cxx_standard}) + set(simdutf_standard ${cxx_standard}) + break() + endif() + endforeach() + + if("${simdutf_standard}" STREQUAL "") + # Building simdutf would fail outright without a C++17 compiler, and + # even with one it would go unused if no C++17-or-later standard is + # tested. Say so and fall back to the scalar validator rather than + # failing the build. + if(NOT compiler_supports_cpp_17) + set(simdutf_reason "the compiler does not support C++17") + else() + set(simdutf_reason "no tested standard is C++17 or later (testing ${msg_standards})") + endif() + message(WARNING + "JSON_TestSimdutf is enabled, but ${simdutf_reason}. simdutf requires C++17, so it " + "is not fetched and JSON_USE_SIMDUTF is not defined: the tests run against the " + "built-in scalar UTF-8 validator instead. Set JSON_TestStandards to include 17 or " + "later, or build with a compiler that supports C++17.") + else() + if (CMAKE_VERSION VERSION_LESS 3.18) + message(FATAL_ERROR "JSON_TestSimdutf requires CMake 3.18 or later (simdutf's minimum).") + endif() + + include(FetchContent) + + # simdutf builds its tests and tools by default, and its tests pull + # further dependencies of their own; only the library is needed here + set(SIMDUTF_TESTS OFF CACHE BOOL "" FORCE) + set(SIMDUTF_TOOLS OFF CACHE BOOL "" FORCE) + set(SIMDUTF_BENCHMARKS OFF CACHE BOOL "" FORCE) + set(SIMDUTF_ICONV OFF CACHE BOOL "" FORCE) + + FetchContent_Declare(simdutf + URL https://github.com/simdutf/simdutf/archive/refs/tags/v${JSON_SIMDUTF_VERSION}.tar.gz + DOWNLOAD_EXTRACT_TIMESTAMP TRUE + ) + FetchContent_MakeAvailable(simdutf) + + target_compile_definitions(test_main PUBLIC JSON_USE_SIMDUTF) + target_link_libraries(test_main PUBLIC simdutf::simdutf) + + # simdutf.h requires C++17; below that the library keeps its scalar + # validator, so any C++11/14 test targets exercise the fallback and the + # C++17-and-later ones exercise simdutf. Both must agree. + message(STATUS "UTF-8 validation delegated to simdutf ${JSON_SIMDUTF_VERSION} for C++17 and later (JSON_USE_SIMDUTF)") + endif() +endif() + # *DO* use json_test_set_test_options() above this line json_test_should_build_32bit_test(json_32bit_test json_32bit_test_only "${JSON_32bitTest}") diff --git a/tests/src/unit-class_lexer.cpp b/tests/src/unit-class_lexer.cpp index 64baf3da6..e89497738 100644 --- a/tests/src/unit-class_lexer.cpp +++ b/tests/src/unit-class_lexer.cpp @@ -12,6 +12,11 @@ #include using nlohmann::json; +#include // strtod +#include // stringstream +#include // string +#include // vector + namespace { // shortcut to scan a string literal @@ -224,3 +229,431 @@ TEST_CASE("lexer class") CHECK((scan_string("/**//**//**/", true) == json::lexer::token_type::end_of_input)); } } + +TEST_CASE("lexer number fast path") +{ + // The contiguous fast path (used for pointer/string input) must agree with + // the streaming byte path (used for std::istream) on token type, numeric + // value, and round-trip text for every well-formed number, and reject the + // same malformed numbers with the same message. + SECTION("contiguous vs streaming parity") + { + const std::vector numbers = + { + "0", "-0", "1", "-1", "42", "-42", "10", "100", "1234567890", + "0.0", "-0.0", "3.14", "-3.14", "0.5", "-0.001", "123.456789", + "1e0", "1E0", "1e10", "1e-10", "1e+10", "1.5e3", "-2.5E-4", + "9223372036854775807", // INT64_MAX -> unsigned + "9223372036854775808", // INT64_MAX + 1 -> unsigned + "18446744073709551615", // UINT64_MAX -> unsigned + "18446744073709551616", // UINT64_MAX + 1 -> float + "-9223372036854775808", // INT64_MIN -> integer + "-9223372036854775809", // INT64_MIN - 1 -> float + "123456789012345678901234567890", // huge -> float + "0.30000000000000004", "2.2250738585072014e-308", "1e308", + // high-precision / wide-exponent values that exercise the + // std::from_chars (Eisel-Lemire) path beyond the Clinger subset + "1.7976931348623157e308", "1.2345678901234567e-250", + "9007199254740993", "5e-324", "1e-320" + }; + + for (const auto& n : numbers) + { + const std::string doc = "[" + n + "]"; + + // contiguous fast path + const json a = json::parse(doc); + // streaming byte path + std::stringstream ss(doc); + const json b = json::parse(ss); + + CAPTURE(n); + CHECK(a == b); + CHECK(a.dump() == b.dump()); + CHECK(a[0].type() == b[0].type()); + } + } + + SECTION("significant-digit gate for the Clinger fast path") + { + // Clinger's fast path needs a significand below 2^53, so it cannot + // succeed once the mantissa has 17 or more significant digits (the + // significand would be at least 10^16). The lexer skips the attempt + // there. That is only allowed to save work: every value must still come + // out bit-exactly, and both scanners must agree. In particular the gate + // must not fire for tokens whose leading zeros merely look like extra + // digits - "0.1234567890123456" has 16 significant digits, not 17. + const std::vector numbers = + { + "1234567890123456", // 16 significant digits + "12345678901234567", // 17 -> attempt skipped + "123456789012345678", // 18 -> attempt skipped + "0.1234567890123456", // 16: the leading "0" is not significant + "0.12345678901234567", // 17 + "0.00000000000000001", // 1, in a long token + "0.000000000000000012345678901234", // 14, in a long token + "-0.0000000000000000000001", // 1, negative + "1.0000000000000000", // 17: trailing zeros are significant here + "10000000000000000", // 17 + "9007199254740992", // 2^53 + "9007199254740993", // 2^53 + 1 + "-65.613616999999977", // canada.json shape + "1.2345678901234567e-250", // 17 with an exponent + "1.234567890123456e-250", // 16 with an exponent + "1e10", "0.0", "-0.0", "0e0", "0.000123" + }; + + for (const auto& n : numbers) + { + CAPTURE(n); + const std::string doc = "[" + n + "]"; + + const json a = json::parse(doc); // contiguous fast path + std::stringstream ss(doc); + const json b = json::parse(ss); // streaming byte path + + CHECK(a[0].type() == b[0].type()); + CHECK(a == b); + + if (a[0].is_number_float()) + { + const double expected = std::strtod(n.c_str(), nullptr); + CHECK(a[0].get() == expected); + CHECK(b[0].get() == expected); + } + } + } + + SECTION("token type classification") + { + CHECK((scan_string("0") == json::lexer::token_type::value_unsigned)); + CHECK((scan_string("-1") == json::lexer::token_type::value_integer)); + CHECK((scan_string("1.5") == json::lexer::token_type::value_float)); + CHECK((scan_string("1e5") == json::lexer::token_type::value_float)); + CHECK((scan_string("18446744073709551615") == json::lexer::token_type::value_unsigned)); + CHECK((scan_string("18446744073709551616") == json::lexer::token_type::value_float)); + CHECK((scan_string("-9223372036854775808") == json::lexer::token_type::value_integer)); + CHECK((scan_string("-9223372036854775809") == json::lexer::token_type::value_float)); + } + + SECTION("malformed numbers are rejected identically") + { + for (const char* bad : + {"-", "1.", "1e", "1e+", "1.2e", "01", "-01", "1..2", "1.2.3" + }) + { + CAPTURE(bad); + // the contiguous fast path must decline and let the byte path report + const std::string doc = std::string("[") + bad + "]"; + CHECK_FALSE(json::accept(doc)); + std::stringstream ss(doc); + CHECK_FALSE(json::accept(ss)); + } + } + +#if !defined(JSON_NOEXCEPTION) + // these sections parse invalid input, which aborts when exceptions are off + SECTION("exhaustive grammar parity with the streaming path") + { + // The JSON number grammar is encoded twice: once as the scan_number() + // state machine and once as the contiguous fast path. Enumerate every + // short string over the number alphabet and require the two encodings to + // agree exactly - on acceptance, on the reported error, and on the parsed + // value - so they cannot drift apart. + const std::string alphabet = "01.eE+-"; + + // full outcome of parsing @a doc, so a mismatch in type, value, or error + // message is caught, not just a mismatch in acceptance + const auto outcome = [](const std::string & doc, bool streaming) -> std::string + { + try + { + if (streaming) + { + std::stringstream ss(doc); + const json j = json::parse(ss); + return std::string(j[0].type_name()) + '|' + j.dump(); + } + const json j = json::parse(doc); + return std::string(j[0].type_name()) + '|' + j.dump(); + } + catch (const json::parse_error& e) + { + return {e.what()}; + } + }; + + std::vector mismatches; + std::vector tokens{""}; + for (std::size_t length = 1; length <= 4; ++length) + { + std::vector next; + next.reserve(tokens.size() * alphabet.size()); + for (const auto& prefix : tokens) + { + for (const char c : alphabet) + { + next.push_back(prefix + c); + } + } + tokens = next; + + for (const auto& token : tokens) + { + const std::string doc = "[" + token + "]"; + if (outcome(doc, false) != outcome(doc, true)) + { + mismatches.push_back(doc); + } + } + } + + // 7 + 49 + 343 + 2401 tokens + CHECK(tokens.size() == 2401); + CAPTURE(mismatches); + CHECK(mismatches.empty()); + } + + SECTION("error positions match the streaming path") + { + // Rejecting identically is not enough: the fast path must also report the + // error at the same position as the byte path. A number directly followed + // by a newline is the interesting case, because the byte path reaches the + // newline (which resets the column) and then ungets it. + // returns the parse_error message, or "" if the document parsed + const auto contiguous_error = [](const std::string & doc) -> std::string + { + try + { + const json j = json::parse(doc); + static_cast(j); + } + catch (const json::parse_error& e) + { + return {e.what()}; + } + return {}; + }; + const auto streaming_error = [](const std::string & doc) -> std::string + { + try + { + std::stringstream ss(doc); + const json j = json::parse(ss); + static_cast(j); + } + catch (const json::parse_error& e) + { + return {e.what()}; + } + return {}; + }; + + for (const char* bad : + {"[01\n]", "[00\n]", "[-01\n]", "{1\n}", "[1\n2]", "[1.2.3\n]", + "[1 \n2]", "[\n1\n2]", "1\n2", "[01\r\n]", "[1e\n]", "[-\n]" + }) + { + CAPTURE(bad); + const std::string doc = bad; + const std::string contiguous_what = contiguous_error(doc); + + CHECK_FALSE(contiguous_what.empty()); + CHECK(contiguous_what == streaming_error(doc)); + } + + // A number terminated by a newline must report the same position as the + // same number terminated by anything else: scan_number() reads the + // terminator and ungets it, so the reported column is the one reached + // after the number's last character - not the 0 that an unget() across + // the newline used to leave behind. + CHECK(contiguous_error("[01\n]") == contiguous_error("[01 ]")); + CHECK(contiguous_error("[01\n]") == + "[json.exception.parse_error.101] parse error at line 1, column 3: " + "syntax error while parsing array - unexpected number literal; expected ']'"); + + // the same for a multi-character token, where the column of the last + // character (the '3' of "-2.5e3") differs from the column it starts at + CHECK(contiguous_error("null -2.5e3\nfalse") == contiguous_error("null -2.5e3 false")); + CHECK(contiguous_error("null -2.5e3\nfalse") == + "[json.exception.parse_error.101] parse error at line 1, column 11: " + "syntax error while parsing value - unexpected number literal; expected end of input"); + } +#endif +} + +TEST_CASE("lexer string fast path") +{ + // Build a byte string from explicit values: a hex escape in a string + // literal swallows every following hex digit, which makes sequences like + // "\xC3\xA9b" mean something other than they look like. + const auto bytes = [](std::initializer_list values) + { + std::string result; + for (const int value : values) + { + result.push_back(static_cast(value)); + } + return result; + }; + +#if !defined(JSON_NOEXCEPTION) + // the full outcome of parsing @a doc: the parsed value, or the exact error + // message, so a mismatch in either is caught. Only usable with exceptions + // on: parsing invalid input aborts when they are off. + const auto outcome = [](const std::string & doc, bool streaming) -> std::string + { + try + { + if (streaming) + { + std::stringstream ss(doc); + const json j = json::parse(ss); + return j.dump(); + } + const json j = json::parse(doc); + return j.dump(); + } + // not just parse_error: if a bulk scanner ever let ill-formed UTF-8 + // through, dump() would throw type_error.316, and that has to surface + // as a reported mismatch rather than as an uncaught exception + catch (const json::exception& e) + { + return {e.what()}; + } + }; +#endif + + // once at the start of the string, once past the first 8-byte SWAR word, so + // the bulk scanner sees each case with and without a run behind it + const std::vector offsets{0, 9}; + +#if !defined(JSON_NOEXCEPTION) + SECTION("exhaustive contiguous vs streaming parity") + { + // ordinary ASCII, both specials, a control byte, characters that make + // the preceding backslash a valid escape, a UTF-8 lead byte of each + // length, a continuation byte, and a byte that is never valid + const std::vector alphabet = + { + "a", "\"", "\\", "n", "u", "0", bytes({0x01}), + bytes({0xC3}), bytes({0xA9}), bytes({0xE4}), bytes({0xF0}), + bytes({0x80}), bytes({0xFF}) + }; + + std::vector mismatches; + std::vector tokens{""}; + for (std::size_t length = 1; length <= 3; ++length) + { + std::vector next; + next.reserve(tokens.size() * alphabet.size()); + for (const auto& prefix : tokens) + { + for (const auto& symbol : alphabet) + { + next.push_back(prefix + symbol); + } + } + tokens = next; + + for (const auto& token : tokens) + { + for (const std::size_t offset : offsets) + { + const std::string doc = "[\"" + std::string(offset, 'a') + token + "\"]"; + if (outcome(doc, false) != outcome(doc, true)) + { + mismatches.push_back(doc); + } + } + } + } + + // 13 + 169 + 2197 tokens, each at two offsets + CHECK(tokens.size() == 2197); + CAPTURE(mismatches); + CHECK(mismatches.empty()); + } + + SECTION("special bytes at every offset of the SWAR stride") + { + // The bulk scanner consumes 8 bytes at a time and then a tail; place + // every kind of byte that ends a run at each offset across two words, + // so multibyte sequences also straddle the word boundary. + const std::vector specials = + { + "\"", "\\", bytes({0x01}), bytes({0x1F}), bytes({0x7F}), + bytes({0xC3, 0xA9}), bytes({0xE4, 0xB8, 0xAD}), bytes({0xF0, 0x9F, 0x98, 0x80}), + bytes({0xFF}), bytes({0xC3}), bytes({0xE4, 0xB8}) + }; + + std::vector mismatches; + for (std::size_t offset = 0; offset <= 17; ++offset) + { + for (const auto& special : specials) + { + const std::string doc = "[\"" + std::string(offset, 'a') + special + "\"]"; + if (outcome(doc, false) != outcome(doc, true)) + { + mismatches.push_back(doc); + } + } + } + CAPTURE(mismatches); + CHECK(mismatches.empty()); + } +#endif + + // json::accept() never throws, so the ranges stay covered without exceptions + SECTION("UTF-8 ranges are accepted and rejected as documented") + { + // The bulk validator must accept exactly what the byte-at-a-time + // scanner accepts, so pin the boundaries of every range it recognizes. + // aggregate, only ever brace-initialized below; default member + // initializers would stop it being an aggregate in C++11 + struct utf8_case // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init) + { + std::string sequence; + bool valid; + const char* description; + }; + const std::vector cases = + { + {bytes({0xC2, 0x80}), true, "U+0080, shortest two-byte"}, + {bytes({0xDF, 0xBF}), true, "U+07FF, longest two-byte"}, + {bytes({0xC1, 0xBF}), false, "overlong two-byte"}, + {bytes({0xC2, 0x7F}), false, "two-byte with bad continuation"}, + {bytes({0xE0, 0xA0, 0x80}), true, "U+0800, shortest three-byte"}, + {bytes({0xE0, 0x9F, 0xBF}), false, "overlong three-byte"}, + {bytes({0xED, 0x9F, 0xBF}), true, "U+D7FF, just below the surrogates"}, + {bytes({0xED, 0xA0, 0x80}), false, "surrogate U+D800"}, + {bytes({0xED, 0xBF, 0xBF}), false, "surrogate U+DFFF"}, + {bytes({0xEE, 0x80, 0x80}), true, "U+E000, just above the surrogates"}, + {bytes({0xEF, 0xBF, 0xBF}), true, "U+FFFF"}, + {bytes({0xF0, 0x90, 0x80, 0x80}), true, "U+10000, shortest four-byte"}, + {bytes({0xF0, 0x8F, 0xBF, 0xBF}), false, "overlong four-byte"}, + {bytes({0xF4, 0x8F, 0xBF, 0xBF}), true, "U+10FFFF, highest code point"}, + {bytes({0xF4, 0x90, 0x80, 0x80}), false, "above U+10FFFF"}, + {bytes({0xF5, 0x80, 0x80, 0x80}), false, "lead byte out of range"}, + {bytes({0x80}), false, "bare continuation byte"}, + {bytes({0xFF}), false, "byte that never appears in UTF-8"}, + {bytes({0xC3}), false, "truncated two-byte"}, + {bytes({0xE4, 0xB8}), false, "truncated three-byte"}, + {bytes({0xF0, 0x9F, 0x98}), false, "truncated four-byte"} + }; + + for (const auto& test_case : cases) + { + CAPTURE(test_case.description); + for (const std::size_t offset : offsets) + { + CAPTURE(offset); + const std::string doc = "[\"" + std::string(offset, 'a') + test_case.sequence + "\"]"; + CHECK(json::accept(doc) == test_case.valid); +#if !defined(JSON_NOEXCEPTION) + CHECK(outcome(doc, false) == outcome(doc, true)); +#endif + } + } + } +} diff --git a/tests/src/unit-user_defined_input.cpp b/tests/src/unit-user_defined_input.cpp index 823e82862..f07a8a608 100644 --- a/tests/src/unit-user_defined_input.cpp +++ b/tests/src/unit-user_defined_input.cpp @@ -18,7 +18,12 @@ #include using nlohmann::json; +#include // array +#include // size_t +#include // uint8_t #include +#include // string +#include // vector #if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) #include @@ -212,6 +217,66 @@ TEST_CASE("Parse with heterogeneous iterator and sentinel types") CHECK(j2.at(0) == 1); } +// A type whose data() hands out raw bytes but whose size() counts something +// else - here fixed-size records. Reading [data(), data() + size()) as bytes +// would silently truncate the input, so data() and size() alone must not be +// taken as evidence of contiguous byte storage. +struct record_buffer +{ + using value_type = std::array; + + std::string bytes; + + const char* data() const noexcept + { + return bytes.data(); + } + std::size_t size() const noexcept + { + return bytes.size() / sizeof(value_type); + } + const char* begin() const noexcept + { + return bytes.data(); + } + const char* end() const noexcept + { + return bytes.data() + bytes.size(); + } +}; + +TEST_CASE("Contiguous byte containers take the pointer adapter") +{ + // Containers with contiguous single-byte storage are routed through the + // pointer-based adapter so the bulk fast paths apply in every standard, not + // only in C++20 where the library iterators model std::contiguous_iterator. + CHECK(nlohmann::detail::is_contiguous_byte_container::value); + CHECK(nlohmann::detail::is_contiguous_byte_container>::value); + CHECK(nlohmann::detail::is_contiguous_byte_container>::value); + CHECK(nlohmann::detail::is_contiguous_byte_container>::value); + + // input_adapter() takes its container by forwarding reference, so the trait + // is also asked about reference types + CHECK(nlohmann::detail::is_contiguous_byte_container::value); + CHECK(nlohmann::detail::is_contiguous_byte_container::value); + + // everything else keeps the iterator-based adapter + CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container>::value); + CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container>::value); + CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container::value); + + // including a type that has data() and size() but whose size() does not + // count the units data() points at: its value_type says so + CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container::value); + + // and such a container still parses through its iterators, in full - taking + // it for a byte container would stop after data() + size() bytes + const record_buffer buffer{"[1,2,3,4,5]"}; + CHECK(buffer.data() == buffer.bytes.data()); + CHECK(buffer.size() * sizeof(record_buffer::value_type) < buffer.bytes.size()); + CHECK(json::parse(buffer) == json({1, 2, 3, 4, 5})); +} + #if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20) // JSON_HAS_CPP_20 (do not remove; see note at top of file) TEST_CASE("Parse with std::counted_iterator and std::default_sentinel_t") @@ -228,6 +293,180 @@ TEST_CASE("Parse with std::counted_iterator and std::default_sentinel_t") const std::counted_iterator first2(json_str.begin(), len); CHECK(json::accept(first2, std::default_sentinel)); } + +TEST_CASE("std::counted_iterator reaches the contiguous fast paths") +{ + // A sized sentinel makes the remaining element count computable in O(1), so + // std::counted_iterator over a contiguous iterator must reach the same bulk + // string/number scanners as a plain pointer - not just the byte-at-a-time + // fallback (see #5268 for the equivalent memcpy fast path). +#if JSON_HAS_RANGES + // JSON_HAS_RANGES is 0 on standard libraries with an incomplete + // (libstdc++ < 11, libc++ < 16), where the adapter deliberately falls back + // to the byte-at-a-time scanner; everything below still has to work there. + using adapter_type = nlohmann::detail::iterator_input_adapter, std::default_sentinel_t>; + CHECK(adapter_type::supports_bulk_scan); + CHECK(adapter_type::supports_seek); +#endif + + // exercise every fast path: long ASCII run, multibyte UTF-8, escapes, and + // integer/floating-point numbers + const std::string json_str = + R"({"ascii":"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",)" + "\"utf8\":\"\xe4\xb8\xad\xe6\x96\x87\xf0\x9f\x98\x80\xc3\xa9\"," + R"("escaped":"aéb\n\\","ints":[0,-1,18446744073709551615,-9223372036854775808],)" + R"("floats":[1.5,-2.25e3,0.30000000000000004]})"; + const auto len = static_cast>(json_str.size()); + + const std::counted_iterator first(json_str.data(), len); + const json j = json::parse(first, std::default_sentinel); + + // parsing through the pointer adapter must give exactly the same result + CHECK(j == json::parse(json_str)); + +#if !defined(JSON_NOEXCEPTION) + // Diagnostics that quote the offending token are reconstructed from the + // already-consumed input (supports_seek), a path a sized sentinel only + // reaches now; check a few that include the "last read" text. Parsing + // invalid input aborts when exceptions are off, hence the guard. + // Raw strings and explicit bytes: an escaped literal and two literals + // written next to each other both read as mistakes to static analysis. + const auto byte = [](int value) + { + return std::string(1, static_cast(value)); + }; + const std::vector diagnostic_docs = + { + "1\nx", + "truX", + "[tru]", + R"("abc)", + R"(["\ud834"])", + R"(["a)" + byte(0x01) + R"(b"])", + R"([")" + byte(0xC3) + byte(0x28) + R"("])", + "[1e]", + R"(["aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaX)" + }; + + for (const auto& text : diagnostic_docs) + { + CAPTURE(text); + const std::counted_iterator it(text.data(), static_cast>(text.size())); + std::string counted_message; + std::string string_message; + try + { + const json counted_result = json::parse(it, std::default_sentinel); + static_cast(counted_result); + } + catch (const json::parse_error& e) + { + counted_message = e.what(); + } + try + { + const json string_result = json::parse(text); + static_cast(string_result); + } + catch (const json::parse_error& e) + { + string_message = e.what(); + } + CHECK_FALSE(counted_message.empty()); + CHECK(counted_message == string_message); + } + + // and errors must still be reported identically + const std::string bad = "[01\n]"; + const std::counted_iterator bad_first(bad.data(), static_cast>(bad.size())); + std::string counted_what; + std::string string_what; + try + { + const json counted_result = json::parse(bad_first, std::default_sentinel); + static_cast(counted_result); + } + catch (const json::parse_error& e) + { + counted_what = e.what(); + } + try + { + const json string_result = json::parse(bad); + static_cast(string_result); + } + catch (const json::parse_error& e) + { + string_what = e.what(); + } + CHECK_FALSE(counted_what.empty()); + CHECK(counted_what == string_what); +#endif +} + +#if !defined(JSON_NOEXCEPTION) +// several cases below are truncated on purpose, and parsing invalid input +// aborts when exceptions are off +TEST_CASE("std::counted_iterator bulk scanning stops at the counted end") +{ + // The count, not the size of the underlying buffer, is the end of the + // input: the bulk scanners must never look at the bytes behind it, even + // though they are readable. Each case is compared against parsing the + // equivalent prefix as a std::string. + const auto via_counted = [](const std::string & buf, std::size_t n) -> std::string + { + const std::counted_iterator first(buf.data(), static_cast>(n)); + try + { + const json j = json::parse(first, std::default_sentinel); + return "OK|" + j.dump(); + } + catch (const json::parse_error& e) + { + return {e.what()}; + } + }; + const auto via_prefix = [](const std::string & buf, std::size_t n) -> std::string + { + try + { + const json j = json::parse(buf.substr(0, n)); + return "OK|" + j.dump(); + } + catch (const json::parse_error& e) + { + return {e.what()}; + } + }; + + struct testcase // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init) + { + const char* buffer; + std::size_t count; + }; + const std::vector cases = + { + {"[\"abc\"]____TRAILING____", 7}, // exact fit, tail hidden + {"[\"abcdefghijklmnop\"]____", 8}, // cut inside a string + {"[\"abc\"]____", 6}, // cut just before the closing quote + {"[12345]xxxxx", 4}, // cut inside a number + {"[123]999999", 5}, // number ends exactly at the count + {"[\"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\"]", 12}, // closing quote only behind the count + {"[\"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\"]", 19}, // cut inside an 8-byte SWAR stride + {"[\"\xe4\xb8\xad\xe6\x96\x87\"]", 5}, // cut inside a UTF-8 sequence + {"[\"\xe4\xb8\xad\xe6\x96\x87\"]____", 10}, // complete UTF-8, tail hidden + {"[1.25e3]TRAILINGDIGITS999", 7}, // number token reaches the count + }; + + for (const auto& tc : cases) + { + CAPTURE(tc.buffer); + CAPTURE(tc.count); + const std::string buffer = tc.buffer; + CHECK(via_counted(buffer, tc.count) == via_prefix(buffer, tc.count)); + } +} +#endif #endif } // namespace