From 0e729be261c106a137e1a2c6e9b88938e3809cd1 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:01:48 +0200 Subject: [PATCH 01/17] Load string scan words with memcpy and use MSVC bit-scan intrinsics Signed-off-by: Niels Lohmann --- include/nlohmann/detail/bit_ops.hpp | 23 +++++++++++++++++++---- tests/src/unit-class_lexer.cpp | 7 +++++++ 2 files changed, 26 insertions(+), 4 deletions(-) diff --git a/include/nlohmann/detail/bit_ops.hpp b/include/nlohmann/detail/bit_ops.hpp index c570814d9..b3bf08557 100644 --- a/include/nlohmann/detail/bit_ops.hpp +++ b/include/nlohmann/detail/bit_ops.hpp @@ -9,8 +9,9 @@ #pragma once #include // uint64_t -#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) - #include // __umulh, _umul128 +#include // memcpy +#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) && (!defined(__SIZEOF_INT128__) || (!defined(__GNUC__) && !defined(__clang__))) + #include // __umulh, _umul128, _BitScanForward64, _BitScanReverse64 #endif #include @@ -28,6 +29,10 @@ inline int count_leading_zeros(std::uint64_t x) noexcept { #if defined(__GNUC__) || defined(__clang__) return __builtin_clzll(x); +#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + unsigned long index = 0; + _BitScanReverse64(&index, x); + return 63 - static_cast(index); #else int n = 0; for (int shift = 32; shift != 0; shift >>= 1) @@ -47,6 +52,10 @@ inline int count_trailing_zeros(std::uint64_t x) noexcept { #if defined(__GNUC__) || defined(__clang__) return __builtin_ctzll(x); +#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + unsigned long index = 0; + _BitScanForward64(&index, x); + return static_cast(index); #else int n = 0; for (int shift = 32; shift != 0; shift >>= 1) @@ -94,14 +103,20 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc #endif } -/// eight bytes as a little-endian word (compilers fold this into one load on -/// little-endian targets) +/// eight bytes as a little-endian word (a single load on little-endian targets) inline std::uint64_t read_eight_bytes(const unsigned char* b) noexcept { +#if defined(_MSC_VER) || defined(__x86_64__) || defined(__i386__) || (defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__) + // the byte order already matches (all MSVC targets are little-endian) + std::uint64_t result = 0; + std::memcpy(&result, b, sizeof(result)); + return result; +#else return static_cast(b[0]) | (static_cast(b[1]) << 8u) | (static_cast(b[2]) << 16u) | (static_cast(b[3]) << 24u) | (static_cast(b[4]) << 32u) | (static_cast(b[5]) << 40u) | (static_cast(b[6]) << 48u) | (static_cast(b[7]) << 56u); +#endif } /// eight bytes as a little-endian word diff --git a/tests/src/unit-class_lexer.cpp b/tests/src/unit-class_lexer.cpp index 82690a669..e5f82b53c 100644 --- a/tests/src/unit-class_lexer.cpp +++ b/tests/src/unit-class_lexer.cpp @@ -1792,4 +1792,11 @@ TEST_CASE("string scanning kernels") CHECK(nlohmann::detail::count_trailing_zeros(bit) == k); CHECK(nlohmann::detail::count_trailing_zeros(bit | (bit << 1u) | 0x8000000000000000u) == k); } + + // eight bytes as a little-endian word, at any alignment + const unsigned char bytes[16] = {0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF}; + CHECK(nlohmann::detail::read_eight_bytes(bytes) == 0x0807060504030201u); + CHECK(nlohmann::detail::read_eight_bytes(bytes + 1) == 0x0908070605040302u); + CHECK(nlohmann::detail::read_eight_bytes(bytes + 8) == 0xFF0F0E0D0C0B0A09u); + CHECK(nlohmann::detail::read_eight_bytes(reinterpret_cast(bytes) + 3) == 0x0B0A090807060504u); } From 653edd738a976f435c5ad72e35320f2eeb73e1a6 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:01:49 +0200 Subject: [PATCH 02/17] Reword float conversion notes for the string type and number_float_t docs Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/number_float_t.md | 8 ++++---- .../docs/features/types/template_parameters.md | 13 +++++++++---- 2 files changed, 13 insertions(+), 8 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/number_float_t.md b/docs/mkdocs/docs/api/basic_json/number_float_t.md index aa41f33f0..1cde41960 100644 --- a/docs/mkdocs/docs/api/basic_json/number_float_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_float_t.md @@ -23,10 +23,10 @@ type to use. ## Template parameters `NumberFloatType` -: the type to store floating-point numbers. The parser converts `#!cpp float`, `#!cpp double`, and a - `#!cpp long double` that is IEEE 754 binary64 itself and other `#!cpp long double` formats with - `#!cpp std::from_chars` or `#!cpp std::strtold`, and serialization falls back to `#!cpp std::snprintf`, so the - type must be `#!cpp float`, `#!cpp double`, or `#!cpp long double`. The +: the type to store floating-point numbers. The type must be `#!cpp float`, `#!cpp double`, or + `#!cpp long double`. The parser converts `#!cpp float`, `#!cpp double`, and a `#!cpp long double` that is IEEE 754 + binary64 itself. It converts other `#!cpp long double` formats with `#!cpp std::from_chars` where available, or + with `#!cpp std::strtold` otherwise. Serialization falls back to `#!cpp std::snprintf`. The [binary formats](../../features/binary_formats/index.md) additionally require `#!cpp float` or `#!cpp double`, because they have no encoding for `#!cpp long double`. See [Template Parameter Requirements](../../features/types/template_parameters.md#numberfloattype). diff --git a/docs/mkdocs/docs/features/types/template_parameters.md b/docs/mkdocs/docs/features/types/template_parameters.md index 3a24d74d8..b68fed8ba 100644 --- a/docs/mkdocs/docs/features/types/template_parameters.md +++ b/docs/mkdocs/docs/features/types/template_parameters.md @@ -353,16 +353,21 @@ using array_t = ArrayType>; ### Always required - A member type `value_type` that is one byte wide and `char`-compatible. The library stores and processes UTF-8 - encoded `char` data and passes `data()` to functions that take a `#!cpp const char*`, such as `#!cpp std::strtod`. + encoded `char` data and passes `data()` to functions that take a `#!cpp const char*`, such as `#!cpp std::strtold` + (only used to parse a `#!cpp long double` that is not IEEE 754 binary64, see + [`NumberFloatType`](#numberfloattype)). `#!cpp std::wstring`, `#!cpp std::u16string`, and `#!cpp std::u32string` are **not** valid choices; see the FAQ on [wide string handling](../../home/faq.md#wide-string-handling). - Constructors: default, copy, move, from `#!cpp const char*` (which must not be `#!cpp explicit`), from `#!cpp (const char*, size_type)`, and from `#!cpp (size_type, char)`; and copy or move assignment. - Member functions `size()`, `clear()`, `resize(n, c)`, `data()`, `push_back(char)`, and `operator[]` (const and non-const, returning references). `c_str()` and `back()` are **not** required. -- `data()` must return a pointer to a contiguous, **null-terminated** buffer -- the parser may hand it to - `#!cpp std::strtod`, which reads up to the null character. A type whose `data()` is not null-terminated does not - fail to compile; it can silently misparse floating-point numbers. +- `data()` must return a pointer to a contiguous, **null-terminated** buffer. `#!cpp float`, `#!cpp double`, and a + `#!cpp long double` that is IEEE 754 binary64 are converted by the library itself and do not depend on this. For any + other `NumberFloatType` (a `#!cpp long double` of another format), the parser falls back to `#!cpp std::strtold` when + `#!cpp std::from_chars` is not available or declines the token, and `std::strtold` reads up to the null character. A type whose `data()` + is not null-terminated does not fail to compile; with such a `NumberFloatType` it can silently misparse + floating-point numbers. - `append(const char*, size_type)`, used by [`dump`](../../api/basic_json/dump.md), and `append(const StringType&)`, used by the CBOR reader for indefinite-length strings. The library's internal string concatenation additionally has to append a `#!cpp char` and a `#!cpp const char*`; for each it selects between `append(arg)`, `#!cpp operator+=`, From e1a85099ef1bb34cfe85be5a2bd0362516e115d4 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:01:50 +0200 Subject: [PATCH 03/17] Regenerate the amalgamated headers Signed-off-by: Niels Lohmann --- single_include/nlohmann/json.hpp | 23 +++++++++++++++++++---- 1 file changed, 19 insertions(+), 4 deletions(-) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 93ffdd25b..7ba4bc8b8 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -8793,8 +8793,9 @@ NLOHMANN_JSON_NAMESPACE_END #include // uint64_t -#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) - #include // __umulh, _umul128 +#include // memcpy +#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) && (!defined(__SIZEOF_INT128__) || (!defined(__GNUC__) && !defined(__clang__))) + #include // __umulh, _umul128, _BitScanForward64, _BitScanReverse64 #endif // #include @@ -8813,6 +8814,10 @@ inline int count_leading_zeros(std::uint64_t x) noexcept { #if defined(__GNUC__) || defined(__clang__) return __builtin_clzll(x); +#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + unsigned long index = 0; + _BitScanReverse64(&index, x); + return 63 - static_cast(index); #else int n = 0; for (int shift = 32; shift != 0; shift >>= 1) @@ -8832,6 +8837,10 @@ inline int count_trailing_zeros(std::uint64_t x) noexcept { #if defined(__GNUC__) || defined(__clang__) return __builtin_ctzll(x); +#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + unsigned long index = 0; + _BitScanForward64(&index, x); + return static_cast(index); #else int n = 0; for (int shift = 32; shift != 0; shift >>= 1) @@ -8879,14 +8888,20 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc #endif } -/// eight bytes as a little-endian word (compilers fold this into one load on -/// little-endian targets) +/// eight bytes as a little-endian word (a single load on little-endian targets) inline std::uint64_t read_eight_bytes(const unsigned char* b) noexcept { +#if defined(_MSC_VER) || defined(__x86_64__) || defined(__i386__) || (defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__) + // the byte order already matches (all MSVC targets are little-endian) + std::uint64_t result = 0; + std::memcpy(&result, b, sizeof(result)); + return result; +#else return static_cast(b[0]) | (static_cast(b[1]) << 8u) | (static_cast(b[2]) << 16u) | (static_cast(b[3]) << 24u) | (static_cast(b[4]) << 32u) | (static_cast(b[5]) << 40u) | (static_cast(b[6]) << 48u) | (static_cast(b[7]) << 56u); +#endif } /// eight bytes as a little-endian word From 86feea50abbc4d8f61339348f13a3c6129256485 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:04:13 +0200 Subject: [PATCH 04/17] Select the Zmij conversion by a trait, not by the double overload The non-template overloads for double asserted binary64 doubles wherever json.hpp was included, so the library no longer compiled where double is not IEEE 754 binary64 (AVR, -fshort-double). A trait now picks Zmij for any binary64 type, including a long double of that format (MSVC, Apple Arm), and Grisu2 for the others. Remove the unused write_short_decimal, powers_of_ten_16, zmij::decimal, zmij::to_decimal, and shortest_digits(double). Signed-off-by: Niels Lohmann --- .../nlohmann/detail/conversions/to_chars.hpp | 172 +++++------------- include/nlohmann/detail/conversions/zmij.hpp | 19 -- tests/src/unit-to_chars.cpp | 77 +++++++- 3 files changed, 120 insertions(+), 148 deletions(-) diff --git a/include/nlohmann/detail/conversions/to_chars.hpp b/include/nlohmann/detail/conversions/to_chars.hpp index 0e16152c3..4324e40a5 100644 --- a/include/nlohmann/detail/conversions/to_chars.hpp +++ b/include/nlohmann/detail/conversions/to_chars.hpp @@ -939,88 +939,6 @@ void grisu2(char* buf, int& len, int& decimal_exponent, FloatType value) grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus); } -/*! -@brief the shortest digits of a positive finite float (other than double): Grisu2 -*/ -template -JSON_HEDLEY_NON_NULL(1) -void shortest_digits(char* buf, int& len, int& decimal_exponent, FloatType value) -{ - grisu2(buf, len, decimal_exponent, value); -} - -/*! -@brief the shortest digits of a positive finite double: the conversion of -Zmij (see zmij.hpp), which always finds the shortest digits that read back as -the same value (Grisu2 does not for about one double in a thousand), and the -closest of them if there are several - -v = buf * 10^decimal_exponent, as for grisu2() -*/ -JSON_HEDLEY_NON_NULL(1) -inline void shortest_digits(char* buf, int& len, int& decimal_exponent, double value) -{ - static_assert(std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53, - "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); - JSON_ASSERT(std::isfinite(value)); - JSON_ASSERT(value > 0); - - std::uint64_t bits = 0; - std::memcpy(&bits, &value, sizeof(bits)); - zmij::decimal d = zmij::to_decimal(bits); - // without trailing zeros (up to 16): 8, 4, 2, 1 at a time - while (d.significand % 100000000 == 0) - { - d.significand /= 100000000; - d.exponent += 8; - } - if (d.significand % 10000 == 0) - { - d.significand /= 10000; - d.exponent += 4; - } - if (d.significand % 100 == 0) - { - d.significand /= 100; - d.exponent += 2; - } - if (d.significand % 10 == 0) - { - d.significand /= 10; - d.exponent += 1; - } - // at most 17 digits, written from the back two at a time - static constexpr const char* pairs = - "00010203040506070809101112131415161718192021222324252627282930313233343536373839" - "40414243444546474849505152535455565758596061626364656667686970717273747576777879" - "8081828384858687888990919293949596979899"; - std::array digits{}; - std::size_t n = digits.size(); - while (d.significand >= 100) - { - const std::uint64_t two_digits = d.significand % 100; // a variable: GCC calls a cast of the remainder useless where std::uint64_t is std::size_t - const auto i = static_cast(two_digits) * 2; - d.significand /= 100; - n -= 2; - digits[n] = pairs[i]; - digits[n + 1] = pairs[i + 1]; - } - if (d.significand >= 10) - { - const auto i = static_cast(d.significand) * 2; - n -= 2; - digits[n] = pairs[i]; - digits[n + 1] = pairs[i + 1]; - } - else - { - digits[--n] = static_cast('0' + d.significand); - } - len = static_cast(digits.size() - n); - std::memcpy(buf, digits.data() + n, static_cast(len)); - decimal_exponent = d.exponent; -} - /*! @brief appends a decimal representation of e to buf @return a pointer to the element following the exponent. @@ -1423,53 +1341,30 @@ inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcep return end + (three ? 5 : 4); } -/// the powers of ten up to 10^16 -inline const std::array& powers_of_ten_16() noexcept -{ - static const std::array powers = - { - { - 1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, 10000000000u, - 100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, 10000000000000000u - } - }; - return powers; -} - /*! -@brief digits * 10^exp, as write_decimal() writes it, for the digits of a -double that need no conversion (count digits, at most 15, the first not 0; -trailing zeros allowed): extended to 16 digits and written by write_shortest() +@brief whether FloatType is an IEEE 754 binary64 type (a double, or a long double +that has the same format, as with MSVC and on Apple's Arm CPUs) -@return a pointer past the text; up to 41 bytes at @a first are written - (some beyond the returned end) +These are the types the conversion of Zmij (see zmij.hpp) is used for; all +others (binary32, or a format the library does not know) use Grisu2. */ -JSON_HEDLEY_NON_NULL(1) -JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_short_decimal(char* first, std::uint64_t digits, int count, int exp) noexcept +template +constexpr bool has_binary64_format() noexcept { - JSON_ASSERT(digits >= powers_of_ten_16()[static_cast(count - 1)] && count <= 15); - const int scale = 16 - count; - return write_shortest(first, zmij::shortest_decimal{digits * powers_of_ten_16()[static_cast(scale)], exp - scale - 1, 0, false}); + return std::numeric_limits::is_iec559 + && std::numeric_limits::digits == 53 + && std::numeric_limits::max_exponent == 1024 + && sizeof(FloatType) == sizeof(std::uint64_t); } -/// as write_short_decimal(), counting the digits (not 0, less than 10^15) -JSON_HEDLEY_NON_NULL(1) -JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noexcept -{ - JSON_ASSERT(digits != 0 && digits < 1000000000000000u); - // floor(log10(2^bits)) + 1 digits, or one less - const int log2_bound = ((64 - count_leading_zeros(digits)) * 1233) >> 12; - const int count = log2_bound + (digits >= powers_of_ten_16()[static_cast(log2_bound)] ? 1 : 0); - return write_short_decimal(first, digits, count, exp); -} +template +struct is_binary64 : std::integral_constant()> {}; -/// a positive finite float (other than double): Grisu2 and format_buffer() +/// a positive finite float (other than binary64): Grisu2 and format_buffer() template JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL -char* write_positive(char* first, const char* last, FloatType value) +char* write_positive_grisu2(char* first, const char* last, FloatType value) { JSON_ASSERT(last - first >= std::numeric_limits::max_digits10); static_cast(last); // (only used in the assertion) @@ -1480,7 +1375,7 @@ char* write_positive(char* first, const char* last, FloatType value) // len is the length of the buffer, i.e., the number of decimal digits. int len = 0; int decimal_exponent = 0; - shortest_digits(first, len, decimal_exponent, value); + grisu2(first, len, decimal_exponent, value); JSON_ASSERT(len <= std::numeric_limits::max_digits10); @@ -1496,15 +1391,16 @@ char* write_positive(char* first, const char* last, FloatType value) return format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp); } -/// a positive finite double: the shortest digits (Zmij), laid out by +/// a positive finite binary64 number: the shortest digits (Zmij), laid out by /// write_shortest() (through a local buffer if [first, last) is shorter than /// the 41 bytes it may write) +template JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_positive(char* first, const char* last, double value) +char* write_positive_zmij(char* first, const char* last, FloatType value) { - static_assert(std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53, - "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); + static_assert(is_binary64::value, + "internal error: the conversion of Zmij needs IEEE 754 binary64 numbers"); std::uint64_t bits = 0; std::memcpy(&bits, &value, sizeof(bits)); const zmij::shortest_decimal d = zmij::to_shortest(bits); @@ -1519,6 +1415,34 @@ inline char* write_positive(char* first, const char* last, double value) return first + len; } +/// a positive finite binary64 number: Zmij (as a long double has the format of +/// a double here, its bits are those of the double of the same value) +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value, std::true_type /*is_binary64*/) +{ + return write_positive_zmij(first, last, value); +} + +/// a positive finite float of any other format: Grisu2 +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value, std::false_type /*is_binary64*/) +{ + return write_positive_grisu2(first, last, value); +} + +/// a positive finite float: Zmij for binary64 numbers, Grisu2 otherwise +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value) +{ + return write_positive(first, last, value, is_binary64 {}); +} + } // namespace dtoa_impl /*! diff --git a/include/nlohmann/detail/conversions/zmij.hpp b/include/nlohmann/detail/conversions/zmij.hpp index ffff8380e..40e477f7b 100644 --- a/include/nlohmann/detail/conversions/zmij.hpp +++ b/include/nlohmann/detail/conversions/zmij.hpp @@ -36,13 +36,6 @@ computed from the compressed tables of Zmij beyond it. namespace zmij { -/// significand * 10^exponent -struct decimal -{ - std::uint64_t significand; - int exponent; -}; - /// the compressed powers of ten of Zmij inline const std::array& pow10_minor() noexcept { @@ -221,18 +214,6 @@ JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexc return shortest_decimal{integral, dec_exp, static_cast(digit), !round_up && !round_down}; } -/// The shortest decimal in the rounding interval of a positive finite double -/// given by its bits, as one number. The significand can end in zeros. -inline decimal to_decimal(std::uint64_t bits) noexcept -{ - const shortest_decimal d = to_shortest(bits); - if (d.has_digit) - { - return decimal{(d.integral * 10) + d.digit, d.exponent}; - } - return decimal{d.integral, d.exponent + 1}; -} - } // namespace zmij } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/tests/src/unit-to_chars.cpp b/tests/src/unit-to_chars.cpp index 8ced43ad0..d8eb44434 100644 --- a/tests/src/unit-to_chars.cpp +++ b/tests/src/unit-to_chars.cpp @@ -15,6 +15,7 @@ #include using nlohmann::detail::dtoa_impl::reinterpret_bits; +#include #include #include #include @@ -666,13 +667,24 @@ void check_shortest(double v) const std::string text(buf.data(), end); CAPTURE(text) CHECK(parse_double(text) == v); - // the layout is that of format_buffer() for the same digits + // the layout is that of format_buffer() for the digits of Zmij + const auto sd = nlohmann::detail::zmij::to_shortest(reinterpret_bits(v)); + const std::uint64_t significand = sd.has_digit ? (sd.integral * 10) + sd.digit : sd.integral; + int exponent = sd.has_digit ? sd.exponent : sd.exponent + 1; + std::string significand_digits = std::to_string(significand); + while (significand_digits.size() > 1 && significand_digits.back() == '0') + { + significand_digits.pop_back(); + ++exponent; + } std::array reference{}; - int len = 0; - int exponent = 0; - nlohmann::detail::dtoa_impl::shortest_digits(reference.data(), len, exponent, v); - const char* const reference_end = nlohmann::detail::dtoa_impl::format_buffer(reference.data(), len, exponent, -4, 15); + std::copy(significand_digits.begin(), significand_digits.end(), reference.begin()); + const char* const reference_end = nlohmann::detail::dtoa_impl::format_buffer(reference.data(), static_cast(significand_digits.size()), exponent, -4, 15); CHECK(text == std::string(reference.data(), static_cast(reference_end - reference.data()))); + // and write_positive() is what to_chars() calls + std::array positive{}; + const char* const positive_end = nlohmann::detail::dtoa_impl::write_positive(positive.data(), positive.data() + positive.size(), v); + CHECK(text == std::string(positive.data(), static_cast(positive_end - positive.data()))); const auto de = digits_and_exponent(text); const std::string& digits = de.first; if (digits.size() > 1) @@ -785,3 +797,58 @@ TEST_CASE("shortest digits of doubles") } } } + +TEST_CASE("choice of the conversion") +{ + using nlohmann::detail::dtoa_impl::is_binary64; + + SECTION("by the format of the type") + { + // Zmij needs binary64 numbers; everything else uses Grisu2 + static_assert(!is_binary64::value, "float is not binary64"); + static_assert(is_binary64::value == (std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53), + "double is binary64 where it is IEEE 754 with 53 digits"); + static_assert(!is_binary64::value, "integers are not binary64"); + static_assert(is_binary64::value == (std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53 && sizeof(long double) == 8), + "long double is binary64 where it has the format of a double"); + CHECK(!is_binary64::value); + CHECK(is_binary64::value); + } + + SECTION("float: Grisu2, double: Zmij") + { + // 5.3165205877497296e+16 is one of the doubles for which Grisu2 does not find the shortest digits + constexpr double value = 5.3165205877497296e+16; + std::array buf{}; + const char* const last = buf.data() + buf.size(); + + char* end = nlohmann::detail::dtoa_impl::write_positive(buf.data(), last, value); + CHECK(std::string(buf.data(), end) == "5.31652058774973e+16"); + end = nlohmann::detail::dtoa_impl::write_positive_grisu2(buf.data(), last, value); + CHECK(std::string(buf.data(), end) == "5.3165205877497296e+16"); + + constexpr float f = 1.1754944e-38f; + end = nlohmann::detail::dtoa_impl::write_positive(buf.data(), last, f); + const std::string dispatched(buf.data(), end); + end = nlohmann::detail::dtoa_impl::write_positive_grisu2(buf.data(), last, f); + CHECK(dispatched == std::string(buf.data(), end)); + } + + SECTION("long double with the format of a double: Zmij") + { + // (on platforms where long double is wider, Grisu2 does not apply either: the snprintf fallback does) + if (std::numeric_limits::digits == 53 && std::numeric_limits::is_iec559) + { + using long_double_json = nlohmann::json::with_float_t; + for (const double d : + { + 5.3165205877497296e+16, 1.0, 0.1, 123456.789, 2.2250738585072014e-308, 1.7976931348623157e+308, -5.3165205877497296e+16 + }) + { + CAPTURE(d) + CHECK(long_double_json(static_cast(d)).dump() == nlohmann::json(d).dump()); + } + CHECK(long_double_json(5.3165205877497296e+16L).dump() == "5.31652058774973e+16"); + } + } +} From 35b28d4918dae617690a0b95197f2efabb1a1d64 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:04:14 +0200 Subject: [PATCH 05/17] Credit Zmij's authors in to_chars.hpp, the README, and the license page Signed-off-by: Niels Lohmann --- README.md | 2 +- docs/mkdocs/docs/home/license.md | 2 +- include/nlohmann/detail/conversions/to_chars.hpp | 1 + 3 files changed, 3 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 9c641284d..f6392d057 100644 --- a/README.md +++ b/README.md @@ -1394,7 +1394,7 @@ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR I - The class contains the UTF-8 Decoder from Bjoern Hoehrmann which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2008-2009 [Björn Hoehrmann](https://bjoern.hoehrmann.de/) - The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2009 [Florian Loitsch](https://florian.loitsch.com/) -- The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut) +- The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, including the conversion of the digits to text by Xiang JunBo and the SIMD instruction sequence of Dougall Johnson, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut) - The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/). - The class contains parts of [Google Abseil](https://github.com/abseil/abseil-cpp) which is licensed under the [Apache 2.0 License](https://opensource.org/licenses/Apache-2.0). - The class contains an adapted version of the Eisel-Lemire algorithm, its table of powers of five, and its digit comparison for long numbers from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors diff --git a/docs/mkdocs/docs/home/license.md b/docs/mkdocs/docs/home/license.md index d3ce12e28..715c6ca51 100644 --- a/docs/mkdocs/docs/home/license.md +++ b/docs/mkdocs/docs/home/license.md @@ -18,7 +18,7 @@ The class contains the UTF-8 Decoder from Bjoern Hoehrmann which is licensed und The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2009 [Florian Loitsch](https://florian.loitsch.com/) -The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut) +The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, including the conversion of the digits to text by Xiang JunBo and the SIMD instruction sequence of Dougall Johnson, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut) The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/). diff --git a/include/nlohmann/detail/conversions/to_chars.hpp b/include/nlohmann/detail/conversions/to_chars.hpp index 4324e40a5..db50795b5 100644 --- a/include/nlohmann/detail/conversions/to_chars.hpp +++ b/include/nlohmann/detail/conversions/to_chars.hpp @@ -4,6 +4,7 @@ // |_____|_____|_____|_|___| https://github.com/nlohmann/json // // SPDX-FileCopyrightText: 2009 Florian Loitsch +// SPDX-FileCopyrightText: 2025 Victor Zverovich // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT From 3168d591e8ecbcb4ccf67ea0afa802a79125e972 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:07:45 +0200 Subject: [PATCH 06/17] Regenerate the amalgamated headers Signed-off-by: Niels Lohmann --- single_include/nlohmann/json.hpp | 192 ++++++++----------------------- 1 file changed, 49 insertions(+), 143 deletions(-) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 846e21331..f4174a9cf 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -25081,6 +25081,7 @@ NLOHMANN_JSON_NAMESPACE_END // |_____|_____|_____|_|___| https://github.com/nlohmann/json // // SPDX-FileCopyrightText: 2009 Florian Loitsch +// SPDX-FileCopyrightText: 2025 Victor Zverovich // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT @@ -25156,13 +25157,6 @@ computed from the compressed tables of Zmij beyond it. namespace zmij { -/// significand * 10^exponent -struct decimal -{ - std::uint64_t significand; - int exponent; -}; - /// the compressed powers of ten of Zmij inline const std::array& pow10_minor() noexcept { @@ -25341,18 +25335,6 @@ JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexc return shortest_decimal{integral, dec_exp, static_cast(digit), !round_up && !round_down}; } -/// The shortest decimal in the rounding interval of a positive finite double -/// given by its bits, as one number. The significand can end in zeros. -inline decimal to_decimal(std::uint64_t bits) noexcept -{ - const shortest_decimal d = to_shortest(bits); - if (d.has_digit) - { - return decimal{(d.integral * 10) + d.digit, d.exponent}; - } - return decimal{d.integral, d.exponent + 1}; -} - } // namespace zmij } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -26260,88 +26242,6 @@ void grisu2(char* buf, int& len, int& decimal_exponent, FloatType value) grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus); } -/*! -@brief the shortest digits of a positive finite float (other than double): Grisu2 -*/ -template -JSON_HEDLEY_NON_NULL(1) -void shortest_digits(char* buf, int& len, int& decimal_exponent, FloatType value) -{ - grisu2(buf, len, decimal_exponent, value); -} - -/*! -@brief the shortest digits of a positive finite double: the conversion of -Zmij (see zmij.hpp), which always finds the shortest digits that read back as -the same value (Grisu2 does not for about one double in a thousand), and the -closest of them if there are several - -v = buf * 10^decimal_exponent, as for grisu2() -*/ -JSON_HEDLEY_NON_NULL(1) -inline void shortest_digits(char* buf, int& len, int& decimal_exponent, double value) -{ - static_assert(std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53, - "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); - JSON_ASSERT(std::isfinite(value)); - JSON_ASSERT(value > 0); - - std::uint64_t bits = 0; - std::memcpy(&bits, &value, sizeof(bits)); - zmij::decimal d = zmij::to_decimal(bits); - // without trailing zeros (up to 16): 8, 4, 2, 1 at a time - while (d.significand % 100000000 == 0) - { - d.significand /= 100000000; - d.exponent += 8; - } - if (d.significand % 10000 == 0) - { - d.significand /= 10000; - d.exponent += 4; - } - if (d.significand % 100 == 0) - { - d.significand /= 100; - d.exponent += 2; - } - if (d.significand % 10 == 0) - { - d.significand /= 10; - d.exponent += 1; - } - // at most 17 digits, written from the back two at a time - static constexpr const char* pairs = - "00010203040506070809101112131415161718192021222324252627282930313233343536373839" - "40414243444546474849505152535455565758596061626364656667686970717273747576777879" - "8081828384858687888990919293949596979899"; - std::array digits{}; - std::size_t n = digits.size(); - while (d.significand >= 100) - { - const std::uint64_t two_digits = d.significand % 100; // a variable: GCC calls a cast of the remainder useless where std::uint64_t is std::size_t - const auto i = static_cast(two_digits) * 2; - d.significand /= 100; - n -= 2; - digits[n] = pairs[i]; - digits[n + 1] = pairs[i + 1]; - } - if (d.significand >= 10) - { - const auto i = static_cast(d.significand) * 2; - n -= 2; - digits[n] = pairs[i]; - digits[n + 1] = pairs[i + 1]; - } - else - { - digits[--n] = static_cast('0' + d.significand); - } - len = static_cast(digits.size() - n); - std::memcpy(buf, digits.data() + n, static_cast(len)); - decimal_exponent = d.exponent; -} - /*! @brief appends a decimal representation of e to buf @return a pointer to the element following the exponent. @@ -26744,53 +26644,30 @@ inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcep return end + (three ? 5 : 4); } -/// the powers of ten up to 10^16 -inline const std::array& powers_of_ten_16() noexcept -{ - static const std::array powers = - { - { - 1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, 10000000000u, - 100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, 10000000000000000u - } - }; - return powers; -} - /*! -@brief digits * 10^exp, as write_decimal() writes it, for the digits of a -double that need no conversion (count digits, at most 15, the first not 0; -trailing zeros allowed): extended to 16 digits and written by write_shortest() +@brief whether FloatType is an IEEE 754 binary64 type (a double, or a long double +that has the same format, as with MSVC and on Apple's Arm CPUs) -@return a pointer past the text; up to 41 bytes at @a first are written - (some beyond the returned end) +These are the types the conversion of Zmij (see zmij.hpp) is used for; all +others (binary32, or a format the library does not know) use Grisu2. */ -JSON_HEDLEY_NON_NULL(1) -JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_short_decimal(char* first, std::uint64_t digits, int count, int exp) noexcept +template +constexpr bool has_binary64_format() noexcept { - JSON_ASSERT(digits >= powers_of_ten_16()[static_cast(count - 1)] && count <= 15); - const int scale = 16 - count; - return write_shortest(first, zmij::shortest_decimal{digits * powers_of_ten_16()[static_cast(scale)], exp - scale - 1, 0, false}); + return std::numeric_limits::is_iec559 + && std::numeric_limits::digits == 53 + && std::numeric_limits::max_exponent == 1024 + && sizeof(FloatType) == sizeof(std::uint64_t); } -/// as write_short_decimal(), counting the digits (not 0, less than 10^15) -JSON_HEDLEY_NON_NULL(1) -JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noexcept -{ - JSON_ASSERT(digits != 0 && digits < 1000000000000000u); - // floor(log10(2^bits)) + 1 digits, or one less - const int log2_bound = ((64 - count_leading_zeros(digits)) * 1233) >> 12; - const int count = log2_bound + (digits >= powers_of_ten_16()[static_cast(log2_bound)] ? 1 : 0); - return write_short_decimal(first, digits, count, exp); -} +template +struct is_binary64 : std::integral_constant()> {}; -/// a positive finite float (other than double): Grisu2 and format_buffer() +/// a positive finite float (other than binary64): Grisu2 and format_buffer() template JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL -char* write_positive(char* first, const char* last, FloatType value) +char* write_positive_grisu2(char* first, const char* last, FloatType value) { JSON_ASSERT(last - first >= std::numeric_limits::max_digits10); static_cast(last); // (only used in the assertion) @@ -26801,7 +26678,7 @@ char* write_positive(char* first, const char* last, FloatType value) // len is the length of the buffer, i.e., the number of decimal digits. int len = 0; int decimal_exponent = 0; - shortest_digits(first, len, decimal_exponent, value); + grisu2(first, len, decimal_exponent, value); JSON_ASSERT(len <= std::numeric_limits::max_digits10); @@ -26817,15 +26694,16 @@ char* write_positive(char* first, const char* last, FloatType value) return format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp); } -/// a positive finite double: the shortest digits (Zmij), laid out by +/// a positive finite binary64 number: the shortest digits (Zmij), laid out by /// write_shortest() (through a local buffer if [first, last) is shorter than /// the 41 bytes it may write) +template JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_positive(char* first, const char* last, double value) +char* write_positive_zmij(char* first, const char* last, FloatType value) { - static_assert(std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53, - "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); + static_assert(is_binary64::value, + "internal error: the conversion of Zmij needs IEEE 754 binary64 numbers"); std::uint64_t bits = 0; std::memcpy(&bits, &value, sizeof(bits)); const zmij::shortest_decimal d = zmij::to_shortest(bits); @@ -26840,6 +26718,34 @@ inline char* write_positive(char* first, const char* last, double value) return first + len; } +/// a positive finite binary64 number: Zmij (as a long double has the format of +/// a double here, its bits are those of the double of the same value) +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value, std::true_type /*is_binary64*/) +{ + return write_positive_zmij(first, last, value); +} + +/// a positive finite float of any other format: Grisu2 +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value, std::false_type /*is_binary64*/) +{ + return write_positive_grisu2(first, last, value); +} + +/// a positive finite float: Zmij for binary64 numbers, Grisu2 otherwise +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value) +{ + return write_positive(first, last, value, is_binary64 {}); +} + } // namespace dtoa_impl /*! From c006db93d4bf3d58e5ddf50b2a62cfeb22062cd7 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:10:11 +0200 Subject: [PATCH 07/17] Reject integer lengths in json_document::parse parse(ptr, len) compiled: len converted to allow_exceptions, and ptr was read as a C string, past the end of a buffer without a terminating NUL. Delete the overloads of parse, parse_copy, accept, and read that take an integer other than bool where the flags are expected. Signed-off-by: Niels Lohmann --- .../docs/api/basic_json_document/accept.md | 5 +++ .../docs/api/basic_json_document/parse.md | 6 +++ .../api/basic_json_document/parse_copy.md | 3 ++ .../docs/api/basic_json_document/read.md | 3 ++ include/nlohmann/detail/view/input.hpp | 7 +++ include/nlohmann/json_view.hpp | 20 +++++++++ tests/src/unit-json_view.cpp | 44 +++++++++++++++++++ 7 files changed, 88 insertions(+) diff --git a/docs/mkdocs/docs/api/basic_json_document/accept.md b/docs/mkdocs/docs/api/basic_json_document/accept.md index 5aa1fc93d..322350130 100644 --- a/docs/mkdocs/docs/api/basic_json_document/accept.md +++ b/docs/mkdocs/docs/api/basic_json_document/accept.md @@ -43,6 +43,11 @@ input's own copy (for inputs that are always read into a buffer) throws. Linear in the length of the input. +## Notes + +An integer argument that is not a `#!cpp bool` where the flags are expected, such as `#!cpp accept(ptr, len)`, does not +compile; see [`parse`](parse.md#notes). + ## Examples ??? example diff --git a/docs/mkdocs/docs/api/basic_json_document/parse.md b/docs/mkdocs/docs/api/basic_json_document/parse.md index a3c44f4e0..7c46e17ea 100644 --- a/docs/mkdocs/docs/api/basic_json_document/parse.md +++ b/docs/mkdocs/docs/api/basic_json_document/parse.md @@ -100,6 +100,12 @@ See [`owns_source`](owns_source.md) to check which happened after a call, and th **Numbers.** As for [`BasicJsonType::parse()`](../basic_json/parse.md), an integer literal too large for the 64-bit integer type becomes a floating-point value. +**No lengths.** An integer argument that is not a `#!cpp bool` where the flags are expected -- for example +`#!cpp parse(ptr, len)` -- does not compile (the overload is deleted). Such a call would convert `len` to +`allow_exceptions` and read `ptr` as a null-terminated string, past the end of a buffer that has none. To parse a +buffer of a given length, pass a pair of pointers: `#!cpp parse(ptr, ptr + len)`. The same holds for +[`parse_copy`](parse_copy.md), [`accept`](accept.md), and [`read`](read.md). + ## Examples ??? example "Example: (1) borrowed vs. owned input, and errors identical to `BasicJsonType::parse()`" diff --git a/docs/mkdocs/docs/api/basic_json_document/parse_copy.md b/docs/mkdocs/docs/api/basic_json_document/parse_copy.md index ed4714d22..7cb51ad02 100644 --- a/docs/mkdocs/docs/api/basic_json_document/parse_copy.md +++ b/docs/mkdocs/docs/api/basic_json_document/parse_copy.md @@ -51,6 +51,9 @@ Linear in the length of the input. only differs in that the input is always copied rather than sometimes borrowed. Prefer [`parse()`](parse.md) when the input's lifetime already covers the document's, since it avoids the copy for borrowed inputs. +An integer argument that is not a `#!cpp bool` where the flags are expected, such as `#!cpp parse_copy(ptr, len)`, does not +compile; see [`parse`](parse.md#notes). + ## Examples ??? example diff --git a/docs/mkdocs/docs/api/basic_json_document/read.md b/docs/mkdocs/docs/api/basic_json_document/read.md index 2267e46bd..ff37014bd 100644 --- a/docs/mkdocs/docs/api/basic_json_document/read.md +++ b/docs/mkdocs/docs/api/basic_json_document/read.md @@ -49,6 +49,9 @@ whether or not the new parse succeeds; take fresh views from [`root()`](root.md) `input` is borrowed or owned by the same rules as [`parse()`](parse.md#notes); a document can borrow on one call and own on the next, since ownership is decided freshly each time. +An integer argument that is not a `#!cpp bool` where the flags are expected, such as `#!cpp read(ptr, len)`, does not +compile; see [`parse`](parse.md#notes). + ## Examples ??? example diff --git a/include/nlohmann/detail/view/input.hpp b/include/nlohmann/detail/view/input.hpp index e8a5e3148..3fafa2ecf 100644 --- a/include/nlohmann/detail/view/input.hpp +++ b/include/nlohmann/detail/view/input.hpp @@ -59,6 +59,13 @@ struct classify_input // NOLINTEND(readability-avoid-nested-conditional-operator) }; +/// an integer type other than bool: a length passed where a flag is expected +template +struct is_integer_not_bool : std::is_integral {}; + +template<> +struct is_integer_not_bool : std::false_type {}; + /// std::basic_string guarantees a NUL at data()[size()] (the parser's sentinel) template struct is_std_string : std::false_type {}; diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 5aec36ef9..178020cbb 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -291,6 +291,13 @@ class basic_json_document return d; } + /// parse(ptr, len) does not compile: len would convert to allow_exceptions + /// and ptr be read as a C string (as for the overloads of parse_copy, + /// accept, and read below) + template + static typename std::enable_if::value, basic_json_document>::type + parse(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// parse [first, last) template::iterator_category>::value, int>::type = 0> @@ -318,6 +325,10 @@ class basic_json_document return d; } + template + static typename std::enable_if::value, basic_json_document>::type + parse_copy(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// check whether the input is valid JSON (the result of basic_json::accept) template static bool accept(InputType&& input, const bool ignore_comments = false, const bool ignore_trailing_commas = false) @@ -327,6 +338,10 @@ class basic_json_document return !d.is_discarded(); } + template + static typename std::enable_if::value, bool>::type + accept(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// parse into this document, reusing its memory template // flawfinder: ignore (a member function, not POSIX read()) @@ -339,6 +354,11 @@ class basic_json_document std::integral_constant::value> {}); } + template + // flawfinder: ignore (a member function, not POSIX read()) + typename std::enable_if::value, void>::type + read(InputType&& input, IntegerType value, Flags&&... flags) = delete; + //////////// // access // //////////// diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index d953d9903..9df459d3a 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -15,12 +15,14 @@ using nlohmann::json_document; using nlohmann::json_view; using nlohmann::ordered_json_document; +#include #include #include #include #include #include #include +#include #include #include @@ -30,6 +32,16 @@ using nlohmann::ordered_json_document; namespace { +// detection of calls that must not compile +template +using parse_call_t = decltype(json_document::parse(std::declval()...)); +template +using parse_copy_call_t = decltype(json_document::parse_copy(std::declval()...)); +template +using accept_call_t = decltype(json_document::accept(std::declval()...)); +template +using read_call_t = decltype(std::declval().read(std::declval()...)); + #if !defined(JSON_NOEXCEPTION) // the exception parse() throws for a text, or "" if it accepts it std::string parse_exception(const std::string& text, bool comments = false, bool trailing_commas = false) @@ -329,6 +341,38 @@ TEST_CASE("json_view") CHECK(json_document::parse("", false).is_discarded()); } + SECTION("integer arguments do not compile") + { + using nlohmann::detail::is_detected; + + // a length is not a flag: parse(ptr, len) would convert len to + // allow_exceptions and read ptr as a C string, which need not end + static_assert(is_detected::value, "parse(ptr, bool) is valid"); + static_assert(is_detected::value, "parse(ptr, bool, bool, bool) is valid"); + static_assert(is_detected::value, "parse(first, last) is valid"); + static_assert(is_detected::value, "parse(first, last, bool) is valid"); + static_assert(!is_detected::value, "parse(ptr, len) must not compile"); + static_assert(!is_detected::value, "parse(ptr, int) must not compile"); + static_assert(!is_detected::value, "parse(ptr, char) must not compile"); + static_assert(!is_detected::value, "parse(ptr, len, bool) must not compile"); + static_assert(!is_detected::value, "parse(string, len) must not compile"); + static_assert(!is_detected&, std::size_t>::value, "parse(vector, len) must not compile"); + + static_assert(is_detected::value, "parse_copy(ptr, bool) is valid"); + static_assert(!is_detected::value, "parse_copy(ptr, len) must not compile"); + + static_assert(is_detected::value, "accept(ptr, bool) is valid"); + static_assert(!is_detected::value, "accept(ptr, len) must not compile"); + + static_assert(is_detected::value, "read(ptr, bool) is valid"); + static_assert(!is_detected::value, "read(ptr, len) must not compile"); + + // the valid calls still work + const char* const text = "[1]"; + CHECK(json_document::parse(text, true).root().is_array()); + CHECK(json_document::accept(text, true, true)); + } + SECTION("document lifetime and reuse") { json_document d; From beb83ae03b0b680f5c9f7ac52b6130b275151133 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:12:05 +0200 Subject: [PATCH 08/17] Copy const rvalue strings and document the accepted inputs json_document::parse(std::move(const_string)) failed to compile with "no matching read_kind": only a non-const rvalue std::string can be moved from. Treat a const rvalue as a copied byte container. parse() does not accept everything BasicJsonType::parse() does: a FILE* and pointers to or arrays of wide characters are rejected at compile time. List the supported inputs instead. Signed-off-by: Niels Lohmann --- .../docs/api/basic_json_document/parse.md | 29 ++++++++++++------- include/nlohmann/detail/view/input.hpp | 10 +++---- tests/src/unit-json_view.cpp | 13 +++++++++ 3 files changed, 37 insertions(+), 15 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json_document/parse.md b/docs/mkdocs/docs/api/basic_json_document/parse.md index 7c46e17ea..219634370 100644 --- a/docs/mkdocs/docs/api/basic_json_document/parse.md +++ b/docs/mkdocs/docs/api/basic_json_document/parse.md @@ -19,24 +19,33 @@ static basic_json_document parse(IteratorType first, IteratorType last, 1. Deserialize from a compatible input, borrowing or owning it depending on its value category and type (see Notes). 2. Deserialize from a pair of input iterators. -Both overloads accept exactly what [`BasicJsonType::parse()`](../basic_json/parse.md) accepts, with the same +Both overloads accept the same JSON text as [`BasicJsonType::parse()`](../basic_json/parse.md), with the same `ignore_comments`/`ignore_trailing_commas` options, but build a [`basic_json_document`](index.md) (a flat index into -the input) instead of a tree of `BasicJsonType` values. +the input) instead of a tree of `BasicJsonType` values. The input must be byte-oriented (see the template parameters +below): not every input type of `BasicJsonType::parse()` is supported. ## Template parameters `InputType` -: A compatible input, for instance: +: A byte-oriented input, one of: - - a `#!cpp std::string`, `#!cpp std::string_view`, or a C-style array of characters - - a pointer to a null-terminated string of single byte characters + - a `#!cpp std::string`, `#!cpp std::string_view`, or a C-style array of single-byte characters + - a pointer to a null-terminated string of single-byte characters (`#!cpp char`, `#!cpp signed char`, + `#!cpp unsigned char`, `#!cpp std::uint8_t`) - a container for which `#!cpp obj.data()` and `#!cpp obj.size()` give contiguous single-byte access, e.g. `#!cpp std::vector` or `#!cpp std::vector` - - an `#!cpp std::istream` object, or anything else [`BasicJsonType::parse()`](../basic_json/parse.md) accepts + - an `#!cpp std::istream` object + - a wide string object (`#!cpp std::wstring`, `#!cpp std::u16string`, `#!cpp std::u32string`), which is converted + to UTF-8 + + Other inputs are not supported: a `#!cpp FILE*`, and pointers to or arrays of wide characters (`#!cpp wchar_t`, + `#!cpp char16_t`, `#!cpp char32_t`) are rejected at compile time by a `#!cpp static_assert`. (Use + [`BasicJsonType::parse()`](../basic_json/parse.md) for these.) `IteratorType` -: a compatible iterator type, for instance a pair of pointers such as `ptr` and `ptr + len`, or a pair of - `#!cpp std::string::iterator` +: an input iterator type, for instance a pair of pointers such as `ptr` and `ptr + len`, or a pair of + `#!cpp std::string::iterator`; the iterators of single-byte characters are borrowed or read like the byte inputs + above, those of wide characters are converted to UTF-8 ## Parameters @@ -84,8 +93,8 @@ Linear in the length of the input. | `input` | ownership | |--------------------------------------------------------------------------------------|--------------------------------------------------------------| | lvalue byte container (`std::string`, `std::vector`, ...), `std::string_view`, C string, character array | **borrowed** -- `input` must outlive the document | -| rvalue `#!cpp std::string` | **owned**, moved in without a copy | -| rvalue byte container other than `#!cpp std::string` | **owned**, copied | +| non-const rvalue `#!cpp std::string` | **owned**, moved in without a copy | +| other rvalue byte container (including a `#!cpp const` rvalue `#!cpp std::string`) | **owned**, copied | | stream, wide string, or anything else read through the general input adapter | **owned**, read into a buffer (a stream is read to its end) | For overload (2), a pair of pointers to single-byte integers (e.g. `#!cpp const char*`, `#!cpp std::uint8_t*`) is diff --git a/include/nlohmann/detail/view/input.hpp b/include/nlohmann/detail/view/input.hpp index 3fafa2ecf..ac3257ee3 100644 --- a/include/nlohmann/detail/view/input.hpp +++ b/include/nlohmann/detail/view/input.hpp @@ -9,7 +9,7 @@ #pragma once #include // basic_string, char_traits, string -#include // decay, integral_constant, is_array, is_lvalue_reference, is_pointer, is_same, remove_reference +#include // decay, integral_constant, is_array, is_const, is_integral, is_lvalue_reference, is_pointer, is_same, remove_reference #include // forward #include @@ -28,12 +28,12 @@ namespace view /// how a document takes its input enum class input_kind { - move_string, ///< rvalue std::string: owned without a copy + move_string, ///< non-const rvalue std::string: owned without a copy c_string, ///< const char* (NUL-terminated): borrowed char_array, ///< char array (e.g. a string literal): borrowed borrow_range, ///< lvalue contiguous byte container, or std::string_view: borrowed - copy_range, ///< rvalue contiguous byte container: copied - adapter, ///< anything else parse() accepts (streams, wide strings, ...): read into a buffer + copy_range, ///< rvalue contiguous byte container (a const rvalue std::string too): copied + adapter, ///< streams, wide strings, and the rest of what the library's input adapter reads: read into a buffer }; template @@ -52,7 +52,7 @@ struct classify_input static constexpr input_kind value = std::is_array::value ? input_kind::char_array : std::is_pointer::value ? input_kind::c_string - : (is_rvalue && std::is_same::value) ? input_kind::move_string + : (is_rvalue && !std::is_const::value && std::is_same::value) ? input_kind::move_string : (is_bytes && (!is_rvalue || is_string_view)) ? input_kind::borrow_range : is_bytes ? input_kind::copy_range : input_kind::adapter; diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 9df459d3a..3b38e078f 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -314,6 +314,19 @@ TEST_CASE("json_view") CHECK(from_rvalue.owns_source()); CHECK(from_rvalue.root().materialize() == expected); CHECK(json_document::parse(std::vector(text.begin(), text.end())).owns_source()); + // a const rvalue cannot be moved from, and is not borrowed (it may be a + // temporary): it is copied, as is a const rvalue of any container + const std::string const_text = text; + const json_document from_const_rvalue = json_document::parse(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) + CHECK(from_const_rvalue.owns_source()); + CHECK(from_const_rvalue.source().data() != const_text.data()); + CHECK(from_const_rvalue.root().materialize() == expected); + json_document read_const_rvalue; + read_const_rvalue.read(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) + CHECK(read_const_rvalue.owns_source()); + CHECK(read_const_rvalue.root().materialize() == expected); + const std::vector const_chars(text.begin(), text.end()); + CHECK(json_document::parse(std::move(const_chars)).owns_source()); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) CHECK(json_document::parse_copy(text).owns_source()); CHECK(json_document::parse_copy(text).root().materialize() == expected); std::istringstream stream(text); From 7beac68df726f48c43fc5c1e41cb84005abef905 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:18:50 +0200 Subject: [PATCH 09/17] Remove the conversion to bool from json_view explicit operator bool meant "refers to a value", which silently differs from what a basic_json converts to. Use !v.is_discarded() instead. Signed-off-by: Niels Lohmann --- .../api/basic_json_view/basic_json_view.md | 5 +- docs/mkdocs/docs/api/basic_json_view/index.md | 1 - .../docs/api/basic_json_view/is_discarded.md | 6 +-- .../docs/api/basic_json_view/operator_bool.md | 46 ------------------- .../basic_json_view__basic_json_view.cpp | 4 +- .../basic_json_view__basic_json_view.output | 4 +- .../basic_json_view__type_predicates.cpp | 4 +- .../basic_json_view__type_predicates.output | 4 +- docs/mkdocs/mkdocs.yml | 1 - include/nlohmann/json_view.hpp | 6 --- tests/src/unit-json_view.cpp | 8 +++- 11 files changed, 19 insertions(+), 70 deletions(-) delete mode 100644 docs/mkdocs/docs/api/basic_json_view/operator_bool.md diff --git a/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md b/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md index 143ea2bea..bd10eae6f 100644 --- a/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md +++ b/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md @@ -4,8 +4,8 @@ basic_json_view() noexcept = default; ``` -Creates an invalid (discarded) view: [`type()`](type.md) is `#!cpp value_t::discarded`, -[`is_discarded()`](is_discarded.md) is `#!cpp true`, and `#!cpp explicit operator bool()` is `#!cpp false`. +Creates an invalid (discarded) view: [`type()`](type.md) is `#!cpp value_t::discarded` and +[`is_discarded()`](is_discarded.md) is `#!cpp true`. This is the only constructor a caller can use directly. Every other view is obtained from a [`basic_json_document`](../basic_json_document/index.md), via [`root()`](../basic_json_document/root.md) or (once @@ -43,7 +43,6 @@ placeholder for "no value yet" and later be assigned a real view. ## See also - [is_discarded](is_discarded.md) - return whether the view is invalid -- [operator bool](operator_bool.md) - return whether the view refers to a value - [root](../basic_json_document/root.md) - the view of a document's root value ## Version history diff --git a/docs/mkdocs/docs/api/basic_json_view/index.md b/docs/mkdocs/docs/api/basic_json_view/index.md index 564406667..8ce113ceb 100644 --- a/docs/mkdocs/docs/api/basic_json_view/index.md +++ b/docs/mkdocs/docs/api/basic_json_view/index.md @@ -62,7 +62,6 @@ element access, iteration, `get()`, JSON Pointer support, `dump()`, or compar - [**is_primitive**](is_primitive.md) - return whether the type is primitive - [**is_structured**](is_structured.md) - return whether the type is structured - [**is_discarded**](is_discarded.md) - return whether the view is invalid -- [**operator bool**](operator_bool.md) - return whether the view refers to a value ### Capacity diff --git a/docs/mkdocs/docs/api/basic_json_view/is_discarded.md b/docs/mkdocs/docs/api/basic_json_view/is_discarded.md index 4cc11077d..22210a5ad 100644 --- a/docs/mkdocs/docs/api/basic_json_view/is_discarded.md +++ b/docs/mkdocs/docs/api/basic_json_view/is_discarded.md @@ -23,8 +23,9 @@ Constant. ## Notes -`#!cpp v.is_discarded()` and `#!cpp !static_cast(v)` are equivalent; use whichever reads better at the call -site. +A `basic_json_view` is not convertible to `#!cpp bool`: such a conversion would mean "refers to a value", whereas +`basic_json` converts to the `#!cpp bool` it holds, so the same code would silently behave differently. Test +`#!cpp !v.is_discarded()` explicitly. ## Examples @@ -45,7 +46,6 @@ site. ## See also -- [operator bool](operator_bool.md) - return whether the view refers to a value - [(constructor)](basic_json_view.md) - the default constructor creates a discarded view - [is_discarded (basic_json_document)](../basic_json_document/is_discarded.md) - return whether the last parse failed - [`BasicJsonType::is_discarded`](../basic_json/is_discarded.md) - the corresponding function of `basic_json` diff --git a/docs/mkdocs/docs/api/basic_json_view/operator_bool.md b/docs/mkdocs/docs/api/basic_json_view/operator_bool.md deleted file mode 100644 index bc13c100c..000000000 --- a/docs/mkdocs/docs/api/basic_json_view/operator_bool.md +++ /dev/null @@ -1,46 +0,0 @@ -# nlohmann::basic_json_view::operator bool - -```cpp -explicit operator bool() const noexcept; -``` - -Returns whether this view refers to a value, i.e. the negation of [`is_discarded()`](is_discarded.md). Being -`#!cpp explicit`, this conversion is only considered in a boolean context (`#!cpp if (v)`, `#!cpp !v`, `#!cpp v && -...`), not for implicit conversions to other types. - -## Return value - -`#!cpp true` if the view refers to a value, `#!cpp false` if it is [discarded](is_discarded.md). - -## Exception safety - -No-throw guarantee: this function never throws exceptions. - -## Complexity - -Constant. - -## Examples - -??? example - - The example below classifies several parsed documents by the type of their root value, without materializing any - of them into a `BasicJsonType` value. - - ```cpp - --8<-- "examples/basic_json_view__type_predicates.cpp" - ``` - - Output: - - ```json - --8<-- "examples/basic_json_view__type_predicates.output" - ``` - -## See also - -- [is_discarded](is_discarded.md) - return whether the view is invalid - -## Version history - -- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp index af08f22e0..3d813717d 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp +++ b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp @@ -8,10 +8,10 @@ int main() // the default constructor is the only public one: it creates an invalid // (discarded) view, useful as a "no value yet" placeholder nlohmann::json_view v; - std::cout << static_cast(v) << ' ' << v.is_discarded() << '\n'; + std::cout << v.is_discarded() << '\n'; // views are trivially copyable handles (two pointers); the document owns // the actual data nlohmann::json_view copy = v; - std::cout << static_cast(copy) << '\n'; + std::cout << copy.is_discarded() << '\n'; } diff --git a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output index 8192d6145..bb101b641 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output +++ b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output @@ -1,2 +1,2 @@ -false true -false +true +true diff --git a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp index 1e3ed62e2..edc37ff52 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp +++ b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp @@ -35,8 +35,8 @@ int main() // parse without exceptions, are both discarded nlohmann::json_view invalid; json_document failed = json_document::parse("not json", /* allow_exceptions */ false); - std::cout << static_cast(invalid) << ' ' << invalid.is_discarded() << '\n'; - std::cout << static_cast(failed.root()) << ' ' << failed.root().is_discarded() << '\n'; + std::cout << invalid.is_discarded() << '\n'; + std::cout << failed.root().is_discarded() << '\n'; // type() returns the same value_t enumeration as basic_json::type() std::cout << (d_object.root().type() == nlohmann::json::value_t::object) << '\n'; diff --git a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output index 285d31ca0..99e264447 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output +++ b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output @@ -7,6 +7,6 @@ true true true true false false -false true -false true +true +true true diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 8324a303b..ce07e3d02 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -266,7 +266,6 @@ nav: - 'is_string': api/basic_json_view/is_string.md - 'is_structured': api/basic_json_view/is_structured.md - 'materialize': api/basic_json_view/materialize.md - - 'operator bool': api/basic_json_view/operator_bool.md - 'size': api/basic_json_view/size.md - 'source_offset': api/basic_json_view/source_offset.md - 'type': api/basic_json_view/type.md diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 178020cbb..66a48960e 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -156,12 +156,6 @@ class basic_json_view return type() == value_t::discarded; } - /// false for discarded views - explicit operator bool() const noexcept - { - return m_node != nullptr; - } - ////////////// // capacity // ////////////// diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 3b38e078f..4e598c617 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -41,6 +41,8 @@ template using accept_call_t = decltype(json_document::accept(std::declval()...)); template using read_call_t = decltype(std::declval().read(std::declval()...)); +template +using bool_conversion_t = decltype(static_cast(std::declval())); #if !defined(JSON_NOEXCEPTION) // the exception parse() throws for a text, or "" if it accepts it @@ -155,7 +157,6 @@ TEST_CASE("json_view") CHECK(v.is_primitive() == j.is_primitive()); CHECK(v.is_structured() == j.is_structured()); CHECK(!v.is_discarded()); - CHECK(static_cast(v)); CHECK(v.size() == j.size()); CHECK(v.empty() == j.empty()); CHECK(v.materialize() == j); @@ -163,7 +164,6 @@ TEST_CASE("json_view") const json_view invalid{}; CHECK(invalid.is_discarded()); - CHECK(!static_cast(invalid)); CHECK(invalid.type() == json::value_t::discarded); CHECK(invalid.size() == 0); CHECK(invalid.empty()); @@ -380,6 +380,10 @@ TEST_CASE("json_view") static_assert(is_detected::value, "read(ptr, bool) is valid"); static_assert(!is_detected::value, "read(ptr, len) must not compile"); + // json_view has no conversion to bool: unlike basic_json's, it would + // mean "exists", not "is not null"; use is_discarded() + static_assert(!is_detected::value, "json_view must not convert to bool"); + // the valid calls still work const char* const text = "[1]"; CHECK(json_document::parse(text, true).root().is_array()); From 14afaca083e6b8dbe7fa045066563ba062135c1c Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:22:39 +0200 Subject: [PATCH 10/17] Delete json_document::root() on temporary documents auto v = json_document::parse(text).root() compiled and left the view dangling. Delete the overload for rvalue documents, take a named document in the tests, and document the lifetime rule. Signed-off-by: Niels Lohmann --- .../docs/api/basic_json_document/root.md | 19 ++++- docs/mkdocs/docs/features/json_view.md | 3 + include/nlohmann/json_view.hpp | 5 +- tests/src/unit-json_view.cpp | 75 ++++++++++++++----- 4 files changed, 80 insertions(+), 22 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json_document/root.md b/docs/mkdocs/docs/api/basic_json_document/root.md index 85682dbcd..3ccedeb4a 100644 --- a/docs/mkdocs/docs/api/basic_json_document/root.md +++ b/docs/mkdocs/docs/api/basic_json_document/root.md @@ -1,10 +1,15 @@ # nlohmann::basic_json_document::root ```cpp -view_type root() const noexcept; +// (1) +view_type root() const& noexcept; + +// (2) +view_type root() const&& = delete; ``` -Returns a view of the root value of the document. +1. Returns a view of the root value of the document. +2. Deleted: the view of a temporary document would dangle. ## Return value @@ -21,6 +26,16 @@ Constant. ## Notes +**Lifetime.** A view refers into the document, so the document must outlive it. `root()` can therefore only be called +on a document that has a name (an lvalue); calling it on a temporary does not compile: + +```cpp +auto v = json_document::parse(text).root(); // error: the document is destroyed at the end of the statement + +auto doc = json_document::parse(text); // OK: keep the document alive +auto v = doc.root(); +``` + `root()` is a cheap handle into the document's index, not a copy of anything; call it as often as needed. The returned view is valid under the same conditions as any other view of the document -- see [Object inspection](../basic_json_view/index.md) -- in particular, it is invalidated by the next diff --git a/docs/mkdocs/docs/features/json_view.md b/docs/mkdocs/docs/features/json_view.md index 6b66c42ca..c6280ef62 100644 --- a/docs/mkdocs/docs/features/json_view.md +++ b/docs/mkdocs/docs/features/json_view.md @@ -76,6 +76,9 @@ Moving the document itself is fine and does **not** invalidate its views: the in that keeps its address across the move. Take a fresh view from [`root()`](../api/basic_json_document/root.md) whenever any of the other conditions above was not met. +Because a view dies with its document, [`root()`](../api/basic_json_document/root.md) is not callable on a temporary +document: `#!cpp auto v = json_document::parse(text).root();` does not compile. Give the document a name first. + ??? example "Example: borrowed and owned documents, and when views become invalid" ```cpp diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 66a48960e..e2cb57c76 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -358,7 +358,7 @@ class basic_json_document //////////// /// the root value (discarded if parsing failed without exceptions) - view_type root() const noexcept + view_type root() const& noexcept { if (!m_data || m_data->discarded) { @@ -367,6 +367,9 @@ class basic_json_document return view_type(m_data.get(), m_data->tape); } + /// deleted: the view of a temporary document would dangle + view_type root() const&& = delete; + bool is_discarded() const noexcept { return !m_data || m_data->discarded; diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 4e598c617..ffd44ca17 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -32,6 +32,22 @@ using nlohmann::ordered_json_document; namespace { +// the value of a text, through a named document: the views of a temporary +// document would dangle (root() of an rvalue document does not compile) +template +auto materialized(Args&& ... args) -> decltype(std::declval().materialize()) +{ + const Document d = Document::parse(std::forward(args)...); + return d.root().materialize(); +} + +template +auto materialized_copy(Input&& input) -> decltype(std::declval().materialize()) +{ + const Document d = Document::parse_copy(std::forward(input)); + return d.root().materialize(); +} + // detection of calls that must not compile template using parse_call_t = decltype(json_document::parse(std::declval()...)); @@ -41,6 +57,8 @@ template using accept_call_t = decltype(json_document::accept(std::declval()...)); template using read_call_t = decltype(std::declval().read(std::declval()...)); +template +using root_call_t = decltype(std::declval().root()); template using bool_conversion_t = decltype(static_cast(std::declval())); @@ -179,19 +197,19 @@ TEST_CASE("json_view") std::string text; g.value(text, 0); CAPTURE(text) - CHECK(json_document::parse(text).root().materialize() == json::parse(text)); + CHECK(materialized(text) == json::parse(text)); // member order as ordered_json::parse keeps it - CHECK(ordered_json_document::parse(text).root().materialize().dump() == ordered_json::parse(text).dump()); + CHECK(materialized(text).dump() == ordered_json::parse(text).dump()); } // duplicate keys: the last value, at the position of the first key - CHECK(json_document::parse(R"({"a":1,"b":2,"a":3})").root().materialize() == json::parse(R"({"a":1,"b":2,"a":3})")); - CHECK(ordered_json_document::parse(R"({"a":1,"b":2,"a":3})").root().materialize().dump() == R"({"a":3,"b":2})"); + CHECK(materialized(R"({"a":1,"b":2,"a":3})") == json::parse(R"({"a":1,"b":2,"a":3})")); + CHECK(materialized(R"({"a":1,"b":2,"a":3})").dump() == R"({"a":3,"b":2})"); // very deep nesting (iterative, as parse()) const std::string deep = std::string(100000, '[') + std::string(100000, ']'); - CHECK(json_document::parse(deep).root().materialize() == json::parse(deep)); + CHECK(materialized(deep) == json::parse(deep)); #if JSON_DIAGNOSTICS // the parents are set, so errors name the path - const json m = json_document::parse(R"({"a":{"b":[1]}})").root().materialize(); + const json m = materialized(R"({"a":{"b":[1]}})"); CHECK_THROWS_WITH_AS(m.at("a").at("b").at(0).at("x"), "[json.exception.type_error.304] (/a/b/0) cannot use at() with number", json::type_error&); #endif } @@ -264,7 +282,7 @@ TEST_CASE("json_view") CHECK(json_document::accept(text)); if (accepted) { - CHECK(float_document::parse(text).root().materialize() == json_float::parse(text)); + CHECK(materialized(text) == json_float::parse(text)); } } float_document f; @@ -278,7 +296,7 @@ TEST_CASE("json_view") CHECK(json_document::accept(with_nul) == json::accept(with_nul)); const std::string nul_in_comment("[1, // c\0\n2]", 12); CHECK(json_document::accept(nul_in_comment, true) == json::accept(nul_in_comment, true)); - CHECK(json_document::parse("\xEF\xBB\xBF[1]").root().materialize() == json::parse("\xEF\xBB\xBF[1]")); + CHECK(materialized("\xEF\xBB\xBF[1]") == json::parse("\xEF\xBB\xBF[1]")); #if !defined(JSON_NOEXCEPTION) CHECK(view_exception("\xEF\xBB") == parse_exception("\xEF\xBB")); #endif @@ -294,18 +312,18 @@ TEST_CASE("json_view") CHECK(!borrowed.owns_source()); CHECK(borrowed.source().data() == text.data()); CHECK(borrowed.root().materialize() == expected); - CHECK(json_document::parse(text.c_str()).root().materialize() == expected); - CHECK(json_document::parse(R"([1, "two", {"three": 3.5}])").root().materialize() == expected); - CHECK(json_document::parse(text.data(), text.data() + text.size()).root().materialize() == expected); + CHECK(materialized(text.c_str()) == expected); + CHECK(materialized(R"([1, "two", {"three": 3.5}])") == expected); + CHECK(materialized(text.data(), text.data() + text.size()) == expected); const std::vector chars(text.begin(), text.end()); CHECK(!json_document::parse(chars).owns_source()); - CHECK(json_document::parse(chars).root().materialize() == expected); + CHECK(materialized(chars) == expected); const std::vector bytes(text.begin(), text.end()); - CHECK(json_document::parse(bytes).root().materialize() == expected); + CHECK(materialized(bytes) == expected); #ifdef JSON_HAS_CPP_17 const std::string_view sv = text; CHECK(!json_document::parse(sv).owns_source()); - CHECK(json_document::parse(sv).root().materialize() == expected); + CHECK(materialized(sv) == expected); #endif // owned @@ -328,14 +346,14 @@ TEST_CASE("json_view") const std::vector const_chars(text.begin(), text.end()); CHECK(json_document::parse(std::move(const_chars)).owns_source()); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) CHECK(json_document::parse_copy(text).owns_source()); - CHECK(json_document::parse_copy(text).root().materialize() == expected); + CHECK(materialized_copy(text) == expected); std::istringstream stream(text); const json_document from_stream = json_document::parse(stream); CHECK(from_stream.owns_source()); CHECK(from_stream.root().materialize() == expected); const std::list list(text.begin(), text.end()); CHECK(json_document::parse(list.begin(), list.end()).owns_source()); - CHECK(json_document::parse(list.begin(), list.end()).root().materialize() == expected); + CHECK(materialized(list.begin(), list.end()) == expected); // iterator pairs: pointers are borrowed, and so are contiguous library // iterators where the input adapter detects them (C++20) @@ -346,10 +364,10 @@ TEST_CASE("json_view") CHECK((from_iterators.source().data() == chars.data()) == contiguous); CHECK(from_iterators.root().materialize() == expected); const std::string padded = "x" + text + "x"; - CHECK(json_document::parse(padded.begin() + 1, padded.end() - 1).root().materialize() == expected); + CHECK(materialized(padded.begin() + 1, padded.end() - 1) == expected); CHECK(json_document::parse(chars.cbegin(), chars.cbegin(), false).is_discarded()); const std::wstring wide = L"[\"\u00e4\u20ac\", 1]"; - CHECK(json_document::parse(wide).root().materialize() == json::parse(wide)); + CHECK(materialized(wide) == json::parse(wide)); CHECK(json_document::parse(static_cast(nullptr), false).is_discarded()); CHECK(json_document::parse("", false).is_discarded()); } @@ -386,10 +404,29 @@ TEST_CASE("json_view") // the valid calls still work const char* const text = "[1]"; - CHECK(json_document::parse(text, true).root().is_array()); + CHECK(materialized(text, true) == json::parse(text)); CHECK(json_document::accept(text, true, true)); } + SECTION("root of a temporary document does not compile") + { + using nlohmann::detail::is_detected; + + // the view would dangle: auto v = json_document::parse(text).root(); + static_assert(is_detected::value, "root() of an lvalue is valid"); + static_assert(is_detected::value, "root() of a const lvalue is valid"); + static_assert(!is_detected::value, "root() of an rvalue must not compile"); + static_assert(!is_detected < root_call_t, json_document && >::value, "root() of an rvalue must not compile"); + static_assert(!is_detected < root_call_t, const json_document && >::value, "root() of a const rvalue must not compile"); + static_assert(!is_detected::value, "root() of an rvalue must not compile"); + + // a named document is fine, also after a move + json_document d = json_document::parse("[1]"); + CHECK(d.root().size() == 1); + const json_document moved = std::move(d); + CHECK(moved.root().size() == 1); + } + SECTION("document lifetime and reuse") { json_document d; From 320a463af72b2a724a2b588b0ae2384c457b1dbf Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:26:32 +0200 Subject: [PATCH 11/17] Store node integers by halves on big-endian targets The non-little-endian path of the node writer wrote the integer's native word over len and next, so len got the high half there. Compose and split the value explicitly (len is the low half, next the high half); the little-endian path stays a plain memcpy. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/view/builder.hpp | 5 +- include/nlohmann/detail/view/node.hpp | 12 ++++- tests/src/unit-json_view_builder.cpp | 58 ++++++++++++++++++++++++ 3 files changed, 73 insertions(+), 2 deletions(-) diff --git a/include/nlohmann/detail/view/builder.hpp b/include/nlohmann/detail/view/builder.hpp index 1cd01fef6..5ce68220d 100644 --- a/include/nlohmann/detail/view/builder.hpp +++ b/include/nlohmann/detail/view/builder.hpp @@ -801,7 +801,10 @@ indent_done: n->flags = flags; n->extra = extra; n->off = static_cast(off); - set_integer_bits(*n, second); + // len is the low half of the second word, next the high half + // (not a native word over both, which swaps them on big-endian) + n->len = static_cast(second); + n->next = static_cast(second >> 32); #endif return n; } diff --git a/include/nlohmann/detail/view/node.hpp b/include/nlohmann/detail/view/node.hpp index 471c1f844..5818066ae 100644 --- a/include/nlohmann/detail/view/node.hpp +++ b/include/nlohmann/detail/view/node.hpp @@ -57,17 +57,27 @@ NLOHMANN_VIEW_ALWAYS_INLINE bool is_container(const node& n) noexcept return static_cast(n.kind) - 1u <= 1u; } -/// the converted value of an integer node (stored in len/next) +/// the converted value of an integer node: len is its low half, next its high +/// half (on little-endian targets the two words are the value in memory) NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept { +#if NLOHMANN_VIEW_LITTLE_ENDIAN std::uint64_t v = 0; std::memcpy(&v, reinterpret_cast(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) return v; +#else + return static_cast(n.len) | (static_cast(n.next) << 32); +#endif } NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept { +#if NLOHMANN_VIEW_LITTLE_ENDIAN std::memcpy(reinterpret_cast(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) +#else + n.len = static_cast(v); + n.next = static_cast(v >> 32); +#endif } /// token length of a number node diff --git a/tests/src/unit-json_view_builder.cpp b/tests/src/unit-json_view_builder.cpp index 9d3ea943b..c585eb919 100644 --- a/tests/src/unit-json_view_builder.cpp +++ b/tests/src/unit-json_view_builder.cpp @@ -401,3 +401,61 @@ TEST_CASE("json_view builder") } } } + +TEST_CASE("json_view node integer bits") +{ + using nlohmann::detail::view::integer_bits; + using nlohmann::detail::view::set_integer_bits; + + // an integer lives in len (low half) and next (high half), on any byte + // order; a big-endian target must not store the native word over both + SECTION("set_integer_bits and integer_bits") + { + node n = {}; + for (const std::uint64_t v : + { + std::uint64_t{0}, std::uint64_t{1}, std::uint64_t{0xFFFFFFFFu}, std::uint64_t{0x100000000u}, + std::uint64_t{0x0000000200000003u}, std::uint64_t{0x0123456789ABCDEFu}, std::uint64_t{0xFFFFFFFFFFFFFFFEu} + }) + { + CAPTURE(v) + n.kind = 0x5A; + n.flags = 0xA5; + n.extra = 0x1234; + n.off = 0x89ABCDEFu; + set_integer_bits(n, v); + CHECK(integer_bits(n) == v); + CHECK(n.len == static_cast(v)); + CHECK(n.next == static_cast(v >> 32)); + // the other fields are untouched + CHECK(n.kind == 0x5A); + CHECK(n.flags == 0xA5); + CHECK(n.extra == 0x1234); + CHECK(n.off == 0x89ABCDEFu); + } + } + + SECTION("parsed integers") + { + struct integer_case + { + const char* text; + std::uint64_t bits; + }; + for (const integer_case c : + { + integer_case{"[8589934595]", 0x0000000200000003u}, integer_case{"[4294967296]", 0x100000000u}, integer_case{"[4294967295]", 0xFFFFFFFFu}, + integer_case{"[-2]", 0xFFFFFFFFFFFFFFFEu}, integer_case{"[-4294967297]", 0xFFFFFFFEFFFFFFFFu}, integer_case{"[18446744073709551615]", 0xFFFFFFFFFFFFFFFFu}, + integer_case{"[7]", 7u} + }) + { + CAPTURE(c.text) + const built b = build(c.text, false, false, true); + REQUIRE(b.ok); + const node& n = b.data->tape[1]; + CHECK(integer_bits(n) == c.bits); + CHECK(n.len == static_cast(c.bits)); + CHECK(n.next == static_cast(c.bits >> 32)); + } + } +} From a1b5b348abcf1c9a8eb770689a19a058eb836fec Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:28:19 +0200 Subject: [PATCH 12/17] Check the node array size for overflow reserve(n) and the growth of the index computed n * sizeof(node) without a check, which wraps around on 32-bit targets for inputs of about 1 GiB and allocates a too small array. Throw std::bad_alloc for a count beyond the address space and clamp the growth step to it. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/view/builder.hpp | 14 +++++++-- .../nlohmann/detail/view/document_data.hpp | 22 ++++++++++++-- tests/src/unit-json_view_builder.cpp | 30 +++++++++++++++++++ 3 files changed, 62 insertions(+), 4 deletions(-) diff --git a/include/nlohmann/detail/view/builder.hpp b/include/nlohmann/detail/view/builder.hpp index 5ce68220d..ccb54c3d7 100644 --- a/include/nlohmann/detail/view/builder.hpp +++ b/include/nlohmann/detail/view/builder.hpp @@ -9,7 +9,7 @@ #pragma once -#include // find, find_if, max +#include // find, find_if, max, min #include // array #include // size_t, ptrdiff_t #include // int64_t, uint8_t, uint16_t, uint32_t, uint64_t @@ -195,8 +195,18 @@ class builder const std::uint64_t done = static_cast(at - b) + 1; const std::uint64_t guess = static_cast(n) * static_cast(e - b + 1) / done; const std::uint64_t grown = guess + (guess / 4) + 64; // a variable: GCC calls a cast of the sum useless where std::uint64_t is std::size_t + // (n is below 2^32: the input is smaller than 4 GiB; the sum cannot wrap) + const std::uint64_t wanted = (std::max)(grown, static_cast(n) + (n / 2) + 64); + const std::uint64_t limit = document_data::max_nodes(); doc.tape_size = n; - doc.reserve((std::max)(static_cast(grown), n + (n / 2) + 64)); + // LCOV_EXCL_START (a node array that fills the address space) + if (NLOHMANN_VIEW_UNLIKELY(n >= limit)) + { + document_data::throw_bad_alloc(); // no room for another node + } + // LCOV_EXCL_STOP + // (a count beyond the limit is cut: the index does not grow beyond what can be addressed) + doc.reserve(static_cast((std::min)(wanted, limit))); return doc.tape; } diff --git a/include/nlohmann/detail/view/document_data.hpp b/include/nlohmann/detail/view/document_data.hpp index 66275702c..aa3958e41 100644 --- a/include/nlohmann/detail/view/document_data.hpp +++ b/include/nlohmann/detail/view/document_data.hpp @@ -11,7 +11,8 @@ #include // array #include // size_t #include // memcpy -#include // operator new, placement new +#include // numeric_limits +#include // bad_alloc, operator new, placement new #include // string #include @@ -84,13 +85,30 @@ struct document_data tape_cap = inline_cap; } - /// make room for n nodes; keeps the first tape_size nodes + /// the largest node count whose size in bytes fits a std::size_t + static constexpr std::size_t max_nodes() noexcept + { + return (std::numeric_limits::max)() / sizeof(node); + } + + [[noreturn]] NLOHMANN_VIEW_NOINLINE static void throw_bad_alloc() + { + NLOHMANN_VIEW_THROW(std::bad_alloc()); + } + + /// make room for n nodes; keeps the first tape_size nodes (throws + /// std::bad_alloc for a count that does not fit the address space, + /// instead of wrapping around in n * sizeof(node)) void reserve(std::size_t n) { if (n <= tape_cap) { return; } + if (NLOHMANN_VIEW_UNLIKELY(n > max_nodes())) + { + throw_bad_alloc(); + } node* fresh = static_cast(::operator new (n * sizeof(node))); if (tape_size != 0) { diff --git a/tests/src/unit-json_view_builder.cpp b/tests/src/unit-json_view_builder.cpp index c585eb919..8b8518c38 100644 --- a/tests/src/unit-json_view_builder.cpp +++ b/tests/src/unit-json_view_builder.cpp @@ -19,8 +19,10 @@ using nlohmann::json; #include #include +#include #include #include +#include #include #include #include @@ -459,3 +461,31 @@ TEST_CASE("json_view node integer bits") } } } + +TEST_CASE("json_view node array size limit") +{ + // a node array larger than the address space is refused, not wrapped to a + // small allocation (the size computation overflows on 32-bit targets, and + // for absurd counts everywhere) + std::unique_ptr d(document_data::create(0)); + d->reserve(8); + REQUIRE(d->tape_cap >= 8); + d->tape_size = 2; + const std::size_t cap = d->tape_cap; + node* const tape = d->tape; + +#if !defined(JSON_NOEXCEPTION) + const std::size_t too_many = document_data::max_nodes() + 1; + CHECK_THROWS_AS(d->reserve(too_many), std::bad_alloc&); + CHECK_THROWS_AS(d->reserve((std::numeric_limits::max)()), std::bad_alloc&); + // the array is unchanged + CHECK(d->tape == tape); + CHECK(d->tape_cap == cap); + CHECK(d->tape_size == 2); +#endif + + // the largest count that fits is not refused by the check (nothing is + // allocated for a count that is already there) + d->reserve(cap); + CHECK(d->tape == tape); +} From 078903cbb92b1acfe3a760ed348f038d4cefb66a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:30:00 +0200 Subject: [PATCH 13/17] State the exact input size limit of json_document The check rejects inputs of 0xFFFFFFF0 bytes or more, but the exception message and the documentation said 4 GiB. Name the limit once (max_input_size), and state 4 GiB minus 16 bytes in the message and the documentation. Test the limit with a container that only claims the size. Signed-off-by: Niels Lohmann --- .../docs/api/basic_json_document/parse.md | 4 +-- docs/mkdocs/docs/features/json_view.md | 2 +- docs/mkdocs/docs/home/architecture.md | 3 +- docs/mkdocs/docs/home/exceptions.md | 4 +-- include/nlohmann/detail/view/errors.hpp | 5 ++- include/nlohmann/detail/view/node.hpp | 5 +++ include/nlohmann/json_view.hpp | 4 +-- tests/src/unit-json_view.cpp | 35 +++++++++++++++++++ 8 files changed, 51 insertions(+), 11 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json_document/parse.md b/docs/mkdocs/docs/api/basic_json_document/parse.md index 219634370..972c63752 100644 --- a/docs/mkdocs/docs/api/basic_json_document/parse.md +++ b/docs/mkdocs/docs/api/basic_json_document/parse.md @@ -79,8 +79,8 @@ discarded; see [`is_discarded`](is_discarded.md). Throws the same exception [`BasicJsonType::parse()`](../basic_json/parse.md) throws for the same input and options -- the same exception id, message, and position -- because on a failing input the library's own parser is run on the same bytes to produce the diagnostic. Additionally throws -[`out_of_range.416`](../../home/exceptions.md#jsonexceptionout_of_range416) if the input is 4 GiB or larger, a size -[`BasicJsonType::parse()`](../basic_json/parse.md) does not reject. +[`out_of_range.416`](../../home/exceptions.md#jsonexceptionout_of_range416) if the input is 4294967280 bytes (4 GiB +minus 16 bytes) or larger, a size [`BasicJsonType::parse()`](../basic_json/parse.md) does not reject. ## Complexity diff --git a/docs/mkdocs/docs/features/json_view.md b/docs/mkdocs/docs/features/json_view.md index c6280ef62..6b49ee2e3 100644 --- a/docs/mkdocs/docs/features/json_view.md +++ b/docs/mkdocs/docs/features/json_view.md @@ -111,7 +111,7 @@ document: `#!cpp auto v = json_document::parse(text).root();` does not compile. - **Only 64-bit integers.** `basic_json_document` requires `BasicJsonType::number_integer_t` and `number_unsigned_t` to both be 64 bits wide; this is a compile-time `#!cpp static_assert`. -- **A 4 GiB input limit.** An input of 4 GiB or more throws +- **A 4 GiB input limit.** An input of 4294967280 bytes (4 GiB minus 16 bytes) or more throws [`out_of_range.416`](../home/exceptions.md#jsonexceptionout_of_range416), a limit `#!cpp basic_json::parse()` does not have. - **A stream is always read to its end.** There is no partial/streaming read of an `#!cpp std::istream`. diff --git a/docs/mkdocs/docs/home/architecture.md b/docs/mkdocs/docs/home/architecture.md index 457f0d536..f7e6c4f25 100644 --- a/docs/mkdocs/docs/home/architecture.md +++ b/docs/mkdocs/docs/home/architecture.md @@ -210,7 +210,8 @@ packet-beta - **Navigation** needs no pointers: the elements of an array or object follow its node, and the node after a value's subtree is `next` nodes further for an array or object, and the next node otherwise (`document_data::after`). Views step from element to element this way and skip whole subtrees in constant time. -- **Offsets** are 32 bits wide, so a document is limited to 4 GiB (`out_of_range.416`). +- **Offsets** are 32 bits wide, so a document is limited to 4294967279 bytes, 4 GiB minus 16 bytes (a margin below + 2^32 for positions one scanner step past the end of the text; `out_of_range.416`). For example, `#!json {"a": [1, 2.5]}` becomes five nodes. Each node's elements follow it, and `next` leads from an array or object past its subtree: diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index 56c3cf71e..e5f8ed45b 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -1048,12 +1048,12 @@ MessagePack's ext type and BSON's binary subtype are each stored in a single byt [`basic_json_document::parse()`](../api/basic_json_document/parse.md) and the other parsing functions of [`basic_json_document`](../api/basic_json_document/index.md) index a value's position in the source text in 32 bits, -so they do not support an input of 4 GiB or more. +so they do not support an input of 4294967280 bytes (4 GiB minus 16 bytes) or more. !!! failure "Example message" ``` - [json.exception.out_of_range.416] input of 4 GiB or more is not supported by json_document + [json.exception.out_of_range.416] input of 4294967280 bytes or more is not supported by json_document ``` !!! note diff --git a/include/nlohmann/detail/view/errors.hpp b/include/nlohmann/detail/view/errors.hpp index ff5816efd..8d67a3973 100644 --- a/include/nlohmann/detail/view/errors.hpp +++ b/include/nlohmann/detail/view/errors.hpp @@ -55,9 +55,8 @@ template { if (f.code == error_code::input_too_large) { - // LCOV_EXCL_START (4 GiB) - NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4 GiB or more is not supported by json_document", nullptr)); - // LCOV_EXCL_STOP + // (the limit is detail::view::max_input_size: 4 GiB minus 16 bytes) + NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4294967280 bytes or more is not supported by json_document", nullptr)); } const BasicJsonType accepted = BasicJsonType::parse(src, src + size, nullptr, true, ignore_comments, ignore_trailing_commas); // LCOV_EXCL_START (only if parse() accepts what the view rejects: a bug) diff --git a/include/nlohmann/detail/view/node.hpp b/include/nlohmann/detail/view/node.hpp index 5818066ae..b36fc8c2a 100644 --- a/include/nlohmann/detail/view/node.hpp +++ b/include/nlohmann/detail/view/node.hpp @@ -29,6 +29,11 @@ static_assert(static_cast(value_t::null) == 0 && static_cast(value_t::number_unsigned) == 6 && static_cast(value_t::number_float) == 7, "the node format depends on the numbering of value_t"); +/// The largest input a document accepts, in bytes. Offsets and node counts are +/// 32 bits wide; the limit keeps 16 bytes (the width of the scanner's steps) +/// below 2^32, so that a position one step past the end of the text fits. +static constexpr std::size_t max_input_size = 0xFFFFFFEFu; + /// node flags struct node_flags { diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index e2cb57c76..923895792 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -475,9 +475,9 @@ class basic_json_document d.discarded = true; detail::view::parse_failure failure; bool ok = false; - if (NLOHMANN_VIEW_UNLIKELY(size >= 0xFFFFFFF0u)) + if (NLOHMANN_VIEW_UNLIKELY(size > detail::view::max_input_size)) { - failure.code = detail::view::error_code::input_too_large; // LCOV_EXCL_LINE (4 GiB) + failure.code = detail::view::error_code::input_too_large; } else { diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index ffd44ca17..ad00cab14 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -48,6 +48,25 @@ auto materialized_copy(Input&& input) -> decltype(std::declval using parse_call_t = decltype(json_document::parse(std::declval()...)); @@ -427,6 +446,22 @@ TEST_CASE("json_view") CHECK(moved.root().size() == 1); } + SECTION("input size limit") + { + // 32-bit offsets: the limit is 4 GiB minus 16 bytes (a margin below 2^32), + // which is what the exception message and the documentation say + const std::size_t limit = nlohmann::detail::view::max_input_size; + CHECK(limit == std::size_t{4294967279u}); + + const oversized_input input{limit + 1}; + CHECK(!json_document::accept(input)); + CHECK(json_document::parse(input, false).is_discarded()); +#if !defined(JSON_NOEXCEPTION) + json_document d; + CHECK_THROWS_WITH_AS(d = json_document::parse(input), "[json.exception.out_of_range.416] input of 4294967280 bytes or more is not supported by json_document", json::out_of_range&); +#endif + } + SECTION("document lifetime and reuse") { json_document d; From e3dafe1dbacd5f09bd067eaec0de7913167fc5e7 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:30:12 +0200 Subject: [PATCH 14/17] Remove the unused legacy_discarded_value_comparison from abi_config Nothing reads it: the view does not compare values. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/abi_config.hpp | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/include/nlohmann/detail/abi_config.hpp b/include/nlohmann/detail/abi_config.hpp index 0e99c81ad..510a67385 100644 --- a/include/nlohmann/detail/abi_config.hpp +++ b/include/nlohmann/detail/abi_config.hpp @@ -15,10 +15,11 @@ namespace detail { /*! -@brief the configuration macros that change the library's behavior +@brief the configuration macros that json_view.hpp reads json.hpp undefines these macros at its end (see macro_unscope.hpp), so code -that builds on the library after it (json_view.hpp) reads them here. Like the +that builds on the library after it (json_view.hpp) reads them here. A macro +is added when the view starts to depend on it. Like the macros, they are part of the ABI namespace, so they always match the basic_json they are used with. */ @@ -26,8 +27,6 @@ struct abi_config { /// JSON_STRICT_NUL_HANDLING: a null byte is an error, not the end of input static constexpr bool strict_nul_handling = JSON_STRICT_NUL_HANDLING != 0; - /// JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON - static constexpr bool legacy_discarded_value_comparison = JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON != 0; }; } // namespace detail From 0c76b316fcfad377df88d0a0b00d32bc17bb2e34 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:30:24 +0200 Subject: [PATCH 15/17] Align the banner comment of json_view.hpp Signed-off-by: Niels Lohmann --- include/nlohmann/json_view.hpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 923895792..90ac019ea 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -16,7 +16,7 @@ * the read-only part of the basic_json interface; materialize() turns a * * subtree into the basic_json value that parse() would produce. * * * - * The source text must outlive a document that borrows it (lvalue byte * + * The source text must outlive a document that borrows it (lvalue byte * * containers, C strings); rvalue strings, streams, and other inputs are * * owned by the document. * \****************************************************************************/ From aa7f0b02a01c258351d2d57af84b072b42bb7db4 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:32:20 +0200 Subject: [PATCH 16/17] Fuzz json_document with exact-size buffers and parse options The fuzzer only parsed a std::string with default options. Also parse an exact-size byte vector, which has no NUL after its last byte and takes the bounds-checked path, and derive ignore_comments and ignore_trailing_commas from the first input byte, comparing against json::parse and json::accept with the same options. Signed-off-by: Niels Lohmann --- tests/fuzzing.md | 5 +- tests/src/fuzzer-parse_json_view.cpp | 123 +++++++++++++++++++-------- 2 files changed, 93 insertions(+), 35 deletions(-) diff --git a/tests/fuzzing.md b/tests/fuzzing.md index de2d66716..ca305cf48 100644 --- a/tests/fuzzing.md +++ b/tests/fuzzing.md @@ -6,7 +6,10 @@ Each parser of the library (JSON, BJData, BON8, BSON, CBOR, MessagePack, and UBJ Additionally, `parse_json_view_fuzzer` (`tests/src/fuzzer-parse_json_view.cpp`) cross-checks `json_document`/`json_view` (the zero-copy, read-only view declared in `json_view.hpp`) against `basic_json` on the same JSON text: it asserts that `json_document::accept` agrees with `json::accept`, that an accepted input materializes to the same value `json::parse` -produces, and that a rejected input makes both parsers throw with an identical `what()`. It takes plain JSON text, so it +produces, and that a rejected input makes both parsers throw with an identical `what()`. It checks this for a +`std::string` input (borrowed, with a NUL after the last byte) and for an exact-size `std::vector` +(borrowed, with nothing after the last byte), and for the `ignore_comments` and `ignore_trailing_commas` options, which +are taken from the low bits of the first input byte (the byte stays part of the text). It takes plain JSON text, so it reuses the `corpus_json` corpus rather than a format of its own. ## What the fuzzers check diff --git a/tests/src/fuzzer-parse_json_view.cpp b/tests/src/fuzzer-parse_json_view.cpp index 59e32a556..9b9f5a02a 100644 --- a/tests/src/fuzzer-parse_json_view.cpp +++ b/tests/src/fuzzer-parse_json_view.cpp @@ -9,7 +9,9 @@ /* This file implements a parser test suitable for fuzz testing. It checks that json_document (the zero-copy, read-only view of a parsed JSON text declared in -json_view.hpp) agrees with basic_json on every input: +json_view.hpp) agrees with basic_json on every input, for the parse options +selected by the low bits of the first input byte (bit 0: ignore_comments, bit 1: +ignore_trailing_commas; the byte stays part of the text): - json_document::accept(data) must equal json::accept(data) - if the input is accepted, json_document::parse(data).root().materialize() @@ -18,12 +20,20 @@ json_view.hpp) agrees with basic_json on every input: enabled) must throw a json::parse_error or json::out_of_range whose what() is identical to the one json::parse(data) throws +This is checked for two kinds of input: a std::string, which the document +borrows and which ends in the NUL the parser uses as sentinel, and an +exact-size byte vector, which has no NUL after its last byte and takes the +parser's bounds-checked path (AddressSanitizer reports any read past the end). + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ #include +#include +#include #include +#include #include #include @@ -35,65 +45,110 @@ drivers. using json = nlohmann::json; using json_document = nlohmann::json_document; -// see http://llvm.org/docs/LibFuzzer.html -extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) +namespace { - // json_document::accept only has a single-argument overload; wrap the raw - // bytes in a (borrowed) std::string so the same bytes can be handed to it - const std::string input(reinterpret_cast(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) +// what json::parse does with a text: the value, or the message of the exception +struct reference_result +{ + bool accepted = false; + json value{}; + std::string what{}; // NOLINT(readability-redundant-member-init) +}; - const bool accepted_by_json = json::accept(data, data + size); - const bool accepted_by_view = json_document::accept(input); +reference_result parse_reference(const std::uint8_t* data, std::size_t size, bool comments, bool trailing_commas) +{ + reference_result r; + r.accepted = json::accept(data, data + size, comments, trailing_commas); + bool json_threw = false; + try + { + r.value = json::parse(data, data + size, nullptr, true, comments, trailing_commas); + } + catch (const json::parse_error& e) + { + r.what = e.what(); + json_threw = true; + } + catch (const json::out_of_range& e) + { + r.what = e.what(); + json_threw = true; + } + // json::accept and json::parse must agree + assert(json_threw == !r.accepted); + static_cast(json_threw); + return r; +} +// json_document must agree with the reference for this input (a container +// that json_document::parse borrows) +template +void check_input(const Input& input, const reference_result& expected, bool comments, bool trailing_commas) +{ // json_document::accept must agree with json::accept on every input - assert(accepted_by_json == accepted_by_view); + const bool accepted_by_view = json_document::accept(input, comments, trailing_commas); + assert(expected.accepted == accepted_by_view); + static_cast(accepted_by_view); - if (accepted_by_json) + if (expected.accepted) { // both parsers must agree on the resulting value - json const j1 = json::parse(data, data + size); - json_document const doc = json_document::parse(input); + json_document const doc = json_document::parse(input, true, comments, trailing_commas); + assert(!doc.is_discarded()); json const j2 = doc.root().materialize(); - assert(j1 == j2); + assert(expected.value == j2); + static_cast(j2); + + // (without exceptions, the same document) + json_document const quiet = json_document::parse(input, false, comments, trailing_commas); + assert(!quiet.is_discarded()); + assert(quiet.node_count() == doc.node_count()); } else { // both parsers must reject the input the same way when exceptions are used - std::string expected_what; - bool json_threw = false; - try - { - static_cast(json::parse(data, data + size)); - } - catch (const json::parse_error& e) - { - expected_what = e.what(); - json_threw = true; - } - catch (const json::out_of_range& e) - { - expected_what = e.what(); - json_threw = true; - } - assert(json_threw); - bool view_threw = false; try { - static_cast(json_document::parse(input)); + static_cast(json_document::parse(input, true, comments, trailing_commas)); } catch (const json::parse_error& e) { - assert(e.what() == expected_what); + assert(e.what() == expected.what); view_threw = true; } catch (const json::out_of_range& e) { - assert(e.what() == expected_what); + assert(e.what() == expected.what); view_threw = true; } assert(view_threw); + static_cast(view_threw); + + // and without exceptions, the document is discarded + json_document const quiet = json_document::parse(input, false, comments, trailing_commas); + assert(quiet.is_discarded()); + static_cast(quiet); } +} +} // namespace + +// see http://llvm.org/docs/LibFuzzer.html +extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) +{ + // the parse options are taken from the low bits of the first byte + const bool comments = size > 0 && (data[0] & 1U) != 0; + const bool trailing_commas = size > 0 && (data[0] & 2U) != 0; + + const reference_result expected = parse_reference(data, size, comments, trailing_commas); + + // a std::string: borrowed, with the NUL of std::string as sentinel + const std::string input(reinterpret_cast(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + check_input(input, expected, comments, trailing_commas); + + // an exact-size byte vector: borrowed, with nothing after its last byte + const std::vector exact(data, data + size); + check_input(exact, expected, comments, trailing_commas); // return 0 - non-zero return values are reserved for future use return 0; From 417c187d1a1a6faad48b199136bbc49098a06487 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:35:53 +0200 Subject: [PATCH 17/17] Regenerate the amalgamated headers Signed-off-by: Niels Lohmann --- single_include/nlohmann/json.hpp | 7 +- single_include/nlohmann/json_view.hpp | 117 ++++++++++++++++++++------ 2 files changed, 96 insertions(+), 28 deletions(-) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 079a8f6ed..e4fd9eeb8 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7511,10 +7511,11 @@ namespace detail { /*! -@brief the configuration macros that change the library's behavior +@brief the configuration macros that json_view.hpp reads json.hpp undefines these macros at its end (see macro_unscope.hpp), so code -that builds on the library after it (json_view.hpp) reads them here. Like the +that builds on the library after it (json_view.hpp) reads them here. A macro +is added when the view starts to depend on it. Like the macros, they are part of the ABI namespace, so they always match the basic_json they are used with. */ @@ -7522,8 +7523,6 @@ struct abi_config { /// JSON_STRICT_NUL_HANDLING: a null byte is an error, not the end of input static constexpr bool strict_nul_handling = JSON_STRICT_NUL_HANDLING != 0; - /// JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON - static constexpr bool legacy_discarded_value_comparison = JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON != 0; }; } // namespace detail diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 6e2acab56..3abcec76d 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -16,7 +16,7 @@ * the read-only part of the basic_json interface; materialize() turns a * * subtree into the basic_json value that parse() would produce. * * * - * The source text must outlive a document that borrows it (lvalue byte * + * The source text must outlive a document that borrows it (lvalue byte * * containers, C strings); rvalue strings, streams, and other inputs are * * owned by the document. * \****************************************************************************/ @@ -51,7 +51,7 @@ -#include // find, find_if, max +#include // find, find_if, max, min #include // array #include // size_t, ptrdiff_t #include // int64_t, uint8_t, uint16_t, uint32_t, uint64_t @@ -75,7 +75,8 @@ #include // array #include // size_t #include // memcpy -#include // operator new, placement new +#include // numeric_limits +#include // bad_alloc, operator new, placement new #include // string // #include @@ -185,6 +186,11 @@ static_assert(static_cast(value_t::null) == 0 && static_cast(value_t::number_unsigned) == 6 && static_cast(value_t::number_float) == 7, "the node format depends on the numbering of value_t"); +/// The largest input a document accepts, in bytes. Offsets and node counts are +/// 32 bits wide; the limit keeps 16 bytes (the width of the scanner's steps) +/// below 2^32, so that a position one step past the end of the text fits. +static constexpr std::size_t max_input_size = 0xFFFFFFEFu; + /// node flags struct node_flags { @@ -213,17 +219,27 @@ NLOHMANN_VIEW_ALWAYS_INLINE bool is_container(const node& n) noexcept return static_cast(n.kind) - 1u <= 1u; } -/// the converted value of an integer node (stored in len/next) +/// the converted value of an integer node: len is its low half, next its high +/// half (on little-endian targets the two words are the value in memory) NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept { +#if NLOHMANN_VIEW_LITTLE_ENDIAN std::uint64_t v = 0; std::memcpy(&v, reinterpret_cast(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) return v; +#else + return static_cast(n.len) | (static_cast(n.next) << 32); +#endif } NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept { +#if NLOHMANN_VIEW_LITTLE_ENDIAN std::memcpy(reinterpret_cast(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) +#else + n.len = static_cast(v); + n.next = static_cast(v >> 32); +#endif } /// token length of a number node @@ -319,13 +335,30 @@ struct document_data tape_cap = inline_cap; } - /// make room for n nodes; keeps the first tape_size nodes + /// the largest node count whose size in bytes fits a std::size_t + static constexpr std::size_t max_nodes() noexcept + { + return (std::numeric_limits::max)() / sizeof(node); + } + + [[noreturn]] NLOHMANN_VIEW_NOINLINE static void throw_bad_alloc() + { + NLOHMANN_VIEW_THROW(std::bad_alloc()); + } + + /// make room for n nodes; keeps the first tape_size nodes (throws + /// std::bad_alloc for a count that does not fit the address space, + /// instead of wrapping around in n * sizeof(node)) void reserve(std::size_t n) { if (n <= tape_cap) { return; } + if (NLOHMANN_VIEW_UNLIKELY(n > max_nodes())) + { + throw_bad_alloc(); + } node* fresh = static_cast(::operator new (n * sizeof(node))); if (tape_size != 0) { @@ -732,8 +765,18 @@ class builder const std::uint64_t done = static_cast(at - b) + 1; const std::uint64_t guess = static_cast(n) * static_cast(e - b + 1) / done; const std::uint64_t grown = guess + (guess / 4) + 64; // a variable: GCC calls a cast of the sum useless where std::uint64_t is std::size_t + // (n is below 2^32: the input is smaller than 4 GiB; the sum cannot wrap) + const std::uint64_t wanted = (std::max)(grown, static_cast(n) + (n / 2) + 64); + const std::uint64_t limit = document_data::max_nodes(); doc.tape_size = n; - doc.reserve((std::max)(static_cast(grown), n + (n / 2) + 64)); + // LCOV_EXCL_START (a node array that fills the address space) + if (NLOHMANN_VIEW_UNLIKELY(n >= limit)) + { + document_data::throw_bad_alloc(); // no room for another node + } + // LCOV_EXCL_STOP + // (a count beyond the limit is cut: the index does not grow beyond what can be addressed) + doc.reserve(static_cast((std::min)(wanted, limit))); return doc.tape; } @@ -1338,7 +1381,10 @@ indent_done: n->flags = flags; n->extra = extra; n->off = static_cast(off); - set_integer_bits(*n, second); + // len is the low half of the second word, next the high half + // (not a native word over both, which swaps them on big-endian) + n->len = static_cast(second); + n->next = static_cast(second >> 32); #endif return n; } @@ -1633,9 +1679,8 @@ template { if (f.code == error_code::input_too_large) { - // LCOV_EXCL_START (4 GiB) - NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4 GiB or more is not supported by json_document", nullptr)); - // LCOV_EXCL_STOP + // (the limit is detail::view::max_input_size: 4 GiB minus 16 bytes) + NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4294967280 bytes or more is not supported by json_document", nullptr)); } const BasicJsonType accepted = BasicJsonType::parse(src, src + size, nullptr, true, ignore_comments, ignore_trailing_commas); // LCOV_EXCL_START (only if parse() accepts what the view rejects: a bug) @@ -1674,7 +1719,7 @@ NLOHMANN_JSON_NAMESPACE_END #include // basic_string, char_traits, string -#include // decay, integral_constant, is_array, is_lvalue_reference, is_pointer, is_same, remove_reference +#include // decay, integral_constant, is_array, is_const, is_integral, is_lvalue_reference, is_pointer, is_same, remove_reference #include // forward // #include @@ -1694,12 +1739,12 @@ namespace view /// how a document takes its input enum class input_kind { - move_string, ///< rvalue std::string: owned without a copy + move_string, ///< non-const rvalue std::string: owned without a copy c_string, ///< const char* (NUL-terminated): borrowed char_array, ///< char array (e.g. a string literal): borrowed borrow_range, ///< lvalue contiguous byte container, or std::string_view: borrowed - copy_range, ///< rvalue contiguous byte container: copied - adapter, ///< anything else parse() accepts (streams, wide strings, ...): read into a buffer + copy_range, ///< rvalue contiguous byte container (a const rvalue std::string too): copied + adapter, ///< streams, wide strings, and the rest of what the library's input adapter reads: read into a buffer }; template @@ -1718,13 +1763,20 @@ struct classify_input static constexpr input_kind value = std::is_array::value ? input_kind::char_array : std::is_pointer::value ? input_kind::c_string - : (is_rvalue && std::is_same::value) ? input_kind::move_string + : (is_rvalue && !std::is_const::value && std::is_same::value) ? input_kind::move_string : (is_bytes && (!is_rvalue || is_string_view)) ? input_kind::borrow_range : is_bytes ? input_kind::copy_range : input_kind::adapter; // NOLINTEND(readability-avoid-nested-conditional-operator) }; +/// an integer type other than bool: a length passed where a flag is expected +template +struct is_integer_not_bool : std::is_integral {}; + +template<> +struct is_integer_not_bool : std::false_type {}; + /// std::basic_string guarantees a NUL at data()[size()] (the parser's sentinel) template struct is_std_string : std::false_type {}; @@ -2188,12 +2240,6 @@ class basic_json_view return type() == value_t::discarded; } - /// false for discarded views - explicit operator bool() const noexcept - { - return m_node != nullptr; - } - ////////////// // capacity // ////////////// @@ -2323,6 +2369,13 @@ class basic_json_document return d; } + /// parse(ptr, len) does not compile: len would convert to allow_exceptions + /// and ptr be read as a C string (as for the overloads of parse_copy, + /// accept, and read below) + template + static typename std::enable_if::value, basic_json_document>::type + parse(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// parse [first, last) template::iterator_category>::value, int>::type = 0> @@ -2350,6 +2403,10 @@ class basic_json_document return d; } + template + static typename std::enable_if::value, basic_json_document>::type + parse_copy(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// check whether the input is valid JSON (the result of basic_json::accept) template static bool accept(InputType&& input, const bool ignore_comments = false, const bool ignore_trailing_commas = false) @@ -2359,6 +2416,10 @@ class basic_json_document return !d.is_discarded(); } + template + static typename std::enable_if::value, bool>::type + accept(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// parse into this document, reusing its memory template // flawfinder: ignore (a member function, not POSIX read()) @@ -2371,12 +2432,17 @@ class basic_json_document std::integral_constant::value> {}); } + template + // flawfinder: ignore (a member function, not POSIX read()) + typename std::enable_if::value, void>::type + read(InputType&& input, IntegerType value, Flags&&... flags) = delete; + //////////// // access // //////////// /// the root value (discarded if parsing failed without exceptions) - view_type root() const noexcept + view_type root() const& noexcept { if (!m_data || m_data->discarded) { @@ -2385,6 +2451,9 @@ class basic_json_document return view_type(m_data.get(), m_data->tape); } + /// deleted: the view of a temporary document would dangle + view_type root() const&& = delete; + bool is_discarded() const noexcept { return !m_data || m_data->discarded; @@ -2490,9 +2559,9 @@ class basic_json_document d.discarded = true; detail::view::parse_failure failure; bool ok = false; - if (NLOHMANN_VIEW_UNLIKELY(size >= 0xFFFFFFF0u)) + if (NLOHMANN_VIEW_UNLIKELY(size > detail::view::max_input_size)) { - failure.code = detail::view::error_code::input_too_large; // LCOV_EXCL_LINE (4 GiB) + failure.code = detail::view::error_code::input_too_large; } else {