From 0e729be261c106a137e1a2c6e9b88938e3809cd1 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:01:48 +0200 Subject: [PATCH 01/26] Load string scan words with memcpy and use MSVC bit-scan intrinsics Signed-off-by: Niels Lohmann --- include/nlohmann/detail/bit_ops.hpp | 23 +++++++++++++++++++---- tests/src/unit-class_lexer.cpp | 7 +++++++ 2 files changed, 26 insertions(+), 4 deletions(-) diff --git a/include/nlohmann/detail/bit_ops.hpp b/include/nlohmann/detail/bit_ops.hpp index c570814d9..b3bf08557 100644 --- a/include/nlohmann/detail/bit_ops.hpp +++ b/include/nlohmann/detail/bit_ops.hpp @@ -9,8 +9,9 @@ #pragma once #include // uint64_t -#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) - #include // __umulh, _umul128 +#include // memcpy +#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) && (!defined(__SIZEOF_INT128__) || (!defined(__GNUC__) && !defined(__clang__))) + #include // __umulh, _umul128, _BitScanForward64, _BitScanReverse64 #endif #include @@ -28,6 +29,10 @@ inline int count_leading_zeros(std::uint64_t x) noexcept { #if defined(__GNUC__) || defined(__clang__) return __builtin_clzll(x); +#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + unsigned long index = 0; + _BitScanReverse64(&index, x); + return 63 - static_cast(index); #else int n = 0; for (int shift = 32; shift != 0; shift >>= 1) @@ -47,6 +52,10 @@ inline int count_trailing_zeros(std::uint64_t x) noexcept { #if defined(__GNUC__) || defined(__clang__) return __builtin_ctzll(x); +#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + unsigned long index = 0; + _BitScanForward64(&index, x); + return static_cast(index); #else int n = 0; for (int shift = 32; shift != 0; shift >>= 1) @@ -94,14 +103,20 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc #endif } -/// eight bytes as a little-endian word (compilers fold this into one load on -/// little-endian targets) +/// eight bytes as a little-endian word (a single load on little-endian targets) inline std::uint64_t read_eight_bytes(const unsigned char* b) noexcept { +#if defined(_MSC_VER) || defined(__x86_64__) || defined(__i386__) || (defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__) + // the byte order already matches (all MSVC targets are little-endian) + std::uint64_t result = 0; + std::memcpy(&result, b, sizeof(result)); + return result; +#else return static_cast(b[0]) | (static_cast(b[1]) << 8u) | (static_cast(b[2]) << 16u) | (static_cast(b[3]) << 24u) | (static_cast(b[4]) << 32u) | (static_cast(b[5]) << 40u) | (static_cast(b[6]) << 48u) | (static_cast(b[7]) << 56u); +#endif } /// eight bytes as a little-endian word diff --git a/tests/src/unit-class_lexer.cpp b/tests/src/unit-class_lexer.cpp index 82690a669..e5f82b53c 100644 --- a/tests/src/unit-class_lexer.cpp +++ b/tests/src/unit-class_lexer.cpp @@ -1792,4 +1792,11 @@ TEST_CASE("string scanning kernels") CHECK(nlohmann::detail::count_trailing_zeros(bit) == k); CHECK(nlohmann::detail::count_trailing_zeros(bit | (bit << 1u) | 0x8000000000000000u) == k); } + + // eight bytes as a little-endian word, at any alignment + const unsigned char bytes[16] = {0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF}; + CHECK(nlohmann::detail::read_eight_bytes(bytes) == 0x0807060504030201u); + CHECK(nlohmann::detail::read_eight_bytes(bytes + 1) == 0x0908070605040302u); + CHECK(nlohmann::detail::read_eight_bytes(bytes + 8) == 0xFF0F0E0D0C0B0A09u); + CHECK(nlohmann::detail::read_eight_bytes(reinterpret_cast(bytes) + 3) == 0x0B0A090807060504u); } From 653edd738a976f435c5ad72e35320f2eeb73e1a6 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:01:49 +0200 Subject: [PATCH 02/26] Reword float conversion notes for the string type and number_float_t docs Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/number_float_t.md | 8 ++++---- .../docs/features/types/template_parameters.md | 13 +++++++++---- 2 files changed, 13 insertions(+), 8 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/number_float_t.md b/docs/mkdocs/docs/api/basic_json/number_float_t.md index aa41f33f0..1cde41960 100644 --- a/docs/mkdocs/docs/api/basic_json/number_float_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_float_t.md @@ -23,10 +23,10 @@ type to use. ## Template parameters `NumberFloatType` -: the type to store floating-point numbers. The parser converts `#!cpp float`, `#!cpp double`, and a - `#!cpp long double` that is IEEE 754 binary64 itself and other `#!cpp long double` formats with - `#!cpp std::from_chars` or `#!cpp std::strtold`, and serialization falls back to `#!cpp std::snprintf`, so the - type must be `#!cpp float`, `#!cpp double`, or `#!cpp long double`. The +: the type to store floating-point numbers. The type must be `#!cpp float`, `#!cpp double`, or + `#!cpp long double`. The parser converts `#!cpp float`, `#!cpp double`, and a `#!cpp long double` that is IEEE 754 + binary64 itself. It converts other `#!cpp long double` formats with `#!cpp std::from_chars` where available, or + with `#!cpp std::strtold` otherwise. Serialization falls back to `#!cpp std::snprintf`. The [binary formats](../../features/binary_formats/index.md) additionally require `#!cpp float` or `#!cpp double`, because they have no encoding for `#!cpp long double`. See [Template Parameter Requirements](../../features/types/template_parameters.md#numberfloattype). diff --git a/docs/mkdocs/docs/features/types/template_parameters.md b/docs/mkdocs/docs/features/types/template_parameters.md index 3a24d74d8..b68fed8ba 100644 --- a/docs/mkdocs/docs/features/types/template_parameters.md +++ b/docs/mkdocs/docs/features/types/template_parameters.md @@ -353,16 +353,21 @@ using array_t = ArrayType>; ### Always required - A member type `value_type` that is one byte wide and `char`-compatible. The library stores and processes UTF-8 - encoded `char` data and passes `data()` to functions that take a `#!cpp const char*`, such as `#!cpp std::strtod`. + encoded `char` data and passes `data()` to functions that take a `#!cpp const char*`, such as `#!cpp std::strtold` + (only used to parse a `#!cpp long double` that is not IEEE 754 binary64, see + [`NumberFloatType`](#numberfloattype)). `#!cpp std::wstring`, `#!cpp std::u16string`, and `#!cpp std::u32string` are **not** valid choices; see the FAQ on [wide string handling](../../home/faq.md#wide-string-handling). - Constructors: default, copy, move, from `#!cpp const char*` (which must not be `#!cpp explicit`), from `#!cpp (const char*, size_type)`, and from `#!cpp (size_type, char)`; and copy or move assignment. - Member functions `size()`, `clear()`, `resize(n, c)`, `data()`, `push_back(char)`, and `operator[]` (const and non-const, returning references). `c_str()` and `back()` are **not** required. -- `data()` must return a pointer to a contiguous, **null-terminated** buffer -- the parser may hand it to - `#!cpp std::strtod`, which reads up to the null character. A type whose `data()` is not null-terminated does not - fail to compile; it can silently misparse floating-point numbers. +- `data()` must return a pointer to a contiguous, **null-terminated** buffer. `#!cpp float`, `#!cpp double`, and a + `#!cpp long double` that is IEEE 754 binary64 are converted by the library itself and do not depend on this. For any + other `NumberFloatType` (a `#!cpp long double` of another format), the parser falls back to `#!cpp std::strtold` when + `#!cpp std::from_chars` is not available or declines the token, and `std::strtold` reads up to the null character. A type whose `data()` + is not null-terminated does not fail to compile; with such a `NumberFloatType` it can silently misparse + floating-point numbers. - `append(const char*, size_type)`, used by [`dump`](../../api/basic_json/dump.md), and `append(const StringType&)`, used by the CBOR reader for indefinite-length strings. The library's internal string concatenation additionally has to append a `#!cpp char` and a `#!cpp const char*`; for each it selects between `append(arg)`, `#!cpp operator+=`, From e1a85099ef1bb34cfe85be5a2bd0362516e115d4 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:01:50 +0200 Subject: [PATCH 03/26] Regenerate the amalgamated headers Signed-off-by: Niels Lohmann --- single_include/nlohmann/json.hpp | 23 +++++++++++++++++++---- 1 file changed, 19 insertions(+), 4 deletions(-) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 93ffdd25b..7ba4bc8b8 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -8793,8 +8793,9 @@ NLOHMANN_JSON_NAMESPACE_END #include // uint64_t -#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) - #include // __umulh, _umul128 +#include // memcpy +#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) && (!defined(__SIZEOF_INT128__) || (!defined(__GNUC__) && !defined(__clang__))) + #include // __umulh, _umul128, _BitScanForward64, _BitScanReverse64 #endif // #include @@ -8813,6 +8814,10 @@ inline int count_leading_zeros(std::uint64_t x) noexcept { #if defined(__GNUC__) || defined(__clang__) return __builtin_clzll(x); +#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + unsigned long index = 0; + _BitScanReverse64(&index, x); + return 63 - static_cast(index); #else int n = 0; for (int shift = 32; shift != 0; shift >>= 1) @@ -8832,6 +8837,10 @@ inline int count_trailing_zeros(std::uint64_t x) noexcept { #if defined(__GNUC__) || defined(__clang__) return __builtin_ctzll(x); +#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + unsigned long index = 0; + _BitScanForward64(&index, x); + return static_cast(index); #else int n = 0; for (int shift = 32; shift != 0; shift >>= 1) @@ -8879,14 +8888,20 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc #endif } -/// eight bytes as a little-endian word (compilers fold this into one load on -/// little-endian targets) +/// eight bytes as a little-endian word (a single load on little-endian targets) inline std::uint64_t read_eight_bytes(const unsigned char* b) noexcept { +#if defined(_MSC_VER) || defined(__x86_64__) || defined(__i386__) || (defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__) + // the byte order already matches (all MSVC targets are little-endian) + std::uint64_t result = 0; + std::memcpy(&result, b, sizeof(result)); + return result; +#else return static_cast(b[0]) | (static_cast(b[1]) << 8u) | (static_cast(b[2]) << 16u) | (static_cast(b[3]) << 24u) | (static_cast(b[4]) << 32u) | (static_cast(b[5]) << 40u) | (static_cast(b[6]) << 48u) | (static_cast(b[7]) << 56u); +#endif } /// eight bytes as a little-endian word From 86feea50abbc4d8f61339348f13a3c6129256485 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:04:13 +0200 Subject: [PATCH 04/26] Select the Zmij conversion by a trait, not by the double overload The non-template overloads for double asserted binary64 doubles wherever json.hpp was included, so the library no longer compiled where double is not IEEE 754 binary64 (AVR, -fshort-double). A trait now picks Zmij for any binary64 type, including a long double of that format (MSVC, Apple Arm), and Grisu2 for the others. Remove the unused write_short_decimal, powers_of_ten_16, zmij::decimal, zmij::to_decimal, and shortest_digits(double). Signed-off-by: Niels Lohmann --- .../nlohmann/detail/conversions/to_chars.hpp | 172 +++++------------- include/nlohmann/detail/conversions/zmij.hpp | 19 -- tests/src/unit-to_chars.cpp | 77 +++++++- 3 files changed, 120 insertions(+), 148 deletions(-) diff --git a/include/nlohmann/detail/conversions/to_chars.hpp b/include/nlohmann/detail/conversions/to_chars.hpp index 0e16152c3..4324e40a5 100644 --- a/include/nlohmann/detail/conversions/to_chars.hpp +++ b/include/nlohmann/detail/conversions/to_chars.hpp @@ -939,88 +939,6 @@ void grisu2(char* buf, int& len, int& decimal_exponent, FloatType value) grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus); } -/*! -@brief the shortest digits of a positive finite float (other than double): Grisu2 -*/ -template -JSON_HEDLEY_NON_NULL(1) -void shortest_digits(char* buf, int& len, int& decimal_exponent, FloatType value) -{ - grisu2(buf, len, decimal_exponent, value); -} - -/*! -@brief the shortest digits of a positive finite double: the conversion of -Zmij (see zmij.hpp), which always finds the shortest digits that read back as -the same value (Grisu2 does not for about one double in a thousand), and the -closest of them if there are several - -v = buf * 10^decimal_exponent, as for grisu2() -*/ -JSON_HEDLEY_NON_NULL(1) -inline void shortest_digits(char* buf, int& len, int& decimal_exponent, double value) -{ - static_assert(std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53, - "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); - JSON_ASSERT(std::isfinite(value)); - JSON_ASSERT(value > 0); - - std::uint64_t bits = 0; - std::memcpy(&bits, &value, sizeof(bits)); - zmij::decimal d = zmij::to_decimal(bits); - // without trailing zeros (up to 16): 8, 4, 2, 1 at a time - while (d.significand % 100000000 == 0) - { - d.significand /= 100000000; - d.exponent += 8; - } - if (d.significand % 10000 == 0) - { - d.significand /= 10000; - d.exponent += 4; - } - if (d.significand % 100 == 0) - { - d.significand /= 100; - d.exponent += 2; - } - if (d.significand % 10 == 0) - { - d.significand /= 10; - d.exponent += 1; - } - // at most 17 digits, written from the back two at a time - static constexpr const char* pairs = - "00010203040506070809101112131415161718192021222324252627282930313233343536373839" - "40414243444546474849505152535455565758596061626364656667686970717273747576777879" - "8081828384858687888990919293949596979899"; - std::array digits{}; - std::size_t n = digits.size(); - while (d.significand >= 100) - { - const std::uint64_t two_digits = d.significand % 100; // a variable: GCC calls a cast of the remainder useless where std::uint64_t is std::size_t - const auto i = static_cast(two_digits) * 2; - d.significand /= 100; - n -= 2; - digits[n] = pairs[i]; - digits[n + 1] = pairs[i + 1]; - } - if (d.significand >= 10) - { - const auto i = static_cast(d.significand) * 2; - n -= 2; - digits[n] = pairs[i]; - digits[n + 1] = pairs[i + 1]; - } - else - { - digits[--n] = static_cast('0' + d.significand); - } - len = static_cast(digits.size() - n); - std::memcpy(buf, digits.data() + n, static_cast(len)); - decimal_exponent = d.exponent; -} - /*! @brief appends a decimal representation of e to buf @return a pointer to the element following the exponent. @@ -1423,53 +1341,30 @@ inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcep return end + (three ? 5 : 4); } -/// the powers of ten up to 10^16 -inline const std::array& powers_of_ten_16() noexcept -{ - static const std::array powers = - { - { - 1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, 10000000000u, - 100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, 10000000000000000u - } - }; - return powers; -} - /*! -@brief digits * 10^exp, as write_decimal() writes it, for the digits of a -double that need no conversion (count digits, at most 15, the first not 0; -trailing zeros allowed): extended to 16 digits and written by write_shortest() +@brief whether FloatType is an IEEE 754 binary64 type (a double, or a long double +that has the same format, as with MSVC and on Apple's Arm CPUs) -@return a pointer past the text; up to 41 bytes at @a first are written - (some beyond the returned end) +These are the types the conversion of Zmij (see zmij.hpp) is used for; all +others (binary32, or a format the library does not know) use Grisu2. */ -JSON_HEDLEY_NON_NULL(1) -JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_short_decimal(char* first, std::uint64_t digits, int count, int exp) noexcept +template +constexpr bool has_binary64_format() noexcept { - JSON_ASSERT(digits >= powers_of_ten_16()[static_cast(count - 1)] && count <= 15); - const int scale = 16 - count; - return write_shortest(first, zmij::shortest_decimal{digits * powers_of_ten_16()[static_cast(scale)], exp - scale - 1, 0, false}); + return std::numeric_limits::is_iec559 + && std::numeric_limits::digits == 53 + && std::numeric_limits::max_exponent == 1024 + && sizeof(FloatType) == sizeof(std::uint64_t); } -/// as write_short_decimal(), counting the digits (not 0, less than 10^15) -JSON_HEDLEY_NON_NULL(1) -JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noexcept -{ - JSON_ASSERT(digits != 0 && digits < 1000000000000000u); - // floor(log10(2^bits)) + 1 digits, or one less - const int log2_bound = ((64 - count_leading_zeros(digits)) * 1233) >> 12; - const int count = log2_bound + (digits >= powers_of_ten_16()[static_cast(log2_bound)] ? 1 : 0); - return write_short_decimal(first, digits, count, exp); -} +template +struct is_binary64 : std::integral_constant()> {}; -/// a positive finite float (other than double): Grisu2 and format_buffer() +/// a positive finite float (other than binary64): Grisu2 and format_buffer() template JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL -char* write_positive(char* first, const char* last, FloatType value) +char* write_positive_grisu2(char* first, const char* last, FloatType value) { JSON_ASSERT(last - first >= std::numeric_limits::max_digits10); static_cast(last); // (only used in the assertion) @@ -1480,7 +1375,7 @@ char* write_positive(char* first, const char* last, FloatType value) // len is the length of the buffer, i.e., the number of decimal digits. int len = 0; int decimal_exponent = 0; - shortest_digits(first, len, decimal_exponent, value); + grisu2(first, len, decimal_exponent, value); JSON_ASSERT(len <= std::numeric_limits::max_digits10); @@ -1496,15 +1391,16 @@ char* write_positive(char* first, const char* last, FloatType value) return format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp); } -/// a positive finite double: the shortest digits (Zmij), laid out by +/// a positive finite binary64 number: the shortest digits (Zmij), laid out by /// write_shortest() (through a local buffer if [first, last) is shorter than /// the 41 bytes it may write) +template JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_positive(char* first, const char* last, double value) +char* write_positive_zmij(char* first, const char* last, FloatType value) { - static_assert(std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53, - "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); + static_assert(is_binary64::value, + "internal error: the conversion of Zmij needs IEEE 754 binary64 numbers"); std::uint64_t bits = 0; std::memcpy(&bits, &value, sizeof(bits)); const zmij::shortest_decimal d = zmij::to_shortest(bits); @@ -1519,6 +1415,34 @@ inline char* write_positive(char* first, const char* last, double value) return first + len; } +/// a positive finite binary64 number: Zmij (as a long double has the format of +/// a double here, its bits are those of the double of the same value) +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value, std::true_type /*is_binary64*/) +{ + return write_positive_zmij(first, last, value); +} + +/// a positive finite float of any other format: Grisu2 +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value, std::false_type /*is_binary64*/) +{ + return write_positive_grisu2(first, last, value); +} + +/// a positive finite float: Zmij for binary64 numbers, Grisu2 otherwise +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value) +{ + return write_positive(first, last, value, is_binary64 {}); +} + } // namespace dtoa_impl /*! diff --git a/include/nlohmann/detail/conversions/zmij.hpp b/include/nlohmann/detail/conversions/zmij.hpp index ffff8380e..40e477f7b 100644 --- a/include/nlohmann/detail/conversions/zmij.hpp +++ b/include/nlohmann/detail/conversions/zmij.hpp @@ -36,13 +36,6 @@ computed from the compressed tables of Zmij beyond it. namespace zmij { -/// significand * 10^exponent -struct decimal -{ - std::uint64_t significand; - int exponent; -}; - /// the compressed powers of ten of Zmij inline const std::array& pow10_minor() noexcept { @@ -221,18 +214,6 @@ JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexc return shortest_decimal{integral, dec_exp, static_cast(digit), !round_up && !round_down}; } -/// The shortest decimal in the rounding interval of a positive finite double -/// given by its bits, as one number. The significand can end in zeros. -inline decimal to_decimal(std::uint64_t bits) noexcept -{ - const shortest_decimal d = to_shortest(bits); - if (d.has_digit) - { - return decimal{(d.integral * 10) + d.digit, d.exponent}; - } - return decimal{d.integral, d.exponent + 1}; -} - } // namespace zmij } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/tests/src/unit-to_chars.cpp b/tests/src/unit-to_chars.cpp index 8ced43ad0..d8eb44434 100644 --- a/tests/src/unit-to_chars.cpp +++ b/tests/src/unit-to_chars.cpp @@ -15,6 +15,7 @@ #include using nlohmann::detail::dtoa_impl::reinterpret_bits; +#include #include #include #include @@ -666,13 +667,24 @@ void check_shortest(double v) const std::string text(buf.data(), end); CAPTURE(text) CHECK(parse_double(text) == v); - // the layout is that of format_buffer() for the same digits + // the layout is that of format_buffer() for the digits of Zmij + const auto sd = nlohmann::detail::zmij::to_shortest(reinterpret_bits(v)); + const std::uint64_t significand = sd.has_digit ? (sd.integral * 10) + sd.digit : sd.integral; + int exponent = sd.has_digit ? sd.exponent : sd.exponent + 1; + std::string significand_digits = std::to_string(significand); + while (significand_digits.size() > 1 && significand_digits.back() == '0') + { + significand_digits.pop_back(); + ++exponent; + } std::array reference{}; - int len = 0; - int exponent = 0; - nlohmann::detail::dtoa_impl::shortest_digits(reference.data(), len, exponent, v); - const char* const reference_end = nlohmann::detail::dtoa_impl::format_buffer(reference.data(), len, exponent, -4, 15); + std::copy(significand_digits.begin(), significand_digits.end(), reference.begin()); + const char* const reference_end = nlohmann::detail::dtoa_impl::format_buffer(reference.data(), static_cast(significand_digits.size()), exponent, -4, 15); CHECK(text == std::string(reference.data(), static_cast(reference_end - reference.data()))); + // and write_positive() is what to_chars() calls + std::array positive{}; + const char* const positive_end = nlohmann::detail::dtoa_impl::write_positive(positive.data(), positive.data() + positive.size(), v); + CHECK(text == std::string(positive.data(), static_cast(positive_end - positive.data()))); const auto de = digits_and_exponent(text); const std::string& digits = de.first; if (digits.size() > 1) @@ -785,3 +797,58 @@ TEST_CASE("shortest digits of doubles") } } } + +TEST_CASE("choice of the conversion") +{ + using nlohmann::detail::dtoa_impl::is_binary64; + + SECTION("by the format of the type") + { + // Zmij needs binary64 numbers; everything else uses Grisu2 + static_assert(!is_binary64::value, "float is not binary64"); + static_assert(is_binary64::value == (std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53), + "double is binary64 where it is IEEE 754 with 53 digits"); + static_assert(!is_binary64::value, "integers are not binary64"); + static_assert(is_binary64::value == (std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53 && sizeof(long double) == 8), + "long double is binary64 where it has the format of a double"); + CHECK(!is_binary64::value); + CHECK(is_binary64::value); + } + + SECTION("float: Grisu2, double: Zmij") + { + // 5.3165205877497296e+16 is one of the doubles for which Grisu2 does not find the shortest digits + constexpr double value = 5.3165205877497296e+16; + std::array buf{}; + const char* const last = buf.data() + buf.size(); + + char* end = nlohmann::detail::dtoa_impl::write_positive(buf.data(), last, value); + CHECK(std::string(buf.data(), end) == "5.31652058774973e+16"); + end = nlohmann::detail::dtoa_impl::write_positive_grisu2(buf.data(), last, value); + CHECK(std::string(buf.data(), end) == "5.3165205877497296e+16"); + + constexpr float f = 1.1754944e-38f; + end = nlohmann::detail::dtoa_impl::write_positive(buf.data(), last, f); + const std::string dispatched(buf.data(), end); + end = nlohmann::detail::dtoa_impl::write_positive_grisu2(buf.data(), last, f); + CHECK(dispatched == std::string(buf.data(), end)); + } + + SECTION("long double with the format of a double: Zmij") + { + // (on platforms where long double is wider, Grisu2 does not apply either: the snprintf fallback does) + if (std::numeric_limits::digits == 53 && std::numeric_limits::is_iec559) + { + using long_double_json = nlohmann::json::with_float_t; + for (const double d : + { + 5.3165205877497296e+16, 1.0, 0.1, 123456.789, 2.2250738585072014e-308, 1.7976931348623157e+308, -5.3165205877497296e+16 + }) + { + CAPTURE(d) + CHECK(long_double_json(static_cast(d)).dump() == nlohmann::json(d).dump()); + } + CHECK(long_double_json(5.3165205877497296e+16L).dump() == "5.31652058774973e+16"); + } + } +} From 35b28d4918dae617690a0b95197f2efabb1a1d64 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:04:14 +0200 Subject: [PATCH 05/26] Credit Zmij's authors in to_chars.hpp, the README, and the license page Signed-off-by: Niels Lohmann --- README.md | 2 +- docs/mkdocs/docs/home/license.md | 2 +- include/nlohmann/detail/conversions/to_chars.hpp | 1 + 3 files changed, 3 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 9c641284d..f6392d057 100644 --- a/README.md +++ b/README.md @@ -1394,7 +1394,7 @@ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR I - The class contains the UTF-8 Decoder from Bjoern Hoehrmann which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2008-2009 [Björn Hoehrmann](https://bjoern.hoehrmann.de/) - The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2009 [Florian Loitsch](https://florian.loitsch.com/) -- The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut) +- The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, including the conversion of the digits to text by Xiang JunBo and the SIMD instruction sequence of Dougall Johnson, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut) - The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/). - The class contains parts of [Google Abseil](https://github.com/abseil/abseil-cpp) which is licensed under the [Apache 2.0 License](https://opensource.org/licenses/Apache-2.0). - The class contains an adapted version of the Eisel-Lemire algorithm, its table of powers of five, and its digit comparison for long numbers from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors diff --git a/docs/mkdocs/docs/home/license.md b/docs/mkdocs/docs/home/license.md index d3ce12e28..715c6ca51 100644 --- a/docs/mkdocs/docs/home/license.md +++ b/docs/mkdocs/docs/home/license.md @@ -18,7 +18,7 @@ The class contains the UTF-8 Decoder from Bjoern Hoehrmann which is licensed und The class contains a slightly modified version of the Grisu2 algorithm from Florian Loitsch which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2009 [Florian Loitsch](https://florian.loitsch.com/) -The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut) +The class contains a port of the shortest double-to-decimal conversion of [Żmij](https://github.com/vitaut/zmij) by Victor Zverovich, including the conversion of the digits to text by Xiang JunBo and the SIMD instruction sequence of Dougall Johnson, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above). Copyright © 2025 [Victor Zverovich](https://github.com/vitaut) The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/). diff --git a/include/nlohmann/detail/conversions/to_chars.hpp b/include/nlohmann/detail/conversions/to_chars.hpp index 4324e40a5..db50795b5 100644 --- a/include/nlohmann/detail/conversions/to_chars.hpp +++ b/include/nlohmann/detail/conversions/to_chars.hpp @@ -4,6 +4,7 @@ // |_____|_____|_____|_|___| https://github.com/nlohmann/json // // SPDX-FileCopyrightText: 2009 Florian Loitsch +// SPDX-FileCopyrightText: 2025 Victor Zverovich // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT From 3168d591e8ecbcb4ccf67ea0afa802a79125e972 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:07:45 +0200 Subject: [PATCH 06/26] Regenerate the amalgamated headers Signed-off-by: Niels Lohmann --- single_include/nlohmann/json.hpp | 192 ++++++++----------------------- 1 file changed, 49 insertions(+), 143 deletions(-) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 846e21331..f4174a9cf 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -25081,6 +25081,7 @@ NLOHMANN_JSON_NAMESPACE_END // |_____|_____|_____|_|___| https://github.com/nlohmann/json // // SPDX-FileCopyrightText: 2009 Florian Loitsch +// SPDX-FileCopyrightText: 2025 Victor Zverovich // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT @@ -25156,13 +25157,6 @@ computed from the compressed tables of Zmij beyond it. namespace zmij { -/// significand * 10^exponent -struct decimal -{ - std::uint64_t significand; - int exponent; -}; - /// the compressed powers of ten of Zmij inline const std::array& pow10_minor() noexcept { @@ -25341,18 +25335,6 @@ JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexc return shortest_decimal{integral, dec_exp, static_cast(digit), !round_up && !round_down}; } -/// The shortest decimal in the rounding interval of a positive finite double -/// given by its bits, as one number. The significand can end in zeros. -inline decimal to_decimal(std::uint64_t bits) noexcept -{ - const shortest_decimal d = to_shortest(bits); - if (d.has_digit) - { - return decimal{(d.integral * 10) + d.digit, d.exponent}; - } - return decimal{d.integral, d.exponent + 1}; -} - } // namespace zmij } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -26260,88 +26242,6 @@ void grisu2(char* buf, int& len, int& decimal_exponent, FloatType value) grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus); } -/*! -@brief the shortest digits of a positive finite float (other than double): Grisu2 -*/ -template -JSON_HEDLEY_NON_NULL(1) -void shortest_digits(char* buf, int& len, int& decimal_exponent, FloatType value) -{ - grisu2(buf, len, decimal_exponent, value); -} - -/*! -@brief the shortest digits of a positive finite double: the conversion of -Zmij (see zmij.hpp), which always finds the shortest digits that read back as -the same value (Grisu2 does not for about one double in a thousand), and the -closest of them if there are several - -v = buf * 10^decimal_exponent, as for grisu2() -*/ -JSON_HEDLEY_NON_NULL(1) -inline void shortest_digits(char* buf, int& len, int& decimal_exponent, double value) -{ - static_assert(std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53, - "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); - JSON_ASSERT(std::isfinite(value)); - JSON_ASSERT(value > 0); - - std::uint64_t bits = 0; - std::memcpy(&bits, &value, sizeof(bits)); - zmij::decimal d = zmij::to_decimal(bits); - // without trailing zeros (up to 16): 8, 4, 2, 1 at a time - while (d.significand % 100000000 == 0) - { - d.significand /= 100000000; - d.exponent += 8; - } - if (d.significand % 10000 == 0) - { - d.significand /= 10000; - d.exponent += 4; - } - if (d.significand % 100 == 0) - { - d.significand /= 100; - d.exponent += 2; - } - if (d.significand % 10 == 0) - { - d.significand /= 10; - d.exponent += 1; - } - // at most 17 digits, written from the back two at a time - static constexpr const char* pairs = - "00010203040506070809101112131415161718192021222324252627282930313233343536373839" - "40414243444546474849505152535455565758596061626364656667686970717273747576777879" - "8081828384858687888990919293949596979899"; - std::array digits{}; - std::size_t n = digits.size(); - while (d.significand >= 100) - { - const std::uint64_t two_digits = d.significand % 100; // a variable: GCC calls a cast of the remainder useless where std::uint64_t is std::size_t - const auto i = static_cast(two_digits) * 2; - d.significand /= 100; - n -= 2; - digits[n] = pairs[i]; - digits[n + 1] = pairs[i + 1]; - } - if (d.significand >= 10) - { - const auto i = static_cast(d.significand) * 2; - n -= 2; - digits[n] = pairs[i]; - digits[n + 1] = pairs[i + 1]; - } - else - { - digits[--n] = static_cast('0' + d.significand); - } - len = static_cast(digits.size() - n); - std::memcpy(buf, digits.data() + n, static_cast(len)); - decimal_exponent = d.exponent; -} - /*! @brief appends a decimal representation of e to buf @return a pointer to the element following the exponent. @@ -26744,53 +26644,30 @@ inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcep return end + (three ? 5 : 4); } -/// the powers of ten up to 10^16 -inline const std::array& powers_of_ten_16() noexcept -{ - static const std::array powers = - { - { - 1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, 10000000000u, - 100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, 10000000000000000u - } - }; - return powers; -} - /*! -@brief digits * 10^exp, as write_decimal() writes it, for the digits of a -double that need no conversion (count digits, at most 15, the first not 0; -trailing zeros allowed): extended to 16 digits and written by write_shortest() +@brief whether FloatType is an IEEE 754 binary64 type (a double, or a long double +that has the same format, as with MSVC and on Apple's Arm CPUs) -@return a pointer past the text; up to 41 bytes at @a first are written - (some beyond the returned end) +These are the types the conversion of Zmij (see zmij.hpp) is used for; all +others (binary32, or a format the library does not know) use Grisu2. */ -JSON_HEDLEY_NON_NULL(1) -JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_short_decimal(char* first, std::uint64_t digits, int count, int exp) noexcept +template +constexpr bool has_binary64_format() noexcept { - JSON_ASSERT(digits >= powers_of_ten_16()[static_cast(count - 1)] && count <= 15); - const int scale = 16 - count; - return write_shortest(first, zmij::shortest_decimal{digits * powers_of_ten_16()[static_cast(scale)], exp - scale - 1, 0, false}); + return std::numeric_limits::is_iec559 + && std::numeric_limits::digits == 53 + && std::numeric_limits::max_exponent == 1024 + && sizeof(FloatType) == sizeof(std::uint64_t); } -/// as write_short_decimal(), counting the digits (not 0, less than 10^15) -JSON_HEDLEY_NON_NULL(1) -JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noexcept -{ - JSON_ASSERT(digits != 0 && digits < 1000000000000000u); - // floor(log10(2^bits)) + 1 digits, or one less - const int log2_bound = ((64 - count_leading_zeros(digits)) * 1233) >> 12; - const int count = log2_bound + (digits >= powers_of_ten_16()[static_cast(log2_bound)] ? 1 : 0); - return write_short_decimal(first, digits, count, exp); -} +template +struct is_binary64 : std::integral_constant()> {}; -/// a positive finite float (other than double): Grisu2 and format_buffer() +/// a positive finite float (other than binary64): Grisu2 and format_buffer() template JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL -char* write_positive(char* first, const char* last, FloatType value) +char* write_positive_grisu2(char* first, const char* last, FloatType value) { JSON_ASSERT(last - first >= std::numeric_limits::max_digits10); static_cast(last); // (only used in the assertion) @@ -26801,7 +26678,7 @@ char* write_positive(char* first, const char* last, FloatType value) // len is the length of the buffer, i.e., the number of decimal digits. int len = 0; int decimal_exponent = 0; - shortest_digits(first, len, decimal_exponent, value); + grisu2(first, len, decimal_exponent, value); JSON_ASSERT(len <= std::numeric_limits::max_digits10); @@ -26817,15 +26694,16 @@ char* write_positive(char* first, const char* last, FloatType value) return format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp); } -/// a positive finite double: the shortest digits (Zmij), laid out by +/// a positive finite binary64 number: the shortest digits (Zmij), laid out by /// write_shortest() (through a local buffer if [first, last) is shorter than /// the 41 bytes it may write) +template JSON_HEDLEY_NON_NULL(1, 2) JSON_HEDLEY_RETURNS_NON_NULL -inline char* write_positive(char* first, const char* last, double value) +char* write_positive_zmij(char* first, const char* last, FloatType value) { - static_assert(std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53, - "internal error: the conversion of Zmij needs IEEE 754 binary64 doubles"); + static_assert(is_binary64::value, + "internal error: the conversion of Zmij needs IEEE 754 binary64 numbers"); std::uint64_t bits = 0; std::memcpy(&bits, &value, sizeof(bits)); const zmij::shortest_decimal d = zmij::to_shortest(bits); @@ -26840,6 +26718,34 @@ inline char* write_positive(char* first, const char* last, double value) return first + len; } +/// a positive finite binary64 number: Zmij (as a long double has the format of +/// a double here, its bits are those of the double of the same value) +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value, std::true_type /*is_binary64*/) +{ + return write_positive_zmij(first, last, value); +} + +/// a positive finite float of any other format: Grisu2 +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value, std::false_type /*is_binary64*/) +{ + return write_positive_grisu2(first, last, value); +} + +/// a positive finite float: Zmij for binary64 numbers, Grisu2 otherwise +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* write_positive(char* first, const char* last, FloatType value) +{ + return write_positive(first, last, value, is_binary64 {}); +} + } // namespace dtoa_impl /*! From c006db93d4bf3d58e5ddf50b2a62cfeb22062cd7 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:10:11 +0200 Subject: [PATCH 07/26] Reject integer lengths in json_document::parse parse(ptr, len) compiled: len converted to allow_exceptions, and ptr was read as a C string, past the end of a buffer without a terminating NUL. Delete the overloads of parse, parse_copy, accept, and read that take an integer other than bool where the flags are expected. Signed-off-by: Niels Lohmann --- .../docs/api/basic_json_document/accept.md | 5 +++ .../docs/api/basic_json_document/parse.md | 6 +++ .../api/basic_json_document/parse_copy.md | 3 ++ .../docs/api/basic_json_document/read.md | 3 ++ include/nlohmann/detail/view/input.hpp | 7 +++ include/nlohmann/json_view.hpp | 20 +++++++++ tests/src/unit-json_view.cpp | 44 +++++++++++++++++++ 7 files changed, 88 insertions(+) diff --git a/docs/mkdocs/docs/api/basic_json_document/accept.md b/docs/mkdocs/docs/api/basic_json_document/accept.md index 5aa1fc93d..322350130 100644 --- a/docs/mkdocs/docs/api/basic_json_document/accept.md +++ b/docs/mkdocs/docs/api/basic_json_document/accept.md @@ -43,6 +43,11 @@ input's own copy (for inputs that are always read into a buffer) throws. Linear in the length of the input. +## Notes + +An integer argument that is not a `#!cpp bool` where the flags are expected, such as `#!cpp accept(ptr, len)`, does not +compile; see [`parse`](parse.md#notes). + ## Examples ??? example diff --git a/docs/mkdocs/docs/api/basic_json_document/parse.md b/docs/mkdocs/docs/api/basic_json_document/parse.md index a3c44f4e0..7c46e17ea 100644 --- a/docs/mkdocs/docs/api/basic_json_document/parse.md +++ b/docs/mkdocs/docs/api/basic_json_document/parse.md @@ -100,6 +100,12 @@ See [`owns_source`](owns_source.md) to check which happened after a call, and th **Numbers.** As for [`BasicJsonType::parse()`](../basic_json/parse.md), an integer literal too large for the 64-bit integer type becomes a floating-point value. +**No lengths.** An integer argument that is not a `#!cpp bool` where the flags are expected -- for example +`#!cpp parse(ptr, len)` -- does not compile (the overload is deleted). Such a call would convert `len` to +`allow_exceptions` and read `ptr` as a null-terminated string, past the end of a buffer that has none. To parse a +buffer of a given length, pass a pair of pointers: `#!cpp parse(ptr, ptr + len)`. The same holds for +[`parse_copy`](parse_copy.md), [`accept`](accept.md), and [`read`](read.md). + ## Examples ??? example "Example: (1) borrowed vs. owned input, and errors identical to `BasicJsonType::parse()`" diff --git a/docs/mkdocs/docs/api/basic_json_document/parse_copy.md b/docs/mkdocs/docs/api/basic_json_document/parse_copy.md index ed4714d22..7cb51ad02 100644 --- a/docs/mkdocs/docs/api/basic_json_document/parse_copy.md +++ b/docs/mkdocs/docs/api/basic_json_document/parse_copy.md @@ -51,6 +51,9 @@ Linear in the length of the input. only differs in that the input is always copied rather than sometimes borrowed. Prefer [`parse()`](parse.md) when the input's lifetime already covers the document's, since it avoids the copy for borrowed inputs. +An integer argument that is not a `#!cpp bool` where the flags are expected, such as `#!cpp parse_copy(ptr, len)`, does not +compile; see [`parse`](parse.md#notes). + ## Examples ??? example diff --git a/docs/mkdocs/docs/api/basic_json_document/read.md b/docs/mkdocs/docs/api/basic_json_document/read.md index 2267e46bd..ff37014bd 100644 --- a/docs/mkdocs/docs/api/basic_json_document/read.md +++ b/docs/mkdocs/docs/api/basic_json_document/read.md @@ -49,6 +49,9 @@ whether or not the new parse succeeds; take fresh views from [`root()`](root.md) `input` is borrowed or owned by the same rules as [`parse()`](parse.md#notes); a document can borrow on one call and own on the next, since ownership is decided freshly each time. +An integer argument that is not a `#!cpp bool` where the flags are expected, such as `#!cpp read(ptr, len)`, does not +compile; see [`parse`](parse.md#notes). + ## Examples ??? example diff --git a/include/nlohmann/detail/view/input.hpp b/include/nlohmann/detail/view/input.hpp index e8a5e3148..3fafa2ecf 100644 --- a/include/nlohmann/detail/view/input.hpp +++ b/include/nlohmann/detail/view/input.hpp @@ -59,6 +59,13 @@ struct classify_input // NOLINTEND(readability-avoid-nested-conditional-operator) }; +/// an integer type other than bool: a length passed where a flag is expected +template +struct is_integer_not_bool : std::is_integral {}; + +template<> +struct is_integer_not_bool : std::false_type {}; + /// std::basic_string guarantees a NUL at data()[size()] (the parser's sentinel) template struct is_std_string : std::false_type {}; diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 5aec36ef9..178020cbb 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -291,6 +291,13 @@ class basic_json_document return d; } + /// parse(ptr, len) does not compile: len would convert to allow_exceptions + /// and ptr be read as a C string (as for the overloads of parse_copy, + /// accept, and read below) + template + static typename std::enable_if::value, basic_json_document>::type + parse(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// parse [first, last) template::iterator_category>::value, int>::type = 0> @@ -318,6 +325,10 @@ class basic_json_document return d; } + template + static typename std::enable_if::value, basic_json_document>::type + parse_copy(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// check whether the input is valid JSON (the result of basic_json::accept) template static bool accept(InputType&& input, const bool ignore_comments = false, const bool ignore_trailing_commas = false) @@ -327,6 +338,10 @@ class basic_json_document return !d.is_discarded(); } + template + static typename std::enable_if::value, bool>::type + accept(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// parse into this document, reusing its memory template // flawfinder: ignore (a member function, not POSIX read()) @@ -339,6 +354,11 @@ class basic_json_document std::integral_constant::value> {}); } + template + // flawfinder: ignore (a member function, not POSIX read()) + typename std::enable_if::value, void>::type + read(InputType&& input, IntegerType value, Flags&&... flags) = delete; + //////////// // access // //////////// diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index d953d9903..9df459d3a 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -15,12 +15,14 @@ using nlohmann::json_document; using nlohmann::json_view; using nlohmann::ordered_json_document; +#include #include #include #include #include #include #include +#include #include #include @@ -30,6 +32,16 @@ using nlohmann::ordered_json_document; namespace { +// detection of calls that must not compile +template +using parse_call_t = decltype(json_document::parse(std::declval()...)); +template +using parse_copy_call_t = decltype(json_document::parse_copy(std::declval()...)); +template +using accept_call_t = decltype(json_document::accept(std::declval()...)); +template +using read_call_t = decltype(std::declval().read(std::declval()...)); + #if !defined(JSON_NOEXCEPTION) // the exception parse() throws for a text, or "" if it accepts it std::string parse_exception(const std::string& text, bool comments = false, bool trailing_commas = false) @@ -329,6 +341,38 @@ TEST_CASE("json_view") CHECK(json_document::parse("", false).is_discarded()); } + SECTION("integer arguments do not compile") + { + using nlohmann::detail::is_detected; + + // a length is not a flag: parse(ptr, len) would convert len to + // allow_exceptions and read ptr as a C string, which need not end + static_assert(is_detected::value, "parse(ptr, bool) is valid"); + static_assert(is_detected::value, "parse(ptr, bool, bool, bool) is valid"); + static_assert(is_detected::value, "parse(first, last) is valid"); + static_assert(is_detected::value, "parse(first, last, bool) is valid"); + static_assert(!is_detected::value, "parse(ptr, len) must not compile"); + static_assert(!is_detected::value, "parse(ptr, int) must not compile"); + static_assert(!is_detected::value, "parse(ptr, char) must not compile"); + static_assert(!is_detected::value, "parse(ptr, len, bool) must not compile"); + static_assert(!is_detected::value, "parse(string, len) must not compile"); + static_assert(!is_detected&, std::size_t>::value, "parse(vector, len) must not compile"); + + static_assert(is_detected::value, "parse_copy(ptr, bool) is valid"); + static_assert(!is_detected::value, "parse_copy(ptr, len) must not compile"); + + static_assert(is_detected::value, "accept(ptr, bool) is valid"); + static_assert(!is_detected::value, "accept(ptr, len) must not compile"); + + static_assert(is_detected::value, "read(ptr, bool) is valid"); + static_assert(!is_detected::value, "read(ptr, len) must not compile"); + + // the valid calls still work + const char* const text = "[1]"; + CHECK(json_document::parse(text, true).root().is_array()); + CHECK(json_document::accept(text, true, true)); + } + SECTION("document lifetime and reuse") { json_document d; From 2ba331e4d07b694d92c3759d768461d9e6b5a6b3 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:11:14 +0200 Subject: [PATCH 08/26] Size the dump buffer of a small value by its own extent The source extent of a value was read from the next node, falling back to the rest of the document when that node held a decoded string. dump() of a small value could thus allocate a buffer as large as the document. Skip a few such nodes, cap the fallback estimate, and shrink a buffer that is much larger than its output. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/view/serializer.hpp | 8 ++++- include/nlohmann/json_view.hpp | 28 ++++++++++++----- tests/src/unit-json_view.cpp | 34 +++++++++++++++++++++ 3 files changed, 62 insertions(+), 8 deletions(-) diff --git a/include/nlohmann/detail/view/serializer.hpp b/include/nlohmann/detail/view/serializer.hpp index 26bfb3225..7870601bc 100644 --- a/include/nlohmann/detail/view/serializer.hpp +++ b/include/nlohmann/detail/view/serializer.hpp @@ -44,7 +44,13 @@ class output_buffer void finish() { - m_out.resize(static_cast(m_pos - m_out.data())); + const auto size = static_cast(m_pos - m_out.data()); + m_out.resize(size); + // do not keep a buffer that was sized for a much larger output + if (m_out.capacity() > 1024 && m_out.capacity() / 2 > size) + { + m_out.shrink_to_fit(); + } } NLOHMANN_VIEW_ALWAYS_INLINE void reserve(std::size_t n) diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 882101336..f984bdd33 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -24,6 +24,7 @@ #ifndef INCLUDE_NLOHMANN_JSON_VIEW_HPP_ #define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ +#include // min #include // size_t #include // memcpy, strlen #include // distance, input_iterator_tag, iterator_traits @@ -682,20 +683,33 @@ class basic_json_view } /// the number of source bytes of this value (estimated for values with - /// decoded strings) + /// decoded strings); the estimate sizes the output buffer of dump() std::size_t source_extent() const noexcept { - const node* const next = document_data::after(m_node); - const bool in_source = (m_node->flags & detail::view::node_flags::storage) == 0; - if (!in_source) + const node* const end = m_doc->tape + m_doc->tape_size; + if ((m_node->flags & detail::view::node_flags::storage) != 0) { return m_node->len; } - if (next != m_doc->tape + m_doc->tape_size && (next->flags & detail::view::node_flags::storage) == 0 && next->off >= m_node->off) + // the value ends where the next node in the source begins; nodes + // with decoded strings (their offset is in the arena) are skipped, + // but only a few of them, to keep the walk short + const node* next = document_data::after(m_node); + for (int skipped = 0; next != end && skipped < 16; ++skipped, ++next) { - return next->off - m_node->off; + if ((next->flags & detail::view::node_flags::storage) == 0) + { + return next->off >= m_node->off ? next->off - m_node->off : 0; + } } - return m_doc->size - m_node->off; + if (next == end) + { + return m_doc->size - m_node->off; + } + // the end is unknown: assume a few bytes per node, the output buffer + // grows should the value be larger + const auto nodes = static_cast(document_data::after(m_node) - m_node); + return (std::min)(m_doc->size - m_node->off, static_cast(1024) + nodes * 16); } /// the value of the first member with this key, or a discarded view diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 3f35c6440..07ad71dcd 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -1183,6 +1183,40 @@ TEST_CASE("json_view dump") CHECK(json_document::parse(deep).root().dump() == deep); } + SECTION("the output buffer of a small value is small") + { + // an escaped key after the value: its node lies in the arena, so the + // source extent of the value cannot be read from the next node + const std::string big(100000, 'a'); + const std::string text = R"({"small":1,"list":[1,2,3],"k\n":")" + big + R"("})"; + const json_document d = json_document::parse(text); + const auto small = d.root()["small"].dump(); + CHECK(small == "1"); + CHECK(small.capacity() < 4096); + const auto list = d.root()["list"].dump(); + CHECK(list == "[1,2,3]"); + CHECK(list.capacity() < 4096); + CHECK(d.root()["list"].dump(2).capacity() < 4096); + + // the whole document and the large value are unaffected + CHECK(d.root().dump() == ordered_json::parse(text).dump()); + CHECK(d.root()["k\n"].dump() == "\"" + big + "\""); + } + + SECTION("output that outgrows the estimate") + { + // ensure_ascii writes six bytes for each two-byte character + std::string chars; + for (int i = 0; i < 5000; ++i) + { + chars += "\xC3\xA9"; + } + const json_document d = json_document::parse("{\"a\":\"" + chars + R"(","k\n":1})"); + const json expected = json::parse("\"" + chars + "\""); + CHECK(d.root()["a"].dump(-1, ' ', true) == expected.dump(-1, ' ', true)); + CHECK(d.root()["a"].dump(-1, ' ', true).size() == 2 + 5000 * 6); + } + SECTION("streams and discarded views") { const json_document d = json_document::parse(R"({"a": [1, 2]})"); From 1d5f9a596b86b4546470bdf7bd95a4e5ca1c8d8c Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:11:24 +0200 Subject: [PATCH 09/26] Regenerate the amalgamated headers Signed-off-by: Niels Lohmann --- single_include/nlohmann/json_view.hpp | 36 +++++++++++++++++++++------ 1 file changed, 28 insertions(+), 8 deletions(-) diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 6b924453e..8f0725791 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -24,6 +24,7 @@ #ifndef INCLUDE_NLOHMANN_JSON_VIEW_HPP_ #define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ +#include // min #include // size_t #include // memcpy, strlen #include // distance, input_iterator_tag, iterator_traits @@ -3031,7 +3032,13 @@ class output_buffer void finish() { - m_out.resize(static_cast(m_pos - m_out.data())); + const auto size = static_cast(m_pos - m_out.data()); + m_out.resize(size); + // do not keep a buffer that was sized for a much larger output + if (m_out.capacity() > 1024 && m_out.capacity() / 2 > size) + { + m_out.shrink_to_fit(); + } } NLOHMANN_VIEW_ALWAYS_INLINE void reserve(std::size_t n) @@ -4258,20 +4265,33 @@ class basic_json_view } /// the number of source bytes of this value (estimated for values with - /// decoded strings) + /// decoded strings); the estimate sizes the output buffer of dump() std::size_t source_extent() const noexcept { - const node* const next = document_data::after(m_node); - const bool in_source = (m_node->flags & detail::view::node_flags::storage) == 0; - if (!in_source) + const node* const end = m_doc->tape + m_doc->tape_size; + if ((m_node->flags & detail::view::node_flags::storage) != 0) { return m_node->len; } - if (next != m_doc->tape + m_doc->tape_size && (next->flags & detail::view::node_flags::storage) == 0 && next->off >= m_node->off) + // the value ends where the next node in the source begins; nodes + // with decoded strings (their offset is in the arena) are skipped, + // but only a few of them, to keep the walk short + const node* next = document_data::after(m_node); + for (int skipped = 0; next != end && skipped < 16; ++skipped, ++next) { - return next->off - m_node->off; + if ((next->flags & detail::view::node_flags::storage) == 0) + { + return next->off >= m_node->off ? next->off - m_node->off : 0; + } } - return m_doc->size - m_node->off; + if (next == end) + { + return m_doc->size - m_node->off; + } + // the end is unknown: assume a few bytes per node, the output buffer + // grows should the value be larger + const auto nodes = static_cast(document_data::after(m_node) - m_node); + return (std::min)(m_doc->size - m_node->off, static_cast(1024) + nodes * 16); } /// the value of the first member with this key, or a discarded view From c7df23666f1f6ab834e4bb60fda26785c0c00743 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:11:33 +0200 Subject: [PATCH 10/26] Make view lookups last-wins and chained access safe * Lookups (operator[], at, find, contains, count, value, JSON pointers) return the last member of a duplicate key, as materialize() and parse() keep it. * operator[] on a discarded view returns a discarded view instead of throwing, so v["a"]["b"] is safe for a missing "a". * operator[] and at() take any integer type (not only int and size_t), fixing ambiguous calls with unsigned, long, std::int64_t, ... * Fix the operator[] documentation, which claimed a discarded view for a type mismatch where type_error.305 is thrown. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json_view/at.md | 15 +- docs/mkdocs/docs/api/basic_json_view/begin.md | 2 +- .../docs/api/basic_json_view/contains.md | 4 +- docs/mkdocs/docs/api/basic_json_view/count.md | 8 +- docs/mkdocs/docs/api/basic_json_view/find.md | 8 +- docs/mkdocs/docs/api/basic_json_view/get.md | 6 +- .../docs/api/basic_json_view/is_discarded.md | 5 + docs/mkdocs/docs/api/basic_json_view/items.md | 6 +- .../docs/api/basic_json_view/operator[].md | 58 +++--- docs/mkdocs/docs/api/basic_json_view/value.md | 8 +- .../docs/examples/basic_json_view__items.cpp | 6 +- .../examples/basic_json_view__items.output | 2 +- docs/mkdocs/docs/features/json_view.md | 10 +- include/nlohmann/detail/view/lookup.hpp | 34 +++- include/nlohmann/json_view.hpp | 50 ++++-- tests/src/unit-json_view.cpp | 168 +++++++++++++++++- 16 files changed, 306 insertions(+), 84 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json_view/at.md b/docs/mkdocs/docs/api/basic_json_view/at.md index d1b1732bb..b5deab72c 100644 --- a/docs/mkdocs/docs/api/basic_json_view/at.md +++ b/docs/mkdocs/docs/api/basic_json_view/at.md @@ -8,15 +8,18 @@ basic_json_view at(const string_t& key) const; // (2) basic_json_view at(size_type idx) const; -basic_json_view at(int idx) const; +template +basic_json_view at(IntegerType idx) const; // (3) basic_json_view at(const json_pointer& ptr) const; ``` -1. Returns the value of the object member with key `key` -- the first one, should the key occur more than once (see +1. Returns the value of the object member with key `key` -- the last one, should the key occur more than once (see [Notes on duplicate keys](operator[].md#notes)). -2. Returns the array element at index `idx`. +2. Returns the array element at index `idx`. The template accepts every integer type except `#!cpp bool` and + `#!cpp std::size_t` and forwards to the `size_type` overload, as for [`operator[]`](operator[].md); a negative + `idx` is out of range. 3. Returns the value a JSON pointer `ptr` refers to, starting at this value. ## Parameters @@ -32,7 +35,7 @@ basic_json_view at(const json_pointer& ptr) const; ## Return value -1. the value of the first member with key `key` +1. the value of the last member with key `key` 2. the element at index `idx` 3. the value `ptr` resolves to, starting at this value @@ -72,8 +75,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics ## Complexity 1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after - another, in document order, stopping at the first match. Each comparison first checks the key's length -- - already known from the index, without reading the key bytes -- before comparing its content. + another, in document order, scanning all of them, since the last match is wanted. Each comparison first checks the + key's length -- already known from the index, without reading the key bytes -- before comparing its content. 2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the index (unlike `BasicJsonType`'s array, which is random-access). 3. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at diff --git a/docs/mkdocs/docs/api/basic_json_view/begin.md b/docs/mkdocs/docs/api/basic_json_view/begin.md index ae2665ee9..f9c43802f 100644 --- a/docs/mkdocs/docs/api/basic_json_view/begin.md +++ b/docs/mkdocs/docs/api/basic_json_view/begin.md @@ -24,7 +24,7 @@ Constant. For an object, iteration visits **every** member, including all occurrences of a duplicate key -- unlike [`operator[]`](operator[].md), [`at`](at.md), [`find`](find.md), [`contains`](contains.md), and [`count`](count.md), -which all resolve to the *first* member with a given key. See the +which all resolve to the *last* member with a given key. See the [Notes on duplicate keys](operator[].md#notes) of `operator[]`. Because objects are iterated in document order rather than sorted by key, the order seen here can differ from what diff --git a/docs/mkdocs/docs/api/basic_json_view/contains.md b/docs/mkdocs/docs/api/basic_json_view/contains.md index bbd791887..7c1e706cd 100644 --- a/docs/mkdocs/docs/api/basic_json_view/contains.md +++ b/docs/mkdocs/docs/api/basic_json_view/contains.md @@ -33,8 +33,8 @@ No-throw guarantee: this function never throws exceptions. ## Complexity 1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after - another, in document order, stopping at the first match. Each comparison first checks the key's length -- already - known from the index, without reading the key bytes -- before comparing its content. + another, in document order, scanning all of them, since the last match is wanted. Each comparison first checks the + key's length -- already known from the index, without reading the key bytes -- before comparing its content. 2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at that level or the index into the array -- as for [`operator[]`](operator[].md#complexity) and [`at`](at.md#complexity) with a JSON pointer. diff --git a/docs/mkdocs/docs/api/basic_json_view/count.md b/docs/mkdocs/docs/api/basic_json_view/count.md index 6a599477e..7e0091d90 100644 --- a/docs/mkdocs/docs/api/basic_json_view/count.md +++ b/docs/mkdocs/docs/api/basic_json_view/count.md @@ -23,9 +23,9 @@ No-throw guarantee: this function never throws exceptions. ## Complexity -Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after -another, in document order, stopping at the first match. Each comparison first checks the key's length -- already -known from the index, without reading the key bytes -- before comparing its content. +Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another, +in document order, scanning all of them, since the last match is wanted. Each comparison first checks the key's length +-- already known from the index, without reading the key bytes -- before comparing its content. ## Notes @@ -36,7 +36,7 @@ Unlike [`BasicJsonType::count()`](../basic_json/count.md), whose return value ca an `ObjectType` that allows multiple entries per key, `count()` here never does: it is exactly [`contains()`](contains.md) as `#!cpp 0`/`#!cpp 1`. This holds even if the source text has a duplicate key -- see the [Notes on duplicate keys](operator[].md#notes) of `operator[]` -- because a `#!cpp count() > 1` result would require -counting every member with a matching key, not just finding the first one. +counting every member with a matching key (the lookup functions resolve to the *last* one). ## Examples diff --git a/docs/mkdocs/docs/api/basic_json_view/find.md b/docs/mkdocs/docs/api/basic_json_view/find.md index f7ad92e9f..d2323353a 100644 --- a/docs/mkdocs/docs/api/basic_json_view/find.md +++ b/docs/mkdocs/docs/api/basic_json_view/find.md @@ -6,7 +6,7 @@ iterator find(const char* key) const; iterator find(const string_t& key) const; ``` -Finds a member with key `key` -- the first one, should the key occur more than once (see +Finds a member with key `key` -- the last one, should the key occur more than once (see [Notes on duplicate keys](operator[].md#notes)). If the value is not an object, or no member has this key, [`end()`](end.md) is returned. @@ -25,9 +25,9 @@ No-throw guarantee: this function never throws exceptions. ## Complexity -Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after -another, in document order, stopping at the first match. Each comparison first checks the key's length -- already -known from the index, without reading the key bytes -- before comparing its content. +Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another, +in document order, scanning all of them, since the last match is wanted. Each comparison first checks the key's length +-- already known from the index, without reading the key bytes -- before comparing its content. ## Notes diff --git a/docs/mkdocs/docs/api/basic_json_view/get.md b/docs/mkdocs/docs/api/basic_json_view/get.md index d8ec333c9..c3bb85790 100644 --- a/docs/mkdocs/docs/api/basic_json_view/get.md +++ b/docs/mkdocs/docs/api/basic_json_view/get.md @@ -87,9 +87,9 @@ exception thrown while converting through `materialize()` (the last bullet) is d !!! info "Duplicate keys" `#!cpp std::map`/`#!cpp std::unordered_map` conversions keep the *last* value of a repeated key, like - [`materialize()`](materialize.md) and [`BasicJsonType::parse()`](../basic_json/parse.md) do. This is the opposite - of [`operator[]`](operator[].md)/[`at`](at.md)/[`find`](find.md)/[`contains`](contains.md), which resolve to the - *first* occurrence (see the [Notes on duplicate keys](operator[].md#notes)). + [`materialize()`](materialize.md) and [`BasicJsonType::parse()`](../basic_json/parse.md) do. This is the member + [`operator[]`](operator[].md)/[`at`](at.md)/[`find`](find.md)/[`contains`](contains.md) resolve to, too (see the + [Notes on duplicate keys](operator[].md#notes)). !!! info "No pointers, references, or implicit conversion" diff --git a/docs/mkdocs/docs/api/basic_json_view/is_discarded.md b/docs/mkdocs/docs/api/basic_json_view/is_discarded.md index 4cc11077d..a7fe7fa07 100644 --- a/docs/mkdocs/docs/api/basic_json_view/is_discarded.md +++ b/docs/mkdocs/docs/api/basic_json_view/is_discarded.md @@ -9,6 +9,10 @@ view (see [(constructor)](basic_json_view.md)), and for [`root()`](../basic_json that is itself [discarded](../basic_json_document/is_discarded.md) -- in particular, the root of a failed [`parse()`](../basic_json_document/parse.md) with `allow_exceptions` set to `#!cpp false`. +A discarded view is also what [`operator[]`](operator[].md) returns for a missing key, an index out of range, or a +JSON pointer that cannot be resolved, and for any access on a view that is itself discarded (so a chain such as +`#!cpp v["a"]["b"]` is safe). [`at`](at.md) throws instead. + ## Return value `#!cpp true` if the view is discarded, `#!cpp false` otherwise. @@ -45,6 +49,7 @@ site. ## See also +- [operator[]](operator[].md) - access specified element; yields a discarded view where an element is missing - [operator bool](operator_bool.md) - return whether the view refers to a value - [(constructor)](basic_json_view.md) - the default constructor creates a discarded view - [is_discarded (basic_json_document)](../basic_json_document/is_discarded.md) - return whether the last parse failed diff --git a/docs/mkdocs/docs/api/basic_json_view/items.md b/docs/mkdocs/docs/api/basic_json_view/items.md index d54f8d723..8e13852da 100644 --- a/docs/mkdocs/docs/api/basic_json_view/items.md +++ b/docs/mkdocs/docs/api/basic_json_view/items.md @@ -48,7 +48,7 @@ Constant. As for [`begin()`](begin.md)/[`end()`](end.md), `items()` visits **every** member of an object, including all occurrences of a duplicate key -- unlike [`operator[]`](operator[].md), [`at`](at.md), [`find`](find.md), -[`contains`](contains.md), and [`count`](count.md), which resolve to the *first* member with a given key. See the +[`contains`](contains.md), and [`count`](count.md), which resolve to the *last* member with a given key. See the [Notes on duplicate keys](operator[].md#notes) of `operator[]`. !!! danger "Lifetime issues" @@ -63,8 +63,8 @@ occurrences of a duplicate key -- unlike [`operator[]`](operator[].md), [`at`](a The example below shows a settings object whose source text records every update to a key as a duplicate member, in the order they happened. `items()` walks all of them, so the update history is visible, while - [`operator[]`](operator[].md) only ever sees the *first* one and [`materialize()`](materialize.md) -- like - [`BasicJsonType::parse()`](../basic_json/parse.md) -- keeps only the *last*. + [`operator[]`](operator[].md) sees the *last* one, and so does [`materialize()`](materialize.md) -- like + [`BasicJsonType::parse()`](../basic_json/parse.md). ```cpp --8<-- "examples/basic_json_view__items.cpp" diff --git a/docs/mkdocs/docs/api/basic_json_view/operator[].md b/docs/mkdocs/docs/api/basic_json_view/operator[].md index 2361bf77c..30130db7c 100644 --- a/docs/mkdocs/docs/api/basic_json_view/operator[].md +++ b/docs/mkdocs/docs/api/basic_json_view/operator[].md @@ -8,16 +8,19 @@ basic_json_view operator[](const string_t& key) const; // (2) basic_json_view operator[](size_type idx) const; -basic_json_view operator[](int idx) const; +template +basic_json_view operator[](IntegerType idx) const; // (3) basic_json_view operator[](const json_pointer& ptr) const; ``` -1. Returns the value of the object member with key `key` -- the first one, should the key occur more than once (see +1. Returns the value of the object member with key `key` -- the last one, should the key occur more than once (see the [Notes](#notes) below) -- or a [discarded](is_discarded.md) view if there is no such member. -2. Returns the array element at index `idx`, or a [discarded](is_discarded.md) view if `idx` is out of range. (The - `#!cpp int` overload only exists so that an integer literal is not ambiguous between this overload and 1.) +2. Returns the array element at index `idx`, or a [discarded](is_discarded.md) view if `idx` is out of range. The + template accepts every integer type except `#!cpp bool` and `#!cpp std::size_t` (`#!cpp int`, `#!cpp unsigned`, + `#!cpp long`, `#!cpp std::int64_t`, ...) and forwards to the `size_type` overload, so that an integer argument is + not ambiguous between that overload and 1; a negative `idx` is out of range. 3. Returns the value a JSON pointer `ptr` refers to, starting at this value, or a [discarded](is_discarded.md) view wherever resolving it further is not possible without inserting into or extending the document (see [Return value](#return-value) and [Exceptions](#exceptions) below). @@ -35,10 +38,12 @@ basic_json_view operator[](const json_pointer& ptr) const; ## Return value -1. the value of the first member with key `key`, or a discarded view if `#!cpp is_object()` is `#!cpp false` or no - member has this key -2. the element at index `idx`, or a discarded view if `#!cpp is_array()` is `#!cpp false` or `#!cpp idx >= size()` -3. the value `ptr` resolves to, starting at this value, or a discarded view for exactly the reference tokens where the +1. the value of the last member with key `key`, or a discarded view if no member has this key (or if this view is + [discarded](is_discarded.md)) +2. the element at index `idx`, or a discarded view if `#!cpp idx >= size()` or `idx` is negative (or if this view is + [discarded](is_discarded.md)) +3. the value `ptr` resolves to, starting at this value, or a discarded view (also if this view is + [discarded](is_discarded.md)) for exactly the reference tokens where the **const** overload of [`BasicJsonType::operator[]`](../basic_json/operator%5B%5D.md) invokes undefined behavior for the same pointer and the same document: an object member that does not exist, or an array index that is out of range @@ -49,10 +54,12 @@ Strong exception safety: if an exception is thrown, there are no changes to the ## Exceptions -1. Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the value is not an object -- +1. Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the value is not an object and + not [discarded](is_discarded.md) -- the same exception, with the same message, that the **const** overload of [`BasicJsonType::operator[]`](../basic_json/operator%5B%5D.md) throws for a string argument on a non-object value. -2. Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the value is not an array -- +2. Throws [`type_error.305`](../../home/exceptions.md#jsonexceptiontype_error305) if the value is not an array and + not [discarded](is_discarded.md) -- the same exception, with the same message, that the **const** overload of [`BasicJsonType::operator[]`](../basic_json/operator%5B%5D.md) throws for a numeric argument on a non-array value. 3. Throws the same exceptions, with the same messages, that the **const** overload of @@ -73,9 +80,9 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics ## Complexity 1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after - another, in document order, stopping at the first match. Each comparison first checks the key's length -- - already known from the index, without reading the key bytes -- before comparing its content, so a key of a - different length than `key` is rejected without touching the source text. + another, in document order, scanning all of them, since the last match is wanted. Each comparison first checks the + key's length -- already known from the index, without reading the key bytes -- before comparing its content, so a + key of a different length than `key` is rejected without touching the source text. 2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the index (unlike `BasicJsonType`'s array, which is random-access). 3. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at @@ -84,20 +91,29 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics ## Notes Unlike `BasicJsonType::operator[]`, which is undefined behavior (guarded by a -[runtime assertion](../../features/assertions.md)) for a missing key on a **const** value, this operator always -returns a safe, testable result: a [discarded](is_discarded.md) view, which is `#!cpp false` in a boolean context. +[runtime assertion](../../features/assertions.md)) for a missing key on a **const** value, this operator returns a +safe, testable result for a missing key or an index out of range: a [discarded](is_discarded.md) view, which is +`#!cpp false` in a boolean context. There is also no non-const overload that inserts a missing key or extends an array -- a view never modifies the document. +!!! info "Chained access" + + `#!cpp operator[]` on a [discarded](is_discarded.md) view returns a discarded view and does not throw, so a chain + like `#!cpp v["a"]["b"][0]` is safe even if `"a"` or `"b"` is missing: the first missing step makes the whole + result discarded, which is tested once at the end. Type errors on values that are *not* discarded still throw: a + key on an array or a primitive, or an index on an object or a primitive, is `type_error.305` as for + `BasicJsonType`. [`at`](at.md) still throws for a discarded view, as it does for a missing key. + !!! info "Duplicate keys" If the source text has an object with a duplicate key, `#!cpp operator[]` (and [`at`](at.md), [`find`](find.md), - [`contains`](contains.md), [`count`](count.md)) all resolve to the *first* member with that key, because a - lookup can stop as soon as it finds a match. This is different from - [`materialize()`](materialize.md) (and [`BasicJsonType::parse()`](../basic_json/parse.md)), which replay every - member in order and so end up keeping the *last* value for a repeated key -- there is no reason for them to stop - early. [`begin()`](begin.md)/[`end()`](end.md) and [`items()`](items.md) iterate over *all* members, including - duplicates, in document order. See the example below and [`size()`](size.md#notes). + [`contains`](contains.md), [`count`](count.md), [`value`](value.md), and JSON pointer resolution) all resolve to + the *last* member with that key. This is the member [`materialize()`](materialize.md) (and + [`BasicJsonType::parse()`](../basic_json/parse.md)) keeps, so a lookup in the view and in the materialized value + agree. [`begin()`](begin.md)/[`end()`](end.md) and [`items()`](items.md) iterate over *all* members, including + duplicates, in document order. A lookup scans all members for this: it cannot stop at the first match. See the + example below and [`size()`](size.md#notes). !!! info "JSON pointer resolution" diff --git a/docs/mkdocs/docs/api/basic_json_view/value.md b/docs/mkdocs/docs/api/basic_json_view/value.md index ff63e32f5..c13c0cd63 100644 --- a/docs/mkdocs/docs/api/basic_json_view/value.md +++ b/docs/mkdocs/docs/api/basic_json_view/value.md @@ -12,7 +12,7 @@ T value(const json_pointer& ptr, const T& default_value) const; string_t value(const json_pointer& ptr, const char* default_value) const; ``` -1. Returns the value of the object member with key `key` -- the first one, should the key occur more than once (see +1. Returns the value of the object member with key `key` -- the last one, should the key occur more than once (see [Notes on duplicate keys](operator[].md#notes)) -- converted to `T`, or `default_value` if there is no such member. 2. Returns the value a JSON pointer `ptr` refers to, starting at this value, converted to `T`, or `default_value` if `ptr` cannot be resolved. @@ -39,7 +39,7 @@ equivalent) deduce `string_t`, not `const char*`, for their return type and for ## Return value -1. the first member with key `key`, converted to `T`, or `default_value` +1. the last member with key `key`, converted to `T`, or `default_value` 2. the value `ptr` resolves to, converted to `T`, or `default_value` ## Exception safety @@ -68,8 +68,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics ## Complexity 1. Linear in the number of members: as for [`operator[]`](operator[].md#complexity), members are compared one after - another, in document order, stopping at the first match. Plus the complexity of converting the found member to - `T` (see [`get`](get.md)). + another, in document order, scanning all of them, since the last match is wanted. Plus the complexity of converting + the found member to `T` (see [`get`](get.md)). 2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at that level or the index into the array -- as for the [`operator[]`](operator[].md#complexity) and [`at`](at.md#complexity) overloads that take a JSON pointer. Plus the complexity of converting the resolved value diff --git a/docs/mkdocs/docs/examples/basic_json_view__items.cpp b/docs/mkdocs/docs/examples/basic_json_view__items.cpp index 3f7779c19..6357c7e4b 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__items.cpp +++ b/docs/mkdocs/docs/examples/basic_json_view__items.cpp @@ -7,8 +7,8 @@ int main() { // a settings object whose source text records every update to a key as // a duplicate member. items() visits all of them, in document order, so - // the update history is visible; operator[] only ever sees the first - // one, and materialize() -- like basic_json::parse() -- keeps the last + // the update history is visible; operator[] and materialize() -- like + // basic_json::parse() -- see the last one json_document updates = json_document::parse(R"({"retries": 1, "timeout": 30, "retries": 5})"); const auto settings = updates.root(); @@ -17,6 +17,6 @@ int main() std::cout << item.key() << '=' << item.value().materialize().dump() << '\n'; } - std::cout << "first \"retries\" seen by operator[]: " << settings["retries"].materialize().dump() << '\n'; + std::cout << "last \"retries\" seen by operator[]: " << settings["retries"].materialize().dump() << '\n'; std::cout << "last \"retries\" kept by materialize(): " << settings.materialize()["retries"].dump() << '\n'; } diff --git a/docs/mkdocs/docs/examples/basic_json_view__items.output b/docs/mkdocs/docs/examples/basic_json_view__items.output index 78609c8ee..8f51dbf18 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__items.output +++ b/docs/mkdocs/docs/examples/basic_json_view__items.output @@ -1,5 +1,5 @@ retries=1 timeout=30 retries=5 -first "retries" seen by operator[]: 1 +last "retries" seen by operator[]: 5 last "retries" kept by materialize(): 5 diff --git a/docs/mkdocs/docs/features/json_view.md b/docs/mkdocs/docs/features/json_view.md index 7dbea9241..06a978ee8 100644 --- a/docs/mkdocs/docs/features/json_view.md +++ b/docs/mkdocs/docs/features/json_view.md @@ -126,14 +126,20 @@ whenever any of the other conditions above was not met. members in the order they appear in the source text. `basic_json`'s default `object_t` is a `std::map`, which sorts by key, so iterating a [`materialize()`](../api/basic_json_view/materialize.md)d value can print members in a different order than iterating the view they came from. +- **Chained access is safe.** [`operator[]`](../api/basic_json_view/operator%5B%5D.md) with a missing key, an index + out of range, or an unresolvable JSON pointer returns a [discarded](../api/basic_json_view/is_discarded.md) view, and + `operator[]` on a discarded view returns a discarded view without throwing: `#!cpp v["a"]["b"][0]` can be tested + once at the end. Type errors on values that exist (a key on an array, an index on an object) still throw, and + [`at`](../api/basic_json_view/at.md) throws for every missing value. - **Duplicate keys are visible.** If an object in the source text repeats a key, [`begin()`](../api/basic_json_view/begin.md)/[`end()`](../api/basic_json_view/end.md) and [`items()`](../api/basic_json_view/items.md) visit *every* occurrence (and [`size()`](../api/basic_json_view/size.md) counts all of them), while [`operator[]`](../api/basic_json_view/operator%5B%5D.md), [`at`](../api/basic_json_view/at.md), [`find`](../api/basic_json_view/find.md), [`contains`](../api/basic_json_view/contains.md), and [`count`](../api/basic_json_view/count.md) resolve to the - *first* occurrence, since a lookup can stop as soon as it finds a match. `basic_json::parse()` (and so - [`materialize()`](../api/basic_json_view/materialize.md)) instead keeps only the *last* value for a repeated key. + *last* occurrence -- the one `basic_json::parse()` (and so + [`materialize()`](../api/basic_json_view/materialize.md)) keeps for a repeated key -- which makes a lookup scan all + members instead of stopping at a match. See the [Notes on duplicate keys](../api/basic_json_view/operator%5B%5D.md#notes) of `operator[]`. - **No [`JSON_DIAGNOSTICS`](../api/macros/json_diagnostics.md) path.** Exceptions thrown by `basic_json_view`'s own element access and lookup functions never carry the JSON Pointer path `JSON_DIAGNOSTICS` would otherwise add: the diff --git a/include/nlohmann/detail/view/lookup.hpp b/include/nlohmann/detail/view/lookup.hpp index 71338be32..a58c9a33a 100644 --- a/include/nlohmann/detail/view/lookup.hpp +++ b/include/nlohmann/detail/view/lookup.hpp @@ -11,6 +11,8 @@ #include // size_t #include // uint16_t, uint32_t, uint64_t #include // memcmp, memcpy +#include // numeric_limits +#include // integral_constant, is_integral, is_same #include #include @@ -81,12 +83,14 @@ class short_key std::uint64_t m_b = 0; }; -/// the key node of the first member of an object with the given key, or -/// nullptr; most keys are rejected by their length, from the index alone +/// the key node of the last member of an object with the given key, or +/// nullptr (the last one, as materialize() and parse() keep it); most keys are +/// rejected by their length, from the index alone inline const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept { const node* const end = document_data::child_end(object); const auto* const k = reinterpret_cast(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const node* last = nullptr; if (NLOHMANN_VIEW_LIKELY(n <= 16)) { const short_key probe(k, n); @@ -94,19 +98,37 @@ inline const node* find_member(const document_data& d, const node* object, const { if (m->len == n && probe.matches(reinterpret_cast(d.str(*m)))) // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) { - return m; + last = m; } } - return nullptr; + return last; } for (const node* m = document_data::first_child(object); m != end; m = document_data::after(m + 1)) { if (m->len == n && std::memcmp(d.str(*m), key, n) == 0) { - return m; + last = m; } } - return nullptr; + return last; +} + +/// whether an integer type is accepted as an array index by the view's +/// operator[] and at(): every integer type but bool and size_t, which has its +/// own overload +template +struct is_index_type : std::integral_constant < bool, + std::is_integral::value && !std::is_same::value && !std::is_same::value > +{}; + +/// an integer as an index: negative values, and values that do not fit a +/// size_t, map to the largest size_t (out of range for every array) +template +SizeType to_index(IntegerType idx) noexcept +{ + const IntegerType zero = 0; + const auto result = static_cast(idx); + return (idx < zero || static_cast(result) != idx) ? (std::numeric_limits::max)() : result; } /// the element of an array at an index below its size diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index b51ff446b..147a0ee4b 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -234,13 +234,18 @@ class basic_json_view // element access // //////////////////// - /// the value of the member with this key (the first one, should the key - /// occur more than once); a discarded view if there is none. Throws - /// type_error.305 if this is not an object. + /// the value of the member with this key (the last one, should the key + /// occur more than once); a discarded view if there is none, or if this + /// is a discarded view (so that v["a"]["b"] is safe). Throws type_error.305 + /// if this is any other value but an object. NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view operator[](string_view_t key) const { if (NLOHMANN_VIEW_UNLIKELY(!is_object())) { + if (is_discarded()) + { + return basic_json_view(); + } detail::view::throw_type_error(305, "cannot use operator[] with a string argument with ", type_name()); } return lookup(key); @@ -257,31 +262,43 @@ class basic_json_view } /// the element at this index; a discarded view if the index is out of - /// range. Throws type_error.305 if this is not an array. + /// range, or if this is a discarded view. Throws type_error.305 if this is + /// any other value but an array. basic_json_view operator[](size_type idx) const { if (NLOHMANN_VIEW_UNLIKELY(!is_array())) { + if (is_discarded()) + { + return basic_json_view(); + } detail::view::throw_type_error(305, "cannot use operator[] with a numeric argument with ", type_name()); } return idx < m_node->len ? basic_json_view(m_doc, detail::view::element_at(m_node, idx)) : basic_json_view(); } - /// (an int argument would be ambiguous between size_type and const char*) - basic_json_view operator[](int idx) const + /// any other integer type (int, unsigned, long, std::int64_t, ...; a + /// single overload for size_type alone would be ambiguous for all of them + /// and for const char*); negative values are out of range + template < typename IntegerType, typename std::enable_if < detail::view::is_index_type::value, int >::type = 0 > + basic_json_view operator[](IntegerType idx) const { - return operator[](static_cast(idx)); + return operator[](detail::view::to_index(idx)); } /// the value a JSON pointer refers to; a discarded view if a key is - /// missing or an index is out of range. Other errors throw what const - /// basic_json::operator[] throws. + /// missing or an index is out of range, or if this is a discarded view. + /// Other errors throw what const basic_json::operator[] throws. basic_json_view operator[](const json_pointer& ptr) const { + if (NLOHMANN_VIEW_UNLIKELY(is_discarded())) + { + return basic_json_view(); + } return detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::unchecked); } - /// the value of the member with this key (the first one, should the key + /// the value of the member with this key (the last one, should the key /// occur more than once). Throws type_error.304 if this is not an object, /// and out_of_range.403 if there is no such member. basic_json_view at(string_view_t key) const @@ -323,9 +340,12 @@ class basic_json_view return basic_json_view(m_doc, detail::view::element_at(m_node, idx)); } - basic_json_view at(int idx) const + /// any other integer type, see operator[]; negative values are out of + /// range + template < typename IntegerType, typename std::enable_if < detail::view::is_index_type::value, int >::type = 0 > + basic_json_view at(IntegerType idx) const { - return at(static_cast(idx)); + return at(detail::view::to_index(idx)); } /// the value a JSON pointer refers to; throws what basic_json::at() @@ -336,7 +356,7 @@ class basic_json_view } /// the member with this key converted to T, or the default value if there - /// is no such member (the first one, should the key occur more than + /// is no such member (the last one, should the key occur more than /// once). Throws type_error.306 if this is not an object. template < typename T, typename std::enable_if < !std::is_same::type, const char*>::value, int >::type = 0 > T value(string_view_t key, const T& default_value) const @@ -401,7 +421,7 @@ class basic_json_view // lookup // //////////// - /// an iterator to the member with this key (the first one, should the + /// an iterator to the member with this key (the last one, should the /// key occur more than once), or end(); end() also for non-objects iterator find(string_view_t key) const { @@ -579,7 +599,7 @@ class basic_json_view : m_doc(d), m_node(n) {} - /// the value of the first member with this key, or a discarded view + /// the value of the last member with this key, or a discarded view /// (object required) NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view lookup(string_view_t key) const noexcept { diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index e208c2abb..55da948e0 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -23,6 +23,7 @@ using nlohmann::ordered_json_view; #include #include #include +#include #include #include #include @@ -442,7 +443,7 @@ std::string exception_of(F f) // compares a view with the ordered_json value materialize() gives for it: // types, sizes, elements and members (by index, key, and iteration), in -// document order; duplicate keys are found as their first occurrence +// document order; duplicate keys are found as their last occurrence void check_access(const ordered_json_view& v, const ordered_json& j) { REQUIRE(v.type() == j.type()); @@ -483,11 +484,23 @@ void check_access(const ordered_json_view& v, const ordered_json& j) const std::string key(it.key().data(), it.key().size()); CHECK(v.contains(key)); CHECK(v.count(key) == 1); - if (std::find(keys.begin(), keys.end(), key) != keys.end()) + if (std::find(keys.begin(), keys.end(), key) == keys.end()) { - continue; // a duplicate: lookups find the first one + keys.push_back(key); + } + // lookups find the last member with the key, which is this one if + // there is no later one + auto next = it; + ++next; + bool is_last = true; + for (; next != v.end(); ++next) + { + is_last = is_last && next.key() != it.key(); + } + if (!is_last) + { + continue; } - keys.push_back(key); CHECK(v.find(key) == it); CHECK(v[key].materialize() == it->materialize()); CHECK(v.at(key).materialize() == it.value().materialize()); @@ -578,15 +591,35 @@ TEST_CASE("json_view element access and iteration") #endif } - SECTION("duplicate keys: lookups find the first member, iteration all") + SECTION("duplicate keys: lookups find the last member, iteration all") { const json_document d = json_document::parse(R"({"a":1,"b":2,"a":3})"); const json_view v = d.root(); CHECK(v.size() == 3); - CHECK(v["a"].materialize() == 1); - CHECK(v.at("a").materialize() == 1); - CHECK(v.find("a") == v.begin()); + CHECK(v["a"].materialize() == 3); + CHECK(v.at("a").materialize() == 3); + CHECK(v.find("a") == std::next(v.begin(), 2)); + CHECK(v.find("a").value().materialize() == 3); + CHECK(v.find("b") == std::next(v.begin())); CHECK(v.count("a") == 1); + CHECK(v.contains("a")); + CHECK(v.value("a", 0) == 3); + CHECK(v["a"].materialize() == v.materialize()["a"]); // as materialize() + // keys of every length class (the 16-byte short compare and memcmp) + for (const std::size_t n : + { + 0u, 1u, 3u, 7u, 8u, 15u, 16u, 17u, 40u + }) + { + const std::string key(n, 'k'); + const json_document dk = json_document::parse("{\"" + key + "\":1,\"" + key + "x\":2,\"" + key + "\":3,\"" + key + "\":4}"); + CAPTURE(n) + CHECK(dk.root()[key].materialize() == 4); + CHECK(dk.root().at(key).materialize() == 4); + CHECK(dk.root().find(key) == std::next(dk.root().begin(), 3)); + CHECK(dk.root().value(key, 0) == 4); + CHECK(dk.root()[key + "x"].materialize() == 2); + } std::string order; for (auto it = v.begin(); it != v.end(); ++it) { @@ -644,7 +677,124 @@ TEST_CASE("json_view element access and iteration") const json_view invalid{}; CHECK(invalid.begin() == invalid.end()); CHECK(std::string(invalid.type_name()) == "discarded"); - CHECK_THROWS_WITH_AS(invalid["a"], "[json.exception.type_error.305] cannot use operator[] with a string argument with discarded", json::type_error&); + } + + SECTION("chained access is safe: operator[] of a discarded view is discarded") + { + const json_document d = json_document::parse(R"({"a":{"b":[10,20]},"s":"str"})"); + const json_view v = d.root(); + // missing keys and indexes + CHECK(!v["x"]); + CHECK(!v["x"]["y"]); + CHECK(!v["x"]["y"]["z"]); + CHECK(!v["x"][0]); + CHECK(!v["x"][0u][1L]); + CHECK(!v["a"]["b"][2]); + CHECK(!v["a"]["b"][2]["c"]); + CHECK(!v["a"]["b"][2][json_view::json_pointer("/c")]); + CHECK(!v["x"][json_view::json_pointer("/a/b")]); + CHECK(!v["x"][json_view::json_pointer("")]); + CHECK(v["x"]["y"].is_discarded()); + CHECK(!v["x"][std::string("y")]); + // a resolvable path still resolves + CHECK(v["a"]["b"][1].materialize() == 20); + CHECK(v[json_view::json_pointer("/a/b/1")].materialize() == 20); + // the discarded view of an unresolved pointer is discarded too + CHECK(!v[json_view::json_pointer("/x/y")]["z"]); + CHECK(!v[json_view::json_pointer("/a/b/5")][0]); +#if !defined(JSON_NOEXCEPTION) + // type errors on values that are not discarded stay + CHECK_THROWS_WITH_AS(v[0], "[json.exception.type_error.305] cannot use operator[] with a numeric argument with object", json::type_error&); + CHECK_THROWS_WITH_AS(v["a"]["b"]["c"], "[json.exception.type_error.305] cannot use operator[] with a string argument with array", json::type_error&); + CHECK_THROWS_WITH_AS(v["s"]["c"], "[json.exception.type_error.305] cannot use operator[] with a string argument with string", json::type_error&); + CHECK_THROWS_WITH_AS(v["s"][0], "[json.exception.type_error.305] cannot use operator[] with a numeric argument with string", json::type_error&); + CHECK_THROWS_AS(v["a"]["b"][0]["c"], json::type_error&); + CHECK_THROWS_AS(v["s"][json_view::json_pointer("/x")], json::out_of_range&); + // at() keeps throwing on a discarded view + const json_view invalid{}; + CHECK(!invalid["a"]); + CHECK(!invalid[0]); + CHECK(!invalid[json_view::json_pointer("/a")]); + CHECK_THROWS_WITH_AS(invalid.at("a"), "[json.exception.type_error.304] cannot use at() with discarded", json::type_error&); + CHECK_THROWS_WITH_AS(invalid.at(0), "[json.exception.type_error.304] cannot use at() with discarded", json::type_error&); + CHECK_THROWS_AS(v.at("x").at("y"), json::out_of_range&); + CHECK_THROWS_AS(v["x"].at("y"), json::type_error&); + CHECK_THROWS_AS(v.at(json_view::json_pointer("/x/y")), json::out_of_range&); + CHECK_THROWS_AS(v["x"].at(json_view::json_pointer("/y")), json::out_of_range&); +#endif + } + + SECTION("integer types as array indexes") + { + const json_document d = json_document::parse("[10,20,30]"); + const json_view v = d.root(); + const json j = v.materialize(); + // (compile-time: no overload is ambiguous) + CHECK(v[0].materialize() == 10); + CHECK(v[1].materialize() == 20); + CHECK(v[0u].materialize() == 10); + CHECK(v[1u].materialize() == 20); + CHECK(v[1L].materialize() == 20); + CHECK(v[2UL].materialize() == 30); + CHECK(v[1LL].materialize() == 20); + CHECK(v[2ULL].materialize() == 30); + CHECK(v[static_cast(1)].materialize() == 20); + CHECK(v[static_cast(2)].materialize() == 30); + CHECK(v[static_cast(1)].materialize() == 20); + CHECK(v[static_cast(2)].materialize() == 30); + CHECK(v[std::int8_t(1)].materialize() == 20); + CHECK(v[std::int16_t(2)].materialize() == 30); + CHECK(v[std::int32_t(1)].materialize() == 20); + CHECK(v[std::int64_t(2)].materialize() == 30); + CHECK(v[std::uint32_t(0)].materialize() == 10); + CHECK(v[std::uint64_t(1)].materialize() == 20); + CHECK(v[std::size_t(2)].materialize() == 30); + CHECK(v[std::ptrdiff_t(1)].materialize() == 20); + CHECK(j[0u] == 10); // as basic_json + + CHECK(v.at(0).materialize() == 10); + CHECK(v.at(1u).materialize() == 20); + CHECK(v.at(1L).materialize() == 20); + CHECK(v.at(2LL).materialize() == 30); + CHECK(v.at(2ULL).materialize() == 30); + CHECK(v.at(static_cast(1)).materialize() == 20); + CHECK(v.at(static_cast(2)).materialize() == 30); + CHECK(v.at(std::int32_t(0)).materialize() == 10); + CHECK(v.at(std::uint32_t(0)).materialize() == 10); + CHECK(v.at(std::int64_t(0)).materialize() == 10); + CHECK(v.at(std::uint64_t(1)).materialize() == 20); + CHECK(v.at(std::size_t(2)).materialize() == 30); + CHECK(j.at(std::uint32_t(0)) == 10); // as basic_json + + // out of range, including negative values (no wrap-around) + CHECK(!v[3]); + CHECK(!v[3u]); + CHECK(!v[3L]); + CHECK(!v[-1]); + CHECK(!v[-1L]); + CHECK(!v[-1LL]); + CHECK(!v[static_cast(-1)]); + CHECK(!v[std::int64_t(-3)]); + CHECK(!v[(std::numeric_limits::min)()]); + CHECK(!v[(std::numeric_limits::max)()]); + CHECK(!v[(std::numeric_limits::max)()]); + CHECK(!v[std::numeric_limits::max()]); +#if !defined(JSON_NOEXCEPTION) + CHECK_THROWS_WITH_AS(v.at(3), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&); + CHECK_THROWS_WITH_AS(v.at(3u), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&); + CHECK_THROWS_WITH_AS(v.at(std::int64_t(3)), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&); + CHECK_THROWS_AS(v.at(-1), json::out_of_range&); + CHECK_THROWS_AS(v.at(-1L), json::out_of_range&); + CHECK_THROWS_AS(v.at(std::int64_t(-1)), json::out_of_range&); + CHECK_THROWS_AS(v.at((std::numeric_limits::min)()), json::out_of_range&); + CHECK_THROWS_AS(v.at((std::numeric_limits::max)()), json::out_of_range&); + // not an array + const json_document o = json_document::parse("{}"); + CHECK_THROWS_AS(o.root()[0u], json::type_error&); + CHECK_THROWS_AS(o.root()[1L], json::type_error&); + CHECK_THROWS_AS(o.root().at(std::uint32_t(0)), json::type_error&); + CHECK_THROWS_AS(o.root().at(std::int64_t(0)), json::type_error&); +#endif } SECTION("iterators") From 4faa3fcca92742fb50483237e05d664bcd703745 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:11:34 +0200 Subject: [PATCH 11/26] Regenerate the amalgamated headers Signed-off-by: Niels Lohmann --- single_include/nlohmann/json_view.hpp | 84 ++++++++++++++++++++------- 1 file changed, 63 insertions(+), 21 deletions(-) diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 66b7059d1..bccb77c0b 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -2046,6 +2046,8 @@ NLOHMANN_JSON_NAMESPACE_END #include // size_t #include // uint16_t, uint32_t, uint64_t #include // memcmp, memcpy +#include // numeric_limits +#include // integral_constant, is_integral, is_same // #include // #include @@ -2119,12 +2121,14 @@ class short_key std::uint64_t m_b = 0; }; -/// the key node of the first member of an object with the given key, or -/// nullptr; most keys are rejected by their length, from the index alone +/// the key node of the last member of an object with the given key, or +/// nullptr (the last one, as materialize() and parse() keep it); most keys are +/// rejected by their length, from the index alone inline const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept { const node* const end = document_data::child_end(object); const auto* const k = reinterpret_cast(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const node* last = nullptr; if (NLOHMANN_VIEW_LIKELY(n <= 16)) { const short_key probe(k, n); @@ -2132,19 +2136,37 @@ inline const node* find_member(const document_data& d, const node* object, const { if (m->len == n && probe.matches(reinterpret_cast(d.str(*m)))) // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) { - return m; + last = m; } } - return nullptr; + return last; } for (const node* m = document_data::first_child(object); m != end; m = document_data::after(m + 1)) { if (m->len == n && std::memcmp(d.str(*m), key, n) == 0) { - return m; + last = m; } } - return nullptr; + return last; +} + +/// whether an integer type is accepted as an array index by the view's +/// operator[] and at(): every integer type but bool and size_t, which has its +/// own overload +template +struct is_index_type : std::integral_constant < bool, + std::is_integral::value && !std::is_same::value && !std::is_same::value > +{}; + +/// an integer as an index: negative values, and values that do not fit a +/// size_t, map to the largest size_t (out of range for every array) +template +SizeType to_index(IntegerType idx) noexcept +{ + const IntegerType zero = 0; + const auto result = static_cast(idx); + return (idx < zero || static_cast(result) != idx) ? (std::numeric_limits::max)() : result; } /// the element of an array at an index below its size @@ -2971,13 +2993,18 @@ class basic_json_view // element access // //////////////////// - /// the value of the member with this key (the first one, should the key - /// occur more than once); a discarded view if there is none. Throws - /// type_error.305 if this is not an object. + /// the value of the member with this key (the last one, should the key + /// occur more than once); a discarded view if there is none, or if this + /// is a discarded view (so that v["a"]["b"] is safe). Throws type_error.305 + /// if this is any other value but an object. NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view operator[](string_view_t key) const { if (NLOHMANN_VIEW_UNLIKELY(!is_object())) { + if (is_discarded()) + { + return basic_json_view(); + } detail::view::throw_type_error(305, "cannot use operator[] with a string argument with ", type_name()); } return lookup(key); @@ -2994,31 +3021,43 @@ class basic_json_view } /// the element at this index; a discarded view if the index is out of - /// range. Throws type_error.305 if this is not an array. + /// range, or if this is a discarded view. Throws type_error.305 if this is + /// any other value but an array. basic_json_view operator[](size_type idx) const { if (NLOHMANN_VIEW_UNLIKELY(!is_array())) { + if (is_discarded()) + { + return basic_json_view(); + } detail::view::throw_type_error(305, "cannot use operator[] with a numeric argument with ", type_name()); } return idx < m_node->len ? basic_json_view(m_doc, detail::view::element_at(m_node, idx)) : basic_json_view(); } - /// (an int argument would be ambiguous between size_type and const char*) - basic_json_view operator[](int idx) const + /// any other integer type (int, unsigned, long, std::int64_t, ...; a + /// single overload for size_type alone would be ambiguous for all of them + /// and for const char*); negative values are out of range + template < typename IntegerType, typename std::enable_if < detail::view::is_index_type::value, int >::type = 0 > + basic_json_view operator[](IntegerType idx) const { - return operator[](static_cast(idx)); + return operator[](detail::view::to_index(idx)); } /// the value a JSON pointer refers to; a discarded view if a key is - /// missing or an index is out of range. Other errors throw what const - /// basic_json::operator[] throws. + /// missing or an index is out of range, or if this is a discarded view. + /// Other errors throw what const basic_json::operator[] throws. basic_json_view operator[](const json_pointer& ptr) const { + if (NLOHMANN_VIEW_UNLIKELY(is_discarded())) + { + return basic_json_view(); + } return detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::unchecked); } - /// the value of the member with this key (the first one, should the key + /// the value of the member with this key (the last one, should the key /// occur more than once). Throws type_error.304 if this is not an object, /// and out_of_range.403 if there is no such member. basic_json_view at(string_view_t key) const @@ -3060,9 +3099,12 @@ class basic_json_view return basic_json_view(m_doc, detail::view::element_at(m_node, idx)); } - basic_json_view at(int idx) const + /// any other integer type, see operator[]; negative values are out of + /// range + template < typename IntegerType, typename std::enable_if < detail::view::is_index_type::value, int >::type = 0 > + basic_json_view at(IntegerType idx) const { - return at(static_cast(idx)); + return at(detail::view::to_index(idx)); } /// the value a JSON pointer refers to; throws what basic_json::at() @@ -3073,7 +3115,7 @@ class basic_json_view } /// the member with this key converted to T, or the default value if there - /// is no such member (the first one, should the key occur more than + /// is no such member (the last one, should the key occur more than /// once). Throws type_error.306 if this is not an object. template < typename T, typename std::enable_if < !std::is_same::type, const char*>::value, int >::type = 0 > T value(string_view_t key, const T& default_value) const @@ -3138,7 +3180,7 @@ class basic_json_view // lookup // //////////// - /// an iterator to the member with this key (the first one, should the + /// an iterator to the member with this key (the last one, should the /// key occur more than once), or end(); end() also for non-objects iterator find(string_view_t key) const { @@ -3316,7 +3358,7 @@ class basic_json_view : m_doc(d), m_node(n) {} - /// the value of the first member with this key, or a discarded view + /// the value of the last member with this key, or a discarded view /// (object required) NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view lookup(string_view_t key) const noexcept { From beb83ae03b0b680f5c9f7ac52b6130b275151133 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:12:05 +0200 Subject: [PATCH 12/26] Copy const rvalue strings and document the accepted inputs json_document::parse(std::move(const_string)) failed to compile with "no matching read_kind": only a non-const rvalue std::string can be moved from. Treat a const rvalue as a copied byte container. parse() does not accept everything BasicJsonType::parse() does: a FILE* and pointers to or arrays of wide characters are rejected at compile time. List the supported inputs instead. Signed-off-by: Niels Lohmann --- .../docs/api/basic_json_document/parse.md | 29 ++++++++++++------- include/nlohmann/detail/view/input.hpp | 10 +++---- tests/src/unit-json_view.cpp | 13 +++++++++ 3 files changed, 37 insertions(+), 15 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json_document/parse.md b/docs/mkdocs/docs/api/basic_json_document/parse.md index 7c46e17ea..219634370 100644 --- a/docs/mkdocs/docs/api/basic_json_document/parse.md +++ b/docs/mkdocs/docs/api/basic_json_document/parse.md @@ -19,24 +19,33 @@ static basic_json_document parse(IteratorType first, IteratorType last, 1. Deserialize from a compatible input, borrowing or owning it depending on its value category and type (see Notes). 2. Deserialize from a pair of input iterators. -Both overloads accept exactly what [`BasicJsonType::parse()`](../basic_json/parse.md) accepts, with the same +Both overloads accept the same JSON text as [`BasicJsonType::parse()`](../basic_json/parse.md), with the same `ignore_comments`/`ignore_trailing_commas` options, but build a [`basic_json_document`](index.md) (a flat index into -the input) instead of a tree of `BasicJsonType` values. +the input) instead of a tree of `BasicJsonType` values. The input must be byte-oriented (see the template parameters +below): not every input type of `BasicJsonType::parse()` is supported. ## Template parameters `InputType` -: A compatible input, for instance: +: A byte-oriented input, one of: - - a `#!cpp std::string`, `#!cpp std::string_view`, or a C-style array of characters - - a pointer to a null-terminated string of single byte characters + - a `#!cpp std::string`, `#!cpp std::string_view`, or a C-style array of single-byte characters + - a pointer to a null-terminated string of single-byte characters (`#!cpp char`, `#!cpp signed char`, + `#!cpp unsigned char`, `#!cpp std::uint8_t`) - a container for which `#!cpp obj.data()` and `#!cpp obj.size()` give contiguous single-byte access, e.g. `#!cpp std::vector` or `#!cpp std::vector` - - an `#!cpp std::istream` object, or anything else [`BasicJsonType::parse()`](../basic_json/parse.md) accepts + - an `#!cpp std::istream` object + - a wide string object (`#!cpp std::wstring`, `#!cpp std::u16string`, `#!cpp std::u32string`), which is converted + to UTF-8 + + Other inputs are not supported: a `#!cpp FILE*`, and pointers to or arrays of wide characters (`#!cpp wchar_t`, + `#!cpp char16_t`, `#!cpp char32_t`) are rejected at compile time by a `#!cpp static_assert`. (Use + [`BasicJsonType::parse()`](../basic_json/parse.md) for these.) `IteratorType` -: a compatible iterator type, for instance a pair of pointers such as `ptr` and `ptr + len`, or a pair of - `#!cpp std::string::iterator` +: an input iterator type, for instance a pair of pointers such as `ptr` and `ptr + len`, or a pair of + `#!cpp std::string::iterator`; the iterators of single-byte characters are borrowed or read like the byte inputs + above, those of wide characters are converted to UTF-8 ## Parameters @@ -84,8 +93,8 @@ Linear in the length of the input. | `input` | ownership | |--------------------------------------------------------------------------------------|--------------------------------------------------------------| | lvalue byte container (`std::string`, `std::vector`, ...), `std::string_view`, C string, character array | **borrowed** -- `input` must outlive the document | -| rvalue `#!cpp std::string` | **owned**, moved in without a copy | -| rvalue byte container other than `#!cpp std::string` | **owned**, copied | +| non-const rvalue `#!cpp std::string` | **owned**, moved in without a copy | +| other rvalue byte container (including a `#!cpp const` rvalue `#!cpp std::string`) | **owned**, copied | | stream, wide string, or anything else read through the general input adapter | **owned**, read into a buffer (a stream is read to its end) | For overload (2), a pair of pointers to single-byte integers (e.g. `#!cpp const char*`, `#!cpp std::uint8_t*`) is diff --git a/include/nlohmann/detail/view/input.hpp b/include/nlohmann/detail/view/input.hpp index 3fafa2ecf..ac3257ee3 100644 --- a/include/nlohmann/detail/view/input.hpp +++ b/include/nlohmann/detail/view/input.hpp @@ -9,7 +9,7 @@ #pragma once #include // basic_string, char_traits, string -#include // decay, integral_constant, is_array, is_lvalue_reference, is_pointer, is_same, remove_reference +#include // decay, integral_constant, is_array, is_const, is_integral, is_lvalue_reference, is_pointer, is_same, remove_reference #include // forward #include @@ -28,12 +28,12 @@ namespace view /// how a document takes its input enum class input_kind { - move_string, ///< rvalue std::string: owned without a copy + move_string, ///< non-const rvalue std::string: owned without a copy c_string, ///< const char* (NUL-terminated): borrowed char_array, ///< char array (e.g. a string literal): borrowed borrow_range, ///< lvalue contiguous byte container, or std::string_view: borrowed - copy_range, ///< rvalue contiguous byte container: copied - adapter, ///< anything else parse() accepts (streams, wide strings, ...): read into a buffer + copy_range, ///< rvalue contiguous byte container (a const rvalue std::string too): copied + adapter, ///< streams, wide strings, and the rest of what the library's input adapter reads: read into a buffer }; template @@ -52,7 +52,7 @@ struct classify_input static constexpr input_kind value = std::is_array::value ? input_kind::char_array : std::is_pointer::value ? input_kind::c_string - : (is_rvalue && std::is_same::value) ? input_kind::move_string + : (is_rvalue && !std::is_const::value && std::is_same::value) ? input_kind::move_string : (is_bytes && (!is_rvalue || is_string_view)) ? input_kind::borrow_range : is_bytes ? input_kind::copy_range : input_kind::adapter; diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 9df459d3a..3b38e078f 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -314,6 +314,19 @@ TEST_CASE("json_view") CHECK(from_rvalue.owns_source()); CHECK(from_rvalue.root().materialize() == expected); CHECK(json_document::parse(std::vector(text.begin(), text.end())).owns_source()); + // a const rvalue cannot be moved from, and is not borrowed (it may be a + // temporary): it is copied, as is a const rvalue of any container + const std::string const_text = text; + const json_document from_const_rvalue = json_document::parse(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) + CHECK(from_const_rvalue.owns_source()); + CHECK(from_const_rvalue.source().data() != const_text.data()); + CHECK(from_const_rvalue.root().materialize() == expected); + json_document read_const_rvalue; + read_const_rvalue.read(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) + CHECK(read_const_rvalue.owns_source()); + CHECK(read_const_rvalue.root().materialize() == expected); + const std::vector const_chars(text.begin(), text.end()); + CHECK(json_document::parse(std::move(const_chars)).owns_source()); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) CHECK(json_document::parse_copy(text).owns_source()); CHECK(json_document::parse_copy(text).root().materialize() == expected); std::istringstream stream(text); From 7beac68df726f48c43fc5c1e41cb84005abef905 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:18:50 +0200 Subject: [PATCH 13/26] Remove the conversion to bool from json_view explicit operator bool meant "refers to a value", which silently differs from what a basic_json converts to. Use !v.is_discarded() instead. Signed-off-by: Niels Lohmann --- .../api/basic_json_view/basic_json_view.md | 5 +- docs/mkdocs/docs/api/basic_json_view/index.md | 1 - .../docs/api/basic_json_view/is_discarded.md | 6 +-- .../docs/api/basic_json_view/operator_bool.md | 46 ------------------- .../basic_json_view__basic_json_view.cpp | 4 +- .../basic_json_view__basic_json_view.output | 4 +- .../basic_json_view__type_predicates.cpp | 4 +- .../basic_json_view__type_predicates.output | 4 +- docs/mkdocs/mkdocs.yml | 1 - include/nlohmann/json_view.hpp | 6 --- tests/src/unit-json_view.cpp | 8 +++- 11 files changed, 19 insertions(+), 70 deletions(-) delete mode 100644 docs/mkdocs/docs/api/basic_json_view/operator_bool.md diff --git a/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md b/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md index 143ea2bea..bd10eae6f 100644 --- a/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md +++ b/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md @@ -4,8 +4,8 @@ basic_json_view() noexcept = default; ``` -Creates an invalid (discarded) view: [`type()`](type.md) is `#!cpp value_t::discarded`, -[`is_discarded()`](is_discarded.md) is `#!cpp true`, and `#!cpp explicit operator bool()` is `#!cpp false`. +Creates an invalid (discarded) view: [`type()`](type.md) is `#!cpp value_t::discarded` and +[`is_discarded()`](is_discarded.md) is `#!cpp true`. This is the only constructor a caller can use directly. Every other view is obtained from a [`basic_json_document`](../basic_json_document/index.md), via [`root()`](../basic_json_document/root.md) or (once @@ -43,7 +43,6 @@ placeholder for "no value yet" and later be assigned a real view. ## See also - [is_discarded](is_discarded.md) - return whether the view is invalid -- [operator bool](operator_bool.md) - return whether the view refers to a value - [root](../basic_json_document/root.md) - the view of a document's root value ## Version history diff --git a/docs/mkdocs/docs/api/basic_json_view/index.md b/docs/mkdocs/docs/api/basic_json_view/index.md index 564406667..8ce113ceb 100644 --- a/docs/mkdocs/docs/api/basic_json_view/index.md +++ b/docs/mkdocs/docs/api/basic_json_view/index.md @@ -62,7 +62,6 @@ element access, iteration, `get()`, JSON Pointer support, `dump()`, or compar - [**is_primitive**](is_primitive.md) - return whether the type is primitive - [**is_structured**](is_structured.md) - return whether the type is structured - [**is_discarded**](is_discarded.md) - return whether the view is invalid -- [**operator bool**](operator_bool.md) - return whether the view refers to a value ### Capacity diff --git a/docs/mkdocs/docs/api/basic_json_view/is_discarded.md b/docs/mkdocs/docs/api/basic_json_view/is_discarded.md index 4cc11077d..22210a5ad 100644 --- a/docs/mkdocs/docs/api/basic_json_view/is_discarded.md +++ b/docs/mkdocs/docs/api/basic_json_view/is_discarded.md @@ -23,8 +23,9 @@ Constant. ## Notes -`#!cpp v.is_discarded()` and `#!cpp !static_cast(v)` are equivalent; use whichever reads better at the call -site. +A `basic_json_view` is not convertible to `#!cpp bool`: such a conversion would mean "refers to a value", whereas +`basic_json` converts to the `#!cpp bool` it holds, so the same code would silently behave differently. Test +`#!cpp !v.is_discarded()` explicitly. ## Examples @@ -45,7 +46,6 @@ site. ## See also -- [operator bool](operator_bool.md) - return whether the view refers to a value - [(constructor)](basic_json_view.md) - the default constructor creates a discarded view - [is_discarded (basic_json_document)](../basic_json_document/is_discarded.md) - return whether the last parse failed - [`BasicJsonType::is_discarded`](../basic_json/is_discarded.md) - the corresponding function of `basic_json` diff --git a/docs/mkdocs/docs/api/basic_json_view/operator_bool.md b/docs/mkdocs/docs/api/basic_json_view/operator_bool.md deleted file mode 100644 index bc13c100c..000000000 --- a/docs/mkdocs/docs/api/basic_json_view/operator_bool.md +++ /dev/null @@ -1,46 +0,0 @@ -# nlohmann::basic_json_view::operator bool - -```cpp -explicit operator bool() const noexcept; -``` - -Returns whether this view refers to a value, i.e. the negation of [`is_discarded()`](is_discarded.md). Being -`#!cpp explicit`, this conversion is only considered in a boolean context (`#!cpp if (v)`, `#!cpp !v`, `#!cpp v && -...`), not for implicit conversions to other types. - -## Return value - -`#!cpp true` if the view refers to a value, `#!cpp false` if it is [discarded](is_discarded.md). - -## Exception safety - -No-throw guarantee: this function never throws exceptions. - -## Complexity - -Constant. - -## Examples - -??? example - - The example below classifies several parsed documents by the type of their root value, without materializing any - of them into a `BasicJsonType` value. - - ```cpp - --8<-- "examples/basic_json_view__type_predicates.cpp" - ``` - - Output: - - ```json - --8<-- "examples/basic_json_view__type_predicates.output" - ``` - -## See also - -- [is_discarded](is_discarded.md) - return whether the view is invalid - -## Version history - -- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp index af08f22e0..3d813717d 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp +++ b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp @@ -8,10 +8,10 @@ int main() // the default constructor is the only public one: it creates an invalid // (discarded) view, useful as a "no value yet" placeholder nlohmann::json_view v; - std::cout << static_cast(v) << ' ' << v.is_discarded() << '\n'; + std::cout << v.is_discarded() << '\n'; // views are trivially copyable handles (two pointers); the document owns // the actual data nlohmann::json_view copy = v; - std::cout << static_cast(copy) << '\n'; + std::cout << copy.is_discarded() << '\n'; } diff --git a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output index 8192d6145..bb101b641 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output +++ b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output @@ -1,2 +1,2 @@ -false true -false +true +true diff --git a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp index 1e3ed62e2..edc37ff52 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp +++ b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp @@ -35,8 +35,8 @@ int main() // parse without exceptions, are both discarded nlohmann::json_view invalid; json_document failed = json_document::parse("not json", /* allow_exceptions */ false); - std::cout << static_cast(invalid) << ' ' << invalid.is_discarded() << '\n'; - std::cout << static_cast(failed.root()) << ' ' << failed.root().is_discarded() << '\n'; + std::cout << invalid.is_discarded() << '\n'; + std::cout << failed.root().is_discarded() << '\n'; // type() returns the same value_t enumeration as basic_json::type() std::cout << (d_object.root().type() == nlohmann::json::value_t::object) << '\n'; diff --git a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output index 285d31ca0..99e264447 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output +++ b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output @@ -7,6 +7,6 @@ true true true true false false -false true -false true +true +true true diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 8324a303b..ce07e3d02 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -266,7 +266,6 @@ nav: - 'is_string': api/basic_json_view/is_string.md - 'is_structured': api/basic_json_view/is_structured.md - 'materialize': api/basic_json_view/materialize.md - - 'operator bool': api/basic_json_view/operator_bool.md - 'size': api/basic_json_view/size.md - 'source_offset': api/basic_json_view/source_offset.md - 'type': api/basic_json_view/type.md diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 178020cbb..66a48960e 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -156,12 +156,6 @@ class basic_json_view return type() == value_t::discarded; } - /// false for discarded views - explicit operator bool() const noexcept - { - return m_node != nullptr; - } - ////////////// // capacity // ////////////// diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 3b38e078f..4e598c617 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -41,6 +41,8 @@ template using accept_call_t = decltype(json_document::accept(std::declval()...)); template using read_call_t = decltype(std::declval().read(std::declval()...)); +template +using bool_conversion_t = decltype(static_cast(std::declval())); #if !defined(JSON_NOEXCEPTION) // the exception parse() throws for a text, or "" if it accepts it @@ -155,7 +157,6 @@ TEST_CASE("json_view") CHECK(v.is_primitive() == j.is_primitive()); CHECK(v.is_structured() == j.is_structured()); CHECK(!v.is_discarded()); - CHECK(static_cast(v)); CHECK(v.size() == j.size()); CHECK(v.empty() == j.empty()); CHECK(v.materialize() == j); @@ -163,7 +164,6 @@ TEST_CASE("json_view") const json_view invalid{}; CHECK(invalid.is_discarded()); - CHECK(!static_cast(invalid)); CHECK(invalid.type() == json::value_t::discarded); CHECK(invalid.size() == 0); CHECK(invalid.empty()); @@ -380,6 +380,10 @@ TEST_CASE("json_view") static_assert(is_detected::value, "read(ptr, bool) is valid"); static_assert(!is_detected::value, "read(ptr, len) must not compile"); + // json_view has no conversion to bool: unlike basic_json's, it would + // mean "exists", not "is not null"; use is_discarded() + static_assert(!is_detected::value, "json_view must not convert to bool"); + // the valid calls still work const char* const text = "[1]"; CHECK(json_document::parse(text, true).root().is_array()); From 14afaca083e6b8dbe7fa045066563ba062135c1c Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:22:39 +0200 Subject: [PATCH 14/26] Delete json_document::root() on temporary documents auto v = json_document::parse(text).root() compiled and left the view dangling. Delete the overload for rvalue documents, take a named document in the tests, and document the lifetime rule. Signed-off-by: Niels Lohmann --- .../docs/api/basic_json_document/root.md | 19 ++++- docs/mkdocs/docs/features/json_view.md | 3 + include/nlohmann/json_view.hpp | 5 +- tests/src/unit-json_view.cpp | 75 ++++++++++++++----- 4 files changed, 80 insertions(+), 22 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json_document/root.md b/docs/mkdocs/docs/api/basic_json_document/root.md index 85682dbcd..3ccedeb4a 100644 --- a/docs/mkdocs/docs/api/basic_json_document/root.md +++ b/docs/mkdocs/docs/api/basic_json_document/root.md @@ -1,10 +1,15 @@ # nlohmann::basic_json_document::root ```cpp -view_type root() const noexcept; +// (1) +view_type root() const& noexcept; + +// (2) +view_type root() const&& = delete; ``` -Returns a view of the root value of the document. +1. Returns a view of the root value of the document. +2. Deleted: the view of a temporary document would dangle. ## Return value @@ -21,6 +26,16 @@ Constant. ## Notes +**Lifetime.** A view refers into the document, so the document must outlive it. `root()` can therefore only be called +on a document that has a name (an lvalue); calling it on a temporary does not compile: + +```cpp +auto v = json_document::parse(text).root(); // error: the document is destroyed at the end of the statement + +auto doc = json_document::parse(text); // OK: keep the document alive +auto v = doc.root(); +``` + `root()` is a cheap handle into the document's index, not a copy of anything; call it as often as needed. The returned view is valid under the same conditions as any other view of the document -- see [Object inspection](../basic_json_view/index.md) -- in particular, it is invalidated by the next diff --git a/docs/mkdocs/docs/features/json_view.md b/docs/mkdocs/docs/features/json_view.md index 6b66c42ca..c6280ef62 100644 --- a/docs/mkdocs/docs/features/json_view.md +++ b/docs/mkdocs/docs/features/json_view.md @@ -76,6 +76,9 @@ Moving the document itself is fine and does **not** invalidate its views: the in that keeps its address across the move. Take a fresh view from [`root()`](../api/basic_json_document/root.md) whenever any of the other conditions above was not met. +Because a view dies with its document, [`root()`](../api/basic_json_document/root.md) is not callable on a temporary +document: `#!cpp auto v = json_document::parse(text).root();` does not compile. Give the document a name first. + ??? example "Example: borrowed and owned documents, and when views become invalid" ```cpp diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 66a48960e..e2cb57c76 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -358,7 +358,7 @@ class basic_json_document //////////// /// the root value (discarded if parsing failed without exceptions) - view_type root() const noexcept + view_type root() const& noexcept { if (!m_data || m_data->discarded) { @@ -367,6 +367,9 @@ class basic_json_document return view_type(m_data.get(), m_data->tape); } + /// deleted: the view of a temporary document would dangle + view_type root() const&& = delete; + bool is_discarded() const noexcept { return !m_data || m_data->discarded; diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 4e598c617..ffd44ca17 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -32,6 +32,22 @@ using nlohmann::ordered_json_document; namespace { +// the value of a text, through a named document: the views of a temporary +// document would dangle (root() of an rvalue document does not compile) +template +auto materialized(Args&& ... args) -> decltype(std::declval().materialize()) +{ + const Document d = Document::parse(std::forward(args)...); + return d.root().materialize(); +} + +template +auto materialized_copy(Input&& input) -> decltype(std::declval().materialize()) +{ + const Document d = Document::parse_copy(std::forward(input)); + return d.root().materialize(); +} + // detection of calls that must not compile template using parse_call_t = decltype(json_document::parse(std::declval()...)); @@ -41,6 +57,8 @@ template using accept_call_t = decltype(json_document::accept(std::declval()...)); template using read_call_t = decltype(std::declval().read(std::declval()...)); +template +using root_call_t = decltype(std::declval().root()); template using bool_conversion_t = decltype(static_cast(std::declval())); @@ -179,19 +197,19 @@ TEST_CASE("json_view") std::string text; g.value(text, 0); CAPTURE(text) - CHECK(json_document::parse(text).root().materialize() == json::parse(text)); + CHECK(materialized(text) == json::parse(text)); // member order as ordered_json::parse keeps it - CHECK(ordered_json_document::parse(text).root().materialize().dump() == ordered_json::parse(text).dump()); + CHECK(materialized(text).dump() == ordered_json::parse(text).dump()); } // duplicate keys: the last value, at the position of the first key - CHECK(json_document::parse(R"({"a":1,"b":2,"a":3})").root().materialize() == json::parse(R"({"a":1,"b":2,"a":3})")); - CHECK(ordered_json_document::parse(R"({"a":1,"b":2,"a":3})").root().materialize().dump() == R"({"a":3,"b":2})"); + CHECK(materialized(R"({"a":1,"b":2,"a":3})") == json::parse(R"({"a":1,"b":2,"a":3})")); + CHECK(materialized(R"({"a":1,"b":2,"a":3})").dump() == R"({"a":3,"b":2})"); // very deep nesting (iterative, as parse()) const std::string deep = std::string(100000, '[') + std::string(100000, ']'); - CHECK(json_document::parse(deep).root().materialize() == json::parse(deep)); + CHECK(materialized(deep) == json::parse(deep)); #if JSON_DIAGNOSTICS // the parents are set, so errors name the path - const json m = json_document::parse(R"({"a":{"b":[1]}})").root().materialize(); + const json m = materialized(R"({"a":{"b":[1]}})"); CHECK_THROWS_WITH_AS(m.at("a").at("b").at(0).at("x"), "[json.exception.type_error.304] (/a/b/0) cannot use at() with number", json::type_error&); #endif } @@ -264,7 +282,7 @@ TEST_CASE("json_view") CHECK(json_document::accept(text)); if (accepted) { - CHECK(float_document::parse(text).root().materialize() == json_float::parse(text)); + CHECK(materialized(text) == json_float::parse(text)); } } float_document f; @@ -278,7 +296,7 @@ TEST_CASE("json_view") CHECK(json_document::accept(with_nul) == json::accept(with_nul)); const std::string nul_in_comment("[1, // c\0\n2]", 12); CHECK(json_document::accept(nul_in_comment, true) == json::accept(nul_in_comment, true)); - CHECK(json_document::parse("\xEF\xBB\xBF[1]").root().materialize() == json::parse("\xEF\xBB\xBF[1]")); + CHECK(materialized("\xEF\xBB\xBF[1]") == json::parse("\xEF\xBB\xBF[1]")); #if !defined(JSON_NOEXCEPTION) CHECK(view_exception("\xEF\xBB") == parse_exception("\xEF\xBB")); #endif @@ -294,18 +312,18 @@ TEST_CASE("json_view") CHECK(!borrowed.owns_source()); CHECK(borrowed.source().data() == text.data()); CHECK(borrowed.root().materialize() == expected); - CHECK(json_document::parse(text.c_str()).root().materialize() == expected); - CHECK(json_document::parse(R"([1, "two", {"three": 3.5}])").root().materialize() == expected); - CHECK(json_document::parse(text.data(), text.data() + text.size()).root().materialize() == expected); + CHECK(materialized(text.c_str()) == expected); + CHECK(materialized(R"([1, "two", {"three": 3.5}])") == expected); + CHECK(materialized(text.data(), text.data() + text.size()) == expected); const std::vector chars(text.begin(), text.end()); CHECK(!json_document::parse(chars).owns_source()); - CHECK(json_document::parse(chars).root().materialize() == expected); + CHECK(materialized(chars) == expected); const std::vector bytes(text.begin(), text.end()); - CHECK(json_document::parse(bytes).root().materialize() == expected); + CHECK(materialized(bytes) == expected); #ifdef JSON_HAS_CPP_17 const std::string_view sv = text; CHECK(!json_document::parse(sv).owns_source()); - CHECK(json_document::parse(sv).root().materialize() == expected); + CHECK(materialized(sv) == expected); #endif // owned @@ -328,14 +346,14 @@ TEST_CASE("json_view") const std::vector const_chars(text.begin(), text.end()); CHECK(json_document::parse(std::move(const_chars)).owns_source()); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) CHECK(json_document::parse_copy(text).owns_source()); - CHECK(json_document::parse_copy(text).root().materialize() == expected); + CHECK(materialized_copy(text) == expected); std::istringstream stream(text); const json_document from_stream = json_document::parse(stream); CHECK(from_stream.owns_source()); CHECK(from_stream.root().materialize() == expected); const std::list list(text.begin(), text.end()); CHECK(json_document::parse(list.begin(), list.end()).owns_source()); - CHECK(json_document::parse(list.begin(), list.end()).root().materialize() == expected); + CHECK(materialized(list.begin(), list.end()) == expected); // iterator pairs: pointers are borrowed, and so are contiguous library // iterators where the input adapter detects them (C++20) @@ -346,10 +364,10 @@ TEST_CASE("json_view") CHECK((from_iterators.source().data() == chars.data()) == contiguous); CHECK(from_iterators.root().materialize() == expected); const std::string padded = "x" + text + "x"; - CHECK(json_document::parse(padded.begin() + 1, padded.end() - 1).root().materialize() == expected); + CHECK(materialized(padded.begin() + 1, padded.end() - 1) == expected); CHECK(json_document::parse(chars.cbegin(), chars.cbegin(), false).is_discarded()); const std::wstring wide = L"[\"\u00e4\u20ac\", 1]"; - CHECK(json_document::parse(wide).root().materialize() == json::parse(wide)); + CHECK(materialized(wide) == json::parse(wide)); CHECK(json_document::parse(static_cast(nullptr), false).is_discarded()); CHECK(json_document::parse("", false).is_discarded()); } @@ -386,10 +404,29 @@ TEST_CASE("json_view") // the valid calls still work const char* const text = "[1]"; - CHECK(json_document::parse(text, true).root().is_array()); + CHECK(materialized(text, true) == json::parse(text)); CHECK(json_document::accept(text, true, true)); } + SECTION("root of a temporary document does not compile") + { + using nlohmann::detail::is_detected; + + // the view would dangle: auto v = json_document::parse(text).root(); + static_assert(is_detected::value, "root() of an lvalue is valid"); + static_assert(is_detected::value, "root() of a const lvalue is valid"); + static_assert(!is_detected::value, "root() of an rvalue must not compile"); + static_assert(!is_detected < root_call_t, json_document && >::value, "root() of an rvalue must not compile"); + static_assert(!is_detected < root_call_t, const json_document && >::value, "root() of a const rvalue must not compile"); + static_assert(!is_detected::value, "root() of an rvalue must not compile"); + + // a named document is fine, also after a move + json_document d = json_document::parse("[1]"); + CHECK(d.root().size() == 1); + const json_document moved = std::move(d); + CHECK(moved.root().size() == 1); + } + SECTION("document lifetime and reuse") { json_document d; From 320a463af72b2a724a2b588b0ae2384c457b1dbf Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:26:32 +0200 Subject: [PATCH 15/26] Store node integers by halves on big-endian targets The non-little-endian path of the node writer wrote the integer's native word over len and next, so len got the high half there. Compose and split the value explicitly (len is the low half, next the high half); the little-endian path stays a plain memcpy. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/view/builder.hpp | 5 +- include/nlohmann/detail/view/node.hpp | 12 ++++- tests/src/unit-json_view_builder.cpp | 58 ++++++++++++++++++++++++ 3 files changed, 73 insertions(+), 2 deletions(-) diff --git a/include/nlohmann/detail/view/builder.hpp b/include/nlohmann/detail/view/builder.hpp index 1cd01fef6..5ce68220d 100644 --- a/include/nlohmann/detail/view/builder.hpp +++ b/include/nlohmann/detail/view/builder.hpp @@ -801,7 +801,10 @@ indent_done: n->flags = flags; n->extra = extra; n->off = static_cast(off); - set_integer_bits(*n, second); + // len is the low half of the second word, next the high half + // (not a native word over both, which swaps them on big-endian) + n->len = static_cast(second); + n->next = static_cast(second >> 32); #endif return n; } diff --git a/include/nlohmann/detail/view/node.hpp b/include/nlohmann/detail/view/node.hpp index 471c1f844..5818066ae 100644 --- a/include/nlohmann/detail/view/node.hpp +++ b/include/nlohmann/detail/view/node.hpp @@ -57,17 +57,27 @@ NLOHMANN_VIEW_ALWAYS_INLINE bool is_container(const node& n) noexcept return static_cast(n.kind) - 1u <= 1u; } -/// the converted value of an integer node (stored in len/next) +/// the converted value of an integer node: len is its low half, next its high +/// half (on little-endian targets the two words are the value in memory) NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept { +#if NLOHMANN_VIEW_LITTLE_ENDIAN std::uint64_t v = 0; std::memcpy(&v, reinterpret_cast(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) return v; +#else + return static_cast(n.len) | (static_cast(n.next) << 32); +#endif } NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept { +#if NLOHMANN_VIEW_LITTLE_ENDIAN std::memcpy(reinterpret_cast(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) +#else + n.len = static_cast(v); + n.next = static_cast(v >> 32); +#endif } /// token length of a number node diff --git a/tests/src/unit-json_view_builder.cpp b/tests/src/unit-json_view_builder.cpp index 9d3ea943b..c585eb919 100644 --- a/tests/src/unit-json_view_builder.cpp +++ b/tests/src/unit-json_view_builder.cpp @@ -401,3 +401,61 @@ TEST_CASE("json_view builder") } } } + +TEST_CASE("json_view node integer bits") +{ + using nlohmann::detail::view::integer_bits; + using nlohmann::detail::view::set_integer_bits; + + // an integer lives in len (low half) and next (high half), on any byte + // order; a big-endian target must not store the native word over both + SECTION("set_integer_bits and integer_bits") + { + node n = {}; + for (const std::uint64_t v : + { + std::uint64_t{0}, std::uint64_t{1}, std::uint64_t{0xFFFFFFFFu}, std::uint64_t{0x100000000u}, + std::uint64_t{0x0000000200000003u}, std::uint64_t{0x0123456789ABCDEFu}, std::uint64_t{0xFFFFFFFFFFFFFFFEu} + }) + { + CAPTURE(v) + n.kind = 0x5A; + n.flags = 0xA5; + n.extra = 0x1234; + n.off = 0x89ABCDEFu; + set_integer_bits(n, v); + CHECK(integer_bits(n) == v); + CHECK(n.len == static_cast(v)); + CHECK(n.next == static_cast(v >> 32)); + // the other fields are untouched + CHECK(n.kind == 0x5A); + CHECK(n.flags == 0xA5); + CHECK(n.extra == 0x1234); + CHECK(n.off == 0x89ABCDEFu); + } + } + + SECTION("parsed integers") + { + struct integer_case + { + const char* text; + std::uint64_t bits; + }; + for (const integer_case c : + { + integer_case{"[8589934595]", 0x0000000200000003u}, integer_case{"[4294967296]", 0x100000000u}, integer_case{"[4294967295]", 0xFFFFFFFFu}, + integer_case{"[-2]", 0xFFFFFFFFFFFFFFFEu}, integer_case{"[-4294967297]", 0xFFFFFFFEFFFFFFFFu}, integer_case{"[18446744073709551615]", 0xFFFFFFFFFFFFFFFFu}, + integer_case{"[7]", 7u} + }) + { + CAPTURE(c.text) + const built b = build(c.text, false, false, true); + REQUIRE(b.ok); + const node& n = b.data->tape[1]; + CHECK(integer_bits(n) == c.bits); + CHECK(n.len == static_cast(c.bits)); + CHECK(n.next == static_cast(c.bits >> 32)); + } + } +} From a1b5b348abcf1c9a8eb770689a19a058eb836fec Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:28:19 +0200 Subject: [PATCH 16/26] Check the node array size for overflow reserve(n) and the growth of the index computed n * sizeof(node) without a check, which wraps around on 32-bit targets for inputs of about 1 GiB and allocates a too small array. Throw std::bad_alloc for a count beyond the address space and clamp the growth step to it. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/view/builder.hpp | 14 +++++++-- .../nlohmann/detail/view/document_data.hpp | 22 ++++++++++++-- tests/src/unit-json_view_builder.cpp | 30 +++++++++++++++++++ 3 files changed, 62 insertions(+), 4 deletions(-) diff --git a/include/nlohmann/detail/view/builder.hpp b/include/nlohmann/detail/view/builder.hpp index 5ce68220d..ccb54c3d7 100644 --- a/include/nlohmann/detail/view/builder.hpp +++ b/include/nlohmann/detail/view/builder.hpp @@ -9,7 +9,7 @@ #pragma once -#include // find, find_if, max +#include // find, find_if, max, min #include // array #include // size_t, ptrdiff_t #include // int64_t, uint8_t, uint16_t, uint32_t, uint64_t @@ -195,8 +195,18 @@ class builder const std::uint64_t done = static_cast(at - b) + 1; const std::uint64_t guess = static_cast(n) * static_cast(e - b + 1) / done; const std::uint64_t grown = guess + (guess / 4) + 64; // a variable: GCC calls a cast of the sum useless where std::uint64_t is std::size_t + // (n is below 2^32: the input is smaller than 4 GiB; the sum cannot wrap) + const std::uint64_t wanted = (std::max)(grown, static_cast(n) + (n / 2) + 64); + const std::uint64_t limit = document_data::max_nodes(); doc.tape_size = n; - doc.reserve((std::max)(static_cast(grown), n + (n / 2) + 64)); + // LCOV_EXCL_START (a node array that fills the address space) + if (NLOHMANN_VIEW_UNLIKELY(n >= limit)) + { + document_data::throw_bad_alloc(); // no room for another node + } + // LCOV_EXCL_STOP + // (a count beyond the limit is cut: the index does not grow beyond what can be addressed) + doc.reserve(static_cast((std::min)(wanted, limit))); return doc.tape; } diff --git a/include/nlohmann/detail/view/document_data.hpp b/include/nlohmann/detail/view/document_data.hpp index 66275702c..aa3958e41 100644 --- a/include/nlohmann/detail/view/document_data.hpp +++ b/include/nlohmann/detail/view/document_data.hpp @@ -11,7 +11,8 @@ #include // array #include // size_t #include // memcpy -#include // operator new, placement new +#include // numeric_limits +#include // bad_alloc, operator new, placement new #include // string #include @@ -84,13 +85,30 @@ struct document_data tape_cap = inline_cap; } - /// make room for n nodes; keeps the first tape_size nodes + /// the largest node count whose size in bytes fits a std::size_t + static constexpr std::size_t max_nodes() noexcept + { + return (std::numeric_limits::max)() / sizeof(node); + } + + [[noreturn]] NLOHMANN_VIEW_NOINLINE static void throw_bad_alloc() + { + NLOHMANN_VIEW_THROW(std::bad_alloc()); + } + + /// make room for n nodes; keeps the first tape_size nodes (throws + /// std::bad_alloc for a count that does not fit the address space, + /// instead of wrapping around in n * sizeof(node)) void reserve(std::size_t n) { if (n <= tape_cap) { return; } + if (NLOHMANN_VIEW_UNLIKELY(n > max_nodes())) + { + throw_bad_alloc(); + } node* fresh = static_cast(::operator new (n * sizeof(node))); if (tape_size != 0) { diff --git a/tests/src/unit-json_view_builder.cpp b/tests/src/unit-json_view_builder.cpp index c585eb919..8b8518c38 100644 --- a/tests/src/unit-json_view_builder.cpp +++ b/tests/src/unit-json_view_builder.cpp @@ -19,8 +19,10 @@ using nlohmann::json; #include #include +#include #include #include +#include #include #include #include @@ -459,3 +461,31 @@ TEST_CASE("json_view node integer bits") } } } + +TEST_CASE("json_view node array size limit") +{ + // a node array larger than the address space is refused, not wrapped to a + // small allocation (the size computation overflows on 32-bit targets, and + // for absurd counts everywhere) + std::unique_ptr d(document_data::create(0)); + d->reserve(8); + REQUIRE(d->tape_cap >= 8); + d->tape_size = 2; + const std::size_t cap = d->tape_cap; + node* const tape = d->tape; + +#if !defined(JSON_NOEXCEPTION) + const std::size_t too_many = document_data::max_nodes() + 1; + CHECK_THROWS_AS(d->reserve(too_many), std::bad_alloc&); + CHECK_THROWS_AS(d->reserve((std::numeric_limits::max)()), std::bad_alloc&); + // the array is unchanged + CHECK(d->tape == tape); + CHECK(d->tape_cap == cap); + CHECK(d->tape_size == 2); +#endif + + // the largest count that fits is not refused by the check (nothing is + // allocated for a count that is already there) + d->reserve(cap); + CHECK(d->tape == tape); +} From 078903cbb92b1acfe3a760ed348f038d4cefb66a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:30:00 +0200 Subject: [PATCH 17/26] State the exact input size limit of json_document The check rejects inputs of 0xFFFFFFF0 bytes or more, but the exception message and the documentation said 4 GiB. Name the limit once (max_input_size), and state 4 GiB minus 16 bytes in the message and the documentation. Test the limit with a container that only claims the size. Signed-off-by: Niels Lohmann --- .../docs/api/basic_json_document/parse.md | 4 +-- docs/mkdocs/docs/features/json_view.md | 2 +- docs/mkdocs/docs/home/architecture.md | 3 +- docs/mkdocs/docs/home/exceptions.md | 4 +-- include/nlohmann/detail/view/errors.hpp | 5 ++- include/nlohmann/detail/view/node.hpp | 5 +++ include/nlohmann/json_view.hpp | 4 +-- tests/src/unit-json_view.cpp | 35 +++++++++++++++++++ 8 files changed, 51 insertions(+), 11 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json_document/parse.md b/docs/mkdocs/docs/api/basic_json_document/parse.md index 219634370..972c63752 100644 --- a/docs/mkdocs/docs/api/basic_json_document/parse.md +++ b/docs/mkdocs/docs/api/basic_json_document/parse.md @@ -79,8 +79,8 @@ discarded; see [`is_discarded`](is_discarded.md). Throws the same exception [`BasicJsonType::parse()`](../basic_json/parse.md) throws for the same input and options -- the same exception id, message, and position -- because on a failing input the library's own parser is run on the same bytes to produce the diagnostic. Additionally throws -[`out_of_range.416`](../../home/exceptions.md#jsonexceptionout_of_range416) if the input is 4 GiB or larger, a size -[`BasicJsonType::parse()`](../basic_json/parse.md) does not reject. +[`out_of_range.416`](../../home/exceptions.md#jsonexceptionout_of_range416) if the input is 4294967280 bytes (4 GiB +minus 16 bytes) or larger, a size [`BasicJsonType::parse()`](../basic_json/parse.md) does not reject. ## Complexity diff --git a/docs/mkdocs/docs/features/json_view.md b/docs/mkdocs/docs/features/json_view.md index c6280ef62..6b49ee2e3 100644 --- a/docs/mkdocs/docs/features/json_view.md +++ b/docs/mkdocs/docs/features/json_view.md @@ -111,7 +111,7 @@ document: `#!cpp auto v = json_document::parse(text).root();` does not compile. - **Only 64-bit integers.** `basic_json_document` requires `BasicJsonType::number_integer_t` and `number_unsigned_t` to both be 64 bits wide; this is a compile-time `#!cpp static_assert`. -- **A 4 GiB input limit.** An input of 4 GiB or more throws +- **A 4 GiB input limit.** An input of 4294967280 bytes (4 GiB minus 16 bytes) or more throws [`out_of_range.416`](../home/exceptions.md#jsonexceptionout_of_range416), a limit `#!cpp basic_json::parse()` does not have. - **A stream is always read to its end.** There is no partial/streaming read of an `#!cpp std::istream`. diff --git a/docs/mkdocs/docs/home/architecture.md b/docs/mkdocs/docs/home/architecture.md index 457f0d536..f7e6c4f25 100644 --- a/docs/mkdocs/docs/home/architecture.md +++ b/docs/mkdocs/docs/home/architecture.md @@ -210,7 +210,8 @@ packet-beta - **Navigation** needs no pointers: the elements of an array or object follow its node, and the node after a value's subtree is `next` nodes further for an array or object, and the next node otherwise (`document_data::after`). Views step from element to element this way and skip whole subtrees in constant time. -- **Offsets** are 32 bits wide, so a document is limited to 4 GiB (`out_of_range.416`). +- **Offsets** are 32 bits wide, so a document is limited to 4294967279 bytes, 4 GiB minus 16 bytes (a margin below + 2^32 for positions one scanner step past the end of the text; `out_of_range.416`). For example, `#!json {"a": [1, 2.5]}` becomes five nodes. Each node's elements follow it, and `next` leads from an array or object past its subtree: diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index 56c3cf71e..e5f8ed45b 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -1048,12 +1048,12 @@ MessagePack's ext type and BSON's binary subtype are each stored in a single byt [`basic_json_document::parse()`](../api/basic_json_document/parse.md) and the other parsing functions of [`basic_json_document`](../api/basic_json_document/index.md) index a value's position in the source text in 32 bits, -so they do not support an input of 4 GiB or more. +so they do not support an input of 4294967280 bytes (4 GiB minus 16 bytes) or more. !!! failure "Example message" ``` - [json.exception.out_of_range.416] input of 4 GiB or more is not supported by json_document + [json.exception.out_of_range.416] input of 4294967280 bytes or more is not supported by json_document ``` !!! note diff --git a/include/nlohmann/detail/view/errors.hpp b/include/nlohmann/detail/view/errors.hpp index ff5816efd..8d67a3973 100644 --- a/include/nlohmann/detail/view/errors.hpp +++ b/include/nlohmann/detail/view/errors.hpp @@ -55,9 +55,8 @@ template { if (f.code == error_code::input_too_large) { - // LCOV_EXCL_START (4 GiB) - NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4 GiB or more is not supported by json_document", nullptr)); - // LCOV_EXCL_STOP + // (the limit is detail::view::max_input_size: 4 GiB minus 16 bytes) + NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4294967280 bytes or more is not supported by json_document", nullptr)); } const BasicJsonType accepted = BasicJsonType::parse(src, src + size, nullptr, true, ignore_comments, ignore_trailing_commas); // LCOV_EXCL_START (only if parse() accepts what the view rejects: a bug) diff --git a/include/nlohmann/detail/view/node.hpp b/include/nlohmann/detail/view/node.hpp index 5818066ae..b36fc8c2a 100644 --- a/include/nlohmann/detail/view/node.hpp +++ b/include/nlohmann/detail/view/node.hpp @@ -29,6 +29,11 @@ static_assert(static_cast(value_t::null) == 0 && static_cast(value_t::number_unsigned) == 6 && static_cast(value_t::number_float) == 7, "the node format depends on the numbering of value_t"); +/// The largest input a document accepts, in bytes. Offsets and node counts are +/// 32 bits wide; the limit keeps 16 bytes (the width of the scanner's steps) +/// below 2^32, so that a position one step past the end of the text fits. +static constexpr std::size_t max_input_size = 0xFFFFFFEFu; + /// node flags struct node_flags { diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index e2cb57c76..923895792 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -475,9 +475,9 @@ class basic_json_document d.discarded = true; detail::view::parse_failure failure; bool ok = false; - if (NLOHMANN_VIEW_UNLIKELY(size >= 0xFFFFFFF0u)) + if (NLOHMANN_VIEW_UNLIKELY(size > detail::view::max_input_size)) { - failure.code = detail::view::error_code::input_too_large; // LCOV_EXCL_LINE (4 GiB) + failure.code = detail::view::error_code::input_too_large; } else { diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index ffd44ca17..ad00cab14 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -48,6 +48,25 @@ auto materialized_copy(Input&& input) -> decltype(std::declval using parse_call_t = decltype(json_document::parse(std::declval()...)); @@ -427,6 +446,22 @@ TEST_CASE("json_view") CHECK(moved.root().size() == 1); } + SECTION("input size limit") + { + // 32-bit offsets: the limit is 4 GiB minus 16 bytes (a margin below 2^32), + // which is what the exception message and the documentation say + const std::size_t limit = nlohmann::detail::view::max_input_size; + CHECK(limit == std::size_t{4294967279u}); + + const oversized_input input{limit + 1}; + CHECK(!json_document::accept(input)); + CHECK(json_document::parse(input, false).is_discarded()); +#if !defined(JSON_NOEXCEPTION) + json_document d; + CHECK_THROWS_WITH_AS(d = json_document::parse(input), "[json.exception.out_of_range.416] input of 4294967280 bytes or more is not supported by json_document", json::out_of_range&); +#endif + } + SECTION("document lifetime and reuse") { json_document d; From e3dafe1dbacd5f09bd067eaec0de7913167fc5e7 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:30:12 +0200 Subject: [PATCH 18/26] Remove the unused legacy_discarded_value_comparison from abi_config Nothing reads it: the view does not compare values. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/abi_config.hpp | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/include/nlohmann/detail/abi_config.hpp b/include/nlohmann/detail/abi_config.hpp index 0e99c81ad..510a67385 100644 --- a/include/nlohmann/detail/abi_config.hpp +++ b/include/nlohmann/detail/abi_config.hpp @@ -15,10 +15,11 @@ namespace detail { /*! -@brief the configuration macros that change the library's behavior +@brief the configuration macros that json_view.hpp reads json.hpp undefines these macros at its end (see macro_unscope.hpp), so code -that builds on the library after it (json_view.hpp) reads them here. Like the +that builds on the library after it (json_view.hpp) reads them here. A macro +is added when the view starts to depend on it. Like the macros, they are part of the ABI namespace, so they always match the basic_json they are used with. */ @@ -26,8 +27,6 @@ struct abi_config { /// JSON_STRICT_NUL_HANDLING: a null byte is an error, not the end of input static constexpr bool strict_nul_handling = JSON_STRICT_NUL_HANDLING != 0; - /// JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON - static constexpr bool legacy_discarded_value_comparison = JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON != 0; }; } // namespace detail From 0c76b316fcfad377df88d0a0b00d32bc17bb2e34 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:30:24 +0200 Subject: [PATCH 19/26] Align the banner comment of json_view.hpp Signed-off-by: Niels Lohmann --- include/nlohmann/json_view.hpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 923895792..90ac019ea 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -16,7 +16,7 @@ * the read-only part of the basic_json interface; materialize() turns a * * subtree into the basic_json value that parse() would produce. * * * - * The source text must outlive a document that borrows it (lvalue byte * + * The source text must outlive a document that borrows it (lvalue byte * * containers, C strings); rvalue strings, streams, and other inputs are * * owned by the document. * \****************************************************************************/ From aa7f0b02a01c258351d2d57af84b072b42bb7db4 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:32:20 +0200 Subject: [PATCH 20/26] Fuzz json_document with exact-size buffers and parse options The fuzzer only parsed a std::string with default options. Also parse an exact-size byte vector, which has no NUL after its last byte and takes the bounds-checked path, and derive ignore_comments and ignore_trailing_commas from the first input byte, comparing against json::parse and json::accept with the same options. Signed-off-by: Niels Lohmann --- tests/fuzzing.md | 5 +- tests/src/fuzzer-parse_json_view.cpp | 123 +++++++++++++++++++-------- 2 files changed, 93 insertions(+), 35 deletions(-) diff --git a/tests/fuzzing.md b/tests/fuzzing.md index de2d66716..ca305cf48 100644 --- a/tests/fuzzing.md +++ b/tests/fuzzing.md @@ -6,7 +6,10 @@ Each parser of the library (JSON, BJData, BON8, BSON, CBOR, MessagePack, and UBJ Additionally, `parse_json_view_fuzzer` (`tests/src/fuzzer-parse_json_view.cpp`) cross-checks `json_document`/`json_view` (the zero-copy, read-only view declared in `json_view.hpp`) against `basic_json` on the same JSON text: it asserts that `json_document::accept` agrees with `json::accept`, that an accepted input materializes to the same value `json::parse` -produces, and that a rejected input makes both parsers throw with an identical `what()`. It takes plain JSON text, so it +produces, and that a rejected input makes both parsers throw with an identical `what()`. It checks this for a +`std::string` input (borrowed, with a NUL after the last byte) and for an exact-size `std::vector` +(borrowed, with nothing after the last byte), and for the `ignore_comments` and `ignore_trailing_commas` options, which +are taken from the low bits of the first input byte (the byte stays part of the text). It takes plain JSON text, so it reuses the `corpus_json` corpus rather than a format of its own. ## What the fuzzers check diff --git a/tests/src/fuzzer-parse_json_view.cpp b/tests/src/fuzzer-parse_json_view.cpp index 59e32a556..9b9f5a02a 100644 --- a/tests/src/fuzzer-parse_json_view.cpp +++ b/tests/src/fuzzer-parse_json_view.cpp @@ -9,7 +9,9 @@ /* This file implements a parser test suitable for fuzz testing. It checks that json_document (the zero-copy, read-only view of a parsed JSON text declared in -json_view.hpp) agrees with basic_json on every input: +json_view.hpp) agrees with basic_json on every input, for the parse options +selected by the low bits of the first input byte (bit 0: ignore_comments, bit 1: +ignore_trailing_commas; the byte stays part of the text): - json_document::accept(data) must equal json::accept(data) - if the input is accepted, json_document::parse(data).root().materialize() @@ -18,12 +20,20 @@ json_view.hpp) agrees with basic_json on every input: enabled) must throw a json::parse_error or json::out_of_range whose what() is identical to the one json::parse(data) throws +This is checked for two kinds of input: a std::string, which the document +borrows and which ends in the NUL the parser uses as sentinel, and an +exact-size byte vector, which has no NUL after its last byte and takes the +parser's bounds-checked path (AddressSanitizer reports any read past the end). + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ #include +#include +#include #include +#include #include #include @@ -35,65 +45,110 @@ drivers. using json = nlohmann::json; using json_document = nlohmann::json_document; -// see http://llvm.org/docs/LibFuzzer.html -extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) +namespace { - // json_document::accept only has a single-argument overload; wrap the raw - // bytes in a (borrowed) std::string so the same bytes can be handed to it - const std::string input(reinterpret_cast(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) +// what json::parse does with a text: the value, or the message of the exception +struct reference_result +{ + bool accepted = false; + json value{}; + std::string what{}; // NOLINT(readability-redundant-member-init) +}; - const bool accepted_by_json = json::accept(data, data + size); - const bool accepted_by_view = json_document::accept(input); +reference_result parse_reference(const std::uint8_t* data, std::size_t size, bool comments, bool trailing_commas) +{ + reference_result r; + r.accepted = json::accept(data, data + size, comments, trailing_commas); + bool json_threw = false; + try + { + r.value = json::parse(data, data + size, nullptr, true, comments, trailing_commas); + } + catch (const json::parse_error& e) + { + r.what = e.what(); + json_threw = true; + } + catch (const json::out_of_range& e) + { + r.what = e.what(); + json_threw = true; + } + // json::accept and json::parse must agree + assert(json_threw == !r.accepted); + static_cast(json_threw); + return r; +} +// json_document must agree with the reference for this input (a container +// that json_document::parse borrows) +template +void check_input(const Input& input, const reference_result& expected, bool comments, bool trailing_commas) +{ // json_document::accept must agree with json::accept on every input - assert(accepted_by_json == accepted_by_view); + const bool accepted_by_view = json_document::accept(input, comments, trailing_commas); + assert(expected.accepted == accepted_by_view); + static_cast(accepted_by_view); - if (accepted_by_json) + if (expected.accepted) { // both parsers must agree on the resulting value - json const j1 = json::parse(data, data + size); - json_document const doc = json_document::parse(input); + json_document const doc = json_document::parse(input, true, comments, trailing_commas); + assert(!doc.is_discarded()); json const j2 = doc.root().materialize(); - assert(j1 == j2); + assert(expected.value == j2); + static_cast(j2); + + // (without exceptions, the same document) + json_document const quiet = json_document::parse(input, false, comments, trailing_commas); + assert(!quiet.is_discarded()); + assert(quiet.node_count() == doc.node_count()); } else { // both parsers must reject the input the same way when exceptions are used - std::string expected_what; - bool json_threw = false; - try - { - static_cast(json::parse(data, data + size)); - } - catch (const json::parse_error& e) - { - expected_what = e.what(); - json_threw = true; - } - catch (const json::out_of_range& e) - { - expected_what = e.what(); - json_threw = true; - } - assert(json_threw); - bool view_threw = false; try { - static_cast(json_document::parse(input)); + static_cast(json_document::parse(input, true, comments, trailing_commas)); } catch (const json::parse_error& e) { - assert(e.what() == expected_what); + assert(e.what() == expected.what); view_threw = true; } catch (const json::out_of_range& e) { - assert(e.what() == expected_what); + assert(e.what() == expected.what); view_threw = true; } assert(view_threw); + static_cast(view_threw); + + // and without exceptions, the document is discarded + json_document const quiet = json_document::parse(input, false, comments, trailing_commas); + assert(quiet.is_discarded()); + static_cast(quiet); } +} +} // namespace + +// see http://llvm.org/docs/LibFuzzer.html +extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) +{ + // the parse options are taken from the low bits of the first byte + const bool comments = size > 0 && (data[0] & 1U) != 0; + const bool trailing_commas = size > 0 && (data[0] & 2U) != 0; + + const reference_result expected = parse_reference(data, size, comments, trailing_commas); + + // a std::string: borrowed, with the NUL of std::string as sentinel + const std::string input(reinterpret_cast(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + check_input(input, expected, comments, trailing_commas); + + // an exact-size byte vector: borrowed, with nothing after its last byte + const std::vector exact(data, data + size); + check_input(exact, expected, comments, trailing_commas); // return 0 - non-zero return values are reserved for future use return 0; From 417c187d1a1a6faad48b199136bbc49098a06487 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:35:53 +0200 Subject: [PATCH 21/26] Regenerate the amalgamated headers Signed-off-by: Niels Lohmann --- single_include/nlohmann/json.hpp | 7 +- single_include/nlohmann/json_view.hpp | 117 ++++++++++++++++++++------ 2 files changed, 96 insertions(+), 28 deletions(-) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 079a8f6ed..e4fd9eeb8 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7511,10 +7511,11 @@ namespace detail { /*! -@brief the configuration macros that change the library's behavior +@brief the configuration macros that json_view.hpp reads json.hpp undefines these macros at its end (see macro_unscope.hpp), so code -that builds on the library after it (json_view.hpp) reads them here. Like the +that builds on the library after it (json_view.hpp) reads them here. A macro +is added when the view starts to depend on it. Like the macros, they are part of the ABI namespace, so they always match the basic_json they are used with. */ @@ -7522,8 +7523,6 @@ struct abi_config { /// JSON_STRICT_NUL_HANDLING: a null byte is an error, not the end of input static constexpr bool strict_nul_handling = JSON_STRICT_NUL_HANDLING != 0; - /// JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON - static constexpr bool legacy_discarded_value_comparison = JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON != 0; }; } // namespace detail diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 6e2acab56..3abcec76d 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -16,7 +16,7 @@ * the read-only part of the basic_json interface; materialize() turns a * * subtree into the basic_json value that parse() would produce. * * * - * The source text must outlive a document that borrows it (lvalue byte * + * The source text must outlive a document that borrows it (lvalue byte * * containers, C strings); rvalue strings, streams, and other inputs are * * owned by the document. * \****************************************************************************/ @@ -51,7 +51,7 @@ -#include // find, find_if, max +#include // find, find_if, max, min #include // array #include // size_t, ptrdiff_t #include // int64_t, uint8_t, uint16_t, uint32_t, uint64_t @@ -75,7 +75,8 @@ #include // array #include // size_t #include // memcpy -#include // operator new, placement new +#include // numeric_limits +#include // bad_alloc, operator new, placement new #include // string // #include @@ -185,6 +186,11 @@ static_assert(static_cast(value_t::null) == 0 && static_cast(value_t::number_unsigned) == 6 && static_cast(value_t::number_float) == 7, "the node format depends on the numbering of value_t"); +/// The largest input a document accepts, in bytes. Offsets and node counts are +/// 32 bits wide; the limit keeps 16 bytes (the width of the scanner's steps) +/// below 2^32, so that a position one step past the end of the text fits. +static constexpr std::size_t max_input_size = 0xFFFFFFEFu; + /// node flags struct node_flags { @@ -213,17 +219,27 @@ NLOHMANN_VIEW_ALWAYS_INLINE bool is_container(const node& n) noexcept return static_cast(n.kind) - 1u <= 1u; } -/// the converted value of an integer node (stored in len/next) +/// the converted value of an integer node: len is its low half, next its high +/// half (on little-endian targets the two words are the value in memory) NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept { +#if NLOHMANN_VIEW_LITTLE_ENDIAN std::uint64_t v = 0; std::memcpy(&v, reinterpret_cast(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) return v; +#else + return static_cast(n.len) | (static_cast(n.next) << 32); +#endif } NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept { +#if NLOHMANN_VIEW_LITTLE_ENDIAN std::memcpy(reinterpret_cast(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) +#else + n.len = static_cast(v); + n.next = static_cast(v >> 32); +#endif } /// token length of a number node @@ -319,13 +335,30 @@ struct document_data tape_cap = inline_cap; } - /// make room for n nodes; keeps the first tape_size nodes + /// the largest node count whose size in bytes fits a std::size_t + static constexpr std::size_t max_nodes() noexcept + { + return (std::numeric_limits::max)() / sizeof(node); + } + + [[noreturn]] NLOHMANN_VIEW_NOINLINE static void throw_bad_alloc() + { + NLOHMANN_VIEW_THROW(std::bad_alloc()); + } + + /// make room for n nodes; keeps the first tape_size nodes (throws + /// std::bad_alloc for a count that does not fit the address space, + /// instead of wrapping around in n * sizeof(node)) void reserve(std::size_t n) { if (n <= tape_cap) { return; } + if (NLOHMANN_VIEW_UNLIKELY(n > max_nodes())) + { + throw_bad_alloc(); + } node* fresh = static_cast(::operator new (n * sizeof(node))); if (tape_size != 0) { @@ -732,8 +765,18 @@ class builder const std::uint64_t done = static_cast(at - b) + 1; const std::uint64_t guess = static_cast(n) * static_cast(e - b + 1) / done; const std::uint64_t grown = guess + (guess / 4) + 64; // a variable: GCC calls a cast of the sum useless where std::uint64_t is std::size_t + // (n is below 2^32: the input is smaller than 4 GiB; the sum cannot wrap) + const std::uint64_t wanted = (std::max)(grown, static_cast(n) + (n / 2) + 64); + const std::uint64_t limit = document_data::max_nodes(); doc.tape_size = n; - doc.reserve((std::max)(static_cast(grown), n + (n / 2) + 64)); + // LCOV_EXCL_START (a node array that fills the address space) + if (NLOHMANN_VIEW_UNLIKELY(n >= limit)) + { + document_data::throw_bad_alloc(); // no room for another node + } + // LCOV_EXCL_STOP + // (a count beyond the limit is cut: the index does not grow beyond what can be addressed) + doc.reserve(static_cast((std::min)(wanted, limit))); return doc.tape; } @@ -1338,7 +1381,10 @@ indent_done: n->flags = flags; n->extra = extra; n->off = static_cast(off); - set_integer_bits(*n, second); + // len is the low half of the second word, next the high half + // (not a native word over both, which swaps them on big-endian) + n->len = static_cast(second); + n->next = static_cast(second >> 32); #endif return n; } @@ -1633,9 +1679,8 @@ template { if (f.code == error_code::input_too_large) { - // LCOV_EXCL_START (4 GiB) - NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4 GiB or more is not supported by json_document", nullptr)); - // LCOV_EXCL_STOP + // (the limit is detail::view::max_input_size: 4 GiB minus 16 bytes) + NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4294967280 bytes or more is not supported by json_document", nullptr)); } const BasicJsonType accepted = BasicJsonType::parse(src, src + size, nullptr, true, ignore_comments, ignore_trailing_commas); // LCOV_EXCL_START (only if parse() accepts what the view rejects: a bug) @@ -1674,7 +1719,7 @@ NLOHMANN_JSON_NAMESPACE_END #include // basic_string, char_traits, string -#include // decay, integral_constant, is_array, is_lvalue_reference, is_pointer, is_same, remove_reference +#include // decay, integral_constant, is_array, is_const, is_integral, is_lvalue_reference, is_pointer, is_same, remove_reference #include // forward // #include @@ -1694,12 +1739,12 @@ namespace view /// how a document takes its input enum class input_kind { - move_string, ///< rvalue std::string: owned without a copy + move_string, ///< non-const rvalue std::string: owned without a copy c_string, ///< const char* (NUL-terminated): borrowed char_array, ///< char array (e.g. a string literal): borrowed borrow_range, ///< lvalue contiguous byte container, or std::string_view: borrowed - copy_range, ///< rvalue contiguous byte container: copied - adapter, ///< anything else parse() accepts (streams, wide strings, ...): read into a buffer + copy_range, ///< rvalue contiguous byte container (a const rvalue std::string too): copied + adapter, ///< streams, wide strings, and the rest of what the library's input adapter reads: read into a buffer }; template @@ -1718,13 +1763,20 @@ struct classify_input static constexpr input_kind value = std::is_array::value ? input_kind::char_array : std::is_pointer::value ? input_kind::c_string - : (is_rvalue && std::is_same::value) ? input_kind::move_string + : (is_rvalue && !std::is_const::value && std::is_same::value) ? input_kind::move_string : (is_bytes && (!is_rvalue || is_string_view)) ? input_kind::borrow_range : is_bytes ? input_kind::copy_range : input_kind::adapter; // NOLINTEND(readability-avoid-nested-conditional-operator) }; +/// an integer type other than bool: a length passed where a flag is expected +template +struct is_integer_not_bool : std::is_integral {}; + +template<> +struct is_integer_not_bool : std::false_type {}; + /// std::basic_string guarantees a NUL at data()[size()] (the parser's sentinel) template struct is_std_string : std::false_type {}; @@ -2188,12 +2240,6 @@ class basic_json_view return type() == value_t::discarded; } - /// false for discarded views - explicit operator bool() const noexcept - { - return m_node != nullptr; - } - ////////////// // capacity // ////////////// @@ -2323,6 +2369,13 @@ class basic_json_document return d; } + /// parse(ptr, len) does not compile: len would convert to allow_exceptions + /// and ptr be read as a C string (as for the overloads of parse_copy, + /// accept, and read below) + template + static typename std::enable_if::value, basic_json_document>::type + parse(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// parse [first, last) template::iterator_category>::value, int>::type = 0> @@ -2350,6 +2403,10 @@ class basic_json_document return d; } + template + static typename std::enable_if::value, basic_json_document>::type + parse_copy(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// check whether the input is valid JSON (the result of basic_json::accept) template static bool accept(InputType&& input, const bool ignore_comments = false, const bool ignore_trailing_commas = false) @@ -2359,6 +2416,10 @@ class basic_json_document return !d.is_discarded(); } + template + static typename std::enable_if::value, bool>::type + accept(InputType&& input, IntegerType value, Flags&&... flags) = delete; + /// parse into this document, reusing its memory template // flawfinder: ignore (a member function, not POSIX read()) @@ -2371,12 +2432,17 @@ class basic_json_document std::integral_constant::value> {}); } + template + // flawfinder: ignore (a member function, not POSIX read()) + typename std::enable_if::value, void>::type + read(InputType&& input, IntegerType value, Flags&&... flags) = delete; + //////////// // access // //////////// /// the root value (discarded if parsing failed without exceptions) - view_type root() const noexcept + view_type root() const& noexcept { if (!m_data || m_data->discarded) { @@ -2385,6 +2451,9 @@ class basic_json_document return view_type(m_data.get(), m_data->tape); } + /// deleted: the view of a temporary document would dangle + view_type root() const&& = delete; + bool is_discarded() const noexcept { return !m_data || m_data->discarded; @@ -2490,9 +2559,9 @@ class basic_json_document d.discarded = true; detail::view::parse_failure failure; bool ok = false; - if (NLOHMANN_VIEW_UNLIKELY(size >= 0xFFFFFFF0u)) + if (NLOHMANN_VIEW_UNLIKELY(size > detail::view::max_input_size)) { - failure.code = detail::view::error_code::input_too_large; // LCOV_EXCL_LINE (4 GiB) + failure.code = detail::view::error_code::input_too_large; } else { From 45045ebb6de0f819281b54a5d7882b8cfa118c09 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:53:54 +0200 Subject: [PATCH 22/26] Replace the removed operator bool of basic_json_view in the header and its tests A view is tested with is_discarded(); the lookups in at(), value() and contains(json_pointer) and the unit tests no longer rely on the explicit conversion to bool. Signed-off-by: Niels Lohmann --- include/nlohmann/json_view.hpp | 8 ++-- single_include/nlohmann/json_view.hpp | 8 ++-- tests/src/unit-json_view.cpp | 66 +++++++++++++-------------- 3 files changed, 41 insertions(+), 41 deletions(-) diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index e52e5a00a..d71d54488 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -302,7 +302,7 @@ class basic_json_view detail::view::throw_type_error(304, "cannot use at() with ", type_name()); } const basic_json_view r = lookup(key); - if (NLOHMANN_VIEW_UNLIKELY(!r)) + if (NLOHMANN_VIEW_UNLIKELY(r.is_discarded())) { detail::view::throw_out_of_range(403, detail::concat("key '", std::string(key.data(), key.size()), "' not found")); } @@ -360,7 +360,7 @@ class basic_json_view detail::view::throw_type_error(306, "cannot use value() with ", type_name()); } const basic_json_view r = lookup(key); - return r ? r.template get() : default_value; + return r.is_discarded() ? default_value : r.template get(); } string_t value(string_view_t key, const char* default_value) const @@ -379,7 +379,7 @@ class basic_json_view detail::view::throw_type_error(306, "cannot use value() with ", type_name()); } const basic_json_view r = detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::value); - return r ? r.template get() : default_value; + return r.is_discarded() ? default_value : r.template get(); } string_t value(const json_pointer& ptr, const char* default_value) const @@ -457,7 +457,7 @@ class basic_json_view /// basic_json::contains()) bool contains(const json_pointer& ptr) const { - return static_cast(detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::contains)); + return !detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::contains).is_discarded(); } /// 1 if this is an object with a member with this key, else 0 (duplicate diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index a73b94c7e..0d9bc1f15 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -3113,7 +3113,7 @@ class basic_json_view detail::view::throw_type_error(304, "cannot use at() with ", type_name()); } const basic_json_view r = lookup(key); - if (NLOHMANN_VIEW_UNLIKELY(!r)) + if (NLOHMANN_VIEW_UNLIKELY(r.is_discarded())) { detail::view::throw_out_of_range(403, detail::concat("key '", std::string(key.data(), key.size()), "' not found")); } @@ -3171,7 +3171,7 @@ class basic_json_view detail::view::throw_type_error(306, "cannot use value() with ", type_name()); } const basic_json_view r = lookup(key); - return r ? r.template get() : default_value; + return r.is_discarded() ? default_value : r.template get(); } string_t value(string_view_t key, const char* default_value) const @@ -3190,7 +3190,7 @@ class basic_json_view detail::view::throw_type_error(306, "cannot use value() with ", type_name()); } const basic_json_view r = detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::value); - return r ? r.template get() : default_value; + return r.is_discarded() ? default_value : r.template get(); } string_t value(const json_pointer& ptr, const char* default_value) const @@ -3268,7 +3268,7 @@ class basic_json_view /// basic_json::contains()) bool contains(const json_pointer& ptr) const { - return static_cast(detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::contains)); + return !detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::contains).is_discarded(); } /// 1 if this is an object with a member with this key, else 0 (duplicate diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index cf2afcac4..4827e83ff 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -593,7 +593,7 @@ void check_access(const ordered_json_view& v, const ordered_json& j) ++i; } CHECK(i == v.size()); - CHECK(!v[v.size()]); + CHECK(v[v.size()].is_discarded()); std::size_t index = 0; for (const auto& item : v.items()) { @@ -659,7 +659,7 @@ void check_access(const ordered_json_view& v, const ordered_json& j) CHECK(v.back().materialize() == j.back()); } } - CHECK(!v["not a key in the generated documents"]); + CHECK(v["not a key in the generated documents"].is_discarded()); CHECK(v.find("not a key in the generated documents") == v.end()); } else @@ -803,8 +803,8 @@ TEST_CASE("json_view element access and iteration") // where basic_json has undefined behavior, the view answers safely const json_document d = json_document::parse(R"({"a":[]})"); - CHECK(!d.root()["b"]); - CHECK(!d.root()["a"][0]); + CHECK(d.root()["b"].is_discarded()); + CHECK(d.root()["a"][0].is_discarded()); CHECK_THROWS_WITH_AS(d.root()["a"].front(), "[json.exception.invalid_iterator.214] cannot get value", json::invalid_iterator&); CHECK_THROWS_WITH_AS(d.root()["a"].back(), "[json.exception.invalid_iterator.214] cannot get value", json::invalid_iterator&); const json_view invalid{}; @@ -817,24 +817,24 @@ TEST_CASE("json_view element access and iteration") const json_document d = json_document::parse(R"({"a":{"b":[10,20]},"s":"str"})"); const json_view v = d.root(); // missing keys and indexes - CHECK(!v["x"]); - CHECK(!v["x"]["y"]); - CHECK(!v["x"]["y"]["z"]); - CHECK(!v["x"][0]); - CHECK(!v["x"][0u][1L]); - CHECK(!v["a"]["b"][2]); - CHECK(!v["a"]["b"][2]["c"]); - CHECK(!v["a"]["b"][2][json_view::json_pointer("/c")]); - CHECK(!v["x"][json_view::json_pointer("/a/b")]); - CHECK(!v["x"][json_view::json_pointer("")]); + CHECK(v["x"].is_discarded()); CHECK(v["x"]["y"].is_discarded()); - CHECK(!v["x"][std::string("y")]); + CHECK(v["x"]["y"]["z"].is_discarded()); + CHECK(v["x"][0].is_discarded()); + CHECK(v["x"][0u][1L].is_discarded()); + CHECK(v["a"]["b"][2].is_discarded()); + CHECK(v["a"]["b"][2]["c"].is_discarded()); + CHECK(v["a"]["b"][2][json_view::json_pointer("/c")].is_discarded()); + CHECK(v["x"][json_view::json_pointer("/a/b")].is_discarded()); + CHECK(v["x"][json_view::json_pointer("")].is_discarded()); + CHECK(v["x"]["y"].is_discarded()); + CHECK(v["x"][std::string("y")].is_discarded()); // a resolvable path still resolves CHECK(v["a"]["b"][1].materialize() == 20); CHECK(v[json_view::json_pointer("/a/b/1")].materialize() == 20); // the discarded view of an unresolved pointer is discarded too - CHECK(!v[json_view::json_pointer("/x/y")]["z"]); - CHECK(!v[json_view::json_pointer("/a/b/5")][0]); + CHECK(v[json_view::json_pointer("/x/y")]["z"].is_discarded()); + CHECK(v[json_view::json_pointer("/a/b/5")][0].is_discarded()); #if !defined(JSON_NOEXCEPTION) // type errors on values that are not discarded stay CHECK_THROWS_WITH_AS(v[0], "[json.exception.type_error.305] cannot use operator[] with a numeric argument with object", json::type_error&); @@ -845,9 +845,9 @@ TEST_CASE("json_view element access and iteration") CHECK_THROWS_AS(v["s"][json_view::json_pointer("/x")], json::out_of_range&); // at() keeps throwing on a discarded view const json_view invalid{}; - CHECK(!invalid["a"]); - CHECK(!invalid[0]); - CHECK(!invalid[json_view::json_pointer("/a")]); + CHECK(invalid["a"].is_discarded()); + CHECK(invalid[0].is_discarded()); + CHECK(invalid[json_view::json_pointer("/a")].is_discarded()); CHECK_THROWS_WITH_AS(invalid.at("a"), "[json.exception.type_error.304] cannot use at() with discarded", json::type_error&); CHECK_THROWS_WITH_AS(invalid.at(0), "[json.exception.type_error.304] cannot use at() with discarded", json::type_error&); CHECK_THROWS_AS(v.at("x").at("y"), json::out_of_range&); @@ -900,18 +900,18 @@ TEST_CASE("json_view element access and iteration") CHECK(j.at(std::uint32_t(0)) == 10); // as basic_json // out of range, including negative values (no wrap-around) - CHECK(!v[3]); - CHECK(!v[3u]); - CHECK(!v[3L]); - CHECK(!v[-1]); - CHECK(!v[-1L]); - CHECK(!v[-1LL]); - CHECK(!v[static_cast(-1)]); - CHECK(!v[std::int64_t(-3)]); - CHECK(!v[(std::numeric_limits::min)()]); - CHECK(!v[(std::numeric_limits::max)()]); - CHECK(!v[(std::numeric_limits::max)()]); - CHECK(!v[std::numeric_limits::max()]); + CHECK(v[3].is_discarded()); + CHECK(v[3u].is_discarded()); + CHECK(v[3L].is_discarded()); + CHECK(v[-1].is_discarded()); + CHECK(v[-1L].is_discarded()); + CHECK(v[-1LL].is_discarded()); + CHECK(v[static_cast(-1)].is_discarded()); + CHECK(v[std::int64_t(-3)].is_discarded()); + CHECK(v[(std::numeric_limits::min)()].is_discarded()); + CHECK(v[(std::numeric_limits::max)()].is_discarded()); + CHECK(v[(std::numeric_limits::max)()].is_discarded()); + CHECK(v[std::numeric_limits::max()].is_discarded()); #if !defined(JSON_NOEXCEPTION) CHECK_THROWS_WITH_AS(v.at(3), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&); CHECK_THROWS_WITH_AS(v.at(3u), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&); @@ -1344,7 +1344,7 @@ TEST_CASE("json_view JSON pointers") else if (at_error.find("out_of_range.401") != std::string::npos || at_error.find("out_of_range.403") != std::string::npos) // NOLINT(abseil-string-find-str-contains) { // undefined behavior for const basic_json::operator[] - CHECK(!v[p]); + CHECK(v[p].is_discarded()); } else { From 801df39c27355e703ea5612cacc4c219ea4818d6 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:54:04 +0200 Subject: [PATCH 23/26] Use named documents in the json_view tests that called root() on a temporary root() of an rvalue document is deleted, because the views would dangle. Signed-off-by: Niels Lohmann --- tests/src/unit-json_view.cpp | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 4827e83ff..bc43a97bd 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -1193,10 +1193,12 @@ TEST_CASE("json_view values") CAPTURE(token) const std::string text = "[" + token + "]"; const double b = json::parse(text)[0].get(); - CHECK(bits(json_document::parse(text).root()[0].get()) == bits(b)); + const json_document dd = json_document::parse(text); + CHECK(bits(dd.root()[0].get()) == bits(b)); if (std::abs(b) < 1e38) { - CHECK(bits(nlohmann::basic_json_document::parse(text).root()[0].get()) == bits(json_float::parse(text)[0].get())); + const nlohmann::basic_json_document df = nlohmann::basic_json_document::parse(text); + CHECK(bits(df.root()[0].get()) == bits(json_float::parse(text)[0].get())); } } } @@ -1253,7 +1255,8 @@ TEST_CASE("json_view values") CHECK(count == 3); // a duplicate key: the last value, as parse() - CHECK((json_document::parse(R"({"a":1,"a":2})").root().get>() == std::map {{"a", 2}})); + const json_document dup = json_document::parse(R"({"a":1,"a":2})"); + CHECK((dup.root().get>() == std::map {{"a", 2}})); const json_view invalid{}; CHECK_THROWS_WITH_AS(invalid.get(), "[json.exception.type_error.302] type must be number, but is discarded", json::type_error&); From b255e07dd378a7c3cf042865a34473e7c7c2f127 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:54:05 +0200 Subject: [PATCH 24/26] Test discarded views with is_discarded() in the operator[] docs and examples The explicit conversion to bool of basic_json_view is gone. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json_view/operator[].md | 2 +- docs/mkdocs/docs/examples/basic_json_view__operator[].cpp | 8 +++++--- .../examples/basic_json_view__operator[]_json_pointer.cpp | 3 ++- 3 files changed, 8 insertions(+), 5 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json_view/operator[].md b/docs/mkdocs/docs/api/basic_json_view/operator[].md index 30130db7c..53d95bed0 100644 --- a/docs/mkdocs/docs/api/basic_json_view/operator[].md +++ b/docs/mkdocs/docs/api/basic_json_view/operator[].md @@ -93,7 +93,7 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics Unlike `BasicJsonType::operator[]`, which is undefined behavior (guarded by a [runtime assertion](../../features/assertions.md)) for a missing key on a **const** value, this operator returns a safe, testable result for a missing key or an index out of range: a [discarded](is_discarded.md) view, which is -`#!cpp false` in a boolean context. +tested with [`is_discarded`](is_discarded.md). There is also no non-const overload that inserts a missing key or extends an array -- a view never modifies the document. diff --git a/docs/mkdocs/docs/examples/basic_json_view__operator[].cpp b/docs/mkdocs/docs/examples/basic_json_view__operator[].cpp index e17919a9d..1919e06be 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__operator[].cpp +++ b/docs/mkdocs/docs/examples/basic_json_view__operator[].cpp @@ -22,17 +22,19 @@ int main() std::cout << user["name"].materialize().dump(); // operator[] on a missing object key gives a discarded view -- test - // it with a plain "if". The const overload of json::operator[] + // it with is_discarded(). The const overload of json::operator[] // would instead be undefined behavior (guarded by an assertion) for // a missing key - if (const auto email = user["email"]) + const auto email = user["email"]; + if (!email.is_discarded()) { std::cout << " <" << email.materialize().dump() << ">"; } // the same holds for an array index past the end: a discarded view, // not undefined behavior - if (const auto first_tag = user["tags"][0]) + const auto first_tag = user["tags"][0]; + if (!first_tag.is_discarded()) { std::cout << " #" << first_tag.materialize().dump(); } diff --git a/docs/mkdocs/docs/examples/basic_json_view__operator[]_json_pointer.cpp b/docs/mkdocs/docs/examples/basic_json_view__operator[]_json_pointer.cpp index 15f37555d..2ea8130a6 100644 --- a/docs/mkdocs/docs/examples/basic_json_view__operator[]_json_pointer.cpp +++ b/docs/mkdocs/docs/examples/basic_json_view__operator[]_json_pointer.cpp @@ -25,7 +25,8 @@ int main() // a missing key or an out-of-range index along the path gives a // discarded view, exactly where const json::operator[] would be // undefined behavior for the same pointer - if (const auto missing = root[json_pointer("/region/servers/5/metrics/cpu")]) + const auto missing = root[json_pointer("/region/servers/5/metrics/cpu")]; + if (!missing.is_discarded()) { std::cout << missing.materialize().dump() << '\n'; } From 1fee86d75a61f128db484e518dab131ce039ba09 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:56:19 +0200 Subject: [PATCH 25/26] Remove the docset entry of json_view's operator bool Signed-off-by: Niels Lohmann --- docs/docset/docSet.sql | 1 - 1 file changed, 1 deletion(-) diff --git a/docs/docset/docSet.sql b/docs/docset/docSet.sql index 9db3d6a8b..d60c5f573 100644 --- a/docs/docset/docSet.sql +++ b/docs/docset/docSet.sql @@ -163,7 +163,6 @@ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_primitive INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_string', 'Method', 'api/basic_json_view/is_string/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_structured', 'Method', 'api/basic_json_view/is_structured/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::materialize', 'Method', 'api/basic_json_view/materialize/index.html'); -INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator bool', 'Method', 'api/basic_json_view/operator_bool/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::size', 'Method', 'api/basic_json_view/size/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::source_offset', 'Method', 'api/basic_json_view/source_offset/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::type', 'Method', 'api/basic_json_view/type/index.html'); From fec4ff829969a3f993cfd909e8a286c5e35e4bb0 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 17:03:14 +0200 Subject: [PATCH 26/26] Hold the documents of the dump and comparison tests in names root() of a temporary document is deleted: its views would dangle. Signed-off-by: Niels Lohmann --- tests/src/unit-json_view.cpp | 29 ++++++++++++++++++++--------- 1 file changed, 20 insertions(+), 9 deletions(-) diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 9cd787ca1..21e4c0f33 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -1450,10 +1450,12 @@ TEST_CASE("json_view dump") } } many += ']'; - CHECK(json_document::parse(many).root().dump() == json::parse(many).dump()); + const json_document many_document = json_document::parse(many); + CHECK(many_document.root().dump() == json::parse(many).dump()); using json_float = nlohmann::basic_json; - CHECK(nlohmann::basic_json_document::parse("[0.1, 1.5e10, 3.4028235e38]").root().dump() == json_float::parse("[0.1, 1.5e10, 3.4028235e38]").dump()); + const nlohmann::basic_json_document float_document = nlohmann::basic_json_document::parse("[0.1, 1.5e10, 3.4028235e38]"); + CHECK(float_document.root().dump() == json_float::parse("[0.1, 1.5e10, 3.4028235e38]").dump()); } SECTION("members in document order, all of them") @@ -1466,7 +1468,8 @@ TEST_CASE("json_view dump") SECTION("deep nesting") { const std::string deep = std::string(100000, '[') + std::string(100000, ']'); - CHECK(json_document::parse(deep).root().dump() == deep); + const json_document deep_document = json_document::parse(deep); + CHECK(deep_document.root().dump() == deep); } SECTION("the output buffer of a small value is small") @@ -1562,7 +1565,9 @@ TEST_CASE("json_view comparison") { const auto same = [](const char* x, const char* y) { - return json_document::parse(x).root() == json_document::parse(y).root(); + const json_document dx = json_document::parse(x); + const json_document dy = json_document::parse(y); + return dx.root() == dy.root(); }; CHECK(same("1", "1.0")); CHECK(same("[1, -1, 2.5]", "[1.0, -1.0, 25e-1]")); @@ -1577,15 +1582,20 @@ TEST_CASE("json_view comparison") CHECK(same("\"\\u00e9\"", "\"\xc3\xa9\"")); CHECK(!same("null", "false")); CHECK(!same("[]", "{}")); - CHECK(ordered_json_document::parse(R"({"a": 1, "b": 2, "a": 3})").root() == ordered_json_document::parse(R"({"a": 3, "b": 2})").root()); - CHECK(ordered_json_document::parse(R"({"a": 1, "b": 2})").root() != ordered_json_document::parse(R"({"b": 2, "a": 1})").root()); + const ordered_json_document dup = ordered_json_document::parse(R"({"a": 1, "b": 2, "a": 3})"); + const ordered_json_document last = ordered_json_document::parse(R"({"a": 3, "b": 2})"); + CHECK(dup.root() == last.root()); + const ordered_json_document ab = ordered_json_document::parse(R"({"a": 1, "b": 2})"); + const ordered_json_document ba = ordered_json_document::parse(R"({"b": 2, "a": 1})"); + CHECK(ab.root() != ba.root()); // discarded values compare as basic_json's do const json discarded(json::value_t::discarded); CHECK((json_view() == json_view()) == (discarded == discarded)); // NOLINT(readability-container-size-empty): operator== is tested CHECK((json_view() == discarded) == (discarded == discarded)); - CHECK(!(json_view() == json_document::parse("null").root())); // NOLINT(readability-container-size-empty) - CHECK(!(json_document::parse("null").root() == discarded)); + const json_document null_document = json_document::parse("null"); + CHECK(!(json_view() == null_document.root())); // NOLINT(readability-container-size-empty) + CHECK(!(null_document.root() == discarded)); } SECTION("deep nesting") @@ -1596,6 +1606,7 @@ TEST_CASE("json_view comparison") CHECK(a.root() == b.root()); CHECK(a.root() == json::parse(deep)); const std::string other = std::string(100000, '[') + "1" + std::string(100000, ']'); - CHECK(a.root() != json_document::parse(other).root()); + const json_document c = json_document::parse(other); + CHECK(a.root() != c.root()); } }