diff --git a/include/nlohmann/detail/bit_ops.hpp b/include/nlohmann/detail/bit_ops.hpp index 3dc781164..9da853faf 100644 --- a/include/nlohmann/detail/bit_ops.hpp +++ b/include/nlohmann/detail/bit_ops.hpp @@ -13,7 +13,7 @@ #include // __umulh, _umul128 #endif -#include +#include // JSON_HEDLEY_ALWAYS_INLINE, NLOHMANN_JSON_NAMESPACE_BEGIN // Portable bit-level helpers for the number and string scanners. They use // compiler builtins or platform-specific intrinsics where available and plain diff --git a/include/nlohmann/detail/conversions/to_chars.hpp b/include/nlohmann/detail/conversions/to_chars.hpp index 837e8d588..0e16152c3 100644 --- a/include/nlohmann/detail/conversions/to_chars.hpp +++ b/include/nlohmann/detail/conversions/to_chars.hpp @@ -1258,7 +1258,7 @@ inline char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept first[1] = '.'; char* const end = first + (k == 1 ? 1 : k + 1); const int e = n - 1; - const auto ea = static_cast(e < 0 ? -e : e); + const auto ea = e < 0 ? 0u - static_cast(e) : static_cast(e); // (unsigned: no signed overflow to assume) const bool three = ea >= 100; end[0] = 'e'; end[1] = e < 0 ? '-' : '+'; @@ -1298,7 +1298,7 @@ inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcep // NOLINTBEGIN(portability-simd-intrinsics) // the two halves in the 64-bit lanes, each as abcd * 2^32 + efgh, then as // bytes (as eight_digit_bytes(), one lane each) - const __m128i x = _mm_set_epi64x(static_cast(sig - (upper * 100000000u)), static_cast(upper)); + const __m128i x = _mm_set_epi64x(static_cast(sig - (upper * 100000000u)), static_cast(upper)); // NOLINT(runtime/int) const __m128i abcd = _mm_srli_epi64(_mm_mul_epu32(x, _mm_set1_epi64x(109951163)), 40); // 2^40 / 10000 + 1 const __m128i abcd_efgh = _mm_add_epi64(x, _mm_mul_epu32(abcd, _mm_set1_epi64x(4294957296))); // 2^32 - 10000 // 32-bit lanes in the order of the text: abcd, efgh of both halves @@ -1413,7 +1413,7 @@ inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcep first[1] = '.'; char* const end = first + (len == 1 ? 1 : len + 1); const int e = n - 1; - const auto ea = static_cast(e < 0 ? -e : e); + const auto ea = e < 0 ? 0u - static_cast(e) : static_cast(e); // (unsigned: no signed overflow to assume) const bool three = ea >= 100; end[0] = 'e'; end[1] = e < 0 ? '-' : '+'; @@ -1472,6 +1472,7 @@ JSON_HEDLEY_RETURNS_NON_NULL char* write_positive(char* first, const char* last, FloatType value) { JSON_ASSERT(last - first >= std::numeric_limits::max_digits10); + static_cast(last); // (only used in the assertion) // Compute v = buffer * 10^decimal_exponent. // The decimal digits are stored in the buffer, which needs to be interpreted @@ -1513,7 +1514,7 @@ inline char* write_positive(char* first, const char* last, double value) } std::array buf; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read const auto len = static_cast(write_shortest(buf.data(), d) - buf.data()); - JSON_ASSERT(static_cast(last - first) >= len); + JSON_ASSERT(last - first >= static_cast(len)); std::memcpy(first, buf.data(), len); return first + len; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 70d5ac805..72ad2e02a 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -8867,8 +8867,8 @@ NLOHMANN_JSON_NAMESPACE_END #include // __umulh, _umul128 #endif -// #include - +// #include +// JSON_HEDLEY_ALWAYS_INLINE, NLOHMANN_JSON_NAMESPACE_BEGIN // Portable bit-level helpers for the number and string scanners. They use // compiler builtins or platform-specific intrinsics where available and plain @@ -26670,7 +26670,7 @@ inline char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept first[1] = '.'; char* const end = first + (k == 1 ? 1 : k + 1); const int e = n - 1; - const auto ea = static_cast(e < 0 ? -e : e); + const auto ea = e < 0 ? 0u - static_cast(e) : static_cast(e); // (unsigned: no signed overflow to assume) const bool three = ea >= 100; end[0] = 'e'; end[1] = e < 0 ? '-' : '+'; @@ -26710,7 +26710,7 @@ inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcep // NOLINTBEGIN(portability-simd-intrinsics) // the two halves in the 64-bit lanes, each as abcd * 2^32 + efgh, then as // bytes (as eight_digit_bytes(), one lane each) - const __m128i x = _mm_set_epi64x(static_cast(sig - (upper * 100000000u)), static_cast(upper)); + const __m128i x = _mm_set_epi64x(static_cast(sig - (upper * 100000000u)), static_cast(upper)); // NOLINT(runtime/int) const __m128i abcd = _mm_srli_epi64(_mm_mul_epu32(x, _mm_set1_epi64x(109951163)), 40); // 2^40 / 10000 + 1 const __m128i abcd_efgh = _mm_add_epi64(x, _mm_mul_epu32(abcd, _mm_set1_epi64x(4294957296))); // 2^32 - 10000 // 32-bit lanes in the order of the text: abcd, efgh of both halves @@ -26825,7 +26825,7 @@ inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcep first[1] = '.'; char* const end = first + (len == 1 ? 1 : len + 1); const int e = n - 1; - const auto ea = static_cast(e < 0 ? -e : e); + const auto ea = e < 0 ? 0u - static_cast(e) : static_cast(e); // (unsigned: no signed overflow to assume) const bool three = ea >= 100; end[0] = 'e'; end[1] = e < 0 ? '-' : '+'; @@ -26884,6 +26884,7 @@ JSON_HEDLEY_RETURNS_NON_NULL char* write_positive(char* first, const char* last, FloatType value) { JSON_ASSERT(last - first >= std::numeric_limits::max_digits10); + static_cast(last); // (only used in the assertion) // Compute v = buffer * 10^decimal_exponent. // The decimal digits are stored in the buffer, which needs to be interpreted @@ -26925,7 +26926,7 @@ inline char* write_positive(char* first, const char* last, double value) } std::array buf; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read const auto len = static_cast(write_shortest(buf.data(), d) - buf.data()); - JSON_ASSERT(static_cast(last - first) >= len); + JSON_ASSERT(last - first >= static_cast(len)); std::memcpy(first, buf.data(), len); return first + len; } diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 994c490c0..01c463fa3 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -25,7 +25,7 @@ #define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ #include // size_t -#include // uint32_t +#include // uint8_t, uint32_t #include // memcpy, strlen #include // distance, input_iterator_tag, iterator_traits #include // map @@ -82,7 +82,7 @@ #include // array #include // size_t -#include // uint32_t +#include // uint8_t, uint32_t #include // memcpy #include // less #include // map @@ -306,7 +306,9 @@ struct document_data node* inline_tape = nullptr; ///< node array allocated together with this header std::size_t inline_cap = 0; std::string arena; ///< decoded strings that contained escapes + std::size_t arena_size = 0; ///< bytes of decoded strings at base[1] (the arena, or those of a loaded image) std::string owned; ///< owned copy of the input, if any + std::vector owned_image; ///< a loaded image the document owns (the text and the decoded strings point into it) // hash indexes of large objects (see object_index.hpp) static constexpr std::uint32_t index_min_members = 128; @@ -3850,6 +3852,840 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // array +#include // size_t +#include // int64_t, uint8_t, uint16_t, uint32_t, uint64_t +#include // memcmp, memcpy +#include // numeric_limits +#include // string +#include // vector + +// #include +// #include + +// #include + +// #include + +// #include + +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // size_t +#include // int64_t, uint64_t +#include // numeric_limits +#include // string +#include // integral_constant + +// #include +// #include + +// #include + +// #include + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/*! +@brief locate the decimal point and the end of the mantissa of a float token + +Also checks that the token is a JSON number. Tokens of the parser and of edits +always are; an image loaded with image_check::bounds can hold any bytes, which +must not reach the conversion (it expects a well-formed token). +*/ +inline bool float_token_layout(const char* first, const char* last, std::size_t& dot, std::size_t& mantissa_end) noexcept +{ + const auto digit = [last](const char* q) + { + return q != last && is_digit(static_cast(*q)); + }; + const char* p = first; + p += (p != last && *p == '-') ? 1 : 0; + if (!digit(p) || (*p == '0' && digit(p + 1))) + { + return false; + } + while (digit(p)) + { + ++p; + } + dot = std::string::npos; + if (p != last && *p == '.') + { + dot = static_cast(p - first); + if (!digit(++p)) + { + return false; + } + while (digit(p)) + { + ++p; + } + } + mantissa_end = static_cast(p - first); + if (p != last && (*p == 'e' || *p == 'E')) + { + ++p; + p += (p != last && (*p == '+' || *p == '-')) ? 1 : 0; + if (!digit(p)) + { + return false; + } + while (digit(p)) + { + ++p; + } + } + return p == last; +} + +/*! +@brief the value of the float token of a node, as parse() converts it + +Uses the lexer's conversion (detail::convert_float), so that the values are +bit-identical to parse(): float and double are converted without allocation +and independent of the locale. A token that is not a JSON number (only in a +damaged image loaded with image_check::bounds) yields 0. +*/ +template +NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) +{ + const char* const last = first + n.len; + std::size_t dot = 0; + std::size_t mantissa_end = 0; + if (NLOHMANN_VIEW_UNLIKELY(!float_token_layout(first, last, dot, mantissa_end))) + { + return FloatType{}; + } + return convert_float(first, last, dot, mantissa_end); +} + +/*! +@brief the digits of a float token with at most 19 digits, from its layout + +The digit layout recorded while parsing says where the integer digits, the +fraction digits, and the exponent are, so the digits are read eight at a +time without scanning. + +@param[in] p first character of the token +@param[in] e end of the token +@param[in] limit end of the readable memory (the source text) +*/ +NLOHMANN_VIEW_ALWAYS_INLINE float_significand layout_decimal(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +{ + const bool negative = *p == '-'; + p += negative ? 1 : 0; + std::uint64_t w = parse_upto19(p, int_digits, limit); + p += int_digits; + std::int64_t q = 0; + if (frac_digits != 0) + { + w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit); + p += 1 + frac_digits; + q = -static_cast(frac_digits); + } + if (p != e) + { + // [eE][+-]digits; huge exponents saturate (the parser rejected + // overflow). The token is not read beyond e, and the digits are taken + // as unsigned, so that a token that is not well-formed (a damaged + // image loaded with image_check::bounds) yields a wrong value, but no + // overflow. + ++p; + const bool exp_negative = p != e && *p == '-'; + p += (p != e && (*p == '-' || *p == '+')) ? 1 : 0; + std::int64_t exp_value = 0; + for (; p != e; ++p) + { + if (exp_value < 0x10000000) + { + exp_value = (exp_value * 10) + static_cast(*p - '0'); + } + } + q += exp_negative ? -exp_value : exp_value; + } + + float_significand d; + d.w = w; + d.exponent = q; + d.negative = negative; + return d; +} + +/*! +@brief the value of a float token with at most 19 digits, from its layout + +The result is correctly rounded by the lexer's conversion +(detail::decimal_to_float(): Clinger's fast path where both operands are +exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19 +digits), so it is the value parse() produces. +*/ +template +NLOHMANN_VIEW_ALWAYS_INLINE FloatType layout_float(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +{ + return decimal_to_float(layout_decimal(p, e, int_digits, frac_digits, limit)); +} + +/// the value of a float set by an edit: its token (the shortest round-trip +/// text, or "nan", "inf", "-inf") in the edit arena +template +NLOHMANN_VIEW_NOINLINE FloatType edited_float(const char* token, const node& n) +{ + if (token[0] == 'n') + { + return std::numeric_limits::quiet_NaN(); + } + if (token[0] == 'i' || (token[0] == '-' && token[1] == 'i')) + { + return token[0] == 'i' ? std::numeric_limits::infinity() : -std::numeric_limits::infinity(); + } + return float_value(token, n); +} + +/// the value of the float token of a node, as parse() converts it; floats and +/// doubles with at most 19 digits are converted from the digit layout +template +FloatType float_value(const document_data& d, const node& n) +{ + if (NLOHMANN_VIEW_UNLIKELY((n.flags & node_flags::storage) == node_flags::edited)) + { + return edited_float(d.str(n), n); + } + return float_value(d, n, std::integral_constant::value> {}); +} + +template +FloatType float_value(const document_data& d, const node& n, std::true_type /*binary32 or binary64*/) +{ + const unsigned int_digits = n.extra & 0xFFu; + const unsigned frac_digits = n.extra >> 8u; + if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many") + { + // (a float token not written by an edit is in the text) + const auto* const first = reinterpret_cast(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return layout_float(first, first + n.len, int_digits, frac_digits, reinterpret_cast(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + return float_value(d.str(n), n); +} + +template +FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/) +{ + return float_value(d.str(n), n); +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + +// #include + +// #include + + +// Images: a document stored so that loading it needs no parsing. +// +// Layout (little-endian): a 64-byte header, the nodes, the text (the source, +// followed by the number tokens written by edits), a NUL, the decoded strings +// (followed by the strings written by edits), a NUL. The idea is that of +// zero-copy formats such as FlatBuffers (https://github.com/google/flatbuffers) +// and YaFF (https://github.com/yandex/yaff); no code is taken from them. +// check_image follows the idea of FlatBuffers' Verifier (bounds and +// structure) and also checks what the parser guarantees about strings and +// numbers, so that reading and serializing a checked image is safe and yields +// valid JSON. + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// how load() checks an image +enum class image_check +{ + /// everything the parser guarantees: structure and bounds, strings (valid + /// UTF-8; source strings without quotes, backslashes, and control + /// characters), and numbers (well-formed, matching the stored values) + full, + /// structure and bounds only: reading and serializing are safe, but a + /// crafted image can yield invalid UTF-8, strings that serialize to + /// invalid JSON, or numbers that differ from their text + bounds, + /// none: for images from a trusted source only (a damaged image is + /// undefined behavior) + none, +}; + +struct image_header +{ + std::array magic; ///< "NJVI" + std::uint32_t version; ///< 1 + std::uint64_t node_count; + std::uint64_t text_size; + std::uint64_t arena_size; + std::array reserved; ///< zero (for later versions) +}; +static_assert(sizeof(image_header) == 64, "the image header must be 64 bytes"); + +constexpr std::uint32_t image_version = 1; + +/// the largest node count and text or string size of an image (as for parsed +/// documents, offsets and counts must fit 32 bits) +constexpr std::uint64_t image_limit = 0xFFFFFFF0u; + +/// Copy the current structure of an edited document into nodes in document +/// order, as the parser would have written them. Text written by edits is +/// appended to text_tail (number tokens) and arena_tail (strings); floats that +/// are not finite become null, as dump() writes them. +inline void compact_nodes(const document_data& d, std::size_t arena_size, std::vector& out, std::string& text_tail, std::string& arena_tail) +{ + struct frame + { + const node* cur; + const node* end; + std::size_t index; ///< the container's node in out + std::uint32_t count; + bool object; + }; + std::vector stack; + const auto string_node = [&](const node & s) + { + node r = s; + r.extra = 0; + r.flags = static_cast(s.flags & node_flags::storage); + if (r.flags == node_flags::edited) + { + r.off = static_cast(arena_size + arena_tail.size()); + arena_tail.append(d.str(s), s.len); + r.flags = node_flags::escaped; + } + return r; + }; + const auto emit = [&](const node * v) + { + node r = *v; + switch (static_cast(v->kind)) + { + case value_t::object: + case value_t::array: + r.flags = 0; + r.extra = 0; + r.off = (v->flags & (node_flags::moved | node_flags::is_new)) != 0 ? 0 : v->off; + r.len = 0; // counted below + r.next = 0; // set when the container is complete + stack.push_back(frame{d.first_child_edited(v), d.child_end_edited(v), out.size(), 0, v->kind == static_cast(value_t::object)}); + break; + case value_t::string: + r = string_node(*v); + break; + case value_t::number_integer: + case value_t::number_unsigned: + if ((v->flags & node_flags::storage) == node_flags::edited) + { + r.off = static_cast(d.size + text_tail.size()); + text_tail.append(d.str(*v), number_length(*v)); + } + r.flags = 0; + break; + case value_t::number_float: + if ((v->flags & node_flags::storage) == node_flags::edited) + { + const char* const t = d.str(*v); + if (t[0] == 'n' || t[0] == 'i' || (v->len > 1 && t[1] == 'i')) + { + r = node{}; // nan and infinity: null, as dump() writes them + r.kind = static_cast(value_t::null); + break; + } + r.off = static_cast(d.size + text_tail.size()); + text_tail.append(t, v->len); + r.extra = 0xFFFFu; // the digit layout is not recorded + } + r.flags = 0; + break; + case value_t::boolean: + r.flags = static_cast(v->flags & node_flags::is_true); + break; + case value_t::null: + case value_t::binary: + case value_t::discarded: + default: + r.flags = 0; + break; + } + out.push_back(r); + }; + emit(d.tape); + while (!stack.empty()) + { + frame& top = stack.back(); + if (top.cur == top.end) + { + node& c = out[top.index]; + c.len = top.count; + c.next = static_cast(out.size() - top.index); + stack.pop_back(); + continue; + } + ++top.count; + const node* v = nullptr; + if (top.object) + { + out.push_back(string_node(*top.cur)); + v = document_data::deref(top.cur + 1); + top.cur = document_data::after(top.cur + 1); + } + else + { + v = document_data::deref(top.cur); + top.cur = document_data::after(top.cur); + } + emit(v); // may grow the stack (top is not used afterwards) + } +} + +/// the document as an image +inline std::vector save_image(const document_data& d) +{ +#if !NLOHMANN_VIEW_LITTLE_ENDIAN + throw_type_error(320, "json_document images need a little-endian target"); // LCOV_EXCL_LINE +#endif + const std::size_t arena_size = d.arena_size; + const node* nodes = d.tape; + std::size_t count = d.tape_size; + std::vector compacted; + std::string text_tail; + std::string arena_tail; + if (d.edits) + { + compact_nodes(d, arena_size, compacted, text_tail, arena_tail); + nodes = compacted.data(); + count = compacted.size(); + } + const std::size_t text_size = d.size + text_tail.size(); + const std::size_t total_arena = arena_size + arena_tail.size(); + if (NLOHMANN_VIEW_UNLIKELY(text_size >= image_limit || total_arena >= image_limit || count >= image_limit)) + { + // LCOV_EXCL_START (4 GiB) + throw_out_of_range(416, "images of 4 GiB or more are not supported by json_document"); + // LCOV_EXCL_STOP + } + image_header h{}; + h.magic = {{'N', 'J', 'V', 'I'}}; + h.version = image_version; + h.node_count = count; + h.text_size = text_size; + h.arena_size = total_arena; + std::vector image(sizeof(h) + (count * sizeof(node)) + text_size + 1 + total_arena + 1); + std::uint8_t* o = image.data(); + std::memcpy(o, &h, sizeof(h)); + o += sizeof(h); + std::memcpy(o, nodes, count * sizeof(node)); + // the hash indexes are rebuilt by load() + for (std::size_t i = 0; i < count; ++i) + { + if (nodes[i].kind == static_cast(value_t::object) && nodes[i].extra != 0) + { + node n = nodes[i]; + n.extra = 0; + std::memcpy(o + (i * sizeof(node)), &n, sizeof(node)); + } + } + o += count * sizeof(node); + const auto append = [&o](const char* s, std::size_t n) + { + if (n != 0) + { + std::memcpy(o, s, n); + o += n; + } + }; + append(d.src, d.size); + append(text_tail.data(), text_tail.size()); + *o++ = 0; + append(d.base[1], arena_size); + append(arena_tail.data(), arena_tail.size()); + *o = 0; + return image; +} + +/// whether a number node matches its token the way the parser records it +/// (after the bounds check) +inline bool check_number(const node& n, const unsigned char* text) +{ + const std::size_t len = number_length(n); + const unsigned char* const s = text + n.off; + const unsigned char* const e = s + len; + const unsigned char* p = s; + const bool negative = *p == '-'; + p += negative ? 1 : 0; + const unsigned char* const int_start = p; + if (p == e) + { + return false; + } + if (*p == '0') + { + ++p; + } + else if (*p >= '1' && *p <= '9') + { + while (p != e && is_digit(*p)) + { + ++p; + } + } + else + { + return false; + } + const auto int_digits = static_cast(p - int_start); + std::size_t frac_digits = 0; + bool is_float = false; + if (p != e && *p == '.') + { + const unsigned char* const f0 = ++p; + while (p != e && is_digit(*p)) + { + ++p; + } + if (p == f0) + { + return false; + } + frac_digits = static_cast(p - f0); + is_float = true; + } + std::int64_t exponent = 0; + if (p != e && (*p | 0x20u) == 'e') + { + ++p; + const bool exp_negative = p != e && *p == '-'; + p += (p != e && (*p == '+' || *p == '-')) ? 1 : 0; + if (p == e || !is_digit(*p)) + { + return false; + } + while (p != e && is_digit(*p)) + { + exponent = exponent < 100000 ? (exponent * 10) + (*p - '0') : exponent; + ++p; + } + exponent = exp_negative ? -exponent : exponent; + is_float = true; + } + if (p != e) + { + return false; + } + if (n.kind == static_cast(value_t::number_float)) + { + // the digit layout the parser records (or "many", as compaction + // writes it), and a finite value + const auto layout = static_cast((int_digits < 255 ? int_digits : 255) | ((frac_digits < 255 ? frac_digits : 255) << 8u)); + if (n.extra != layout && n.extra != 0xFFFFu) + { + return false; + } + // parse() rejects floats that overflow; as there, only a number whose + // magnitude could reach 1e308 needs the conversion + if (static_cast(int_digits) + exponent > 300) + { + const auto v = float_value(reinterpret_cast(s), n); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return v <= (std::numeric_limits::max)() && v >= -(std::numeric_limits::max)(); + } + return true; + } + // integers: the token's value is the stored one; number_integer nodes of + // edits can be non-negative (as basic_json keeps the type of a value) + const bool integer = n.kind == static_cast(value_t::number_integer); + if (is_float || int_digits > 20 || (negative && !integer)) + { + return false; + } + // (at most 19 digits cannot overflow; 20 digits are compared with 2^64 - 1) + if (int_digits == 20 && std::memcmp(int_start, "18446744073709551615", 20) > 0) + { + return false; + } + std::uint64_t m = 0; + for (const unsigned char* d = int_start; d != int_start + int_digits; ++d) + { + m = (m * 10) + static_cast(*d - '0'); + } + if (integer && m > (negative ? std::uint64_t{1} << 63u : (std::uint64_t{1} << 63u) - 1)) + { + return false; + } + return integer_bits(n) == (negative ? 0 - m : m); +} + +/// Check the nodes of a loaded image against its text and decoded strings: +/// kinds, flags, and `extra`; extents and element counts of arrays and +/// objects; keys; bounds; string contents (source strings as the parser +/// leaves them: no quotes, backslashes, or control characters; all strings +/// valid UTF-8); and number tokens. +inline bool check_image(const node* nodes, std::size_t count, const unsigned char* text, std::size_t text_size, + const unsigned char* arena, std::size_t arena_size, bool full) +{ + struct frame + { + std::size_t end; + std::uint32_t len; + std::uint32_t seen; + bool object; + bool expect_key; + }; + std::vector stack; + const auto check_string = [&](const node & n) -> bool + { + if ((n.flags & ~node_flags::escaped) != 0 || n.extra != 0) + { + return false; + } + const bool decoded = (n.flags & node_flags::escaped) != 0; + const unsigned char* const base = decoded ? arena : text; + const std::size_t limit = decoded ? arena_size : text_size; + if (n.off > limit || n.len > limit - n.off) + { + return false; + } + if (!full) + { + return true; + } + const unsigned char* const b = base + n.off; + return decoded ? valid_utf8_prefix(b, n.len) == n.len : scan_string_run(b, b + n.len) == b + n.len; + }; + // bounds of a number token; the recorded digit layout must lie within it + const auto number_in_bounds = [&](const node & n) -> bool + { + const std::size_t len = number_length(n); + if (len == 0 || n.off > text_size || len > text_size - n.off) + { + return false; + } + if (n.kind != static_cast(value_t::number_float)) + { + return (n.extra >> 8u) == 0; + } + // float_value() reads the sign, the integer digits, and the point and + // fraction digits the layout records (a layout of more than 19 digits + // means the general conversion, which stays within the token) + const std::size_t int_digits = n.extra & 0xFFu; + const std::size_t frac_digits = n.extra >> 8u; + const std::size_t need = (text[n.off] == '-' ? 1u : 0u) + int_digits + (frac_digits != 0 ? frac_digits + 1 : 0); + return int_digits + frac_digits > 19 || need <= len; + }; + std::size_t i = 0; + for (;;) + { + // close finished arrays and objects + while (!stack.empty() && i == stack.back().end) + { + const frame f = stack.back(); + if (f.seen != f.len || (f.object && !f.expect_key)) + { + return false; + } + stack.pop_back(); + if (!stack.empty()) + { + ++stack.back().seen; + stack.back().expect_key = true; + } + } + if (i == count) + { + return stack.empty(); + } + if (i != 0 && stack.empty()) + { + return false; // nodes after the root + } + const node& n = nodes[i]; + if (!stack.empty() && stack.back().object && stack.back().expect_key) + { + if (n.kind != static_cast(value_t::string) || !check_string(n)) + { + return false; + } + stack.back().expect_key = false; + ++i; + continue; + } + bool complete = true; + switch (static_cast(n.kind)) + { + case value_t::null: + // (the offset of a literal is read to size the output of dump()) + if (n.flags != 0 || n.extra != 0 || n.off > text_size) + { + return false; + } + break; + case value_t::boolean: + if ((n.flags & ~node_flags::is_true) != 0 || n.extra != 0 || n.off > text_size) + { + return false; + } + break; + case value_t::string: + if (!check_string(n)) + { + return false; + } + break; + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + if (n.flags != 0 || !number_in_bounds(n) || (full && !check_number(n, text))) + { + return false; + } + break; + case value_t::array: + case value_t::object: + { + const std::size_t limit = stack.empty() ? count : stack.back().end; + if (n.flags != 0 || n.extra != 0 || n.next == 0 || n.next > limit - i || n.off > text_size) + { + return false; + } + stack.push_back(frame{i + n.next, n.len, 0, n.kind == static_cast(value_t::object), true}); + complete = false; + break; + } + case value_t::binary: + case value_t::discarded: + default: + return false; + } + ++i; + if (complete && !stack.empty()) + { + ++stack.back().seen; + stack.back().expect_key = true; + } + } +} + +[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_invalid_image(const char* what) +{ + throw_parse_error(116, concat("invalid json_document image: ", what)); +} + +/// Read an image into d. The text and the decoded strings stay in the image; +/// the nodes are copied (so that they are aligned, and edits can change them). +inline void load_image(document_data& d, const std::uint8_t* image, std::size_t size, image_check check) +{ +#if !NLOHMANN_VIEW_LITTLE_ENDIAN + throw_type_error(320, "json_document images need a little-endian target"); // LCOV_EXCL_LINE +#endif + if (image == nullptr || size < sizeof(image_header)) + { + throw_invalid_image("too short"); + } + image_header h{}; + std::memcpy(&h, image, sizeof(h)); + // (the reserved fields are for later versions) + if (std::memcmp(h.magic.data(), "NJVI", 4) != 0 || h.version != image_version + || (h.reserved[0] | h.reserved[1] | h.reserved[2] | h.reserved[3]) != 0) + { + throw_invalid_image("unknown format"); + } + const std::size_t room = size - sizeof(h); + if (h.node_count == 0 || h.node_count > room / sizeof(node) || h.node_count >= image_limit || h.text_size >= image_limit || h.arena_size >= image_limit) + { + throw_invalid_image("sizes out of range"); + } + const auto count = static_cast(h.node_count); + const auto text_size = static_cast(h.text_size); + const auto arena_size = static_cast(h.arena_size); + const std::size_t text_at = sizeof(h) + (count * sizeof(node)); + // the text, a NUL, the decoded strings, a NUL, and nothing after them + if (size - text_at < 2 || text_size > size - text_at - 2 || arena_size != size - text_at - text_size - 2 + || image[text_at + text_size] != 0 || image[size - 1] != 0) + { + throw_invalid_image("sizes out of range"); + } + + d.discarded = true; + d.edits.reset(); + d.base[2] = nullptr; + d.owned.clear(); + if (d.owned_image.empty() || image != d.owned_image.data()) + { + d.owned_image.clear(); + } + d.arena.clear(); + d.indexes.clear(); + d.index_slots.clear(); + d.large_objects.clear(); + d.tape_size = 0; + d.reserve(count); + std::memcpy(d.tape, image + sizeof(h), count * sizeof(node)); + d.tape_size = count; + const std::uint8_t* const text = image + text_at; + const std::uint8_t* const arena = text + text_size + 1; + d.src = reinterpret_cast(text); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + d.size = text_size; + d.base[0] = d.src; + d.base[1] = reinterpret_cast(arena); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + d.arena_size = arena_size; + if (check != image_check::none && !check_image(d.tape, count, text, text_size, arena, arena_size, check == image_check::full)) + { + throw_invalid_image("the check failed"); + } + // the hash indexes of large objects, as after parsing + for (std::size_t i = 0; i < count; ++i) + { + node& n = d.tape[i]; + if (n.kind == static_cast(value_t::object)) + { + n.extra = 0; + if (n.len >= document_data::index_min_members) + { + d.large_objects.push_back(static_cast(i)); + } + } + } + build_object_indexes(d); + d.discarded = false; +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // #include // __ _____ _____ _____ // __| | __| | | | JSON for Modern C++ @@ -4237,192 +5073,6 @@ NLOHMANN_JSON_NAMESPACE_END // #include // #include -// __ _____ _____ _____ -// __| | __| | | | JSON for Modern C++ -// | | |__ | | | | | | version 3.12.0 -// |_____|_____|_____|_|___| https://github.com/nlohmann/json -// -// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann -// SPDX-License-Identifier: MIT - - - -#include // size_t -#include // int64_t, uint64_t -#include // numeric_limits -#include // string -#include // integral_constant - -// #include -// #include - -// #include - -// #include - -// #include - - -NLOHMANN_JSON_NAMESPACE_BEGIN -namespace detail -{ -namespace view -{ - -/*! -@brief the value of the float token of a node, as parse() converts it - -Uses the lexer's conversion (detail::convert_float), so that the values are -bit-identical to parse(): float and double are converted without allocation -and independent of the locale. The digit layout recorded while parsing locates -the decimal point and the exponent without scanning the token. -*/ -template -NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) -{ - const char* const last = first + n.len; - const std::size_t neg = first[0] == '-' ? 1 : 0; - const std::size_t int_digits = n.extra & 0xFFu; - const std::size_t frac_digits = n.extra >> 8u; - std::size_t dot = std::string::npos; - std::size_t mantissa_end = n.len; - if (int_digits != 255 && frac_digits != 255) - { - dot = frac_digits != 0 ? neg + int_digits : std::string::npos; - mantissa_end = neg + int_digits + (frac_digits != 0 ? 1 + frac_digits : 0); - } - else - { - // more digits than the layout records: locate them - for (std::size_t i = 0; i < n.len; ++i) - { - if (first[i] == '.') - { - dot = i; - } - else if (first[i] == 'e' || first[i] == 'E') - { - mantissa_end = i; - break; - } - } - } - return convert_float(first, last, dot, mantissa_end); -} - -/*! -@brief the digits of a float token with at most 19 digits, from its layout - -The digit layout recorded while parsing says where the integer digits, the -fraction digits, and the exponent are, so the digits are read eight at a -time without scanning. - -@param[in] p first character of the token -@param[in] e end of the token -@param[in] limit end of the readable memory (the source text) -*/ -NLOHMANN_VIEW_ALWAYS_INLINE float_significand layout_decimal(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept -{ - const bool negative = *p == '-'; - p += negative ? 1 : 0; - std::uint64_t w = parse_upto19(p, int_digits, limit); - p += int_digits; - std::int64_t q = 0; - if (frac_digits != 0) - { - w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit); - p += 1 + frac_digits; - q = -static_cast(frac_digits); - } - if (p != e) - { - // [eE][+-]digits; huge exponents saturate (the parser rejected overflow) - ++p; - const bool exp_negative = *p == '-'; - p += (*p == '-' || *p == '+') ? 1 : 0; - std::int64_t exp_value = 0; - for (; p != e; ++p) - { - if (exp_value < 0x10000000) - { - exp_value = (exp_value * 10) + (*p - '0'); - } - } - q += exp_negative ? -exp_value : exp_value; - } - - float_significand d; - d.w = w; - d.exponent = q; - d.negative = negative; - return d; -} - -/*! -@brief the value of a float token with at most 19 digits, from its layout - -The result is correctly rounded by the lexer's conversion -(detail::decimal_to_float(): Clinger's fast path where both operands are -exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19 -digits), so it is the value parse() produces. -*/ -template -NLOHMANN_VIEW_ALWAYS_INLINE FloatType layout_float(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept -{ - return decimal_to_float(layout_decimal(p, e, int_digits, frac_digits, limit)); -} - -/// the value of a float set by an edit: its token (the shortest round-trip -/// text, or "nan", "inf", "-inf") in the edit arena -template -NLOHMANN_VIEW_NOINLINE FloatType edited_float(const char* token, const node& n) -{ - if (token[0] == 'n') - { - return std::numeric_limits::quiet_NaN(); - } - if (token[0] == 'i' || (token[0] == '-' && token[1] == 'i')) - { - return token[0] == 'i' ? std::numeric_limits::infinity() : -std::numeric_limits::infinity(); - } - return float_value(token, n); -} - -/// the value of the float token of a node, as parse() converts it; floats and -/// doubles with at most 19 digits are converted from the digit layout -template -FloatType float_value(const document_data& d, const node& n) -{ - if (NLOHMANN_VIEW_UNLIKELY((n.flags & node_flags::storage) == node_flags::edited)) - { - return edited_float(d.str(n), n); - } - return float_value(d, n, std::integral_constant::value> {}); -} - -template -FloatType float_value(const document_data& d, const node& n, std::true_type /*binary32 or binary64*/) -{ - const unsigned int_digits = n.extra & 0xFFu; - const unsigned frac_digits = n.extra >> 8u; - if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many") - { - // (a float token not written by an edit is in the text) - const auto* const first = reinterpret_cast(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) - return layout_float(first, first + n.len, int_digits, frac_digits, reinterpret_cast(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) - } - return float_value(d.str(n), n); -} - -template -FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/) -{ - return float_value(d.str(n), n); -} - -} // namespace view -} // namespace detail -NLOHMANN_JSON_NAMESPACE_END NLOHMANN_JSON_NAMESPACE_BEGIN @@ -5048,7 +5698,9 @@ class view_serializer m_out.put('"'); } - /// as serializer::dump_escaped() for valid UTF-8 (the view has no other) + /// as serializer::dump_escaped(); strings of a document are valid UTF-8, + /// except in a damaged image loaded with image_check::bounds, for which + /// this throws what basic_json::dump() throws for the string template void write_escaped(const unsigned char* s, std::size_t n) { @@ -5072,12 +5724,13 @@ class view_serializer } std::uint32_t codepoint = s[i]; std::size_t len = 1; - if (codepoint >= 0xC0) + if (codepoint >= 0x80) { - len = 2; - if (codepoint >= 0xE0) + len = validate_one_utf8(s + i, n - i); + if (NLOHMANN_VIEW_UNLIKELY(len == 0)) { - len = codepoint >= 0xF0 ? 4 : 3; + invalid_utf8(s, n); + return; } codepoint &= 0xFFu >> (len + 1); for (std::size_t k = 1; k < len; ++k) @@ -5090,6 +5743,13 @@ class view_serializer } } + /// throw what basic_json::dump() throws for a string that is not valid UTF-8 + NLOHMANN_VIEW_NOINLINE static void invalid_utf8(const unsigned char* s, std::size_t n) + { + const string_t dumped = BasicJsonType(string_t(reinterpret_cast(s), n)).dump(); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + static_cast(dumped); + } + template void write_codepoint(std::uint32_t codepoint, const unsigned char* bytes, std::size_t len) { @@ -6260,7 +6920,7 @@ class basic_json_document /// whether the document holds its own copy of the text bool owns_source() const noexcept { - return m_data && !m_data->owned.empty() && m_data->src == m_data->owned.data(); + return m_data && ((!m_data->owned.empty() && m_data->src == m_data->owned.data()) || !m_data->owned_image.empty()); } /// number of index nodes (values plus object keys) @@ -6269,7 +6929,7 @@ class basic_json_document return m_data ? m_data->tape_size : 0; } - /// bytes held by the document (index, decoded strings, owned text) + /// bytes held by the document (index, decoded strings, owned text or image) std::size_t memory_usage() const noexcept { if (!m_data) @@ -6278,7 +6938,7 @@ class basic_json_document } return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node)) + (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0) - + m_data->arena.capacity() + m_data->owned.capacity() + + m_data->arena.capacity() + m_data->owned.capacity() + m_data->owned_image.capacity() + (m_data->indexes.capacity() * sizeof(document_data::object_index)) + (m_data->index_slots.capacity() * sizeof(std::uint32_t)) + (m_data->large_objects.capacity() * sizeof(std::uint32_t)) + (m_data->edits != nullptr ? m_data->edits->bytes : 0); @@ -6298,8 +6958,10 @@ class basic_json_document // allocate everything first, so that an exception leaves the document // unchanged + // (the decoded strings of a loaded image stay in the image) + const bool arena_in_use = d.base[1] == d.arena.data(); const bool shrink_arena = d.arena.capacity() > d.arena.size(); - std::string arena(shrink_arena ? d.arena : std::string()); + std::string arena(shrink_arena && arena_in_use ? d.arena : std::string()); // (edits link to the nodes of the index, which then stays in place) const bool shrink_tape = d.tape != d.inline_tape && d.tape_size != d.tape_cap && d.edits == nullptr; const bool into_header = d.tape_size <= d.inline_cap; @@ -6315,10 +6977,62 @@ class basic_json_document if (shrink_arena) { d.arena.swap(arena); - d.base[1] = d.arena.data(); + if (arena_in_use) + { + d.base[1] = d.arena.data(); + } } } + //////////// + // images // + //////////// + + /// how load() checks an image (full, bounds, or none) + using image_check = detail::view::image_check; + + /// The document as an image that load() reads without parsing: the node + /// index, the text, and the decoded strings. An edited document is + /// written in its current state (floats that are not finite become null, + /// as in dump()). + std::vector save() const + { + if (NLOHMANN_VIEW_UNLIKELY(!m_data || m_data->discarded)) + { + detail::view::throw_type_error(320, "cannot save a discarded json_document"); + } + return detail::view::save_image(*m_data); + } + + /// Read an image written by save(). The image is borrowed: it must stay + /// alive and unchanged while the document is used. + NLOHMANN_VIEW_NODISCARD + static basic_json_document load(const std::uint8_t* image, std::size_t size, const image_check check = image_check::full) + { + basic_json_document d; + d.ensure_data(nullptr, 0); + detail::view::load_image(*d.m_data, image, size, check); + return d; + } + + /// read an image (borrowed) + NLOHMANN_VIEW_NODISCARD + static basic_json_document load(const std::vector& image, const image_check check = image_check::full) + { + return load(image.data(), image.size(), check); + } + + /// read an image and keep it (no copy) + NLOHMANN_VIEW_NODISCARD + static basic_json_document load(std::vector&& image, const image_check check = image_check::full) + { + basic_json_document d; + d.ensure_data(nullptr, 0); + d.m_data->owned_image = std::move(image); + detail::view::load_image(*d.m_data, d.m_data->owned_image.data(), d.m_data->owned_image.size(), check); + return d; + } + /////////// // edits // /////////// @@ -6502,6 +7216,7 @@ class basic_json_document { d.owned.clear(); } + d.owned_image.clear(); d.src = src; d.size = size; d.tape_size = 0; @@ -6526,6 +7241,7 @@ class basic_json_document { d.base[0] = d.src; d.base[1] = d.arena.data(); + d.arena_size = d.arena.size(); detail::view::build_object_indexes(d); d.discarded = false; return; diff --git a/tests/src/unit-to_chars.cpp b/tests/src/unit-to_chars.cpp index 2745f7f95..8ced43ad0 100644 --- a/tests/src/unit-to_chars.cpp +++ b/tests/src/unit-to_chars.cpp @@ -639,12 +639,22 @@ std::pair digits_and_exponent(const std::string& s) return {digits, e}; } +/// the correctly rounded double of a decimal text +/// (not std::strtod: the C runtimes of some platforms, e.g. MinGW's, round +/// some 16 and 17 digit inputs wrongly) +double parse_double(const std::string& text) +{ + const nlohmann::json j = nlohmann::json::parse(text, nullptr, false); + // (a discarded value: out of range, as strtod's HUGE_VAL) + return j.is_discarded() ? std::numeric_limits::infinity() : j.get(); +} + /// whether the decimal digits * 10^e reads back as v bool reads_back(const std::string& digits, int e, double v) { const std::string text = digits + "e" + std::to_string(e); // (compared bit for bit: v is positive and finite, and -Wfloat-equal) - return reinterpret_bits(std::strtod(text.c_str(), nullptr)) == reinterpret_bits(v); + return reinterpret_bits(parse_double(text)) == reinterpret_bits(v); } /// Check the representation of a positive finite double: it reads back as @@ -655,7 +665,7 @@ void check_shortest(double v) char* end = nlohmann::detail::to_chars(buf.data(), buf.data() + 32, v); const std::string text(buf.data(), end); CAPTURE(text) - CHECK(std::strtod(text.c_str(), nullptr) == v); + CHECK(parse_double(text) == v); // the layout is that of format_buffer() for the same digits std::array reference{}; int len = 0; @@ -691,7 +701,8 @@ void check_shortest(double v) CHECK(!reads_back(std::to_string(candidate), e, v)); } } -#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars) + // (icpc with libstdc++ 11 defines __cpp_lib_to_chars, but has no floating-point std::to_chars) +#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars) && !defined(__INTEL_COMPILER) // the closest of the shortest representations, as std::to_chars finds it std::array std_text{}; const auto r = std::to_chars(std_text.data(), std_text.data() + std_text.size(), v, std::chars_format::scientific);