diff --git a/include/nlohmann/detail/view/serializer.hpp b/include/nlohmann/detail/view/serializer.hpp index 64d0bd933..8059f29da 100644 --- a/include/nlohmann/detail/view/serializer.hpp +++ b/include/nlohmann/detail/view/serializer.hpp @@ -31,7 +31,11 @@ namespace view { /// append-only output buffer: writes through a raw pointer into a string that -/// is resized ahead, and trimmed by finish() +/// is resized ahead, and trimmed by finish(). The estimate is reserved, and the +/// string grows in steps of 64 KiB within it: resize() fills the new bytes +/// with zeros (before C++23, a string cannot grow without), and a small step +/// is filled while the writer is about to use it, in the cache, instead of +/// filling the whole estimate in memory first. template class output_buffer { @@ -75,17 +79,49 @@ class output_buffer m_pos += n; } + /// the write position and the end of the writable space, for a writer + /// that keeps the position in a local variable (set_cursor() hands it back) + char* cursor() const noexcept + { + return m_pos; + } + + char* limit() const noexcept + { + return m_end; + } + + void set_cursor(char* p) noexcept + { + m_pos = p; + } + private: + /// the size of a growth step (a function: std::min() takes a reference, + /// which a static constexpr member does not have before C++17) + static constexpr std::size_t step() noexcept + { + return 65536; + } + static StringType& sized(StringType& out, std::size_t estimate) { - out.resize((std::max)(estimate, static_cast(64))); + out.reserve(estimate); + out.resize((std::min)((std::max)(estimate, static_cast(64)), step())); return out; } NLOHMANN_VIEW_NOINLINE void grow(std::size_t n) { const auto used = static_cast(m_pos - m_out.data()); - m_out.resize((std::max)(m_out.size() * 2, used + n + 256)); + // (a step does not go beyond the reserved estimate, so that a good + // estimate is never copied to a larger allocation) + const std::size_t size = (std::max)((std::min)(m_out.size() + step(), m_out.capacity()), used + n + 256); + if (size > m_out.capacity()) + { + m_out.reserve((std::max)(m_out.capacity() * 2, size)); + } + m_out.resize(size); m_pos = &m_out[0] + used; m_end = &m_out[0] + m_out.size(); } @@ -95,6 +131,105 @@ class output_buffer char* m_end; }; +/// The length of the run at s that dump() writes unchanged without +/// ensure_ascii: all bytes but quotes, backslashes, and control characters. +/// Unlike detail::string_bulk_run(), non-ASCII bytes are not validated: the +/// strings of a document are valid UTF-8 (a damaged image loaded with +/// image_check::bounds can have others, which are then written unchanged). +inline std::size_t plain_output_run(const unsigned char* s, std::size_t n) noexcept +{ + constexpr std::uint64_t ones = 0x0101010101010101ull; + constexpr std::uint64_t high = 0x8080808080808080ull; + std::size_t i = 0; + for (; i + 8 <= n; i += 8) + { + const std::uint64_t v = read_eight_bytes(s + i); + const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' + const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' + const std::uint64_t stop = (((q - ones) & ~q) | ((b - ones) & ~b) | ((v - 0x2020202020202020ull) & ~v)) & high; + if (stop != 0) + { + // the lowest flagged byte is the first stop: borrows only flag bytes above a true one + return i + (static_cast(count_trailing_zeros(stop)) / 8); + } + } + for (; i < n; ++i) + { + if (s[i] == '"' || s[i] == '\\' || s[i] < 0x20) + { + return i; + } + } + return n; +} + +/// A stack that starts in a buffer of the caller (a local array) and moves to +/// the heap (a vector of the caller) only when that is full, so that dumps of +/// shallow documents need no allocation. The top is a pointer, as in +/// std::vector. The address of the stack never escapes (the growth gets the +/// vector and returns the new storage), so its pointers stay in registers. +template +class small_stack +{ + public: + small_stack(T* buffer, std::size_t capacity, std::vector& heap) noexcept + : m_begin(buffer), m_top(buffer), m_end(buffer + capacity), m_heap(&heap) + {} + small_stack(const small_stack&) = delete; + small_stack(small_stack&&) = delete; + small_stack& operator=(const small_stack&) = delete; + small_stack& operator=(small_stack&&) = delete; + ~small_stack() = default; + + NLOHMANN_VIEW_ALWAYS_INLINE void push_back(const T& x) + { + if (NLOHMANN_VIEW_UNLIKELY(m_top == m_end)) + { + const std::size_t used = size(); + const std::size_t capacity = 2 * static_cast(m_end - m_begin); + m_begin = grow(*m_heap, m_begin, used, capacity); + m_top = m_begin + used; + m_end = m_begin + capacity; + } + *m_top++ = x; + } + + NLOHMANN_VIEW_ALWAYS_INLINE T& back() noexcept + { + return m_top[-1]; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void pop_back() noexcept + { + --m_top; + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool empty() const noexcept + { + return m_top == m_begin; + } + + NLOHMANN_VIEW_ALWAYS_INLINE std::size_t size() const noexcept + { + return static_cast(m_top - m_begin); + } + + private: + /// the used entries moved to heap storage of the given capacity + NLOHMANN_VIEW_NOINLINE static T* grow(std::vector& heap, const T* begin, std::size_t used, std::size_t capacity) + { + std::vector bigger(capacity); + std::copy(begin, begin + used, bigger.begin()); + heap.swap(bigger); + return heap.data(); + } + + T* m_begin; + T* m_top; + T* m_end; + std::vector* m_heap; +}; + /// how the view's dump() writes a value struct dump_style { @@ -129,6 +264,18 @@ class view_serializer void dump(const node* root) { + if (!m_style.pretty && !m_style.ensure_ascii) + { + if (m_style.source_numbers) + { + dump_compact(root); + } + else + { + dump_compact(root); + } + return; + } struct frame { const node* pos; ///< next element, or key of the next member @@ -136,7 +283,9 @@ class view_serializer bool object; bool first; ///< nothing written yet }; - std::vector stack; + std::array buffer; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read + std::vector heap; + small_stack stack(buffer.data(), buffer.size(), heap); const node* n = root; for (;;) { @@ -207,6 +356,289 @@ class view_serializer } private: + /*! + @brief the compact output without ensure_ascii (the default dump()) + + The same walk as dump(), with the write position in a local variable + (stores through char pointers would otherwise force a reload of the + buffer's members after each one), and with strings and number tokens of + the source copied by fixed-size moves of 32 bytes where the source has + that many bytes left, instead of a library call per token. The buffer + keeps 64 bytes of slack for the overshoot. + */ + /// a string that is not a plain string of the source (decoded, or written + /// by an edit), without ensure_ascii: runs without characters to escape + /// are copied + NLOHMANN_VIEW_NOINLINE void write_decoded(const node& n) + { + const auto* const s = reinterpret_cast(m_doc.str(n)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + m_out.put('"'); + for (std::size_t i = 0; i < n.len;) + { + const std::size_t run = plain_output_run(s + i, n.len - i); + if (run != 0) + { + m_out.put(reinterpret_cast(s + i), run); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + i += run; + continue; + } + write_codepoint(s[i], s + i, 1); // a quote, a backslash, or a control character + ++i; + } + m_out.put('"'); + } + + /// the copies of dump_compact() that are not fixed-size moves (long + /// strings, or near the end of the source); out of line, so that the + /// compiler does not merge the fixed-size moves into this call + NLOHMANN_VIEW_NOINLINE static void copy_long(char* to, const char* from, std::size_t n) noexcept + { + std::memcpy(to, from, n); + } + + template + void dump_compact(const node* root) + { + struct frame + { + const node* pos; ///< (editable documents) next element, or key of the next member + const node* end; + bool object; + }; + std::array buffer; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read + std::vector heap; + small_stack stack(buffer.data(), buffer.size(), heap); + const char* const src = m_doc.src; + const char* const src_end = src + m_doc.size; + char* w = m_out.cursor(); + char* lim = m_out.limit(); + // room for n bytes and the slack + const auto room = [&](std::size_t n) + { + if (NLOHMANN_VIEW_UNLIKELY(static_cast(lim - w) < n + 64)) + { + m_out.set_cursor(w); + m_out.reserve(n + 64); + w = m_out.cursor(); + lim = m_out.limit(); + } + }; + // copy n bytes of the source (after room(n)) + const auto copy = [&](const char* from, std::size_t n) + { + if (n <= 32 && static_cast(src_end - from) >= 32) + { + std::memcpy(w, from, 32); + } + else if (n <= 256 && static_cast(src_end - from) >= n + 32) + { + for (std::size_t i = 0; i < n; i += 32) + { + std::memcpy(w + i, from + i, 32); + } + } + else + { + copy_long(w, from, n); + } + w += n; + }; + // a literal of n bytes (after room(n)) + const auto literal = [&](const char* text, std::size_t n) + { + std::memcpy(w, text, n); + w += n; + }; + // a string that is not a plain string of the source (out of line, so + // that the cursor stays in a register here) + const auto escaped = [&](const node & n) + { + m_out.set_cursor(w); + write_decoded(n); + w = m_out.cursor(); + lim = m_out.limit(); + }; + + // Read-only documents: the elements of a container follow it in the + // node array, so the walk goes through the array in order, and a + // frame only needs the end of its container. Editable documents: the + // elements of a moved container live elsewhere, so a frame keeps the + // position of the next element (see navigation). + // The innermost open container is kept in registers (cur; end == + // nullptr: none), the stack holds the ones around it. + frame cur{nullptr, nullptr, false}; + const node* n = root; + for (;;) + { + // write the value at n (read-only documents: and advance n) + bool opened = false; + switch (static_cast(n->kind)) + { + case value_t::string: + if ((n->flags & node_flags::storage) == 0) + { + room(n->len + 2); + *w++ = '"'; + copy(src + n->off, n->len); + *w++ = '"'; + } + else + { + escaped(*n); + } + break; + case value_t::number_integer: + case value_t::number_unsigned: + { + const std::uint32_t len = number_length(*n); + room(len); + if (Editable && (n->flags & node_flags::storage) != 0) + { + copy_long(w, m_doc.str(*n), len); // a canonical token written by an edit + w += len; + break; + } + const char* const token = src + n->off; + if (!SourceNumbers && NLOHMANN_VIEW_UNLIKELY(len == 2 && token[0] == '-' && token[1] == '0')) + { + *w++ = '0'; // parse() reads -0 as the integer 0 + } + else + { + copy(token, len); + } + break; + } + case value_t::number_float: + if (SourceNumbers && (n->flags & node_flags::storage) != node_flags::edited) + { + room(n->len); + copy(src + n->off, n->len); + } + else if (std::is_same::value) + { + room(64); + w = write_double_at(w, *n); + } + else + { + m_out.set_cursor(w); + write_float_node(*n); + w = m_out.cursor(); + lim = m_out.limit(); + } + break; + case value_t::boolean: + room(8); + if ((n->flags & node_flags::is_true) != 0) + { + literal("true", 4); + } + else + { + literal("false", 5); + } + break; + case value_t::object: + case value_t::array: + { + const bool object = n->kind == static_cast(value_t::object); + room(8); + if (n->len == 0) + { + literal(object ? "{}" : "[]", 2); + } + else + { + *w++ = object ? '{' : '['; + stack.push_back(cur); + if (Editable) + { + cur = frame{nav::first(m_doc, n), nav::end(m_doc, n), object}; + } + else + { + cur = frame{nullptr, n + n->next, object}; + } + opened = true; + } + break; + } + case value_t::null: + room(8); + literal("null", 4); + break; + case value_t::binary: // LCOV_EXCL_LINE (not in a document) + case value_t::discarded: // LCOV_EXCL_LINE + default: // LCOV_EXCL_LINE + break; // LCOV_EXCL_LINE + } + if (!Editable) + { + ++n; // the next node: the first element of an opened container, or the node after a scalar + } + + // go to the next value: close finished containers, then separate + // (a container just opened has an element) + if (!opened) + { + for (;;) + { + if (cur.end == nullptr) + { + m_out.set_cursor(w); + m_out.finish(); + return; + } + if ((Editable ? cur.pos : n) != cur.end) + { + break; + } + room(1); + *w++ = cur.object ? '}' : ']'; + cur = stack.back(); + stack.pop_back(); + } + room(1); + *w++ = ','; + } + const node* const at = Editable ? cur.pos : n; + if (cur.object) + { + const node& key = *at; + if ((key.flags & node_flags::storage) == 0) + { + room(key.len + 3); + *w++ = '"'; + copy(src + key.off, key.len); + w[0] = '"'; + w[1] = ':'; + w += 2; + } + else + { + escaped(key); + room(1); + *w++ = ':'; + } + if (Editable) + { + n = nav::value(at + 1); + cur.pos = document_data::after(at + 1); + } + else + { + ++n; + } + } + else if (Editable) + { + n = nav::value(at); + cur.pos = document_data::after(at); + } + } + } + void newline(std::size_t level) { if (m_style.pretty) @@ -258,7 +690,7 @@ class view_serializer } else { - write_float(float_value(m_doc, n)); + write_float_node(n); } break; case value_t::object: // LCOV_EXCL_LINE (containers are written by dump()) @@ -270,6 +702,83 @@ class view_serializer } } + /// a float node as dump() writes it + void write_float_node(const node& n) + { + write_float_node(n, std::is_same {}); + } + + void write_float_node(const node& n, std::false_type /*other*/) + { + write_float(float_value(m_doc, n)); + } + + void write_float_node(const node& n, std::true_type /*double*/) + { + m_out.reserve(64); + m_out.set_cursor(write_double_at(m_out.cursor(), n)); + } + + /*! + @brief (doubles) the float at n as dump() writes it, at w (64 bytes of room) + + A token of at most 15 significant digits is written from its digits, + without a conversion: two decimals of at most 15 digits are farther + apart than the rounding interval of a (normal) double (the argument + behind DBL_DIG), so the token's digits are the shortest ones of its + double, which the library's conversion writes (Zmij). Other tokens are + converted from the digits already read. + */ + char* write_double_at(char* w, const node& n) + { + const unsigned int_digits = n.extra & 0xFFu; + const unsigned frac_digits = n.extra >> 8u; + if ((n.flags & node_flags::storage) != node_flags::edited && int_digits + frac_digits <= 19) + { + const auto* const first = reinterpret_cast(m_doc.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const float_significand d = layout_decimal(first, first + n.len, int_digits, frac_digits, reinterpret_cast(m_doc.src + m_doc.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + // (the exponent keeps the value far from subnormals and overflow) + if (d.w != 0 && d.w < 1000000000000000u && d.exponent >= -290 && d.exponent <= 290) + { + *w = '-'; + w += d.negative ? 1 : 0; + // (without leading zeros, all digits of the token count) + const unsigned char lead = first[d.negative ? 1 : 0]; + return lead != '0' ? ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast(int_digits + frac_digits), static_cast(d.exponent)) + : ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast(d.exponent)); + } + return write_double_value_at(w, decimal_to_float(d)); // (without reading the token again) + } + return write_double_value_at(w, static_cast(float_value(m_doc, n))); + } + + /// n bytes of text at w + static char* write_text_at(char* w, const char* text, std::size_t n) noexcept + { + std::memcpy(w, text, n); + return w + n; + } + + /// a double as dump() writes it, at w (64 bytes of room) + static char* write_double_value_at(char* w, double x) + { + // (from the bits: without the checks of to_chars()) + std::uint64_t bits = 0; + std::memcpy(&bits, &x, sizeof(bits)); + if (NLOHMANN_VIEW_UNLIKELY((bits & 0x7FF0000000000000u) == 0x7FF0000000000000u)) + { + return write_text_at(w, "null", 4); + } + *w = '-'; + w += bits >> 63u; + bits &= ~(std::uint64_t{1} << 63u); + if (bits == 0) + { + return write_text_at(w, "0.0", 3); + } + return ::nlohmann::detail::dtoa_impl::write_shortest(w, ::nlohmann::detail::zmij::to_shortest(bits)); + } + /// as serializer::dump_float() void write_float(number_float_t x) { diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 53ec3dfd7..219cb87bf 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -592,8 +592,10 @@ class basic_json_view style.indent_char = indent_char; style.ensure_ascii = ensure_ascii; style.source_numbers = numbers == number_format::source; - // the compact text is about as long as the source text of the value - const std::size_t estimate = source_extent() + (style.pretty ? source_extent() / 2 : 0) + 64; + // the compact text is about as long as the source text of the value; + // the compact writer keeps 64 bytes of slack, so that it does not grow + // the buffer just before the end + const std::size_t estimate = source_extent() + (style.pretty ? source_extent() / 2 : 0) + 160; detail::view::view_serializer(*m_doc, out, estimate, style).dump(m_node); return out; } diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index b4dd1cd32..00b0450f2 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -5407,7 +5407,11 @@ namespace view { /// append-only output buffer: writes through a raw pointer into a string that -/// is resized ahead, and trimmed by finish() +/// is resized ahead, and trimmed by finish(). The estimate is reserved, and the +/// string grows in steps of 64 KiB within it: resize() fills the new bytes +/// with zeros (before C++23, a string cannot grow without), and a small step +/// is filled while the writer is about to use it, in the cache, instead of +/// filling the whole estimate in memory first. template class output_buffer { @@ -5451,17 +5455,49 @@ class output_buffer m_pos += n; } + /// the write position and the end of the writable space, for a writer + /// that keeps the position in a local variable (set_cursor() hands it back) + char* cursor() const noexcept + { + return m_pos; + } + + char* limit() const noexcept + { + return m_end; + } + + void set_cursor(char* p) noexcept + { + m_pos = p; + } + private: + /// the size of a growth step (a function: std::min() takes a reference, + /// which a static constexpr member does not have before C++17) + static constexpr std::size_t step() noexcept + { + return 65536; + } + static StringType& sized(StringType& out, std::size_t estimate) { - out.resize((std::max)(estimate, static_cast(64))); + out.reserve(estimate); + out.resize((std::min)((std::max)(estimate, static_cast(64)), step())); return out; } NLOHMANN_VIEW_NOINLINE void grow(std::size_t n) { const auto used = static_cast(m_pos - m_out.data()); - m_out.resize((std::max)(m_out.size() * 2, used + n + 256)); + // (a step does not go beyond the reserved estimate, so that a good + // estimate is never copied to a larger allocation) + const std::size_t size = (std::max)((std::min)(m_out.size() + step(), m_out.capacity()), used + n + 256); + if (size > m_out.capacity()) + { + m_out.reserve((std::max)(m_out.capacity() * 2, size)); + } + m_out.resize(size); m_pos = &m_out[0] + used; m_end = &m_out[0] + m_out.size(); } @@ -5471,6 +5507,105 @@ class output_buffer char* m_end; }; +/// The length of the run at s that dump() writes unchanged without +/// ensure_ascii: all bytes but quotes, backslashes, and control characters. +/// Unlike detail::string_bulk_run(), non-ASCII bytes are not validated: the +/// strings of a document are valid UTF-8 (a damaged image loaded with +/// image_check::bounds can have others, which are then written unchanged). +inline std::size_t plain_output_run(const unsigned char* s, std::size_t n) noexcept +{ + constexpr std::uint64_t ones = 0x0101010101010101ull; + constexpr std::uint64_t high = 0x8080808080808080ull; + std::size_t i = 0; + for (; i + 8 <= n; i += 8) + { + const std::uint64_t v = read_eight_bytes(s + i); + const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' + const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' + const std::uint64_t stop = (((q - ones) & ~q) | ((b - ones) & ~b) | ((v - 0x2020202020202020ull) & ~v)) & high; + if (stop != 0) + { + // the lowest flagged byte is the first stop: borrows only flag bytes above a true one + return i + (static_cast(count_trailing_zeros(stop)) / 8); + } + } + for (; i < n; ++i) + { + if (s[i] == '"' || s[i] == '\\' || s[i] < 0x20) + { + return i; + } + } + return n; +} + +/// A stack that starts in a buffer of the caller (a local array) and moves to +/// the heap (a vector of the caller) only when that is full, so that dumps of +/// shallow documents need no allocation. The top is a pointer, as in +/// std::vector. The address of the stack never escapes (the growth gets the +/// vector and returns the new storage), so its pointers stay in registers. +template +class small_stack +{ + public: + small_stack(T* buffer, std::size_t capacity, std::vector& heap) noexcept + : m_begin(buffer), m_top(buffer), m_end(buffer + capacity), m_heap(&heap) + {} + small_stack(const small_stack&) = delete; + small_stack(small_stack&&) = delete; + small_stack& operator=(const small_stack&) = delete; + small_stack& operator=(small_stack&&) = delete; + ~small_stack() = default; + + NLOHMANN_VIEW_ALWAYS_INLINE void push_back(const T& x) + { + if (NLOHMANN_VIEW_UNLIKELY(m_top == m_end)) + { + const std::size_t used = size(); + const std::size_t capacity = 2 * static_cast(m_end - m_begin); + m_begin = grow(*m_heap, m_begin, used, capacity); + m_top = m_begin + used; + m_end = m_begin + capacity; + } + *m_top++ = x; + } + + NLOHMANN_VIEW_ALWAYS_INLINE T& back() noexcept + { + return m_top[-1]; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void pop_back() noexcept + { + --m_top; + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool empty() const noexcept + { + return m_top == m_begin; + } + + NLOHMANN_VIEW_ALWAYS_INLINE std::size_t size() const noexcept + { + return static_cast(m_top - m_begin); + } + + private: + /// the used entries moved to heap storage of the given capacity + NLOHMANN_VIEW_NOINLINE static T* grow(std::vector& heap, const T* begin, std::size_t used, std::size_t capacity) + { + std::vector bigger(capacity); + std::copy(begin, begin + used, bigger.begin()); + heap.swap(bigger); + return heap.data(); + } + + T* m_begin; + T* m_top; + T* m_end; + std::vector* m_heap; +}; + /// how the view's dump() writes a value struct dump_style { @@ -5505,6 +5640,18 @@ class view_serializer void dump(const node* root) { + if (!m_style.pretty && !m_style.ensure_ascii) + { + if (m_style.source_numbers) + { + dump_compact(root); + } + else + { + dump_compact(root); + } + return; + } struct frame { const node* pos; ///< next element, or key of the next member @@ -5512,7 +5659,9 @@ class view_serializer bool object; bool first; ///< nothing written yet }; - std::vector stack; + std::array buffer; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read + std::vector heap; + small_stack stack(buffer.data(), buffer.size(), heap); const node* n = root; for (;;) { @@ -5583,6 +5732,289 @@ class view_serializer } private: + /*! + @brief the compact output without ensure_ascii (the default dump()) + + The same walk as dump(), with the write position in a local variable + (stores through char pointers would otherwise force a reload of the + buffer's members after each one), and with strings and number tokens of + the source copied by fixed-size moves of 32 bytes where the source has + that many bytes left, instead of a library call per token. The buffer + keeps 64 bytes of slack for the overshoot. + */ + /// a string that is not a plain string of the source (decoded, or written + /// by an edit), without ensure_ascii: runs without characters to escape + /// are copied + NLOHMANN_VIEW_NOINLINE void write_decoded(const node& n) + { + const auto* const s = reinterpret_cast(m_doc.str(n)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + m_out.put('"'); + for (std::size_t i = 0; i < n.len;) + { + const std::size_t run = plain_output_run(s + i, n.len - i); + if (run != 0) + { + m_out.put(reinterpret_cast(s + i), run); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + i += run; + continue; + } + write_codepoint(s[i], s + i, 1); // a quote, a backslash, or a control character + ++i; + } + m_out.put('"'); + } + + /// the copies of dump_compact() that are not fixed-size moves (long + /// strings, or near the end of the source); out of line, so that the + /// compiler does not merge the fixed-size moves into this call + NLOHMANN_VIEW_NOINLINE static void copy_long(char* to, const char* from, std::size_t n) noexcept + { + std::memcpy(to, from, n); + } + + template + void dump_compact(const node* root) + { + struct frame + { + const node* pos; ///< (editable documents) next element, or key of the next member + const node* end; + bool object; + }; + std::array buffer; // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init): written before read + std::vector heap; + small_stack stack(buffer.data(), buffer.size(), heap); + const char* const src = m_doc.src; + const char* const src_end = src + m_doc.size; + char* w = m_out.cursor(); + char* lim = m_out.limit(); + // room for n bytes and the slack + const auto room = [&](std::size_t n) + { + if (NLOHMANN_VIEW_UNLIKELY(static_cast(lim - w) < n + 64)) + { + m_out.set_cursor(w); + m_out.reserve(n + 64); + w = m_out.cursor(); + lim = m_out.limit(); + } + }; + // copy n bytes of the source (after room(n)) + const auto copy = [&](const char* from, std::size_t n) + { + if (n <= 32 && static_cast(src_end - from) >= 32) + { + std::memcpy(w, from, 32); + } + else if (n <= 256 && static_cast(src_end - from) >= n + 32) + { + for (std::size_t i = 0; i < n; i += 32) + { + std::memcpy(w + i, from + i, 32); + } + } + else + { + copy_long(w, from, n); + } + w += n; + }; + // a literal of n bytes (after room(n)) + const auto literal = [&](const char* text, std::size_t n) + { + std::memcpy(w, text, n); + w += n; + }; + // a string that is not a plain string of the source (out of line, so + // that the cursor stays in a register here) + const auto escaped = [&](const node & n) + { + m_out.set_cursor(w); + write_decoded(n); + w = m_out.cursor(); + lim = m_out.limit(); + }; + + // Read-only documents: the elements of a container follow it in the + // node array, so the walk goes through the array in order, and a + // frame only needs the end of its container. Editable documents: the + // elements of a moved container live elsewhere, so a frame keeps the + // position of the next element (see navigation). + // The innermost open container is kept in registers (cur; end == + // nullptr: none), the stack holds the ones around it. + frame cur{nullptr, nullptr, false}; + const node* n = root; + for (;;) + { + // write the value at n (read-only documents: and advance n) + bool opened = false; + switch (static_cast(n->kind)) + { + case value_t::string: + if ((n->flags & node_flags::storage) == 0) + { + room(n->len + 2); + *w++ = '"'; + copy(src + n->off, n->len); + *w++ = '"'; + } + else + { + escaped(*n); + } + break; + case value_t::number_integer: + case value_t::number_unsigned: + { + const std::uint32_t len = number_length(*n); + room(len); + if (Editable && (n->flags & node_flags::storage) != 0) + { + copy_long(w, m_doc.str(*n), len); // a canonical token written by an edit + w += len; + break; + } + const char* const token = src + n->off; + if (!SourceNumbers && NLOHMANN_VIEW_UNLIKELY(len == 2 && token[0] == '-' && token[1] == '0')) + { + *w++ = '0'; // parse() reads -0 as the integer 0 + } + else + { + copy(token, len); + } + break; + } + case value_t::number_float: + if (SourceNumbers && (n->flags & node_flags::storage) != node_flags::edited) + { + room(n->len); + copy(src + n->off, n->len); + } + else if (std::is_same::value) + { + room(64); + w = write_double_at(w, *n); + } + else + { + m_out.set_cursor(w); + write_float_node(*n); + w = m_out.cursor(); + lim = m_out.limit(); + } + break; + case value_t::boolean: + room(8); + if ((n->flags & node_flags::is_true) != 0) + { + literal("true", 4); + } + else + { + literal("false", 5); + } + break; + case value_t::object: + case value_t::array: + { + const bool object = n->kind == static_cast(value_t::object); + room(8); + if (n->len == 0) + { + literal(object ? "{}" : "[]", 2); + } + else + { + *w++ = object ? '{' : '['; + stack.push_back(cur); + if (Editable) + { + cur = frame{nav::first(m_doc, n), nav::end(m_doc, n), object}; + } + else + { + cur = frame{nullptr, n + n->next, object}; + } + opened = true; + } + break; + } + case value_t::null: + room(8); + literal("null", 4); + break; + case value_t::binary: // LCOV_EXCL_LINE (not in a document) + case value_t::discarded: // LCOV_EXCL_LINE + default: // LCOV_EXCL_LINE + break; // LCOV_EXCL_LINE + } + if (!Editable) + { + ++n; // the next node: the first element of an opened container, or the node after a scalar + } + + // go to the next value: close finished containers, then separate + // (a container just opened has an element) + if (!opened) + { + for (;;) + { + if (cur.end == nullptr) + { + m_out.set_cursor(w); + m_out.finish(); + return; + } + if ((Editable ? cur.pos : n) != cur.end) + { + break; + } + room(1); + *w++ = cur.object ? '}' : ']'; + cur = stack.back(); + stack.pop_back(); + } + room(1); + *w++ = ','; + } + const node* const at = Editable ? cur.pos : n; + if (cur.object) + { + const node& key = *at; + if ((key.flags & node_flags::storage) == 0) + { + room(key.len + 3); + *w++ = '"'; + copy(src + key.off, key.len); + w[0] = '"'; + w[1] = ':'; + w += 2; + } + else + { + escaped(key); + room(1); + *w++ = ':'; + } + if (Editable) + { + n = nav::value(at + 1); + cur.pos = document_data::after(at + 1); + } + else + { + ++n; + } + } + else if (Editable) + { + n = nav::value(at); + cur.pos = document_data::after(at); + } + } + } + void newline(std::size_t level) { if (m_style.pretty) @@ -5634,7 +6066,7 @@ class view_serializer } else { - write_float(float_value(m_doc, n)); + write_float_node(n); } break; case value_t::object: // LCOV_EXCL_LINE (containers are written by dump()) @@ -5646,6 +6078,83 @@ class view_serializer } } + /// a float node as dump() writes it + void write_float_node(const node& n) + { + write_float_node(n, std::is_same {}); + } + + void write_float_node(const node& n, std::false_type /*other*/) + { + write_float(float_value(m_doc, n)); + } + + void write_float_node(const node& n, std::true_type /*double*/) + { + m_out.reserve(64); + m_out.set_cursor(write_double_at(m_out.cursor(), n)); + } + + /*! + @brief (doubles) the float at n as dump() writes it, at w (64 bytes of room) + + A token of at most 15 significant digits is written from its digits, + without a conversion: two decimals of at most 15 digits are farther + apart than the rounding interval of a (normal) double (the argument + behind DBL_DIG), so the token's digits are the shortest ones of its + double, which the library's conversion writes (Zmij). Other tokens are + converted from the digits already read. + */ + char* write_double_at(char* w, const node& n) + { + const unsigned int_digits = n.extra & 0xFFu; + const unsigned frac_digits = n.extra >> 8u; + if ((n.flags & node_flags::storage) != node_flags::edited && int_digits + frac_digits <= 19) + { + const auto* const first = reinterpret_cast(m_doc.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const float_significand d = layout_decimal(first, first + n.len, int_digits, frac_digits, reinterpret_cast(m_doc.src + m_doc.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + // (the exponent keeps the value far from subnormals and overflow) + if (d.w != 0 && d.w < 1000000000000000u && d.exponent >= -290 && d.exponent <= 290) + { + *w = '-'; + w += d.negative ? 1 : 0; + // (without leading zeros, all digits of the token count) + const unsigned char lead = first[d.negative ? 1 : 0]; + return lead != '0' ? ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast(int_digits + frac_digits), static_cast(d.exponent)) + : ::nlohmann::detail::dtoa_impl::write_short_decimal(w, d.w, static_cast(d.exponent)); + } + return write_double_value_at(w, decimal_to_float(d)); // (without reading the token again) + } + return write_double_value_at(w, static_cast(float_value(m_doc, n))); + } + + /// n bytes of text at w + static char* write_text_at(char* w, const char* text, std::size_t n) noexcept + { + std::memcpy(w, text, n); + return w + n; + } + + /// a double as dump() writes it, at w (64 bytes of room) + static char* write_double_value_at(char* w, double x) + { + // (from the bits: without the checks of to_chars()) + std::uint64_t bits = 0; + std::memcpy(&bits, &x, sizeof(bits)); + if (NLOHMANN_VIEW_UNLIKELY((bits & 0x7FF0000000000000u) == 0x7FF0000000000000u)) + { + return write_text_at(w, "null", 4); + } + *w = '-'; + w += bits >> 63u; + bits &= ~(std::uint64_t{1} << 63u); + if (bits == 0) + { + return write_text_at(w, "0.0", 3); + } + return ::nlohmann::detail::dtoa_impl::write_shortest(w, ::nlohmann::detail::zmij::to_shortest(bits)); + } + /// as serializer::dump_float() void write_float(number_float_t x) { @@ -6564,8 +7073,10 @@ class basic_json_view style.indent_char = indent_char; style.ensure_ascii = ensure_ascii; style.source_numbers = numbers == number_format::source; - // the compact text is about as long as the source text of the value - const std::size_t estimate = source_extent() + (style.pretty ? source_extent() / 2 : 0) + 64; + // the compact text is about as long as the source text of the value; + // the compact writer keeps 64 bytes of slack, so that it does not grow + // the buffer just before the end + const std::size_t estimate = source_extent() + (style.pretty ? source_extent() / 2 : 0) + 160; detail::view::view_serializer(*m_doc, out, estimate, style).dump(m_node); return out; } diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 9494fb3e8..86c37e3f9 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -1147,6 +1147,57 @@ TEST_CASE("json_view dump") CHECK(d.root().dump() == json::parse(text).dump()); CHECK(d.root().dump() == "[1.5,100.0,0,-0.0,1.2345678901234568e+29,18446744073709551615,-9223372036854775808,0.1,1e-07,5e-324]"); CHECK(d.root().dump(-1, ' ', false, json_view::number_format::source) == "[1.50,1E2,-0,-0.0,123456789012345678901234567890,18446744073709551615,-9223372036854775808,0.1,1e-7,5e-324]"); + // also indented, and with ensure_ascii + CHECK(d.root().dump(0, ' ', false, json_view::number_format::source) == "[\n1.50,\n1E2,\n-0,\n-0.0,\n123456789012345678901234567890,\n18446744073709551615,\n-9223372036854775808,\n0.1,\n1e-7,\n5e-324\n]"); + CHECK(d.root().dump(-1, ' ', true, json_view::number_format::source) == "[1.50,1E2,-0,-0.0,123456789012345678901234567890,18446744073709551615,-9223372036854775808,0.1,1e-7,5e-324]"); + + // float tokens of up to 17 significant digits in every spelling: those + // of at most 15 digits are written from their digits, the others + // through the conversion; both as dump() writes them + { + std::mt19937_64 tokens(1170); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) + // a number below n; the remainder is a std::uint64_t, which is + // std::size_t on some platforms and wider on others + const auto draw = [&tokens](std::size_t n) + { + const std::uint64_t r = tokens() % n; + return static_cast(r); + }; + std::string many_tokens = "["; + for (int i = 0; i < 20000; ++i) + { + const std::size_t length = 1 + draw(17); + std::string digits(1, static_cast('1' + draw(9))); + for (std::size_t k = 1; k < length; ++k) + { + digits += static_cast('0' + draw(10)); + } + digits += std::string(draw(4), '0'); // trailing zeros + std::string token = draw(3) == 0 ? "-" : ""; + const std::size_t point = draw(digits.size() + 1); + if (point == 0) + { + token += "0." + std::string(draw(5), '0') + digits; + } + else + { + token += digits.substr(0, point) + (point < digits.size() ? "." + digits.substr(point) : ""); + } + // an exponent that keeps the value between about 1e-320 and 1e300 + const int exponent = static_cast(draw(600)) - 300 - static_cast(point); + if (draw(4) != 0) + { + token += (draw(2) == 0 ? "e" : "E") + std::string(exponent >= 0 && draw(2) == 0 ? "+" : "") + std::to_string(exponent); + } + else if (point == digits.size()) + { + token += ".0"; // (a float, not an integer) + } + many_tokens += (i != 0 ? "," : "") + token; + } + many_tokens += ']'; + CHECK(json_document::parse(many_tokens).root().dump() == json::parse(many_tokens).dump()); + } // random doubles, written as parse() and dump() would std::mt19937_64 rng(1170); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) diff --git a/tests/src/unit-json_view_edit.cpp b/tests/src/unit-json_view_edit.cpp index 5e0ece881..1db051b85 100644 --- a/tests/src/unit-json_view_edit.cpp +++ b/tests/src/unit-json_view_edit.cpp @@ -26,6 +26,7 @@ using ptr_t = ordered_json::json_pointer; #include #include #include +#include #include #include #include @@ -481,6 +482,19 @@ TEST_CASE("json_view edits: views and values") CHECK(d.root().materialize().dump() == json::parse(R"([1.5, 100.0, 0.1, null, null, 18446744073709551615, -9223372036854775808])").dump()); } + SECTION("numbers of other float types") + { + // doubles have their own path to the output; other float types are + // written as basic_json writes them, non-finite values as null + using json_float = nlohmann::basic_json; + using document_float = nlohmann::basic_json_document; + document_float d = document_float::parse("[1.5]"); + d.push_back(d.root(), std::numeric_limits::quiet_NaN()); + d.push_back(d.root(), -std::numeric_limits::infinity()); + CHECK(d.root().dump() == "[1.5,null,null]"); + CHECK(d.root().dump(2) == json_float::parse("[1.5, null, null]").dump(2)); + } + SECTION("nulls become containers, and the root can be replaced") { json_editable_document d = json_editable_document::parse("[null, null]");