mirror of
https://github.com/nlohmann/json.git
synced 2026-10-06 22:47:13 +00:00
Add json_document and json_view: node index, parser, and document
Add json_document and json_view, a read-only, zero-copy index of a JSON text, as the first public slice of the zero-copy view (#5295). A parse produces a flat array of 16-byte nodes in document order, one per value and one per object key. Strings stay in the source text; escaped strings are decoded into an arena. Integers are converted while their digits are in the cache; floats keep only their digit layout and are converted on read. Containers store the size of their subtree, so a reader can step over one in constant time. A document makes a handful of allocations, however many values it has. The parser accepts exactly what json::parse accepts, with every combination of ignore_comments and ignore_trailing_commas, with and without a trailing NUL, and under JSON_STRICT_NUL_HANDLING. It is portable C++11 and does not depend on byte order. basic_json_document adds parse, parse_copy, accept, read (reuses a document's memory), root, is_discarded, source, owns_source, node_count, memory_usage, and shrink_to_fit. basic_json_view adds type, the is_* queries, operator bool, size, empty, materialize, and source_offset. A parse error throws the same exception basic_json::parse would throw for the same input, message and position included. detail::abi_config keeps JSON_STRICT_NUL_HANDLING and JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON readable after json.hpp undefines them, in the ABI namespace so they always match the basic_json in use. A NUL byte that ends a // comment is the end of the input, as in parse() since #5696. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
1 parent
17842694f6
commit
4303ba674d
146 files changed
+9611
-13
No files matched your search
@@ -0,0 +1,97 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint8_t, uint16_t, uint32_t, uint64_t
|
||||
#include <cstring> // memcpy
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
namespace view
|
||||
{
|
||||
|
||||
// the node kinds are value_t values; the tests of is_container() and of the
|
||||
// number kinds depend on this numbering
|
||||
static_assert(static_cast<std::uint8_t>(value_t::null) == 0 && static_cast<std::uint8_t>(value_t::object) == 1
|
||||
&& static_cast<std::uint8_t>(value_t::array) == 2 && static_cast<std::uint8_t>(value_t::string) == 3
|
||||
&& static_cast<std::uint8_t>(value_t::boolean) == 4 && static_cast<std::uint8_t>(value_t::number_integer) == 5
|
||||
&& static_cast<std::uint8_t>(value_t::number_unsigned) == 6 && static_cast<std::uint8_t>(value_t::number_float) == 7,
|
||||
"the node format depends on the numbering of value_t");
|
||||
|
||||
/// node flags
|
||||
struct node_flags
|
||||
{
|
||||
static constexpr std::uint8_t escaped = 1; ///< string payload lives in the decode arena, not the source
|
||||
static constexpr std::uint8_t storage = 3; ///< mask: where a string or number token lives (index into document_data::base)
|
||||
static constexpr std::uint8_t is_true = 4; ///< boolean value
|
||||
};
|
||||
|
||||
/// One entry of the flat index, in document order. An object's members are
|
||||
/// stored as key node followed by the value's subtree. Integers keep their
|
||||
/// converted 64-bit value in the len/next bytes (the node after a scalar is
|
||||
/// always the next one, and the token length follows from `extra`).
|
||||
struct node
|
||||
{
|
||||
std::uint8_t kind; ///< value_t
|
||||
std::uint8_t flags; ///< node_flags
|
||||
std::uint16_t extra; ///< numbers: integer digits (low byte) and fraction digits (high byte), 255 = "many"; otherwise 0
|
||||
std::uint32_t off; ///< source offset (string content, number token, literal, bracket); arena offset if node_flags::escaped
|
||||
std::uint32_t len; ///< string: decoded bytes; float: token bytes; array/object: element count
|
||||
std::uint32_t next; ///< array/object: number of nodes of the subtree (its extent in the enclosing sequence)
|
||||
};
|
||||
static_assert(sizeof(node) == 16, "node must stay 16 bytes");
|
||||
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE bool is_container(const node& n) noexcept
|
||||
{
|
||||
return static_cast<unsigned>(n.kind) - 1u <= 1u;
|
||||
}
|
||||
|
||||
/// the converted value of an integer node (stored in len/next)
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept
|
||||
{
|
||||
std::uint64_t v = 0;
|
||||
std::memcpy(&v, reinterpret_cast<const unsigned char*>(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
return v;
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept
|
||||
{
|
||||
std::memcpy(reinterpret_cast<unsigned char*>(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
}
|
||||
|
||||
/// token length of a number node
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE std::uint32_t number_length(const node& n) noexcept
|
||||
{
|
||||
return n.kind == static_cast<std::uint8_t>(value_t::number_float) ? n.len
|
||||
: (n.extra & 0xFFu) + (n.kind == static_cast<std::uint8_t>(value_t::number_integer) ? 1u : 0u);
|
||||
}
|
||||
|
||||
/// estimated number of nodes for an input of `size` bytes (one node per ~12
|
||||
/// bytes covers typical documents without regrowth)
|
||||
inline std::size_t estimate_nodes(std::size_t size) noexcept
|
||||
{
|
||||
return (size / 12) + 16;
|
||||
}
|
||||
|
||||
/// estimated number of nodes for the input [src, src + size): pretty-printed
|
||||
/// input (whitespace after the first byte) needs about a node per 12 bytes,
|
||||
/// minified input up to one per 4 (yyjson tells the two apart the same way)
|
||||
inline std::size_t estimate_nodes(const char* src, std::size_t size) noexcept
|
||||
{
|
||||
return size >= 2 && (src[1] == ' ' || src[1] == '\n' || src[1] == '\r' || src[1] == '\t') ? estimate_nodes(size) : (size / 4) + 16;
|
||||
}
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
Reference in new issue
Block a user