mirror of
https://github.com/nlohmann/json.git
synced 2026-10-06 22:47:13 +00:00
Add json_document and json_view, a read-only, zero-copy index of a JSON text, as the first public slice of the zero-copy view (#5295). A parse produces a flat array of 16-byte nodes in document order, one per value and one per object key. Strings stay in the source text; escaped strings are decoded into an arena. Integers are converted while their digits are in the cache; floats keep only their digit layout and are converted on read. Containers store the size of their subtree, so a reader can step over one in constant time. A document makes a handful of allocations, however many values it has. The parser accepts exactly what json::parse accepts, with every combination of ignore_comments and ignore_trailing_commas, with and without a trailing NUL, and under JSON_STRICT_NUL_HANDLING. It is portable C++11 and does not depend on byte order. basic_json_document adds parse, parse_copy, accept, read (reuses a document's memory), root, is_discarded, source, owns_source, node_count, memory_usage, and shrink_to_fit. basic_json_view adds type, the is_* queries, operator bool, size, empty, materialize, and source_offset. A parse error throws the same exception basic_json::parse would throw for the same input, message and position included. detail::abi_config keeps JSON_STRICT_NUL_HANDLING and JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON readable after json.hpp undefines them, in the ABI namespace so they always match the basic_json in use. A NUL byte that ends a // comment is the end of the input, as in parse() since #5696. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
98 lines
4.1 KiB
C++
98 lines
4.1 KiB
C++
// __ _____ _____ _____
|
|
// __| | __| | | | JSON for Modern C++
|
|
// | | |__ | | | | | | version 3.12.0
|
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
|
//
|
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
#pragma once
|
|
|
|
#include <cstddef> // size_t
|
|
#include <cstdint> // uint8_t, uint16_t, uint32_t, uint64_t
|
|
#include <cstring> // memcpy
|
|
|
|
#include <nlohmann/json.hpp>
|
|
#include <nlohmann/detail/view/macro_scope.hpp>
|
|
|
|
NLOHMANN_JSON_NAMESPACE_BEGIN
|
|
namespace detail
|
|
{
|
|
namespace view
|
|
{
|
|
|
|
// the node kinds are value_t values; the tests of is_container() and of the
|
|
// number kinds depend on this numbering
|
|
static_assert(static_cast<std::uint8_t>(value_t::null) == 0 && static_cast<std::uint8_t>(value_t::object) == 1
|
|
&& static_cast<std::uint8_t>(value_t::array) == 2 && static_cast<std::uint8_t>(value_t::string) == 3
|
|
&& static_cast<std::uint8_t>(value_t::boolean) == 4 && static_cast<std::uint8_t>(value_t::number_integer) == 5
|
|
&& static_cast<std::uint8_t>(value_t::number_unsigned) == 6 && static_cast<std::uint8_t>(value_t::number_float) == 7,
|
|
"the node format depends on the numbering of value_t");
|
|
|
|
/// node flags
|
|
struct node_flags
|
|
{
|
|
static constexpr std::uint8_t escaped = 1; ///< string payload lives in the decode arena, not the source
|
|
static constexpr std::uint8_t storage = 3; ///< mask: where a string or number token lives (index into document_data::base)
|
|
static constexpr std::uint8_t is_true = 4; ///< boolean value
|
|
};
|
|
|
|
/// One entry of the flat index, in document order. An object's members are
|
|
/// stored as key node followed by the value's subtree. Integers keep their
|
|
/// converted 64-bit value in the len/next bytes (the node after a scalar is
|
|
/// always the next one, and the token length follows from `extra`).
|
|
struct node
|
|
{
|
|
std::uint8_t kind; ///< value_t
|
|
std::uint8_t flags; ///< node_flags
|
|
std::uint16_t extra; ///< numbers: integer digits (low byte) and fraction digits (high byte), 255 = "many"; otherwise 0
|
|
std::uint32_t off; ///< source offset (string content, number token, literal, bracket); arena offset if node_flags::escaped
|
|
std::uint32_t len; ///< string: decoded bytes; float: token bytes; array/object: element count
|
|
std::uint32_t next; ///< array/object: number of nodes of the subtree (its extent in the enclosing sequence)
|
|
};
|
|
static_assert(sizeof(node) == 16, "node must stay 16 bytes");
|
|
|
|
NLOHMANN_VIEW_ALWAYS_INLINE bool is_container(const node& n) noexcept
|
|
{
|
|
return static_cast<unsigned>(n.kind) - 1u <= 1u;
|
|
}
|
|
|
|
/// the converted value of an integer node (stored in len/next)
|
|
NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept
|
|
{
|
|
std::uint64_t v = 0;
|
|
std::memcpy(&v, reinterpret_cast<const unsigned char*>(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
|
return v;
|
|
}
|
|
|
|
NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept
|
|
{
|
|
std::memcpy(reinterpret_cast<unsigned char*>(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
|
}
|
|
|
|
/// token length of a number node
|
|
NLOHMANN_VIEW_ALWAYS_INLINE std::uint32_t number_length(const node& n) noexcept
|
|
{
|
|
return n.kind == static_cast<std::uint8_t>(value_t::number_float) ? n.len
|
|
: (n.extra & 0xFFu) + (n.kind == static_cast<std::uint8_t>(value_t::number_integer) ? 1u : 0u);
|
|
}
|
|
|
|
/// estimated number of nodes for an input of `size` bytes (one node per ~12
|
|
/// bytes covers typical documents without regrowth)
|
|
inline std::size_t estimate_nodes(std::size_t size) noexcept
|
|
{
|
|
return (size / 12) + 16;
|
|
}
|
|
|
|
/// estimated number of nodes for the input [src, src + size): pretty-printed
|
|
/// input (whitespace after the first byte) needs about a node per 12 bytes,
|
|
/// minified input up to one per 4 (yyjson tells the two apart the same way)
|
|
inline std::size_t estimate_nodes(const char* src, std::size_t size) noexcept
|
|
{
|
|
return size >= 2 && (src[1] == ' ' || src[1] == '\n' || src[1] == '\r' || src[1] == '\t') ? estimate_nodes(size) : (size / 4) + 16;
|
|
}
|
|
|
|
} // namespace view
|
|
} // namespace detail
|
|
NLOHMANN_JSON_NAMESPACE_END
|