Merge branch 'json-view/19-edit-set' into json-view/21-images

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann committed 2026-10-09 17:36:35 +02:00
commit 0a99602e37
64 files changed
+2746 -897

No files matched your search

+3 -4
View File
@@ -15,10 +15,11 @@ namespace detail
{
/*!
@brief the configuration macros that change the library's behavior
@brief the configuration macros that json_view.hpp reads
json.hpp undefines these macros at its end (see macro_unscope.hpp), so code
that builds on the library after it (json_view.hpp) reads them here. Like the
that builds on the library after it (json_view.hpp) reads them here. A macro
is added when the view starts to depend on it. Like the
macros, they are part of the ABI namespace, so they always match the
basic_json they are used with.
*/
@@ -26,8 +27,6 @@ struct abi_config
{
/// JSON_STRICT_NUL_HANDLING: a null byte is an error, not the end of input
static constexpr bool strict_nul_handling = JSON_STRICT_NUL_HANDLING != 0;
/// JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
static constexpr bool legacy_discarded_value_comparison = JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON != 0;
};
} // namespace detail
+20 -5
View File
@@ -9,8 +9,9 @@
#pragma once
#include <cstdint> // uint64_t
#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
#include <intrin0.h> // __umulh, _umul128
#include <cstring> // memcpy
#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) && (!defined(__SIZEOF_INT128__) || (!defined(__GNUC__) && !defined(__clang__)))
#include <intrin0.h> // __umulh, _umul128, _BitScanForward64, _BitScanReverse64
#endif
#include <nlohmann/detail/macro_scope.hpp> // JSON_HEDLEY_ALWAYS_INLINE, NLOHMANN_JSON_NAMESPACE_BEGIN
@@ -28,6 +29,10 @@ inline int count_leading_zeros(std::uint64_t x) noexcept
{
#if defined(__GNUC__) || defined(__clang__)
return __builtin_clzll(x);
#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
unsigned long index = 0;
_BitScanReverse64(&index, x);
return 63 - static_cast<int>(index);
#else
int n = 0;
for (int shift = 32; shift != 0; shift >>= 1)
@@ -47,6 +52,10 @@ inline int count_trailing_zeros(std::uint64_t x) noexcept
{
#if defined(__GNUC__) || defined(__clang__)
return __builtin_ctzll(x);
#elif defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
unsigned long index = 0;
_BitScanForward64(&index, x);
return static_cast<int>(index);
#else
int n = 0;
for (int shift = 32; shift != 0; shift >>= 1)
@@ -94,15 +103,21 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc
#endif
}
/// eight bytes as a little-endian word (compilers fold this into one load on
/// little-endian targets; always inlined, as GCC otherwise calls it in the
/// number loops)
/// eight bytes as a little-endian word (a single load on little-endian
/// targets; always inlined, as GCC otherwise calls it in the number loops)
JSON_HEDLEY_ALWAYS_INLINE std::uint64_t read_eight_bytes(const unsigned char* b) noexcept
{
#if defined(_MSC_VER) || defined(__x86_64__) || defined(__i386__) || (defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__)
// the byte order already matches (all MSVC targets are little-endian)
std::uint64_t result = 0;
std::memcpy(&result, b, sizeof(result));
return result;
#else
return static_cast<std::uint64_t>(b[0]) | (static_cast<std::uint64_t>(b[1]) << 8u)
| (static_cast<std::uint64_t>(b[2]) << 16u) | (static_cast<std::uint64_t>(b[3]) << 24u)
| (static_cast<std::uint64_t>(b[4]) << 32u) | (static_cast<std::uint64_t>(b[5]) << 40u)
| (static_cast<std::uint64_t>(b[6]) << 48u) | (static_cast<std::uint64_t>(b[7]) << 56u);
#endif
}
/// eight bytes as a little-endian word
+49 -124
View File
@@ -4,6 +4,7 @@
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2009 Florian Loitsch <https://florian.loitsch.com/>
// SPDX-FileCopyrightText: 2025 Victor Zverovich <https://github.com/vitaut/zmij>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
@@ -939,88 +940,6 @@ void grisu2(char* buf, int& len, int& decimal_exponent, FloatType value)
grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus);
}
/*!
@brief the shortest digits of a positive finite float (other than double): Grisu2
*/
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1)
void shortest_digits(char* buf, int& len, int& decimal_exponent, FloatType value)
{
grisu2(buf, len, decimal_exponent, value);
}
/*!
@brief the shortest digits of a positive finite double: the conversion of
Zmij (see zmij.hpp), which always finds the shortest digits that read back as
the same value (Grisu2 does not for about one double in a thousand), and the
closest of them if there are several
v = buf * 10^decimal_exponent, as for grisu2()
*/
JSON_HEDLEY_NON_NULL(1)
inline void shortest_digits(char* buf, int& len, int& decimal_exponent, double value)
{
static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
JSON_ASSERT(std::isfinite(value));
JSON_ASSERT(value > 0);
std::uint64_t bits = 0;
std::memcpy(&bits, &value, sizeof(bits));
zmij::decimal d = zmij::to_decimal(bits);
// without trailing zeros (up to 16): 8, 4, 2, 1 at a time
while (d.significand % 100000000 == 0)
{
d.significand /= 100000000;
d.exponent += 8;
}
if (d.significand % 10000 == 0)
{
d.significand /= 10000;
d.exponent += 4;
}
if (d.significand % 100 == 0)
{
d.significand /= 100;
d.exponent += 2;
}
if (d.significand % 10 == 0)
{
d.significand /= 10;
d.exponent += 1;
}
// at most 17 digits, written from the back two at a time
static constexpr const char* pairs =
"00010203040506070809101112131415161718192021222324252627282930313233343536373839"
"40414243444546474849505152535455565758596061626364656667686970717273747576777879"
"8081828384858687888990919293949596979899";
std::array<char, 20> digits{};
std::size_t n = digits.size();
while (d.significand >= 100)
{
const std::uint64_t two_digits = d.significand % 100; // a variable: GCC calls a cast of the remainder useless where std::uint64_t is std::size_t
const auto i = static_cast<std::size_t>(two_digits) * 2;
d.significand /= 100;
n -= 2;
digits[n] = pairs[i];
digits[n + 1] = pairs[i + 1];
}
if (d.significand >= 10)
{
const auto i = static_cast<std::size_t>(d.significand) * 2;
n -= 2;
digits[n] = pairs[i];
digits[n + 1] = pairs[i + 1];
}
else
{
digits[--n] = static_cast<char>('0' + d.significand);
}
len = static_cast<int>(digits.size() - n);
std::memcpy(buf, digits.data() + n, static_cast<std::size_t>(len));
decimal_exponent = d.exponent;
}
/*!
@brief appends a decimal representation of e to buf
@return a pointer to the element following the exponent.
@@ -1423,53 +1342,30 @@ inline char* write_shortest(char* first, const zmij::shortest_decimal d) noexcep
return end + (three ? 5 : 4);
}
/// the powers of ten up to 10^16
inline const std::array<std::uint64_t, 17>& powers_of_ten_16() noexcept
{
static const std::array<std::uint64_t, 17> powers =
{
{
1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, 10000000000u,
100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, 10000000000000000u
}
};
return powers;
}
/*!
@brief digits * 10^exp, as write_decimal() writes it, for the digits of a
double that need no conversion (count digits, at most 15, the first not 0;
trailing zeros allowed): extended to 16 digits and written by write_shortest()
@brief whether FloatType is an IEEE 754 binary64 type (a double, or a long double
that has the same format, as with MSVC and on Apple's Arm CPUs)
@return a pointer past the text; up to 41 bytes at @a first are written
(some beyond the returned end)
These are the types the conversion of Zmij (see zmij.hpp) is used for; all
others (binary32, or a format the library does not know) use Grisu2.
*/
JSON_HEDLEY_NON_NULL(1)
JSON_HEDLEY_RETURNS_NON_NULL
inline char* write_short_decimal(char* first, std::uint64_t digits, int count, int exp) noexcept
template<typename FloatType>
constexpr bool has_binary64_format() noexcept
{
JSON_ASSERT(digits >= powers_of_ten_16()[static_cast<std::size_t>(count - 1)] && count <= 15);
const int scale = 16 - count;
return write_shortest(first, zmij::shortest_decimal{digits * powers_of_ten_16()[static_cast<std::size_t>(scale)], exp - scale - 1, 0, false});
return std::numeric_limits<FloatType>::is_iec559
&& std::numeric_limits<FloatType>::digits == 53
&& std::numeric_limits<FloatType>::max_exponent == 1024
&& sizeof(FloatType) == sizeof(std::uint64_t);
}
/// as write_short_decimal(), counting the digits (not 0, less than 10^15)
JSON_HEDLEY_NON_NULL(1)
JSON_HEDLEY_RETURNS_NON_NULL
inline char* write_short_decimal(char* first, std::uint64_t digits, int exp) noexcept
{
JSON_ASSERT(digits != 0 && digits < 1000000000000000u);
// floor(log10(2^bits)) + 1 digits, or one less
const int log2_bound = ((64 - count_leading_zeros(digits)) * 1233) >> 12;
const int count = log2_bound + (digits >= powers_of_ten_16()[static_cast<std::size_t>(log2_bound)] ? 1 : 0);
return write_short_decimal(first, digits, count, exp);
}
template<typename FloatType>
struct is_binary64 : std::integral_constant<bool, has_binary64_format<FloatType>()> {};
/// a positive finite float (other than double): Grisu2 and format_buffer()
/// a positive finite float (other than binary64): Grisu2 and format_buffer()
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
char* write_positive(char* first, const char* last, FloatType value)
char* write_positive_grisu2(char* first, const char* last, FloatType value)
{
JSON_ASSERT(last - first >= std::numeric_limits<FloatType>::max_digits10);
static_cast<void>(last); // (only used in the assertion)
@@ -1480,7 +1376,7 @@ char* write_positive(char* first, const char* last, FloatType value)
// len is the length of the buffer, i.e., the number of decimal digits.
int len = 0;
int decimal_exponent = 0;
shortest_digits(first, len, decimal_exponent, value);
grisu2(first, len, decimal_exponent, value);
JSON_ASSERT(len <= std::numeric_limits<FloatType>::max_digits10);
@@ -1496,15 +1392,16 @@ char* write_positive(char* first, const char* last, FloatType value)
return format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp);
}
/// a positive finite double: the shortest digits (Zmij), laid out by
/// a positive finite binary64 number: the shortest digits (Zmij), laid out by
/// write_shortest() (through a local buffer if [first, last) is shorter than
/// the 41 bytes it may write)
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
inline char* write_positive(char* first, const char* last, double value)
char* write_positive_zmij(char* first, const char* last, FloatType value)
{
static_assert(std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53,
"internal error: the conversion of Zmij needs IEEE 754 binary64 doubles");
static_assert(is_binary64<FloatType>::value,
"internal error: the conversion of Zmij needs IEEE 754 binary64 numbers");
std::uint64_t bits = 0;
std::memcpy(&bits, &value, sizeof(bits));
const zmij::shortest_decimal d = zmij::to_shortest(bits);
@@ -1519,6 +1416,34 @@ inline char* write_positive(char* first, const char* last, double value)
return first + len;
}
/// a positive finite binary64 number: Zmij (as a long double has the format of
/// a double here, its bits are those of the double of the same value)
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
char* write_positive(char* first, const char* last, FloatType value, std::true_type /*is_binary64*/)
{
return write_positive_zmij(first, last, value);
}
/// a positive finite float of any other format: Grisu2
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
char* write_positive(char* first, const char* last, FloatType value, std::false_type /*is_binary64*/)
{
return write_positive_grisu2(first, last, value);
}
/// a positive finite float: Zmij for binary64 numbers, Grisu2 otherwise
template<typename FloatType>
JSON_HEDLEY_NON_NULL(1, 2)
JSON_HEDLEY_RETURNS_NON_NULL
char* write_positive(char* first, const char* last, FloatType value)
{
return write_positive(first, last, value, is_binary64<FloatType> {});
}
} // namespace dtoa_impl
/*!
@@ -36,13 +36,6 @@ computed from the compressed tables of Zmij beyond it.
namespace zmij
{
/// significand * 10^exponent
struct decimal
{
std::uint64_t significand;
int exponent;
};
/// the compressed powers of ten of Zmij
inline const std::array<std::uint64_t, 28>& pow10_minor() noexcept
{
@@ -221,18 +214,6 @@ JSON_HEDLEY_ALWAYS_INLINE shortest_decimal to_shortest(std::uint64_t bits) noexc
return shortest_decimal{integral, dec_exp, static_cast<unsigned char>(digit), !round_up && !round_down};
}
/// The shortest decimal in the rounding interval of a positive finite double
/// given by its bits, as one number. The significand can end in zeros.
inline decimal to_decimal(std::uint64_t bits) noexcept
{
const shortest_decimal d = to_shortest(bits);
if (d.has_digit)
{
return decimal{(d.integral * 10) + d.digit, d.exponent};
}
return decimal{d.integral, d.exponent + 1};
}
} // namespace zmij
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
+16 -3
View File
@@ -9,7 +9,7 @@
#pragma once
#include <algorithm> // find, find_if, max
#include <algorithm> // find, find_if, max, min
#include <array> // array
#include <cstddef> // size_t, ptrdiff_t
#include <cstdint> // int64_t, uint8_t, uint16_t, uint32_t, uint64_t
@@ -202,8 +202,18 @@ class builder
const std::uint64_t done = static_cast<std::uint64_t>(at - b) + 1;
const std::uint64_t guess = static_cast<std::uint64_t>(n) * static_cast<std::uint64_t>(e - b + 1) / done;
const std::uint64_t grown = guess + (guess / 4) + 64; // a variable: GCC calls a cast of the sum useless where std::uint64_t is std::size_t
// (n is below 2^32: the input is smaller than 4 GiB; the sum cannot wrap)
const std::uint64_t wanted = (std::max)(grown, static_cast<std::uint64_t>(n) + (n / 2) + 64);
const std::uint64_t limit = document_data::max_nodes();
doc.tape_size = n;
doc.reserve((std::max)(static_cast<std::size_t>(grown), n + (n / 2) + 64));
// LCOV_EXCL_START (a node array that fills the address space)
if (NLOHMANN_VIEW_UNLIKELY(n >= limit))
{
document_data::throw_bad_alloc(); // no room for another node
}
// LCOV_EXCL_STOP
// (a count beyond the limit is cut: the index does not grow beyond what can be addressed)
doc.reserve(static_cast<std::size_t>((std::min)(wanted, limit)));
return doc.tape;
}
@@ -816,7 +826,10 @@ indent_done:
n->flags = flags;
n->extra = extra;
n->off = static_cast<std::uint32_t>(off);
set_integer_bits(*n, second);
// len is the low half of the second word, next the high half
// (not a native word over both, which swaps them on big-endian)
n->len = static_cast<std::uint32_t>(second);
n->next = static_cast<std::uint32_t>(second >> 32);
#endif
return n;
}
+20 -2
View File
@@ -13,9 +13,10 @@
#include <cstdint> // uint8_t, uint32_t
#include <cstring> // memcpy
#include <functional> // less
#include <limits> // numeric_limits
#include <map> // map
#include <memory> // unique_ptr
#include <new> // operator new, placement new
#include <new> // bad_alloc, operator new, placement new
#include <string> // string
#include <vector> // vector
@@ -123,13 +124,30 @@ struct document_data
tape_cap = inline_cap;
}
/// make room for n nodes; keeps the first tape_size nodes
/// the largest node count whose size in bytes fits a std::size_t
static constexpr std::size_t max_nodes() noexcept
{
return (std::numeric_limits<std::size_t>::max)() / sizeof(node);
}
[[noreturn]] NLOHMANN_VIEW_NOINLINE static void throw_bad_alloc()
{
NLOHMANN_VIEW_THROW(std::bad_alloc());
}
/// make room for n nodes; keeps the first tape_size nodes (throws
/// std::bad_alloc for a count that does not fit the address space,
/// instead of wrapping around in n * sizeof(node))
void reserve(std::size_t n)
{
if (n <= tape_cap)
{
return;
}
if (NLOHMANN_VIEW_UNLIKELY(n > max_nodes()))
{
throw_bad_alloc();
}
node* fresh = static_cast<node*>(::operator new (n * sizeof(node)));
if (tape_size != 0)
{
+209 -84
View File
@@ -17,6 +17,7 @@
#include <string> // string, to_string
#include <type_traits> // decay, enable_if, integral_constant, is_arithmetic, is_convertible, is_floating_point, is_same, is_signed
#include <utility> // forward
#include <vector> // vector
#include <nlohmann/json.hpp>
#include <nlohmann/detail/view/document_data.hpp>
@@ -151,25 +152,23 @@ class editor
{
become_empty(o, value_t::object);
}
// an existing member: assign it (and drop later duplicates, so that
// lookups, iteration, and materialize() agree)
// an existing member: assign the one that lookups find (the last
// one, should the key occur more than once), and drop the others, so
// that lookups, iteration, and materialize() agree. The key stays at
// the position of its first occurrence, as materialize() puts it.
node* slot = nullptr;
bool duplicates = false;
std::size_t matches = 0;
for (const node* k = nav::first(m_doc, o), *end = nav::end(m_doc, o); k != end; k = document_data::after(k + 1))
{
if (key_equals(*k, key))
{
if (slot != nullptr)
{
duplicates = true;
break;
}
slot = const_cast<node*>(nav::value(k + 1)); // NOLINT(cppcoreguidelines-pro-type-const-cast): the nodes belong to this document
++matches;
}
}
if (slot != nullptr)
{
if (duplicates)
if (matches > 1)
{
erase_members(o, key, true);
}
@@ -315,27 +314,38 @@ class editor
return k.len == key.size() && (key.size() == 0 || std::memcmp(m_doc.str(k), key.data(), key.size()) == 0);
}
/// remove the members with this key (all, or all but the first) from an object
std::size_t erase_members(node* o, string_view_t key, bool keep_first)
/// Remove the members with this key from an object: all of them, or all
/// but one. That one stays where the first occurrence is, but holds the
/// value of the last (the one that lookups find, which views may refer to).
std::size_t erase_members(node* o, string_view_t key, bool keep_one)
{
node* const h = block_of(m_doc, o, 0);
node last_value{}; // the entry of the value of the last member
node* const end = h + h->next;
if (keep_one)
{
for (node* r = h + 1; r != end; r += 2)
{
if (key_equals(*r, key))
{
last_value = r[1];
}
}
}
node* w = h + 1;
std::size_t erased = 0;
bool kept = false;
for (node* r = h + 1, *end = h + h->next; r != end; r += 2)
for (node* r = h + 1; r != end; r += 2)
{
const bool match = key_equals(*r, key);
if (match && (kept || !keep_first))
if (match && (kept || !keep_one))
{
++erased;
continue;
}
w[0] = r[0];
w[1] = match ? last_value : r[1];
kept = kept || match;
if (w != r)
{
w[0] = r[0];
w[1] = r[1];
}
w += 2;
}
h->next = static_cast<std::uint32_t>(w - h);
@@ -347,9 +357,10 @@ class editor
/// turn a null into an empty array/object in place
static void become_empty(node* n, value_t k) noexcept
{
const std::uint8_t linked = n->flags & node_flags::linked;
*n = node{};
n->kind = static_cast<std::uint8_t>(k);
n->flags = node_flags::is_new;
n->flags = static_cast<std::uint8_t>(node_flags::is_new | linked);
n->next = 1;
}
@@ -357,13 +368,17 @@ class editor
/// include slot (if known).
void assign(node* slot, const encoded& e, node* parent, bool parent_known)
{
// an entry of a moved sequence links to the slot: it can take any extent
const std::uint8_t linked = slot->flags & node_flags::linked;
if (e.region == nullptr)
{
if (is_container(*slot) && slot->next > 1 && slot != m_doc.tape)
if (is_container(*slot) && slot->next > 1 && slot != m_doc.tape && linked == 0)
{
// The slot spans its old elements in the enclosing sequence, but
// a scalar is one node: the enclosing container first switches to
// links (then the extent of the slot no longer matters).
// links (then the extent of the slot no longer matters). Looking
// for the container is linear in the size of the document, so
// links (which are marked in the slot) avoid it.
node* const p = parent_known ? parent : find_parent(m_doc, slot);
if (p != nullptr && ((p->flags & node_flags::moved) == 0 || moved_capacity(m_doc, p) == 0))
{
@@ -371,6 +386,7 @@ class editor
}
}
*slot = e.scalar;
slot->flags = static_cast<std::uint8_t>(slot->flags | linked);
return;
}
// an array/object: the slot keeps its extent (so that the enclosing
@@ -379,11 +395,21 @@ class editor
const node* const r = e.region;
const std::uint32_t extent = is_container(*slot) ? slot->next : 1;
const bool was_moved = (slot->flags & node_flags::moved) != 0;
// Everything that can throw happens before the slot changes: a slot
// that is a container without the moved flag would show its old
// elements. reserve_moved() makes the set_moved() below, which sets
// the flag, safe; the entry of `regions` exists already (encode()
// added it), so that the assignment at the end does not allocate.
if (!was_moved)
{
reserve_moved(m_doc);
}
slot->kind = r->kind;
slot->extra = 0;
slot->len = r->len;
slot->next = extent;
slot->flags = was_moved ? static_cast<std::uint8_t>(node_flags::moved | node_flags::is_new) : std::uint8_t{0};
// (set_moved() adds the moved flag to a slot that does not have it yet)
slot->flags = static_cast<std::uint8_t>((was_moved ? node_flags::moved | node_flags::is_new : 0) | linked);
set_moved(m_doc, slot, e.region, 0);
edit_state_of(m_doc).regions[e.region] = slot;
}
@@ -600,6 +626,8 @@ class editor
switch (static_cast<value_t>(n.kind))
{
case value_t::string:
// (an editable document only holds valid UTF-8, whatever the check of the other document was)
check_utf8(from.str(n), n.len);
return string_node(from.str(n), n.len);
case value_t::number_integer:
case value_t::number_unsigned:
@@ -663,100 +691,197 @@ class editor
}
}
// The subtrees are walked with an explicit stack (as materialize() does):
// the nesting depth is limited by memory only, not by the call stack.
/// number of nodes of a subtree (containers, keys, scalars)
template<bool E>
static std::size_t count_nodes(const document_data& d, const node* n)
{
if (!is_container(*n))
using walk = navigation<E>;
struct frame
{
return 1;
}
const bool object = n->kind == static_cast<std::uint8_t>(value_t::object);
std::size_t r = 1;
for (const node* c = navigation<E>::first(d, n), *end = navigation<E>::end(d, n); c != end;)
const node* pos; ///< next element, or key of the next member
const node* end;
bool object;
};
std::vector<frame> open;
std::size_t r = 0;
for (;;)
{
const node* const v = object ? c + 1 : c;
r += (object ? 1 : 0) + count_nodes<E>(d, navigation<E>::value(v));
c = document_data::after(v);
++r;
if (is_container(*n))
{
open.push_back(frame{walk::first(d, n), walk::end(d, n), n->kind == static_cast<std::uint8_t>(value_t::object)});
}
// the next value: close finished containers, then step over the key
for (;;)
{
if (open.empty())
{
return r;
}
frame& f = open.back();
if (f.pos == f.end)
{
open.pop_back();
continue;
}
const node* v = f.pos;
if (f.object)
{
++r; // the key
++v;
}
f.pos = document_data::after(v);
n = walk::value(v);
break;
}
}
return r;
}
/// copy a subtree (of any document) as a contiguous sequence; returns its end
/// copy a subtree (of any document) as a contiguous sequence of
/// count_nodes() nodes
template<bool E>
node* fill_nodes(const document_data& d, const node* n, node* out)
void fill_nodes(const document_data& d, const node* n, node* out)
{
if (!is_container(*n))
using walk = navigation<E>;
struct frame
{
*out = copy_scalar(d, *n);
return out + 1;
}
node* const self = out++;
*self = plain_node(static_cast<value_t>(n->kind));
self->len = n->len;
const bool object = n->kind == static_cast<std::uint8_t>(value_t::object);
for (const node* c = navigation<E>::first(d, n), *end = navigation<E>::end(d, n); c != end;)
const node* pos; ///< next element, or key of the next member
const node* end;
bool object;
node* self; ///< the container in the copy
};
std::vector<frame> open;
for (;;)
{
if (object)
if (is_container(*n))
{
*out++ = copy_scalar(d, *c);
++c;
node* const self = out++;
*self = plain_node(static_cast<value_t>(n->kind));
self->len = n->len;
open.push_back(frame{walk::first(d, n), walk::end(d, n), n->kind == static_cast<std::uint8_t>(value_t::object), self});
}
else
{
*out++ = copy_scalar(d, *n);
}
// the next value: close finished containers, then copy the key
for (;;)
{
if (open.empty())
{
return;
}
frame& f = open.back();
if (f.pos == f.end)
{
f.self->next = static_cast<std::uint32_t>(out - f.self);
open.pop_back();
continue;
}
const node* v = f.pos;
if (f.object)
{
*out++ = copy_scalar(d, *v);
++v;
}
f.pos = document_data::after(v);
n = walk::value(v);
break;
}
out = fill_nodes<E>(d, navigation<E>::value(c), out);
c = document_data::after(c);
}
self->next = static_cast<std::uint32_t>(out - self);
return out;
}
static std::size_t count_nodes(const BasicJsonType& j)
{
std::size_t r = 1;
if (j.is_object())
using iterator = typename BasicJsonType::const_iterator;
struct frame
{
for (const auto& member : j.items())
iterator pos;
iterator end;
bool object;
};
std::vector<frame> open;
const BasicJsonType* n = &j;
std::size_t r = 0;
for (;;)
{
++r;
if (n->is_structured())
{
r += 1 + count_nodes(member.value());
open.push_back(frame{n->cbegin(), n->cend(), n->is_object()});
}
for (;;)
{
if (open.empty())
{
return r;
}
frame& f = open.back();
if (f.pos == f.end)
{
open.pop_back();
continue;
}
r += f.object ? 1 : 0; // the key
n = &*f.pos;
++f.pos;
break;
}
}
else if (j.is_array())
{
for (const auto& e : j)
{
r += count_nodes(e);
}
}
return r;
}
node* fill_nodes(const BasicJsonType& j, node* out)
void fill_nodes(const BasicJsonType& j, node* out)
{
if (!j.is_structured())
using iterator = typename BasicJsonType::const_iterator;
struct frame
{
*out = json_scalar(j);
return out + 1;
}
node* const self = out++;
*self = plain_node(j.type());
self->len = static_cast<std::uint32_t>(j.size());
if (j.is_object())
iterator pos;
iterator end;
bool object;
node* self; ///< the container in the copy
};
std::vector<frame> open;
const BasicJsonType* n = &j;
for (;;)
{
for (const auto& member : j.items())
if (n->is_structured())
{
check_utf8(member.key().data(), member.key().size());
*out++ = string_node(member.key().data(), member.key().size());
out = fill_nodes(member.value(), out);
node* const self = out++;
*self = plain_node(n->type());
self->len = static_cast<std::uint32_t>(n->size());
open.push_back(frame{n->cbegin(), n->cend(), n->is_object(), self});
}
else
{
*out++ = json_scalar(*n);
}
for (;;)
{
if (open.empty())
{
return;
}
frame& f = open.back();
if (f.pos == f.end)
{
f.self->next = static_cast<std::uint32_t>(out - f.self);
open.pop_back();
continue;
}
if (f.object)
{
const auto& key = f.pos.key();
check_utf8(key.data(), key.size());
*out++ = string_node(key.data(), key.size());
}
n = &*f.pos;
++f.pos;
break;
}
}
else
{
for (const auto& e : j)
{
out = fill_nodes(e, out);
}
}
self->next = static_cast<std::uint32_t>(out - self);
return out;
}
document_data& m_doc;
+38 -14
View File
@@ -63,6 +63,23 @@ inline node* alloc_nodes(document_data& d, std::size_t k)
return r;
}
/// The capacity of the edit arena after it grows by n bytes (`used` of `cap`
/// are taken): doubled, or what is needed plus some room, but never more than
/// the 4 GiB - 1 bytes that the 32-bit offsets of nodes can address. An error
/// if n more bytes do not fit even then.
inline std::size_t text_capacity(std::size_t cap, std::size_t used, std::size_t n)
{
constexpr std::size_t limit = 0xFFFFFFFFu;
if (NLOHMANN_VIEW_UNLIKELY(used > limit || n > limit - used))
{
throw_out_of_range(416, "edits of 4 GiB or more are not supported by json_document");
}
const std::size_t needed = used + n;
const std::size_t wanted = needed + (std::min)(limit - needed, std::size_t{256});
const std::size_t doubled = cap > limit / 2 ? limit : cap * 2;
return (std::max)(doubled, wanted);
}
/// copy n bytes into the edit arena and return their offset; a new buffer
/// leaves the old one alive, so that string views into it remain valid
inline std::uint32_t append_text(document_data& d, const char* s, std::size_t n)
@@ -70,11 +87,7 @@ inline std::uint32_t append_text(document_data& d, const char* s, std::size_t n)
document_data::edit_state& e = edit_state_of(d);
if (NLOHMANN_VIEW_UNLIKELY(e.text_cap - e.text_used < n))
{
const std::size_t cap = (std::max)(e.text_cap * 2, e.text_used + n + 256);
if (cap > 0xFFFFFFFFu)
{
throw_out_of_range(416, "edits of 4 GiB or more are not supported by json_document"); // LCOV_EXCL_LINE (4 GiB)
}
const std::size_t cap = text_capacity(e.text_cap, e.text_used, n);
std::unique_ptr<char[]> fresh(new char[cap]); // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
if (e.text_used != 0)
{
@@ -101,16 +114,12 @@ inline std::size_t moved_capacity(const document_data& d, const node* n) noexcep
return d.edits->moved_cap[n->off];
}
/// let container n take its elements from `seq` (header node first)
inline void set_moved(document_data& d, node* n, node* seq, std::size_t cap)
/// Make room for one more moved container. This is the part of set_moved()
/// that can throw: a caller that changes a node before it calls set_moved()
/// calls this first, so that a failure leaves the node as it was.
inline void reserve_moved(document_data& d)
{
document_data::edit_state& e = edit_state_of(d);
if ((n->flags & node_flags::moved) != 0)
{
e.moved[n->off] = seq;
e.moved_cap[n->off] = cap;
return;
}
if (e.moved.size() >= 0xFFFFFFFFu)
{
throw_out_of_range(416, "more than 4294967295 edited arrays and objects are not supported by json_document"); // LCOV_EXCL_LINE
@@ -121,6 +130,20 @@ inline void set_moved(document_data& d, node* n, node* seq, std::size_t cap)
e.moved.reserve((2 * e.moved.size()) + 16);
e.moved_cap.reserve((2 * e.moved.size()) + 16);
}
}
/// let container n take its elements from `seq` (header node first); cannot
/// throw if n is moved already or reserve_moved() was called
inline void set_moved(document_data& d, node* n, node* seq, std::size_t cap)
{
document_data::edit_state& e = edit_state_of(d);
if ((n->flags & node_flags::moved) != 0)
{
e.moved[n->off] = seq;
e.moved_cap[n->off] = cap;
return;
}
reserve_moved(d);
e.moved.push_back(seq);
e.moved_cap.push_back(cap);
n->off = static_cast<std::uint32_t>(e.moved.size() - 1);
@@ -146,6 +169,7 @@ inline node* block_of(document_data& d, node* n, std::size_t extra)
set_moved(d, n, nh, cap);
return nh;
}
reserve_moved(d); // (so that set_moved() below cannot throw: the links are marked before)
const bool object = n->kind == static_cast<std::uint8_t>(value_t::object);
const std::size_t used = 1 + (static_cast<std::size_t>(n->len) * (object ? 2 : 1));
const std::size_t cap = used + extra;
@@ -161,7 +185,7 @@ inline node* block_of(document_data& d, node* n, std::size_t extra)
{
*o++ = *c++; // the key
}
make_link(*o, document_data::deref(c));
make_link(*o, const_cast<node*>(document_data::deref(c))); // NOLINT(cppcoreguidelines-pro-type-const-cast): the nodes belong to the document
++o;
c = document_data::after(c);
}
+2 -3
View File
@@ -66,9 +66,8 @@ template<typename BasicJsonType>
{
if (f.code == error_code::input_too_large)
{
// LCOV_EXCL_START (4 GiB)
NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4 GiB or more is not supported by json_document", nullptr));
// LCOV_EXCL_STOP
// (the limit is detail::view::max_input_size: 4 GiB minus 16 bytes)
NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4294967280 bytes or more is not supported by json_document", nullptr));
}
const BasicJsonType accepted = BasicJsonType::parse(src, src + size, nullptr, true, ignore_comments, ignore_trailing_commas);
// LCOV_EXCL_START (only if parse() accepts what the view rejects: a bug)
+12 -5
View File
@@ -9,7 +9,7 @@
#pragma once
#include <string> // basic_string, char_traits, string
#include <type_traits> // decay, integral_constant, is_array, is_lvalue_reference, is_pointer, is_same, remove_reference
#include <type_traits> // decay, integral_constant, is_array, is_const, is_integral, is_lvalue_reference, is_pointer, is_same, remove_reference
#include <utility> // forward
#include <nlohmann/json.hpp>
@@ -28,12 +28,12 @@ namespace view
/// how a document takes its input
enum class input_kind
{
move_string, ///< rvalue std::string: owned without a copy
move_string, ///< non-const rvalue std::string: owned without a copy
c_string, ///< const char* (NUL-terminated): borrowed
char_array, ///< char array (e.g. a string literal): borrowed
borrow_range, ///< lvalue contiguous byte container, or std::string_view: borrowed
copy_range, ///< rvalue contiguous byte container: copied
adapter, ///< anything else parse() accepts (streams, wide strings, ...): read into a buffer
copy_range, ///< rvalue contiguous byte container (a const rvalue std::string too): copied
adapter, ///< streams, wide strings, and the rest of what the library's input adapter reads: read into a buffer
};
template<typename InputType>
@@ -52,13 +52,20 @@ struct classify_input
static constexpr input_kind value =
std::is_array<R>::value ? input_kind::char_array
: std::is_pointer<D>::value ? input_kind::c_string
: (is_rvalue && std::is_same<D, std::string>::value) ? input_kind::move_string
: (is_rvalue && !std::is_const<R>::value && std::is_same<D, std::string>::value) ? input_kind::move_string
: (is_bytes && (!is_rvalue || is_string_view)) ? input_kind::borrow_range
: is_bytes ? input_kind::copy_range
: input_kind::adapter;
// NOLINTEND(readability-avoid-nested-conditional-operator)
};
/// an integer type other than bool: a length passed where a flag is expected
template<typename T>
struct is_integer_not_bool : std::is_integral<T> {};
template<>
struct is_integer_not_bool<bool> : std::false_type {};
/// std::basic_string guarantees a NUL at data()[size()] (the parser's sentinel)
template<typename T>
struct is_std_string : std::false_type {};
+28 -6
View File
@@ -11,6 +11,8 @@
#include <cstddef> // size_t
#include <cstdint> // uint16_t, uint32_t, uint64_t
#include <cstring> // memcmp, memcpy
#include <limits> // numeric_limits
#include <type_traits> // integral_constant, is_integral, is_same
#include <nlohmann/json.hpp>
#include <nlohmann/detail/view/document_data.hpp>
@@ -82,8 +84,9 @@ class short_key
std::uint64_t m_b = 0;
};
/// the key node of the first member of an object with the given key, or
/// nullptr; most keys are rejected by their length, from the index alone
/// the key node of the last member of an object with the given key, or
/// nullptr (the last one, as materialize() and parse() keep it); most keys are
/// rejected by their length, from the index alone
template<bool Editable>
const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept
{
@@ -94,6 +97,7 @@ const node* find_member(const document_data& d, const node* object, const char*
}
const node* const end = nav::end(d, object);
const auto* const k = reinterpret_cast<const unsigned char*>(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
const node* last = nullptr;
if (NLOHMANN_VIEW_LIKELY(n <= 16))
{
const short_key probe(k, n);
@@ -101,19 +105,37 @@ const node* find_member(const document_data& d, const node* object, const char*
{
if (m->len == n && probe.matches(reinterpret_cast<const unsigned char*>(d.str(*m)))) // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
{
return m;
last = m;
}
}
return nullptr;
return last;
}
for (const node* m = nav::first(d, object); m != end; m = document_data::after(m + 1))
{
if (m->len == n && std::memcmp(d.str(*m), key, n) == 0)
{
return m;
last = m;
}
}
return nullptr;
return last;
}
/// whether an integer type is accepted as an array index by the view's
/// operator[] and at(): every integer type but bool and size_t, which has its
/// own overload
template<typename T>
struct is_index_type : std::integral_constant < bool,
std::is_integral<T>::value && !std::is_same<T, bool>::value && !std::is_same<T, std::size_t>::value >
{};
/// an integer as an index: negative values, and values that do not fit a
/// size_t, map to the largest size_t (out of range for every array)
template<typename SizeType, typename IntegerType>
SizeType to_index(IntegerType idx) noexcept
{
const IntegerType zero = 0;
const auto result = static_cast<SizeType>(idx);
return (idx < zero || static_cast<IntegerType>(result) != idx) ? (std::numeric_limits<SizeType>::max)() : result;
}
/// the entry of the element of an array at an index below its size (a link
+21 -3
View File
@@ -29,6 +29,11 @@ static_assert(static_cast<std::uint8_t>(value_t::null) == 0 && static_cast<std::
&& static_cast<std::uint8_t>(value_t::number_unsigned) == 6 && static_cast<std::uint8_t>(value_t::number_float) == 7,
"the node format depends on the numbering of value_t");
/// The largest input a document accepts, in bytes. Offsets and node counts are
/// 32 bits wide; the limit keeps 16 bytes (the width of the scanner's steps)
/// below 2^32, so that a position one step past the end of the text fits.
static constexpr std::size_t max_input_size = 0xFFFFFFEFu;
/// node flags
struct node_flags
{
@@ -38,6 +43,7 @@ struct node_flags
static constexpr std::uint8_t is_true = 4; ///< boolean value
static constexpr std::uint8_t moved = 8; ///< array/object: the elements live in a separate sequence (editable documents)
static constexpr std::uint8_t is_new = 16; ///< written by an edit: no source position
static constexpr std::uint8_t linked = 32; ///< an entry of a moved sequence links to this value (editable documents): its extent in the parsed layout no longer matters
};
/// kind of an entry of an edited sequence that stands for a value stored
@@ -52,7 +58,7 @@ struct node
{
std::uint8_t kind; ///< value_t, or kind_link
std::uint8_t flags; ///< node_flags
std::uint16_t extra; ///< numbers: integer digits (low byte) and fraction digits (high byte), 255 = "many"; objects: number of the hash index; otherwise 0
std::uint16_t extra; ///< numbers: integer digits (low byte) and fraction digits (high byte), 255 = "many"; objects: number of the hash index (1-based, 0 = none); otherwise 0
std::uint32_t off; ///< source offset (string content, number token, literal, bracket); arena offset if escaped/edited; number of the element sequence if moved
std::uint32_t len; ///< string: decoded bytes; float: token bytes; array/object: element count
std::uint32_t next; ///< array/object: number of nodes of the subtree (its extent in the enclosing sequence)
@@ -72,24 +78,36 @@ NLOHMANN_VIEW_ALWAYS_INLINE const node* link_target(const node& n) noexcept
return t;
}
inline void make_link(node& n, const node* target) noexcept
/// let the entry n stand for the value at target (and mark the value)
inline void make_link(node& n, node* target) noexcept
{
target->flags = static_cast<std::uint8_t>(target->flags | node_flags::linked);
n = node{};
n.kind = kind_link;
std::memcpy(reinterpret_cast<unsigned char*>(&n) + 8, static_cast<const void*>(&target), sizeof(const node*)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
}
/// the converted value of an integer node (stored in len/next)
/// the converted value of an integer node: len is its low half, next its high
/// half (on little-endian targets the two words are the value in memory)
NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept
{
#if NLOHMANN_VIEW_LITTLE_ENDIAN
std::uint64_t v = 0;
std::memcpy(&v, reinterpret_cast<const unsigned char*>(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
return v;
#else
return static_cast<std::uint64_t>(n.len) | (static_cast<std::uint64_t>(n.next) << 32);
#endif
}
NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept
{
#if NLOHMANN_VIEW_LITTLE_ENDIAN
std::memcpy(reinterpret_cast<unsigned char*>(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
#else
n.len = static_cast<std::uint32_t>(v);
n.next = static_cast<std::uint32_t>(v >> 32);
#endif
}
/// token length of a number node
+34 -5
View File
@@ -22,8 +22,14 @@
// large objects). An object with document_data::index_min_members members or
// more gets an open-addressing table after parsing; its node stores the
// number of the table (1-based) in `extra`. A slot holds the offset of a key
// node from its object node (0: empty). Of duplicate keys, the first is kept,
// as for the linear search.
// node from its object node (0: empty). Of duplicate keys, the last is kept,
// as for the linear search, and as basic_json::parse() does.
//
// The hash is not seeded, so keys chosen to collide could make the build
// quadratic. A key therefore sits at most index_max_displacement slots away
// from its home slot; if a key would sit further away, the table is dropped
// and the object is searched linearly (like a small one). For the same
// reason, a lookup visits at most index_max_displacement + 1 slots.
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -31,6 +37,11 @@ namespace detail
namespace view
{
/// the farthest a key may sit from its home slot (a table with at most half of
/// its slots in use gives random keys a distance of about 50 for millions of
/// members; and every member costs at most this many steps while building)
constexpr std::size_t index_max_displacement = 64;
/// hash of a key: its bytes, eight at a time, in a fixed byte order
inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept
{
@@ -68,27 +79,44 @@ inline void build_object_index(document_data& d, node* obj)
d.index_slots.resize(start + cap, 0);
std::uint32_t* const slots = d.index_slots.data() + start;
const std::size_t mask = cap - 1;
bool degenerate = false;
for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1))
{
const char* const key = d.str(*k);
const std::uint64_t hash = key_hash(key, k->len); // (a cast of the call would be useless where std::uint64_t is std::size_t)
std::size_t i = static_cast<std::size_t>(hash) & mask;
bool duplicate = false;
std::size_t distance = 0;
while (slots[i] != 0)
{
const node* const other = obj + slots[i];
if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0))
{
duplicate = true; // keep the first
duplicate = true; // keep the last: the key's slot now leads to this member
slots[i] = static_cast<std::uint32_t>(k - obj);
break;
}
if (++distance > index_max_displacement)
{
degenerate = true; // too many keys share a home region
break;
}
i = (i + 1) & mask;
}
if (degenerate)
{
break;
}
if (!duplicate)
{
slots[i] = static_cast<std::uint32_t>(k - obj);
}
}
if (degenerate)
{
d.index_slots.resize(start); // no table: the object is searched linearly
return;
}
d.indexes.push_back(document_data::object_index{start, static_cast<std::uint32_t>(mask)});
obj->extra = static_cast<std::uint16_t>(d.indexes.size());
}
@@ -102,7 +130,7 @@ inline void build_object_indexes(document_data& d)
}
}
/// the key node of the first member with this key of an indexed object, or
/// the key node of the last member with this key of an indexed object, or
/// nullptr
inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept
{
@@ -110,7 +138,7 @@ inline const node* find_indexed(const document_data& d, const node* obj, const c
const std::uint32_t* const slots = d.index_slots.data() + ix.start;
const std::uint64_t hash = key_hash(key, n); // (a cast of the call would be useless where std::uint64_t is std::size_t)
std::size_t i = static_cast<std::size_t>(hash) & ix.mask;
for (;;)
for (std::size_t distance = 0; distance <= index_max_displacement; ++distance)
{
const std::uint32_t s = slots[i];
if (s == 0)
@@ -124,6 +152,7 @@ inline const node* find_indexed(const document_data& d, const node* obj, const c
}
i = (i + 1) & ix.mask;
}
return nullptr; // (no key sits further from its home slot)
}
} // namespace view
+7 -1
View File
@@ -44,7 +44,13 @@ class output_buffer
void finish()
{
m_out.resize(static_cast<std::size_t>(m_pos - m_out.data()));
const auto size = static_cast<std::size_t>(m_pos - m_out.data());
m_out.resize(size);
// do not keep a buffer that was sized for a much larger output
if (m_out.capacity() > 1024 && m_out.capacity() / 2 > size)
{
m_out.shrink_to_fit();
}
}
NLOHMANN_VIEW_ALWAYS_INLINE void reserve(std::size_t n)
+2 -1
View File
@@ -35,7 +35,8 @@
#else
#define NLOHMANN_VIEW_NEON 0
#endif
#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2))
// (x86 only: other targets can define __SSE2__ as well, e.g., WebAssembly with -msse2, but have no <cpuid.h>)
#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__x86_64__) || defined(__i386__) || defined(_M_X64) || defined(_M_IX86)) && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2))
#include <emmintrin.h>
#define NLOHMANN_VIEW_SSE2 1
#else
+131 -39
View File
@@ -7,16 +7,18 @@
// SPDX-License-Identifier: MIT
/****************************************************************************\
* Zero-copy, read-only view of a parsed JSON text. *
* Zero-copy view of a parsed JSON text. *
* *
* json_document::parse() builds a flat index of the values of a JSON text *
* (16 bytes per value) instead of a tree of basic_json values. Strings and *
* numbers stay in the source text; only strings with escapes are decoded, *
* into one buffer. json_view is a handle to one value of the document, with *
* the read-only part of the basic_json interface; materialize() turns a *
* subtree into the basic_json value that parse() would produce. *
* subtree into the basic_json value that parse() would produce. An editable *
* document (json_editable_document) also has set(), push_back(), insert(), *
* and erase(): edits never write to the source text, and views stay valid. *
* *
* The source text must outlive a document that borrows it (lvalue byte *
* The source text must outlive a document that borrows it (lvalue byte *
* containers, C strings); rvalue strings, streams, and other inputs are *
* owned by the document. *
\****************************************************************************/
@@ -24,6 +26,7 @@
#ifndef INCLUDE_NLOHMANN_JSON_VIEW_HPP_
#define INCLUDE_NLOHMANN_JSON_VIEW_HPP_
#include <algorithm> // all_of, min
#include <cstddef> // size_t
#include <cstdint> // uint8_t, uint32_t
#include <cstring> // memcpy, strlen
@@ -181,12 +184,6 @@ class basic_json_view
return type() == value_t::discarded;
}
/// false for discarded views
explicit operator bool() const noexcept
{
return m_node != nullptr;
}
/// the name of the type, as basic_json::type_name()
const char* type_name() const noexcept
{
@@ -246,13 +243,18 @@ class basic_json_view
// element access //
////////////////////
/// the value of the member with this key (the first one, should the key
/// occur more than once); a discarded view if there is none. Throws
/// type_error.305 if this is not an object.
/// the value of the member with this key (the last one, should the key
/// occur more than once); a discarded view if there is none, or if this
/// is a discarded view (so that v["a"]["b"] is safe). Throws type_error.305
/// if this is any other value but an object.
NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view operator[](string_view_t key) const
{
if (NLOHMANN_VIEW_UNLIKELY(!is_object()))
{
if (is_discarded())
{
return basic_json_view();
}
detail::view::throw_type_error(305, "cannot use operator[] with a string argument with ", type_name());
}
return lookup(key);
@@ -269,31 +271,43 @@ class basic_json_view
}
/// the element at this index; a discarded view if the index is out of
/// range. Throws type_error.305 if this is not an array.
/// range, or if this is a discarded view. Throws type_error.305 if this is
/// any other value but an array.
basic_json_view operator[](size_type idx) const
{
if (NLOHMANN_VIEW_UNLIKELY(!is_array()))
{
if (is_discarded())
{
return basic_json_view();
}
detail::view::throw_type_error(305, "cannot use operator[] with a numeric argument with ", type_name());
}
return idx < m_node->len ? basic_json_view(m_doc, navigation::value(detail::view::element_at<Editable>(*m_doc, m_node, idx))) : basic_json_view();
}
/// (an int argument would be ambiguous between size_type and const char*)
basic_json_view operator[](int idx) const
/// any other integer type (int, unsigned, long, std::int64_t, ...; a
/// single overload for size_type alone would be ambiguous for all of them
/// and for const char*); negative values are out of range
template < typename IntegerType, typename std::enable_if < detail::view::is_index_type<IntegerType>::value, int >::type = 0 >
basic_json_view operator[](IntegerType idx) const
{
return operator[](static_cast<size_type>(idx));
return operator[](detail::view::to_index<size_type>(idx));
}
/// the value a JSON pointer refers to; a discarded view if a key is
/// missing or an index is out of range. Other errors throw what const
/// basic_json::operator[] throws.
/// missing or an index is out of range, or if this is a discarded view.
/// Other errors throw what const basic_json::operator[] throws.
basic_json_view operator[](const json_pointer& ptr) const
{
if (NLOHMANN_VIEW_UNLIKELY(is_discarded()))
{
return basic_json_view();
}
return detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::unchecked);
}
/// the value of the member with this key (the first one, should the key
/// the value of the member with this key (the last one, should the key
/// occur more than once). Throws type_error.304 if this is not an object,
/// and out_of_range.403 if there is no such member.
basic_json_view at(string_view_t key) const
@@ -303,7 +317,7 @@ class basic_json_view
detail::view::throw_type_error(304, "cannot use at() with ", type_name());
}
const basic_json_view r = lookup(key);
if (NLOHMANN_VIEW_UNLIKELY(!r))
if (NLOHMANN_VIEW_UNLIKELY(r.is_discarded()))
{
detail::view::throw_out_of_range(403, detail::concat("key '", std::string(key.data(), key.size()), "' not found"));
}
@@ -335,9 +349,12 @@ class basic_json_view
return basic_json_view(m_doc, navigation::value(detail::view::element_at<Editable>(*m_doc, m_node, idx)));
}
basic_json_view at(int idx) const
/// any other integer type, see operator[]; negative values are out of
/// range
template < typename IntegerType, typename std::enable_if < detail::view::is_index_type<IntegerType>::value, int >::type = 0 >
basic_json_view at(IntegerType idx) const
{
return at(static_cast<size_type>(idx));
return at(detail::view::to_index<size_type>(idx));
}
/// the value a JSON pointer refers to; throws what basic_json::at()
@@ -348,7 +365,7 @@ class basic_json_view
}
/// the member with this key converted to T, or the default value if there
/// is no such member (the first one, should the key occur more than
/// is no such member (the last one, should the key occur more than
/// once). Throws type_error.306 if this is not an object.
template < typename T, typename std::enable_if < !std::is_same<typename std::decay<T>::type, const char*>::value, int >::type = 0 >
T value(string_view_t key, const T& default_value) const
@@ -358,7 +375,7 @@ class basic_json_view
detail::view::throw_type_error(306, "cannot use value() with ", type_name());
}
const basic_json_view r = lookup(key);
return r ? r.template get<T>() : default_value;
return r.is_discarded() ? default_value : r.template get<T>();
}
string_t value(string_view_t key, const char* default_value) const
@@ -377,7 +394,7 @@ class basic_json_view
detail::view::throw_type_error(306, "cannot use value() with ", type_name());
}
const basic_json_view r = detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::value);
return r ? r.template get<T>() : default_value;
return r.is_discarded() ? default_value : r.template get<T>();
}
string_t value(const json_pointer& ptr, const char* default_value) const
@@ -413,7 +430,7 @@ class basic_json_view
// lookup //
////////////
/// an iterator to the member with this key (the first one, should the
/// an iterator to the member with this key (the last one, should the
/// key occur more than once), or end(); end() also for non-objects
iterator find(string_view_t key) const
{
@@ -455,7 +472,7 @@ class basic_json_view
/// basic_json::contains())
bool contains(const json_pointer& ptr) const
{
return static_cast<bool>(detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::contains));
return !detail::view::resolve_pointer(*this, detail::json_pointer_access::reference_tokens(ptr), detail::view::pointer_mode::contains).is_discarded();
}
/// 1 if this is an object with a member with this key, else 0 (duplicate
@@ -712,7 +729,7 @@ class basic_json_view
}
/// the number of source bytes of this value (estimated for values with
/// decoded strings)
/// decoded strings); the estimate sizes the output buffer of dump()
std::size_t source_extent() const noexcept
{
if (editable() && m_doc->edits != nullptr)
@@ -720,20 +737,33 @@ class basic_json_view
// positions of moved and new values are not source offsets
return m_node == m_doc->tape ? m_doc->size + m_doc->edits->text_used : 64;
}
const node* const next = document_data::after(m_node);
const bool in_source = (m_node->flags & detail::view::node_flags::storage) == 0;
if (!in_source)
const node* const end = m_doc->tape + m_doc->tape_size;
if ((m_node->flags & detail::view::node_flags::storage) != 0)
{
return m_node->len;
}
if (next != m_doc->tape + m_doc->tape_size && (next->flags & detail::view::node_flags::storage) == 0 && next->off >= m_node->off)
// the value ends where the next node in the source begins; nodes
// with decoded strings (their offset is in the arena) are skipped,
// but only a few of them, to keep the walk short
const node* next = document_data::after(m_node);
for (int skipped = 0; next != end && skipped < 16; ++skipped, ++next)
{
return next->off - m_node->off;
if ((next->flags & detail::view::node_flags::storage) == 0)
{
return next->off >= m_node->off ? next->off - m_node->off : 0;
}
}
return m_doc->size - m_node->off;
if (next == end)
{
return m_doc->size - m_node->off;
}
// the end is unknown: assume a few bytes per node, the output buffer
// grows should the value be larger
const auto nodes = static_cast<std::size_t>(document_data::after(m_node) - m_node);
return (std::min)(m_doc->size - m_node->off, static_cast<std::size_t>(1024) + nodes * 16);
}
/// the value of the first member with this key, or a discarded view
/// the value of the last member with this key, or a discarded view
/// (object required)
NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view lookup(string_view_t key) const noexcept
{
@@ -867,6 +897,13 @@ class basic_json_document
return d;
}
/// parse(ptr, len) does not compile: len would convert to allow_exceptions
/// and ptr be read as a C string (as for the overloads of parse_copy,
/// accept, and read below)
template<typename InputType, typename IntegerType, typename... Flags>
static typename std::enable_if<detail::view::is_integer_not_bool<IntegerType>::value, basic_json_document>::type
parse(InputType&& input, IntegerType value, Flags&&... flags) = delete;
/// parse [first, last)
template<typename IteratorType, typename std::enable_if<
std::is_base_of<std::input_iterator_tag, typename std::iterator_traits<IteratorType>::iterator_category>::value, int>::type = 0>
@@ -894,6 +931,10 @@ class basic_json_document
return d;
}
template<typename InputType, typename IntegerType, typename... Flags>
static typename std::enable_if<detail::view::is_integer_not_bool<IntegerType>::value, basic_json_document>::type
parse_copy(InputType&& input, IntegerType value, Flags&&... flags) = delete;
/// check whether the input is valid JSON (the result of basic_json::accept)
template<typename InputType>
static bool accept(InputType&& input, const bool ignore_comments = false, const bool ignore_trailing_commas = false)
@@ -903,6 +944,10 @@ class basic_json_document
return !d.is_discarded();
}
template<typename InputType, typename IntegerType, typename... Flags>
static typename std::enable_if<detail::view::is_integer_not_bool<IntegerType>::value, bool>::type
accept(InputType&& input, IntegerType value, Flags&&... flags) = delete;
/// parse into this document, reusing its memory
template<typename InputType>
// flawfinder: ignore (a member function, not POSIX read())
@@ -915,12 +960,17 @@ class basic_json_document
std::integral_constant<detail::view::input_kind, detail::view::classify_input<InputType>::value> {});
}
template<typename InputType, typename IntegerType, typename... Flags>
// flawfinder: ignore (a member function, not POSIX read())
typename std::enable_if<detail::view::is_integer_not_bool<IntegerType>::value, void>::type
read(InputType&& input, IntegerType value, Flags&&... flags) = delete;
////////////
// access //
////////////
/// the root value (discarded if parsing failed without exceptions)
view_type root() const noexcept
view_type root() const& noexcept
{
if (!m_data || m_data->discarded)
{
@@ -929,6 +979,9 @@ class basic_json_document
return view_type(m_data.get(), m_data->tape);
}
/// deleted: the view of a temporary document would dangle
view_type root() const&& = delete;
bool is_discarded() const noexcept
{
return !m_data || m_data->discarded;
@@ -988,6 +1041,8 @@ class basic_json_document
// (edits link to the nodes of the index, which then stays in place)
const bool shrink_tape = d.tape != d.inline_tape && d.tape_size != d.tape_cap && d.edits == nullptr;
const bool into_header = d.tape_size <= d.inline_cap;
std::vector<document_data::object_index> indexes(d.indexes.capacity() > d.indexes.size() ? d.indexes : std::vector<document_data::object_index>());
std::vector<std::uint32_t> index_slots(d.index_slots.capacity() > d.index_slots.size() ? d.index_slots : std::vector<std::uint32_t>());
node* fresh = (shrink_tape && !into_header) ? static_cast<node*>(::operator new (d.tape_size * sizeof(node))) : d.inline_tape;
if (shrink_tape)
@@ -1005,6 +1060,14 @@ class basic_json_document
d.base[1] = d.arena.data();
}
}
if (d.indexes.capacity() > d.indexes.size())
{
d.indexes.swap(indexes);
}
if (d.index_slots.capacity() > d.index_slots.size())
{
d.index_slots.swap(index_slots);
}
}
////////////
@@ -1095,7 +1158,9 @@ class basic_json_document
/// set the value at a JSON pointer: its parent must exist; an object
/// member is set (added if missing), an array element assigned, and "-"
/// or the size of the array appends
/// or the size of the array appends. A null parent becomes what
/// basic_json's operator[](json_pointer) makes of it: an array for "-"
/// and for digits (padded with nulls up to the index), an object otherwise.
template<typename V>
view_type set(const json_pointer& ptr, V&& value)
{
@@ -1105,6 +1170,31 @@ class basic_json_document
}
const view_type parent = root().at(ptr.parent_pointer());
const auto& token = ptr.back();
if (parent.is_null())
{
const bool digits = std::all_of(token.begin(), token.end(), [](const char c)
{
return c >= '0' && c <= '9';
});
if (token == "-")
{
return push_back(parent, std::forward<V>(value));
}
if (digits)
{
// (an invalid index is an error before the parent changes)
const std::size_t idx = pointer_index(token);
if (idx >= 0xFFFFFFFFu)
{
detail::view::throw_out_of_range(401, detail::concat("array index ", std::to_string(idx), " is out of range"));
}
for (std::size_t i = 0; i < idx; ++i)
{
push_back(parent, nullptr);
}
return push_back(parent, std::forward<V>(value));
}
}
if (parent.is_array())
{
const std::size_t idx = token == "-" ? parent.size() : pointer_index(token);
@@ -1252,9 +1342,9 @@ class basic_json_document
d.discarded = true;
detail::view::parse_failure failure;
bool ok = false;
if (NLOHMANN_VIEW_UNLIKELY(size >= 0xFFFFFFF0u))
if (NLOHMANN_VIEW_UNLIKELY(size > detail::view::max_input_size))
{
failure.code = detail::view::error_code::input_too_large; // LCOV_EXCL_LINE (4 GiB)
failure.code = detail::view::error_code::input_too_large;
}
else
{
@@ -1266,9 +1356,11 @@ class basic_json_document
d.base[1] = d.arena.data();
d.arena_size = d.arena.size();
detail::view::build_object_indexes(d);
std::vector<std::uint32_t>().swap(d.large_objects); // (only needed while parsing)
d.discarded = false;
return;
}
std::vector<std::uint32_t>().swap(d.large_objects);
if (allow_exceptions)
{
detail::view::throw_parse_failure<BasicJsonType>(failure, src, size, comments, trailing_commas);