Add json_document and json_view: parse, accept, types, materialize

The public classes of the zero-copy view (#5295), in the new header
<nlohmann/json_view.hpp>:

- basic_json_document<BasicJsonType>: parse (borrowing contiguous byte
  inputs, owning rvalue strings, streams, and other inputs), parse_copy,
  accept, read, root, is_discarded, source, owns_source, node_count,
  memory_usage, shrink_to_fit
- basic_json_view<BasicJsonType>: type and the is_* queries, size, empty,
  materialize (the value parse() would produce, built by the same SAX
  handler), source_offset
- the aliases json_document, json_view, ordered_json_document, and
  ordered_json_view

A parse error throws the exception basic_json::parse would throw for the
same input: the library parser is run on the failing input, so messages,
positions, and exception ids are the same. Inputs of 4 GiB or more are
rejected with out_of_range.416.

The single header single_include/nlohmann/json_view.hpp keeps including
json.hpp; make amalgamate, check-amalgamation, include.zip, and release
handle it.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-30 20:57:43 +02:00
parent e96e2982a5
commit cf352d4ef5
10 changed files with 3623 additions and 6 deletions
+81
View File
@@ -0,0 +1,81 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#pragma once
#include <algorithm> // min
#include <cstddef> // size_t
#include <string> // string
#include <nlohmann/json.hpp>
#include <nlohmann/detail/view/builder.hpp>
#include <nlohmann/detail/view/macro_scope.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
{
namespace view
{
// Exceptions are thrown out of line, so that the accessors that may throw stay
// small enough to be inlined.
[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_type_error(int id, const char* prefix, const char* type)
{
NLOHMANN_VIEW_THROW(type_error::create(id, concat(prefix, type), nullptr));
}
[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_out_of_range(int id, const std::string& msg)
{
NLOHMANN_VIEW_THROW(out_of_range::create(id, msg, nullptr));
}
[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_invalid_iterator(int id, const char* msg)
{
NLOHMANN_VIEW_THROW(invalid_iterator::create(id, msg, nullptr));
}
/*!
@brief throw the exception BasicJsonType::parse would throw for this input
The view accepts exactly the inputs parse() accepts, so on a failure the
library parser is run on the same bytes: it throws the exception parse() would
throw, with the same message, position, and "last read" token. The error path
is cold, so this costs nothing on valid input. Should parse() accept the input
nevertheless (a bug), the view's own failure is reported.
*/
template<typename BasicJsonType>
[[noreturn]] NLOHMANN_VIEW_NOINLINE void throw_parse_failure(const parse_failure& f, const char* src, std::size_t size,
bool ignore_comments, bool ignore_trailing_commas)
{
if (f.code == error_code::input_too_large)
{
NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4 GiB or more is not supported by json_document", nullptr));
}
const BasicJsonType accepted = BasicJsonType::parse(src, src + size, nullptr, true, ignore_comments, ignore_trailing_commas);
static_cast<void>(accepted);
position_t pos;
const std::size_t off = (std::min)(f.offset, size);
pos.chars_read_total = off + 1;
std::size_t line_start = 0;
for (std::size_t i = 0; i < off; ++i)
{
if (src[i] == '\n')
{
++pos.lines_read;
line_start = i + 1;
}
}
pos.chars_read_current_line = off + 1 - line_start;
NLOHMANN_VIEW_THROW(parse_error::create(101, pos, "syntax error while parsing value", nullptr));
}
} // namespace view
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
+86
View File
@@ -0,0 +1,86 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#pragma once
#include <string> // basic_string, char_traits, string
#include <type_traits> // decay, integral_constant, is_array, is_lvalue_reference, is_pointer, is_same, remove_reference
#include <utility> // forward
#include <nlohmann/json.hpp>
#include <nlohmann/detail/view/macro_scope.hpp>
#if NLOHMANN_VIEW_HAS_CPP_17
#include <string_view> // string_view
#endif
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
{
namespace view
{
/// how a document takes its input
enum class input_kind
{
move_string, ///< rvalue std::string: owned without a copy
c_string, ///< const char* (NUL-terminated): borrowed
char_array, ///< char array (e.g. a string literal): borrowed
borrow_range, ///< lvalue contiguous byte container, or std::string_view: borrowed
copy_range, ///< rvalue contiguous byte container: copied
adapter, ///< anything else parse() accepts (streams, wide strings, ...): read into a buffer
};
template<typename InputType>
struct classify_input
{
using R = typename std::remove_reference<InputType>::type;
using D = typename std::decay<InputType>::type;
static constexpr bool is_rvalue = !std::is_lvalue_reference<InputType>::value;
static constexpr bool is_bytes = is_contiguous_byte_container<D>::value;
#if NLOHMANN_VIEW_HAS_CPP_17
static constexpr bool is_string_view = std::is_same<D, std::string_view>::value;
#else
static constexpr bool is_string_view = false;
#endif
static constexpr input_kind value =
std::is_array<R>::value ? input_kind::char_array
: std::is_pointer<D>::value ? input_kind::c_string
: (is_rvalue && std::is_same<D, std::string>::value) ? input_kind::move_string
: (is_bytes && (!is_rvalue || is_string_view)) ? input_kind::borrow_range
: is_bytes ? input_kind::copy_range
: input_kind::adapter;
};
/// std::basic_string guarantees a NUL at data()[size()] (the parser's sentinel)
template<typename T>
struct is_std_string : std::false_type {};
template<typename Traits, typename Alloc>
struct is_std_string<std::basic_string<char, Traits, Alloc>> : std::true_type {};
/// drain a json input adapter (UTF-16/32 inputs arrive as UTF-8)
template<typename Adapter>
std::string collect_adapter(Adapter&& ia)
{
std::string buf;
for (;;)
{
const auto ch = ia.get_character();
if (ch == std::char_traits<char>::eof())
{
break;
}
buf.push_back(static_cast<char>(ch));
}
return buf;
}
} // namespace view
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
@@ -37,6 +37,14 @@
#define NLOHMANN_VIEW_NOINLINE
#endif
#if defined(__GNUC__) || defined(__clang__)
#define NLOHMANN_VIEW_NODISCARD __attribute__((warn_unused_result))
#elif defined(_MSC_VER)
#define NLOHMANN_VIEW_NODISCARD _Check_return_
#else
#define NLOHMANN_VIEW_NODISCARD
#endif
// exceptions as in json.hpp (JSON_NOEXCEPTION, JSON_THROW_USER)
#if (defined(__cpp_exceptions) || defined(__EXCEPTIONS) || defined(_CPPUNWIND)) && !defined(JSON_NOEXCEPTION)
#define NLOHMANN_VIEW_THROW(exception) throw exception
@@ -15,6 +15,7 @@
#undef NLOHMANN_VIEW_UNLIKELY
#undef NLOHMANN_VIEW_ALWAYS_INLINE
#undef NLOHMANN_VIEW_NOINLINE
#undef NLOHMANN_VIEW_NODISCARD
#undef NLOHMANN_VIEW_THROW
#undef NLOHMANN_VIEW_LITTLE_ENDIAN
#undef NLOHMANN_VIEW_REPEAT16
@@ -0,0 +1,130 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#pragma once
#include <cstdint> // int64_t, uint8_t
#include <vector> // vector
#include <nlohmann/json.hpp>
#include <nlohmann/detail/view/document_data.hpp>
#include <nlohmann/detail/view/macro_scope.hpp>
#include <nlohmann/detail/view/node.hpp>
#include <nlohmann/detail/view/number.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
{
namespace view
{
/*!
@brief the basic_json value of the subtree at n
The subtree is replayed into the SAX handler that parse() uses to build its
values, so the result is the value parse() would produce: duplicate keys keep
the last value, and with JSON_DIAGNOSTICS the parent pointers are set. It is
iterative, so the nesting depth is limited by memory only, as for parse().
Without a lexer the handler records no source positions
(JSON_DIAGNOSTIC_POSITIONS).
*/
template<typename BasicJsonType>
BasicJsonType materialize(const document_data& d, const node* n)
{
using string_t = typename BasicJsonType::string_t;
using sax_t = json_sax_dom_parser<BasicJsonType, iterator_input_adapter<const char*>>;
BasicJsonType result;
sax_t sax(result, true);
const string_t no_token{};
// the ends of the open containers, and whether they are objects
std::vector<std::pair<const node*, bool>> open;
for (;;)
{
switch (static_cast<value_t>(n->kind))
{
case value_t::object:
case value_t::array:
{
const bool object = n->kind == static_cast<std::uint8_t>(value_t::object);
if (object)
{
sax.start_object(n->len);
}
else
{
sax.start_array(n->len);
}
open.emplace_back(document_data::child_end(n), object);
n = document_data::first_child(n);
break;
}
case value_t::string:
{
string_t s(d.str(*n), n->len);
sax.string(s);
++n;
break;
}
case value_t::number_integer:
sax.number_integer(static_cast<typename BasicJsonType::number_integer_t>(static_cast<std::int64_t>(integer_bits(*n))));
++n;
break;
case value_t::number_unsigned:
sax.number_unsigned(static_cast<typename BasicJsonType::number_unsigned_t>(integer_bits(*n)));
++n;
break;
case value_t::number_float:
sax.number_float(float_value<typename BasicJsonType::number_float_t>(d.str(*n), *n), no_token);
++n;
break;
case value_t::boolean:
sax.boolean((n->flags & node_flags::is_true) != 0);
++n;
break;
case value_t::null:
case value_t::binary:
case value_t::discarded:
default:
sax.null();
++n;
break;
}
for (;;)
{
if (open.empty())
{
return result;
}
if (n != open.back().first)
{
break;
}
if (open.back().second)
{
sax.end_object();
}
else
{
sax.end_array();
}
open.pop_back();
}
if (open.back().second)
{
// the key of the next member
string_t key(d.str(*n), n->len);
sax.key(key);
++n;
}
}
}
} // namespace view
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
+67
View File
@@ -0,0 +1,67 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#pragma once
#include <cstddef> // size_t
#include <string> // string
#include <nlohmann/json.hpp>
#include <nlohmann/detail/view/macro_scope.hpp>
#include <nlohmann/detail/view/node.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
{
namespace view
{
/*!
@brief the value of the float token of a node, as parse() converts it
Uses the lexer's conversion (detail::convert_float), so that the values are
bit-identical to parse(): float and double are converted without allocation
and independent of the locale. The digit layout recorded while parsing locates
the decimal point and the exponent without scanning the token.
*/
template<typename FloatType>
NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n)
{
const char* const last = first + n.len;
const std::size_t neg = first[0] == '-' ? 1 : 0;
const std::size_t int_digits = n.extra & 0xFFu;
const std::size_t frac_digits = n.extra >> 8u;
std::size_t dot = std::string::npos;
std::size_t mantissa_end = n.len;
if (int_digits != 255 && frac_digits != 255)
{
dot = frac_digits != 0 ? neg + int_digits : std::string::npos;
mantissa_end = neg + int_digits + (frac_digits != 0 ? 1 + frac_digits : 0);
}
else
{
// more digits than the layout records: locate them
for (std::size_t i = 0; i < n.len; ++i)
{
if (first[i] == '.')
{
dot = i;
}
else if (first[i] == 'e' || first[i] == 'E')
{
mantissa_end = i;
break;
}
}
}
return convert_float<FloatType>(first, last, dot, mantissa_end);
}
} // namespace view
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END