Merge branch 'json-view/13-view-dump' into json-view/16-view-simd

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann committed 2026-10-09 17:06:43 +02:00
commit 66e875895f
55 files changed
+1339 -653

No files matched your search

+4 -1
View File
@@ -6,7 +6,10 @@ Each parser of the library (JSON, BJData, BON8, BSON, CBOR, MessagePack, and UBJ
Additionally, `parse_json_view_fuzzer` (`tests/src/fuzzer-parse_json_view.cpp`) cross-checks `json_document`/`json_view`
(the zero-copy, read-only view declared in `json_view.hpp`) against `basic_json` on the same JSON text: it asserts that
`json_document::accept` agrees with `json::accept`, that an accepted input materializes to the same value `json::parse`
produces, and that a rejected input makes both parsers throw with an identical `what()`. It takes plain JSON text, so it
produces, and that a rejected input makes both parsers throw with an identical `what()`. It checks this for a
`std::string` input (borrowed, with a NUL after the last byte) and for an exact-size `std::vector<std::uint8_t>`
(borrowed, with nothing after the last byte), and for the `ignore_comments` and `ignore_trailing_commas` options, which
are taken from the low bits of the first input byte (the byte stays part of the text). It takes plain JSON text, so it
reuses the `corpus_json` corpus rather than a format of its own.
## What the fuzzers check
+89 -34
View File
@@ -9,7 +9,9 @@
/*
This file implements a parser test suitable for fuzz testing. It checks that
json_document (the zero-copy, read-only view of a parsed JSON text declared in
json_view.hpp) agrees with basic_json on every input:
json_view.hpp) agrees with basic_json on every input, for the parse options
selected by the low bits of the first input byte (bit 0: ignore_comments, bit 1:
ignore_trailing_commas; the byte stays part of the text):
- json_document::accept(data) must equal json::accept(data)
- if the input is accepted, json_document::parse(data).root().materialize()
@@ -18,12 +20,20 @@ json_view.hpp) agrees with basic_json on every input:
enabled) must throw a json::parse_error or json::out_of_range whose what()
is identical to the one json::parse(data) throws
This is checked for two kinds of input: a std::string, which the document
borrows and which ends in the NUL the parser uses as sentinel, and an
exact-size byte vector, which has no NUL after its last byte and takes the
parser's bounds-checked path (AddressSanitizer reports any read past the end).
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
drivers.
*/
#include <cassert>
#include <cstddef>
#include <cstdint>
#include <string>
#include <vector>
#include <nlohmann/json.hpp>
#include <nlohmann/json_view.hpp>
@@ -35,65 +45,110 @@ drivers.
using json = nlohmann::json;
using json_document = nlohmann::json_document;
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
namespace
{
// json_document::accept only has a single-argument overload; wrap the raw
// bytes in a (borrowed) std::string so the same bytes can be handed to it
const std::string input(reinterpret_cast<const char*>(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
// what json::parse does with a text: the value, or the message of the exception
struct reference_result
{
bool accepted = false;
json value{};
std::string what{}; // NOLINT(readability-redundant-member-init)
};
const bool accepted_by_json = json::accept(data, data + size);
const bool accepted_by_view = json_document::accept(input);
reference_result parse_reference(const std::uint8_t* data, std::size_t size, bool comments, bool trailing_commas)
{
reference_result r;
r.accepted = json::accept(data, data + size, comments, trailing_commas);
bool json_threw = false;
try
{
r.value = json::parse(data, data + size, nullptr, true, comments, trailing_commas);
}
catch (const json::parse_error& e)
{
r.what = e.what();
json_threw = true;
}
catch (const json::out_of_range& e)
{
r.what = e.what();
json_threw = true;
}
// json::accept and json::parse must agree
assert(json_threw == !r.accepted);
static_cast<void>(json_threw);
return r;
}
// json_document must agree with the reference for this input (a container
// that json_document::parse borrows)
template<typename Input>
void check_input(const Input& input, const reference_result& expected, bool comments, bool trailing_commas)
{
// json_document::accept must agree with json::accept on every input
assert(accepted_by_json == accepted_by_view);
const bool accepted_by_view = json_document::accept(input, comments, trailing_commas);
assert(expected.accepted == accepted_by_view);
static_cast<void>(accepted_by_view);
if (accepted_by_json)
if (expected.accepted)
{
// both parsers must agree on the resulting value
json const j1 = json::parse(data, data + size);
json_document const doc = json_document::parse(input);
json_document const doc = json_document::parse(input, true, comments, trailing_commas);
assert(!doc.is_discarded());
json const j2 = doc.root().materialize();
assert(j1 == j2);
assert(expected.value == j2);
static_cast<void>(j2);
// (without exceptions, the same document)
json_document const quiet = json_document::parse(input, false, comments, trailing_commas);
assert(!quiet.is_discarded());
assert(quiet.node_count() == doc.node_count());
}
else
{
// both parsers must reject the input the same way when exceptions are used
std::string expected_what;
bool json_threw = false;
try
{
static_cast<void>(json::parse(data, data + size));
}
catch (const json::parse_error& e)
{
expected_what = e.what();
json_threw = true;
}
catch (const json::out_of_range& e)
{
expected_what = e.what();
json_threw = true;
}
assert(json_threw);
bool view_threw = false;
try
{
static_cast<void>(json_document::parse(input));
static_cast<void>(json_document::parse(input, true, comments, trailing_commas));
}
catch (const json::parse_error& e)
{
assert(e.what() == expected_what);
assert(e.what() == expected.what);
view_threw = true;
}
catch (const json::out_of_range& e)
{
assert(e.what() == expected_what);
assert(e.what() == expected.what);
view_threw = true;
}
assert(view_threw);
static_cast<void>(view_threw);
// and without exceptions, the document is discarded
json_document const quiet = json_document::parse(input, false, comments, trailing_commas);
assert(quiet.is_discarded());
static_cast<void>(quiet);
}
}
} // namespace
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
{
// the parse options are taken from the low bits of the first byte
const bool comments = size > 0 && (data[0] & 1U) != 0;
const bool trailing_commas = size > 0 && (data[0] & 2U) != 0;
const reference_result expected = parse_reference(data, size, comments, trailing_commas);
// a std::string: borrowed, with the NUL of std::string as sentinel
const std::string input(reinterpret_cast<const char*>(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
check_input(input, expected, comments, trailing_commas);
// an exact-size byte vector: borrowed, with nothing after its last byte
const std::vector<std::uint8_t> exact(data, data + size);
check_input(exact, expected, comments, trailing_commas);
// return 0 - non-zero return values are reserved for future use
return 0;
+7
View File
@@ -1792,4 +1792,11 @@ TEST_CASE("string scanning kernels")
CHECK(nlohmann::detail::count_trailing_zeros(bit) == k);
CHECK(nlohmann::detail::count_trailing_zeros(bit | (bit << 1u) | 0x8000000000000000u) == k);
}
// eight bytes as a little-endian word, at any alignment
const unsigned char bytes[16] = {0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF};
CHECK(nlohmann::detail::read_eight_bytes(bytes) == 0x0807060504030201u);
CHECK(nlohmann::detail::read_eight_bytes(bytes + 1) == 0x0908070605040302u);
CHECK(nlohmann::detail::read_eight_bytes(bytes + 8) == 0xFF0F0E0D0C0B0A09u);
CHECK(nlohmann::detail::read_eight_bytes(reinterpret_cast<const char*>(bytes) + 3) == 0x0B0A090807060504u);
}
+377 -46
View File
@@ -19,16 +19,19 @@ using nlohmann::ordered_json_view;
#include <algorithm>
#include <array>
#include <cmath>
#include <cstddef>
#include <cstdint>
#include <cstdio>
#include <cstring>
#include <iomanip>
#include <iterator>
#include <limits>
#include <list>
#include <map>
#include <random>
#include <sstream>
#include <string>
#include <type_traits>
#include <unordered_map>
#include <utility>
#include <vector>
@@ -39,6 +42,55 @@ using nlohmann::ordered_json_view;
namespace
{
// the value of a text, through a named document: the views of a temporary
// document would dangle (root() of an rvalue document does not compile)
template<typename Document, typename... Args>
auto materialized(Args&& ... args) -> decltype(std::declval<typename Document::view_type>().materialize())
{
const Document d = Document::parse(std::forward<Args>(args)...);
return d.root().materialize();
}
template<typename Document, typename Input>
auto materialized_copy(Input&& input) -> decltype(std::declval<typename Document::view_type>().materialize())
{
const Document d = Document::parse_copy(std::forward<Input>(input));
return d.root().materialize();
}
// a "byte container" that claims to hold `size` bytes, to reach the limit on
// the size of the input without allocating gigabytes; nothing past the first
// bytes is ever read, because the size is checked before the parse starts
struct oversized_input
{
using value_type = char;
std::size_t claimed;
const char* data() const
{
return "[1]";
}
std::size_t size() const
{
return claimed;
}
};
// detection of calls that must not compile
template<typename... Args>
using parse_call_t = decltype(json_document::parse(std::declval<Args>()...));
template<typename... Args>
using parse_copy_call_t = decltype(json_document::parse_copy(std::declval<Args>()...));
template<typename... Args>
using accept_call_t = decltype(json_document::accept(std::declval<Args>()...));
template<typename... Args>
using read_call_t = decltype(std::declval<json_document&>().read(std::declval<Args>()...));
template<typename Document>
using root_call_t = decltype(std::declval<Document>().root());
template<typename View>
using bool_conversion_t = decltype(static_cast<bool>(std::declval<View>()));
#if !defined(JSON_NOEXCEPTION)
// the exception parse() throws for a text, or "" if it accepts it
std::string parse_exception(const std::string& text, bool comments = false, bool trailing_commas = false)
@@ -152,7 +204,6 @@ TEST_CASE("json_view")
CHECK(v.is_primitive() == j.is_primitive());
CHECK(v.is_structured() == j.is_structured());
CHECK(!v.is_discarded());
CHECK(static_cast<bool>(v));
CHECK(v.size() == j.size());
CHECK(v.empty() == j.empty());
CHECK(v.materialize() == j);
@@ -160,7 +211,6 @@ TEST_CASE("json_view")
const json_view invalid{};
CHECK(invalid.is_discarded());
CHECK(!static_cast<bool>(invalid));
CHECK(invalid.type() == json::value_t::discarded);
CHECK(invalid.size() == 0);
CHECK(invalid.empty());
@@ -176,19 +226,19 @@ TEST_CASE("json_view")
std::string text;
g.value(text, 0);
CAPTURE(text)
CHECK(json_document::parse(text).root().materialize() == json::parse(text));
CHECK(materialized<json_document>(text) == json::parse(text));
// member order as ordered_json::parse keeps it
CHECK(ordered_json_document::parse(text).root().materialize().dump() == ordered_json::parse(text).dump());
CHECK(materialized<ordered_json_document>(text).dump() == ordered_json::parse(text).dump());
}
// duplicate keys: the last value, at the position of the first key
CHECK(json_document::parse(R"({"a":1,"b":2,"a":3})").root().materialize() == json::parse(R"({"a":1,"b":2,"a":3})"));
CHECK(ordered_json_document::parse(R"({"a":1,"b":2,"a":3})").root().materialize().dump() == R"({"a":3,"b":2})");
CHECK(materialized<json_document>(R"({"a":1,"b":2,"a":3})") == json::parse(R"({"a":1,"b":2,"a":3})"));
CHECK(materialized<ordered_json_document>(R"({"a":1,"b":2,"a":3})").dump() == R"({"a":3,"b":2})");
// very deep nesting (iterative, as parse())
const std::string deep = std::string(100000, '[') + std::string(100000, ']');
CHECK(json_document::parse(deep).root().materialize() == json::parse(deep));
CHECK(materialized<json_document>(deep) == json::parse(deep));
#if JSON_DIAGNOSTICS
// the parents are set, so errors name the path
const json m = json_document::parse(R"({"a":{"b":[1]}})").root().materialize();
const json m = materialized<json_document>(R"({"a":{"b":[1]}})");
CHECK_THROWS_WITH_AS(m.at("a").at("b").at(0).at("x"), "[json.exception.type_error.304] (/a/b/0) cannot use at() with number", json::type_error&);
#endif
}
@@ -261,7 +311,7 @@ TEST_CASE("json_view")
CHECK(json_document::accept(text));
if (accepted)
{
CHECK(float_document::parse(text).root().materialize() == json_float::parse(text));
CHECK(materialized<float_document>(text) == json_float::parse(text));
}
}
float_document f;
@@ -275,7 +325,7 @@ TEST_CASE("json_view")
CHECK(json_document::accept(with_nul) == json::accept(with_nul));
const std::string nul_in_comment("[1, // c\0\n2]", 12);
CHECK(json_document::accept(nul_in_comment, true) == json::accept(nul_in_comment, true));
CHECK(json_document::parse("\xEF\xBB\xBF[1]").root().materialize() == json::parse("\xEF\xBB\xBF[1]"));
CHECK(materialized<json_document>("\xEF\xBB\xBF[1]") == json::parse("\xEF\xBB\xBF[1]"));
#if !defined(JSON_NOEXCEPTION)
CHECK(view_exception("\xEF\xBB") == parse_exception("\xEF\xBB"));
#endif
@@ -291,18 +341,18 @@ TEST_CASE("json_view")
CHECK(!borrowed.owns_source());
CHECK(borrowed.source().data() == text.data());
CHECK(borrowed.root().materialize() == expected);
CHECK(json_document::parse(text.c_str()).root().materialize() == expected);
CHECK(json_document::parse(R"([1, "two", {"three": 3.5}])").root().materialize() == expected);
CHECK(json_document::parse(text.data(), text.data() + text.size()).root().materialize() == expected);
CHECK(materialized<json_document>(text.c_str()) == expected);
CHECK(materialized<json_document>(R"([1, "two", {"three": 3.5}])") == expected);
CHECK(materialized<json_document>(text.data(), text.data() + text.size()) == expected);
const std::vector<char> chars(text.begin(), text.end());
CHECK(!json_document::parse(chars).owns_source());
CHECK(json_document::parse(chars).root().materialize() == expected);
CHECK(materialized<json_document>(chars) == expected);
const std::vector<std::uint8_t> bytes(text.begin(), text.end());
CHECK(json_document::parse(bytes).root().materialize() == expected);
CHECK(materialized<json_document>(bytes) == expected);
#ifdef JSON_HAS_CPP_17
const std::string_view sv = text;
CHECK(!json_document::parse(sv).owns_source());
CHECK(json_document::parse(sv).root().materialize() == expected);
CHECK(materialized<json_document>(sv) == expected);
#endif
// owned
@@ -311,15 +361,28 @@ TEST_CASE("json_view")
CHECK(from_rvalue.owns_source());
CHECK(from_rvalue.root().materialize() == expected);
CHECK(json_document::parse(std::vector<char>(text.begin(), text.end())).owns_source());
// a const rvalue cannot be moved from, and is not borrowed (it may be a
// temporary): it is copied, as is a const rvalue of any container
const std::string const_text = text;
const json_document from_const_rvalue = json_document::parse(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg)
CHECK(from_const_rvalue.owns_source());
CHECK(from_const_rvalue.source().data() != const_text.data());
CHECK(from_const_rvalue.root().materialize() == expected);
json_document read_const_rvalue;
read_const_rvalue.read(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg)
CHECK(read_const_rvalue.owns_source());
CHECK(read_const_rvalue.root().materialize() == expected);
const std::vector<char> const_chars(text.begin(), text.end());
CHECK(json_document::parse(std::move(const_chars)).owns_source()); // NOLINT(performance-move-const-arg,hicpp-move-const-arg)
CHECK(json_document::parse_copy(text).owns_source());
CHECK(json_document::parse_copy(text).root().materialize() == expected);
CHECK(materialized_copy<json_document>(text) == expected);
std::istringstream stream(text);
const json_document from_stream = json_document::parse(stream);
CHECK(from_stream.owns_source());
CHECK(from_stream.root().materialize() == expected);
const std::list<char> list(text.begin(), text.end());
CHECK(json_document::parse(list.begin(), list.end()).owns_source());
CHECK(json_document::parse(list.begin(), list.end()).root().materialize() == expected);
CHECK(materialized<json_document>(list.begin(), list.end()) == expected);
// iterator pairs: pointers are borrowed, and so are contiguous library
// iterators where the input adapter detects them (C++20)
@@ -330,14 +393,85 @@ TEST_CASE("json_view")
CHECK((from_iterators.source().data() == chars.data()) == contiguous);
CHECK(from_iterators.root().materialize() == expected);
const std::string padded = "x" + text + "x";
CHECK(json_document::parse(padded.begin() + 1, padded.end() - 1).root().materialize() == expected);
CHECK(materialized<json_document>(padded.begin() + 1, padded.end() - 1) == expected);
CHECK(json_document::parse(chars.cbegin(), chars.cbegin(), false).is_discarded());
const std::wstring wide = L"[\"\u00e4\u20ac\", 1]";
CHECK(json_document::parse(wide).root().materialize() == json::parse(wide));
CHECK(materialized<json_document>(wide) == json::parse(wide));
CHECK(json_document::parse(static_cast<const char*>(nullptr), false).is_discarded());
CHECK(json_document::parse("", false).is_discarded());
}
SECTION("integer arguments do not compile")
{
using nlohmann::detail::is_detected;
// a length is not a flag: parse(ptr, len) would convert len to
// allow_exceptions and read ptr as a C string, which need not end
static_assert(is_detected<parse_call_t, const char*, bool>::value, "parse(ptr, bool) is valid");
static_assert(is_detected<parse_call_t, const char*, bool, bool, bool>::value, "parse(ptr, bool, bool, bool) is valid");
static_assert(is_detected<parse_call_t, const char*, const char*>::value, "parse(first, last) is valid");
static_assert(is_detected<parse_call_t, const char*, const char*, bool>::value, "parse(first, last, bool) is valid");
static_assert(!is_detected<parse_call_t, const char*, std::size_t>::value, "parse(ptr, len) must not compile");
static_assert(!is_detected<parse_call_t, const char*, int>::value, "parse(ptr, int) must not compile");
static_assert(!is_detected<parse_call_t, const char*, char>::value, "parse(ptr, char) must not compile");
static_assert(!is_detected<parse_call_t, const char*, std::size_t, bool>::value, "parse(ptr, len, bool) must not compile");
static_assert(!is_detected<parse_call_t, const std::string&, std::size_t>::value, "parse(string, len) must not compile");
static_assert(!is_detected<parse_call_t, const std::vector<char>&, std::size_t>::value, "parse(vector, len) must not compile");
static_assert(is_detected<parse_copy_call_t, const char*, bool>::value, "parse_copy(ptr, bool) is valid");
static_assert(!is_detected<parse_copy_call_t, const char*, std::size_t>::value, "parse_copy(ptr, len) must not compile");
static_assert(is_detected<accept_call_t, const char*, bool>::value, "accept(ptr, bool) is valid");
static_assert(!is_detected<accept_call_t, const char*, std::size_t>::value, "accept(ptr, len) must not compile");
static_assert(is_detected<read_call_t, const char*, bool>::value, "read(ptr, bool) is valid");
static_assert(!is_detected<read_call_t, const char*, std::size_t>::value, "read(ptr, len) must not compile");
// json_view has no conversion to bool: unlike basic_json's, it would
// mean "exists", not "is not null"; use is_discarded()
static_assert(!is_detected<bool_conversion_t, json_view>::value, "json_view must not convert to bool");
// the valid calls still work
const char* const text = "[1]";
CHECK(materialized<json_document>(text, true) == json::parse(text));
CHECK(json_document::accept(text, true, true));
}
SECTION("root of a temporary document does not compile")
{
using nlohmann::detail::is_detected;
// the view would dangle: auto v = json_document::parse(text).root();
static_assert(is_detected<root_call_t, json_document&>::value, "root() of an lvalue is valid");
static_assert(is_detected<root_call_t, const json_document&>::value, "root() of a const lvalue is valid");
static_assert(!is_detected<root_call_t, json_document>::value, "root() of an rvalue must not compile");
static_assert(!is_detected < root_call_t, json_document && >::value, "root() of an rvalue must not compile");
static_assert(!is_detected < root_call_t, const json_document && >::value, "root() of a const rvalue must not compile");
static_assert(!is_detected<root_call_t, ordered_json_document>::value, "root() of an rvalue must not compile");
// a named document is fine, also after a move
json_document d = json_document::parse("[1]");
CHECK(d.root().size() == 1);
const json_document moved = std::move(d);
CHECK(moved.root().size() == 1);
}
SECTION("input size limit")
{
// 32-bit offsets: the limit is 4 GiB minus 16 bytes (a margin below 2^32),
// which is what the exception message and the documentation say
const std::size_t limit = nlohmann::detail::view::max_input_size;
CHECK(limit == std::size_t{4294967279u});
const oversized_input input{limit + 1};
CHECK(!json_document::accept(input));
CHECK(json_document::parse(input, false).is_discarded());
#if !defined(JSON_NOEXCEPTION)
json_document d;
CHECK_THROWS_WITH_AS(d = json_document::parse(input), "[json.exception.out_of_range.416] input of 4294967280 bytes or more is not supported by json_document", json::out_of_range&);
#endif
}
SECTION("document lifetime and reuse")
{
json_document d;
@@ -443,7 +577,7 @@ std::string exception_of(F f)
// compares a view with the ordered_json value materialize() gives for it:
// types, sizes, elements and members (by index, key, and iteration), in
// document order; duplicate keys are found as their first occurrence
// document order; duplicate keys are found as their last occurrence
void check_access(const ordered_json_view& v, const ordered_json& j)
{
REQUIRE(v.type() == j.type());
@@ -460,7 +594,7 @@ void check_access(const ordered_json_view& v, const ordered_json& j)
++i;
}
CHECK(i == v.size());
CHECK(!v[v.size()]);
CHECK(v[v.size()].is_discarded());
std::size_t index = 0;
for (const auto& item : v.items())
{
@@ -484,11 +618,23 @@ void check_access(const ordered_json_view& v, const ordered_json& j)
const std::string key(it.key().data(), it.key().size());
CHECK(v.contains(key));
CHECK(v.count(key) == 1);
if (std::find(keys.begin(), keys.end(), key) != keys.end())
if (std::find(keys.begin(), keys.end(), key) == keys.end())
{
continue; // a duplicate: lookups find the first one
keys.push_back(key);
}
// lookups find the last member with the key, which is this one if
// there is no later one
auto next = it;
++next;
bool is_last = true;
for (; next != v.end(); ++next)
{
is_last = is_last && next.key() != it.key();
}
if (!is_last)
{
continue;
}
keys.push_back(key);
CHECK(v.find(key) == it);
CHECK(v[key].materialize() == it->materialize());
CHECK(v.at(key).materialize() == it.value().materialize());
@@ -514,7 +660,7 @@ void check_access(const ordered_json_view& v, const ordered_json& j)
CHECK(v.back().materialize() == j.back());
}
}
CHECK(!v["not a key in the generated documents"]);
CHECK(v["not a key in the generated documents"].is_discarded());
CHECK(v.find("not a key in the generated documents") == v.end());
}
else
@@ -579,15 +725,35 @@ TEST_CASE("json_view element access and iteration")
#endif
}
SECTION("duplicate keys: lookups find the first member, iteration all")
SECTION("duplicate keys: lookups find the last member, iteration all")
{
const json_document d = json_document::parse(R"({"a":1,"b":2,"a":3})");
const json_view v = d.root();
CHECK(v.size() == 3);
CHECK(v["a"].materialize() == 1);
CHECK(v.at("a").materialize() == 1);
CHECK(v.find("a") == v.begin());
CHECK(v["a"].materialize() == 3);
CHECK(v.at("a").materialize() == 3);
CHECK(v.find("a") == std::next(v.begin(), 2));
CHECK(v.find("a").value().materialize() == 3);
CHECK(v.find("b") == std::next(v.begin()));
CHECK(v.count("a") == 1);
CHECK(v.contains("a"));
CHECK(v.value("a", 0) == 3);
CHECK(v["a"].materialize() == v.materialize()["a"]); // as materialize()
// keys of every length class (the 16-byte short compare and memcmp)
for (const std::size_t n :
{
0u, 1u, 3u, 7u, 8u, 15u, 16u, 17u, 40u
})
{
const std::string key(n, 'k');
const json_document dk = json_document::parse("{\"" + key + "\":1,\"" + key + "x\":2,\"" + key + "\":3,\"" + key + "\":4}");
CAPTURE(n)
CHECK(dk.root()[key].materialize() == 4);
CHECK(dk.root().at(key).materialize() == 4);
CHECK(dk.root().find(key) == std::next(dk.root().begin(), 3));
CHECK(dk.root().value(key, 0) == 4);
CHECK(dk.root()[key + "x"].materialize() == 2);
}
std::string order;
for (auto it = v.begin(); it != v.end(); ++it)
{
@@ -638,14 +804,131 @@ TEST_CASE("json_view element access and iteration")
// where basic_json has undefined behavior, the view answers safely
const json_document d = json_document::parse(R"({"a":[]})");
CHECK(!d.root()["b"]);
CHECK(!d.root()["a"][0]);
CHECK(d.root()["b"].is_discarded());
CHECK(d.root()["a"][0].is_discarded());
CHECK_THROWS_WITH_AS(d.root()["a"].front(), "[json.exception.invalid_iterator.214] cannot get value", json::invalid_iterator&);
CHECK_THROWS_WITH_AS(d.root()["a"].back(), "[json.exception.invalid_iterator.214] cannot get value", json::invalid_iterator&);
const json_view invalid{};
CHECK(invalid.begin() == invalid.end());
CHECK(std::string(invalid.type_name()) == "discarded");
CHECK_THROWS_WITH_AS(invalid["a"], "[json.exception.type_error.305] cannot use operator[] with a string argument with discarded", json::type_error&);
}
SECTION("chained access is safe: operator[] of a discarded view is discarded")
{
const json_document d = json_document::parse(R"({"a":{"b":[10,20]},"s":"str"})");
const json_view v = d.root();
// missing keys and indexes
CHECK(v["x"].is_discarded());
CHECK(v["x"]["y"].is_discarded());
CHECK(v["x"]["y"]["z"].is_discarded());
CHECK(v["x"][0].is_discarded());
CHECK(v["x"][0u][1L].is_discarded());
CHECK(v["a"]["b"][2].is_discarded());
CHECK(v["a"]["b"][2]["c"].is_discarded());
CHECK(v["a"]["b"][2][json_view::json_pointer("/c")].is_discarded());
CHECK(v["x"][json_view::json_pointer("/a/b")].is_discarded());
CHECK(v["x"][json_view::json_pointer("")].is_discarded());
CHECK(v["x"]["y"].is_discarded());
CHECK(v["x"][std::string("y")].is_discarded());
// a resolvable path still resolves
CHECK(v["a"]["b"][1].materialize() == 20);
CHECK(v[json_view::json_pointer("/a/b/1")].materialize() == 20);
// the discarded view of an unresolved pointer is discarded too
CHECK(v[json_view::json_pointer("/x/y")]["z"].is_discarded());
CHECK(v[json_view::json_pointer("/a/b/5")][0].is_discarded());
#if !defined(JSON_NOEXCEPTION)
// type errors on values that are not discarded stay
CHECK_THROWS_WITH_AS(v[0], "[json.exception.type_error.305] cannot use operator[] with a numeric argument with object", json::type_error&);
CHECK_THROWS_WITH_AS(v["a"]["b"]["c"], "[json.exception.type_error.305] cannot use operator[] with a string argument with array", json::type_error&);
CHECK_THROWS_WITH_AS(v["s"]["c"], "[json.exception.type_error.305] cannot use operator[] with a string argument with string", json::type_error&);
CHECK_THROWS_WITH_AS(v["s"][0], "[json.exception.type_error.305] cannot use operator[] with a numeric argument with string", json::type_error&);
CHECK_THROWS_AS(v["a"]["b"][0]["c"], json::type_error&);
CHECK_THROWS_AS(v["s"][json_view::json_pointer("/x")], json::out_of_range&);
// at() keeps throwing on a discarded view
const json_view invalid{};
CHECK(invalid["a"].is_discarded());
CHECK(invalid[0].is_discarded());
CHECK(invalid[json_view::json_pointer("/a")].is_discarded());
CHECK_THROWS_WITH_AS(invalid.at("a"), "[json.exception.type_error.304] cannot use at() with discarded", json::type_error&);
CHECK_THROWS_WITH_AS(invalid.at(0), "[json.exception.type_error.304] cannot use at() with discarded", json::type_error&);
CHECK_THROWS_AS(v.at("x").at("y"), json::out_of_range&);
CHECK_THROWS_AS(v["x"].at("y"), json::type_error&);
CHECK_THROWS_AS(v.at(json_view::json_pointer("/x/y")), json::out_of_range&);
CHECK_THROWS_AS(v["x"].at(json_view::json_pointer("/y")), json::out_of_range&);
#endif
}
SECTION("integer types as array indexes")
{
const json_document d = json_document::parse("[10,20,30]");
const json_view v = d.root();
const json j = v.materialize();
// (compile-time: no overload is ambiguous)
CHECK(v[0].materialize() == 10);
CHECK(v[1].materialize() == 20);
CHECK(v[0u].materialize() == 10);
CHECK(v[1u].materialize() == 20);
CHECK(v[1L].materialize() == 20);
CHECK(v[2UL].materialize() == 30);
CHECK(v[1LL].materialize() == 20);
CHECK(v[2ULL].materialize() == 30);
CHECK(v[static_cast<short>(1)].materialize() == 20);
CHECK(v[static_cast<unsigned short>(2)].materialize() == 30);
CHECK(v[static_cast<signed char>(1)].materialize() == 20);
CHECK(v[static_cast<unsigned char>(2)].materialize() == 30);
CHECK(v[std::int8_t(1)].materialize() == 20);
CHECK(v[std::int16_t(2)].materialize() == 30);
CHECK(v[std::int32_t(1)].materialize() == 20);
CHECK(v[std::int64_t(2)].materialize() == 30);
CHECK(v[std::uint32_t(0)].materialize() == 10);
CHECK(v[std::uint64_t(1)].materialize() == 20);
CHECK(v[std::size_t(2)].materialize() == 30);
CHECK(v[std::ptrdiff_t(1)].materialize() == 20);
CHECK(j[0u] == 10); // as basic_json
CHECK(v.at(0).materialize() == 10);
CHECK(v.at(1u).materialize() == 20);
CHECK(v.at(1L).materialize() == 20);
CHECK(v.at(2LL).materialize() == 30);
CHECK(v.at(2ULL).materialize() == 30);
CHECK(v.at(static_cast<short>(1)).materialize() == 20);
CHECK(v.at(static_cast<unsigned short>(2)).materialize() == 30);
CHECK(v.at(std::int32_t(0)).materialize() == 10);
CHECK(v.at(std::uint32_t(0)).materialize() == 10);
CHECK(v.at(std::int64_t(0)).materialize() == 10);
CHECK(v.at(std::uint64_t(1)).materialize() == 20);
CHECK(v.at(std::size_t(2)).materialize() == 30);
CHECK(j.at(std::uint32_t(0)) == 10); // as basic_json
// out of range, including negative values (no wrap-around)
CHECK(v[3].is_discarded());
CHECK(v[3u].is_discarded());
CHECK(v[3L].is_discarded());
CHECK(v[-1].is_discarded());
CHECK(v[-1L].is_discarded());
CHECK(v[-1LL].is_discarded());
CHECK(v[static_cast<short>(-1)].is_discarded());
CHECK(v[std::int64_t(-3)].is_discarded());
CHECK(v[(std::numeric_limits<std::int64_t>::min)()].is_discarded());
CHECK(v[(std::numeric_limits<std::uint64_t>::max)()].is_discarded());
CHECK(v[(std::numeric_limits<std::size_t>::max)()].is_discarded());
CHECK(v[std::numeric_limits<int>::max()].is_discarded());
#if !defined(JSON_NOEXCEPTION)
CHECK_THROWS_WITH_AS(v.at(3), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&);
CHECK_THROWS_WITH_AS(v.at(3u), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&);
CHECK_THROWS_WITH_AS(v.at(std::int64_t(3)), "[json.exception.out_of_range.401] array index 3 is out of range", json::out_of_range&);
CHECK_THROWS_AS(v.at(-1), json::out_of_range&);
CHECK_THROWS_AS(v.at(-1L), json::out_of_range&);
CHECK_THROWS_AS(v.at(std::int64_t(-1)), json::out_of_range&);
CHECK_THROWS_AS(v.at((std::numeric_limits<std::int64_t>::min)()), json::out_of_range&);
CHECK_THROWS_AS(v.at((std::numeric_limits<std::uint64_t>::max)()), json::out_of_range&);
// not an array
const json_document o = json_document::parse("{}");
CHECK_THROWS_AS(o.root()[0u], json::type_error&);
CHECK_THROWS_AS(o.root()[1L], json::type_error&);
CHECK_THROWS_AS(o.root().at(std::uint32_t(0)), json::type_error&);
CHECK_THROWS_AS(o.root().at(std::int64_t(0)), json::type_error&);
#endif
}
SECTION("iterators")
@@ -919,10 +1202,12 @@ TEST_CASE("json_view values")
CAPTURE(token)
const std::string text = "[" + token + "]";
const double b = json::parse(text)[0].get<double>();
CHECK(bits(json_document::parse(text).root()[0].get<double>()) == bits(b));
const json_document dd = json_document::parse(text);
CHECK(bits(dd.root()[0].get<double>()) == bits(b));
if (std::abs(b) < 1e38)
{
CHECK(bits(nlohmann::basic_json_document<json_float>::parse(text).root()[0].get<float>()) == bits(json_float::parse(text)[0].get<float>()));
const nlohmann::basic_json_document<json_float> df = nlohmann::basic_json_document<json_float>::parse(text);
CHECK(bits(df.root()[0].get<float>()) == bits(json_float::parse(text)[0].get<float>()));
}
}
}
@@ -979,7 +1264,8 @@ TEST_CASE("json_view values")
CHECK(count == 3);
// a duplicate key: the last value, as parse()
CHECK((json_document::parse(R"({"a":1,"a":2})").root().get<std::map<std::string, int>>() == std::map<std::string, int> {{"a", 2}}));
const json_document dup = json_document::parse(R"({"a":1,"a":2})");
CHECK((dup.root().get<std::map<std::string, int>>() == std::map<std::string, int> {{"a", 2}}));
const json_view invalid{};
CHECK_THROWS_WITH_AS(invalid.get<int>(), "[json.exception.type_error.302] type must be number, but is discarded", json::type_error&);
@@ -1070,7 +1356,7 @@ TEST_CASE("json_view JSON pointers")
else if (at_error.find("out_of_range.401") != std::string::npos || at_error.find("out_of_range.403") != std::string::npos) // NOLINT(abseil-string-find-str-contains)
{
// undefined behavior for const basic_json::operator[]
CHECK(!v[p]);
CHECK(v[p].is_discarded());
}
else
{
@@ -1164,10 +1450,12 @@ TEST_CASE("json_view dump")
}
}
many += ']';
CHECK(json_document::parse(many).root().dump() == json::parse(many).dump());
const json_document many_document = json_document::parse(many);
CHECK(many_document.root().dump() == json::parse(many).dump());
using json_float = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, float>;
CHECK(nlohmann::basic_json_document<json_float>::parse("[0.1, 1.5e10, 3.4028235e38]").root().dump() == json_float::parse("[0.1, 1.5e10, 3.4028235e38]").dump());
const nlohmann::basic_json_document<json_float> float_document = nlohmann::basic_json_document<json_float>::parse("[0.1, 1.5e10, 3.4028235e38]");
CHECK(float_document.root().dump() == json_float::parse("[0.1, 1.5e10, 3.4028235e38]").dump());
}
SECTION("members in document order, all of them")
@@ -1180,7 +1468,42 @@ TEST_CASE("json_view dump")
SECTION("deep nesting")
{
const std::string deep = std::string(100000, '[') + std::string(100000, ']');
CHECK(json_document::parse(deep).root().dump() == deep);
const json_document deep_document = json_document::parse(deep);
CHECK(deep_document.root().dump() == deep);
}
SECTION("the output buffer of a small value is small")
{
// an escaped key after the value: its node lies in the arena, so the
// source extent of the value cannot be read from the next node
const std::string big(100000, 'a');
const std::string text = R"({"small":1,"list":[1,2,3],"k\n":")" + big + R"("})";
const json_document d = json_document::parse(text);
const auto small = d.root()["small"].dump();
CHECK(small == "1");
CHECK(small.capacity() < 4096);
const auto list = d.root()["list"].dump();
CHECK(list == "[1,2,3]");
CHECK(list.capacity() < 4096);
CHECK(d.root()["list"].dump(2).capacity() < 4096);
// the whole document and the large value are unaffected
CHECK(d.root().dump() == ordered_json::parse(text).dump());
CHECK(d.root()["k\n"].dump() == "\"" + big + "\"");
}
SECTION("output that outgrows the estimate")
{
// ensure_ascii writes six bytes for each two-byte character
std::string chars;
for (int i = 0; i < 5000; ++i)
{
chars += "\xC3\xA9";
}
const json_document d = json_document::parse("{\"a\":\"" + chars + R"(","k\n":1})");
const json expected = json::parse("\"" + chars + "\"");
CHECK(d.root()["a"].dump(-1, ' ', true) == expected.dump(-1, ' ', true));
CHECK(d.root()["a"].dump(-1, ' ', true).size() == 2 + 5000 * 6);
}
SECTION("streams and discarded views")
@@ -1242,7 +1565,9 @@ TEST_CASE("json_view comparison")
{
const auto same = [](const char* x, const char* y)
{
return json_document::parse(x).root() == json_document::parse(y).root();
const json_document dx = json_document::parse(x);
const json_document dy = json_document::parse(y);
return dx.root() == dy.root();
};
CHECK(same("1", "1.0"));
CHECK(same("[1, -1, 2.5]", "[1.0, -1.0, 25e-1]"));
@@ -1257,15 +1582,20 @@ TEST_CASE("json_view comparison")
CHECK(same("\"\\u00e9\"", "\"\xc3\xa9\""));
CHECK(!same("null", "false"));
CHECK(!same("[]", "{}"));
CHECK(ordered_json_document::parse(R"({"a": 1, "b": 2, "a": 3})").root() == ordered_json_document::parse(R"({"a": 3, "b": 2})").root());
CHECK(ordered_json_document::parse(R"({"a": 1, "b": 2})").root() != ordered_json_document::parse(R"({"b": 2, "a": 1})").root());
const ordered_json_document dup = ordered_json_document::parse(R"({"a": 1, "b": 2, "a": 3})");
const ordered_json_document last = ordered_json_document::parse(R"({"a": 3, "b": 2})");
CHECK(dup.root() == last.root());
const ordered_json_document ab = ordered_json_document::parse(R"({"a": 1, "b": 2})");
const ordered_json_document ba = ordered_json_document::parse(R"({"b": 2, "a": 1})");
CHECK(ab.root() != ba.root());
// discarded values compare as basic_json's do
const json discarded(json::value_t::discarded);
CHECK((json_view() == json_view()) == (discarded == discarded)); // NOLINT(readability-container-size-empty): operator== is tested
CHECK((json_view() == discarded) == (discarded == discarded));
CHECK(!(json_view() == json_document::parse("null").root())); // NOLINT(readability-container-size-empty)
CHECK(!(json_document::parse("null").root() == discarded));
const json_document null_document = json_document::parse("null");
CHECK(!(json_view() == null_document.root())); // NOLINT(readability-container-size-empty)
CHECK(!(null_document.root() == discarded));
}
SECTION("deep nesting")
@@ -1276,7 +1606,8 @@ TEST_CASE("json_view comparison")
CHECK(a.root() == b.root());
CHECK(a.root() == json::parse(deep));
const std::string other = std::string(100000, '[') + "1" + std::string(100000, ']');
CHECK(a.root() != json_document::parse(other).root());
const json_document c = json_document::parse(other);
CHECK(a.root() != c.root());
}
}
+88
View File
@@ -20,8 +20,10 @@ using nlohmann::json;
#include <array>
#include <cstdint>
#include <fstream>
#include <limits>
#include <map>
#include <memory>
#include <new>
#include <random>
#include <sstream>
#include <string>
@@ -506,3 +508,89 @@ TEST_CASE("json_view builder: strings across vector blocks")
}
}
}
TEST_CASE("json_view node integer bits")
{
using nlohmann::detail::view::integer_bits;
using nlohmann::detail::view::set_integer_bits;
// an integer lives in len (low half) and next (high half), on any byte
// order; a big-endian target must not store the native word over both
SECTION("set_integer_bits and integer_bits")
{
node n = {};
for (const std::uint64_t v :
{
std::uint64_t{0}, std::uint64_t{1}, std::uint64_t{0xFFFFFFFFu}, std::uint64_t{0x100000000u},
std::uint64_t{0x0000000200000003u}, std::uint64_t{0x0123456789ABCDEFu}, std::uint64_t{0xFFFFFFFFFFFFFFFEu}
})
{
CAPTURE(v)
n.kind = 0x5A;
n.flags = 0xA5;
n.extra = 0x1234;
n.off = 0x89ABCDEFu;
set_integer_bits(n, v);
CHECK(integer_bits(n) == v);
CHECK(n.len == static_cast<std::uint32_t>(v));
CHECK(n.next == static_cast<std::uint32_t>(v >> 32));
// the other fields are untouched
CHECK(n.kind == 0x5A);
CHECK(n.flags == 0xA5);
CHECK(n.extra == 0x1234);
CHECK(n.off == 0x89ABCDEFu);
}
}
SECTION("parsed integers")
{
struct integer_case
{
const char* text;
std::uint64_t bits;
};
for (const integer_case c :
{
integer_case{"[8589934595]", 0x0000000200000003u}, integer_case{"[4294967296]", 0x100000000u}, integer_case{"[4294967295]", 0xFFFFFFFFu},
integer_case{"[-2]", 0xFFFFFFFFFFFFFFFEu}, integer_case{"[-4294967297]", 0xFFFFFFFEFFFFFFFFu}, integer_case{"[18446744073709551615]", 0xFFFFFFFFFFFFFFFFu},
integer_case{"[7]", 7u}
})
{
CAPTURE(c.text)
const built b = build(c.text, false, false, true);
REQUIRE(b.ok);
const node& n = b.data->tape[1];
CHECK(integer_bits(n) == c.bits);
CHECK(n.len == static_cast<std::uint32_t>(c.bits));
CHECK(n.next == static_cast<std::uint32_t>(c.bits >> 32));
}
}
}
TEST_CASE("json_view node array size limit")
{
// a node array larger than the address space is refused, not wrapped to a
// small allocation (the size computation overflows on 32-bit targets, and
// for absurd counts everywhere)
std::unique_ptr<document_data, document_data::deleter> d(document_data::create(0));
d->reserve(8);
REQUIRE(d->tape_cap >= 8);
d->tape_size = 2;
const std::size_t cap = d->tape_cap;
node* const tape = d->tape;
#if !defined(JSON_NOEXCEPTION)
const std::size_t too_many = document_data::max_nodes() + 1;
CHECK_THROWS_AS(d->reserve(too_many), std::bad_alloc&);
CHECK_THROWS_AS(d->reserve((std::numeric_limits<std::size_t>::max)()), std::bad_alloc&);
// the array is unchanged
CHECK(d->tape == tape);
CHECK(d->tape_cap == cap);
CHECK(d->tape_size == 2);
#endif
// the largest count that fits is not refused by the check (nothing is
// allocated for a count that is already there)
d->reserve(cap);
CHECK(d->tape == tape);
}
+72 -5
View File
@@ -15,6 +15,7 @@
#include <nlohmann/json.hpp>
using nlohmann::detail::dtoa_impl::reinterpret_bits;
#include <algorithm>
#include <array>
#include <cmath>
#include <cstdint>
@@ -666,13 +667,24 @@ void check_shortest(double v)
const std::string text(buf.data(), end);
CAPTURE(text)
CHECK(parse_double(text) == v);
// the layout is that of format_buffer() for the same digits
// the layout is that of format_buffer() for the digits of Zmij
const auto sd = nlohmann::detail::zmij::to_shortest(reinterpret_bits<std::uint64_t>(v));
const std::uint64_t significand = sd.has_digit ? (sd.integral * 10) + sd.digit : sd.integral;
int exponent = sd.has_digit ? sd.exponent : sd.exponent + 1;
std::string significand_digits = std::to_string(significand);
while (significand_digits.size() > 1 && significand_digits.back() == '0')
{
significand_digits.pop_back();
++exponent;
}
std::array<char, 64> reference{};
int len = 0;
int exponent = 0;
nlohmann::detail::dtoa_impl::shortest_digits(reference.data(), len, exponent, v);
const char* const reference_end = nlohmann::detail::dtoa_impl::format_buffer(reference.data(), len, exponent, -4, 15);
std::copy(significand_digits.begin(), significand_digits.end(), reference.begin());
const char* const reference_end = nlohmann::detail::dtoa_impl::format_buffer(reference.data(), static_cast<int>(significand_digits.size()), exponent, -4, 15);
CHECK(text == std::string(reference.data(), static_cast<std::size_t>(reference_end - reference.data())));
// and write_positive() is what to_chars() calls
std::array<char, 64> positive{};
const char* const positive_end = nlohmann::detail::dtoa_impl::write_positive(positive.data(), positive.data() + positive.size(), v);
CHECK(text == std::string(positive.data(), static_cast<std::size_t>(positive_end - positive.data())));
const auto de = digits_and_exponent(text);
const std::string& digits = de.first;
if (digits.size() > 1)
@@ -785,3 +797,58 @@ TEST_CASE("shortest digits of doubles")
}
}
}
TEST_CASE("choice of the conversion")
{
using nlohmann::detail::dtoa_impl::is_binary64;
SECTION("by the format of the type")
{
// Zmij needs binary64 numbers; everything else uses Grisu2
static_assert(!is_binary64<float>::value, "float is not binary64");
static_assert(is_binary64<double>::value == (std::numeric_limits<double>::is_iec559 && std::numeric_limits<double>::digits == 53),
"double is binary64 where it is IEEE 754 with 53 digits");
static_assert(!is_binary64<int>::value, "integers are not binary64");
static_assert(is_binary64<long double>::value == (std::numeric_limits<long double>::is_iec559 && std::numeric_limits<long double>::digits == 53 && sizeof(long double) == 8),
"long double is binary64 where it has the format of a double");
CHECK(!is_binary64<float>::value);
CHECK(is_binary64<double>::value);
}
SECTION("float: Grisu2, double: Zmij")
{
// 5.3165205877497296e+16 is one of the doubles for which Grisu2 does not find the shortest digits
constexpr double value = 5.3165205877497296e+16;
std::array<char, 64> buf{};
const char* const last = buf.data() + buf.size();
char* end = nlohmann::detail::dtoa_impl::write_positive(buf.data(), last, value);
CHECK(std::string(buf.data(), end) == "5.31652058774973e+16");
end = nlohmann::detail::dtoa_impl::write_positive_grisu2(buf.data(), last, value);
CHECK(std::string(buf.data(), end) == "5.3165205877497296e+16");
constexpr float f = 1.1754944e-38f;
end = nlohmann::detail::dtoa_impl::write_positive(buf.data(), last, f);
const std::string dispatched(buf.data(), end);
end = nlohmann::detail::dtoa_impl::write_positive_grisu2(buf.data(), last, f);
CHECK(dispatched == std::string(buf.data(), end));
}
SECTION("long double with the format of a double: Zmij")
{
// (on platforms where long double is wider, Grisu2 does not apply either: the snprintf fallback does)
if (std::numeric_limits<long double>::digits == 53 && std::numeric_limits<long double>::is_iec559)
{
using long_double_json = nlohmann::json::with_float_t<long double>;
for (const double d :
{
5.3165205877497296e+16, 1.0, 0.1, 123456.789, 2.2250738585072014e-308, 1.7976931348623157e+308, -5.3165205877497296e+16
})
{
CAPTURE(d)
CHECK(long_double_json(static_cast<long double>(d)).dump() == nlohmann::json(d).dump());
}
CHECK(long_double_json(5.3165205877497296e+16L).dump() == "5.31652058774973e+16");
}
}
}