mirror of
https://github.com/nlohmann/json.git
synced 2026-10-11 08:57:15 +00:00
Scan json_view strings with SIMD and index large objects
Speed up json_view's parser with SIMD scanning and a hash table for large objects. Long runs of string bytes are scanned 16 bytes at a time with NEON (AArch64, GCC and Clang) and SSE2 (x86-64), both baseline instruction sets. Keys keep 16 table checks before the vector loop, because their lengths repeat from record to record; string values get 8, because their lengths vary more. Non-ASCII text is validated 16 bytes at a time with simdjson's "lookup4" check (Keiser and Lemire, 2021), with NEON on AArch64 and, on x86-64, with SSSE3. SSSE3 is not part of baseline x86-64, so the check is compiled for SSSE3 with a function attribute and used only where CPUID reports it, which all x86-64 CPUs since about 2011 do; the answer is cached in a statically initialized atomic, so there is no guard of a local static and no global constructor. The same input is accepted either way. JSON_VIEW_NO_SIMD selects the portable code. On x86-64, string runs are now checked vector-first: one SSE2 compare from the first byte finds the end of most keys and short values, instead of a branch per byte for the first 8-16 bytes. AArch64 keeps the byte-wise steps, where a NEON mask costs more and the branches predict well. Entering an object or array no longer stalls: open() stores the parent's frame field by field instead of building it on the stack and reading it back with wider loads, which waited for the narrower stores to retire. Objects with 128 members or more get an open-addressing hash table built when the object closes, so operator[], at(), find(), contains(), count(), value(), and JSON pointers take constant time on average in such objects; of duplicate keys, the first is kept, as for the linear search. The idea comes from Boost.JSON. simdjson is credited in simd.hpp's SPDX block, the README, and license.md. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
26 files changed
+1090
-26
No files matched your search
@@ -369,6 +369,25 @@ json_test_add_test_for(src/unit-diagnostic-positions.cpp
|
||||
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
|
||||
)
|
||||
|
||||
# the json_view parser again with the portable string scanning instead of
|
||||
# NEON/SSE2, and on x86-64 with the SSSE3 UTF-8 check (JSON_VIEW_USE_SSSE3)
|
||||
json_test_set_test_options(test-json_view_builder_portable
|
||||
COMPILE_DEFINITIONS JSON_VIEW_NO_SIMD
|
||||
)
|
||||
json_test_add_test_for(src/unit-json_view_builder.cpp
|
||||
NAME test-json_view_builder_portable
|
||||
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
|
||||
)
|
||||
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(x86_64|AMD64|amd64)$" AND NOT MSVC)
|
||||
json_test_set_test_options(test-json_view_builder_ssse3
|
||||
COMPILE_DEFINITIONS JSON_VIEW_USE_SSSE3
|
||||
COMPILE_OPTIONS -mssse3
|
||||
)
|
||||
json_test_add_test_for(src/unit-json_view_builder.cpp
|
||||
NAME test-json_view_builder_ssse3
|
||||
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
|
||||
)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# *DO NOT* use json_test_set_test_options() below this line
|
||||
|
||||
@@ -1321,3 +1321,247 @@ TEST_CASE("json_view JSON pointers")
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
TEST_CASE("json_view dump")
|
||||
{
|
||||
SECTION("the output of ordered_json::dump()")
|
||||
{
|
||||
generator g;
|
||||
for (int i = 0; i < 2000; ++i)
|
||||
{
|
||||
std::string text;
|
||||
g.value(text, 0);
|
||||
const ordered_json_document d = ordered_json_document::parse(text);
|
||||
if (has_duplicate_keys(d.root()))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
CAPTURE(text)
|
||||
const ordered_json j = ordered_json::parse(text);
|
||||
for (const int indent :
|
||||
{
|
||||
-1, 0, 2
|
||||
})
|
||||
{
|
||||
for (const bool ensure_ascii :
|
||||
{
|
||||
false, true
|
||||
})
|
||||
{
|
||||
CHECK(d.root().dump(indent, i % 2 == 0 ? ' ' : '\t', ensure_ascii) == j.dump(indent, i % 2 == 0 ? ' ' : '\t', ensure_ascii));
|
||||
}
|
||||
}
|
||||
// also of each element
|
||||
for (const ordered_json_view e : d.root())
|
||||
{
|
||||
CHECK(e.dump() == e.materialize().dump());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("strings")
|
||||
{
|
||||
const std::string text = R"(["plain", "\u0000\u0001\u001f\u007f\u0080é€😀", "\"\\\/\b\f\n\r\t", "aéあ😀b", "long text beyond the eight bytes of a word \n with an escape in the middle"])";
|
||||
const ordered_json_document d = ordered_json_document::parse(text);
|
||||
const ordered_json j = ordered_json::parse(text);
|
||||
CHECK(d.root().dump() == j.dump());
|
||||
CHECK(d.root().dump(-1, ' ', true) == j.dump(-1, ' ', true));
|
||||
CHECK(d.root().dump(4, ' ', true) == j.dump(4, ' ', true));
|
||||
const ordered_json_document keys = ordered_json_document::parse(R"({"é\n": {"\"": [], "": {}}})");
|
||||
CHECK(keys.root().dump(2, ' ', true) == ordered_json::parse(R"({"é\n": {"\"": [], "": {}}})").dump(2, ' ', true));
|
||||
}
|
||||
|
||||
SECTION("numbers")
|
||||
{
|
||||
const std::string text = "[1.50, 1E2, -0, -0.0, 123456789012345678901234567890, 18446744073709551615, -9223372036854775808, 0.1, 1e-7, 5e-324]";
|
||||
const json_document d = json_document::parse(text);
|
||||
CHECK(d.root().dump() == json::parse(text).dump());
|
||||
CHECK(d.root().dump() == "[1.5,100.0,0,-0.0,1.2345678901234568e+29,18446744073709551615,-9223372036854775808,0.1,1e-07,5e-324]");
|
||||
CHECK(d.root().dump(-1, ' ', false, json_view::number_format::source) == "[1.50,1E2,-0,-0.0,123456789012345678901234567890,18446744073709551615,-9223372036854775808,0.1,1e-7,5e-324]");
|
||||
|
||||
// random doubles, written as parse() and dump() would
|
||||
std::mt19937_64 rng(1170); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed)
|
||||
std::string many = "[";
|
||||
for (int i = 0; i < 5000; ++i)
|
||||
{
|
||||
const std::uint64_t bits = rng();
|
||||
double x = 0;
|
||||
std::memcpy(&x, &bits, sizeof(x));
|
||||
if (std::isfinite(x))
|
||||
{
|
||||
many += (many.size() > 1 ? "," : "") + json(x).dump();
|
||||
}
|
||||
}
|
||||
many += ']';
|
||||
CHECK(json_document::parse(many).root().dump() == json::parse(many).dump());
|
||||
|
||||
using json_float = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, float>;
|
||||
CHECK(nlohmann::basic_json_document<json_float>::parse("[0.1, 1.5e10, 3.4028235e38]").root().dump() == json_float::parse("[0.1, 1.5e10, 3.4028235e38]").dump());
|
||||
}
|
||||
|
||||
SECTION("members in document order, all of them")
|
||||
{
|
||||
const json_document d = json_document::parse(R"({"b": 1, "a": 2, "b": 3})");
|
||||
CHECK(d.root().dump() == R"({"b":1,"a":2,"b":3})");
|
||||
CHECK(d.root().dump(1) == "{\n \"b\": 1,\n \"a\": 2,\n \"b\": 3\n}");
|
||||
}
|
||||
|
||||
SECTION("deep nesting")
|
||||
{
|
||||
const std::string deep = std::string(100000, '[') + std::string(100000, ']');
|
||||
CHECK(json_document::parse(deep).root().dump() == deep);
|
||||
}
|
||||
|
||||
SECTION("streams and discarded views")
|
||||
{
|
||||
const json_document d = json_document::parse(R"({"a": [1, 2]})");
|
||||
std::ostringstream compact;
|
||||
compact << d.root();
|
||||
CHECK(compact.str() == R"({"a":[1,2]})");
|
||||
std::ostringstream pretty;
|
||||
pretty << std::setw(2) << std::setfill('.') << d.root() << d.root()["a"];
|
||||
CHECK(pretty.str() == "{\n..\"a\": [\n....1,\n....2\n..]\n}[1,2]");
|
||||
CHECK(json_view().dump() == json(json::value_t::discarded).dump());
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view comparison")
|
||||
{
|
||||
SECTION("equality of the values parse() produces")
|
||||
{
|
||||
generator g;
|
||||
std::vector<std::string> texts;
|
||||
for (int i = 0; i < 600; ++i)
|
||||
{
|
||||
std::string text;
|
||||
g.value(text, 0);
|
||||
texts.push_back(text);
|
||||
// the same value written differently: sorted keys, canonical numbers
|
||||
texts.push_back(json::parse(text).dump(1));
|
||||
}
|
||||
for (std::size_t i = 0; i + 2 < texts.size(); ++i)
|
||||
{
|
||||
for (std::size_t k = i; k < i + 3; ++k)
|
||||
{
|
||||
CAPTURE(texts[i])
|
||||
CAPTURE(texts[k])
|
||||
const json_document a = json_document::parse(texts[i]);
|
||||
const json_document b = json_document::parse(texts[k]);
|
||||
const json ja = json::parse(texts[i]);
|
||||
const json jb = json::parse(texts[k]);
|
||||
CHECK((a.root() == b.root()) == (ja == jb));
|
||||
CHECK((a.root() != b.root()) == (ja != jb));
|
||||
CHECK((a.root() == jb) == (ja == jb));
|
||||
CHECK((jb == a.root()) == (ja == jb));
|
||||
CHECK((a.root() != jb) == (ja != jb));
|
||||
CHECK((jb != a.root()) == (ja != jb));
|
||||
|
||||
// ordered_json compares members in order
|
||||
const ordered_json_document oa = ordered_json_document::parse(texts[i]);
|
||||
const ordered_json_document ob = ordered_json_document::parse(texts[k]);
|
||||
const ordered_json oja = ordered_json::parse(texts[i]);
|
||||
const ordered_json ojb = ordered_json::parse(texts[k]);
|
||||
CHECK((oa.root() == ob.root()) == (oja == ojb));
|
||||
CHECK((oa.root() == ojb) == (oja == ojb));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("numbers, duplicate keys, member order")
|
||||
{
|
||||
const auto same = [](const char* x, const char* y)
|
||||
{
|
||||
return json_document::parse(x).root() == json_document::parse(y).root();
|
||||
};
|
||||
CHECK(same("1", "1.0"));
|
||||
CHECK(same("[1, -1, 2.5]", "[1.0, -1.0, 25e-1]"));
|
||||
CHECK(!same("1", "1.5"));
|
||||
CHECK(same("18446744073709551615", "18446744073709551615"));
|
||||
CHECK(same(R"({"a": 1, "a": 2})", R"({"a": 2})"));
|
||||
CHECK(!same(R"({"a": 1, "a": 2})", R"({"a": 1})"));
|
||||
CHECK(same(R"({"a": 1, "b": 2})", R"({"b": 2, "a": 1})"));
|
||||
CHECK(!same(R"({"a": 1})", R"({"a": 1, "b": 2})"));
|
||||
CHECK(!same("[1, 2]", "[2, 1]"));
|
||||
CHECK(!same("\"a\"", "\"b\""));
|
||||
CHECK(same("\"\\u00e9\"", "\"\xc3\xa9\""));
|
||||
CHECK(!same("null", "false"));
|
||||
CHECK(!same("[]", "{}"));
|
||||
CHECK(ordered_json_document::parse(R"({"a": 1, "b": 2, "a": 3})").root() == ordered_json_document::parse(R"({"a": 3, "b": 2})").root());
|
||||
CHECK(ordered_json_document::parse(R"({"a": 1, "b": 2})").root() != ordered_json_document::parse(R"({"b": 2, "a": 1})").root());
|
||||
|
||||
// discarded values compare as basic_json's do
|
||||
const json discarded(json::value_t::discarded);
|
||||
CHECK((json_view() == json_view()) == (discarded == discarded)); // NOLINT(readability-container-size-empty): operator== is tested
|
||||
CHECK((json_view() == discarded) == (discarded == discarded));
|
||||
CHECK(!(json_view() == json_document::parse("null").root())); // NOLINT(readability-container-size-empty)
|
||||
CHECK(!(json_document::parse("null").root() == discarded));
|
||||
}
|
||||
|
||||
SECTION("deep nesting")
|
||||
{
|
||||
const std::string deep = std::string(100000, '[') + std::string(100000, ']');
|
||||
const json_document a = json_document::parse(deep);
|
||||
const json_document b = json_document::parse(deep);
|
||||
CHECK(a.root() == b.root());
|
||||
CHECK(a.root() == json::parse(deep));
|
||||
const std::string other = std::string(100000, '[') + "1" + std::string(100000, ']');
|
||||
CHECK(a.root() != json_document::parse(other).root());
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view large objects")
|
||||
{
|
||||
// objects with 128 members or more are looked up with a hash index
|
||||
for (const std::size_t members :
|
||||
{
|
||||
127u, 128u, 129u, 10000u
|
||||
})
|
||||
{
|
||||
CAPTURE(members)
|
||||
std::string text = "{";
|
||||
for (std::size_t i = 0; i < members; ++i)
|
||||
{
|
||||
text += (i != 0 ? ",\"" : "\"") + std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\\n" : "") + "\":" + std::to_string(i);
|
||||
}
|
||||
text += R"(,"":"empty key","k1":"a duplicate of an earlier key"})";
|
||||
const json_document d = json_document::parse(text);
|
||||
const json_view v = d.root();
|
||||
const json j = json::parse(text);
|
||||
for (std::size_t i = 0; i < members; ++i)
|
||||
{
|
||||
const std::string key = std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\n" : "");
|
||||
CHECK(v[key].get<std::size_t>() == i);
|
||||
CHECK(v.contains(key));
|
||||
CHECK(v.find(key).key() == key);
|
||||
CHECK(v.at(key).get<std::size_t>() == i);
|
||||
CHECK(!v.contains(key + "x"));
|
||||
}
|
||||
CHECK(v[""].get_string() == "empty key");
|
||||
CHECK(v["k1"].get<int>() == 1); // the first of duplicate keys, as for small objects
|
||||
CHECK(!v.contains("missing"));
|
||||
CHECK_THROWS_WITH_AS(v.at("missing"), "[json.exception.out_of_range.403] key 'missing' not found", json::out_of_range&);
|
||||
CHECK(v == j);
|
||||
CHECK(v.materialize() == j);
|
||||
}
|
||||
|
||||
SECTION("nested, reused, and in arrays")
|
||||
{
|
||||
std::string inner = "{";
|
||||
for (int i = 0; i < 300; ++i)
|
||||
{
|
||||
inner += (i != 0 ? ",\"m" : "\"m") + std::to_string(i) + "\":" + std::to_string(i);
|
||||
}
|
||||
inner += '}';
|
||||
const std::string text = "[" + inner + ",{\"x\":" + inner + "}," + inner + "]";
|
||||
json_document d = json_document::parse(text);
|
||||
CHECK(d.root()[0]["m299"].get<int>() == 299);
|
||||
CHECK(d.root()[1]["x"]["m150"].get<int>() == 150);
|
||||
CHECK(d.root()[2]["m0"].get<int>() == 0);
|
||||
const std::size_t with_index = d.memory_usage();
|
||||
d.read(std::string("{\"small\": 1}"));
|
||||
CHECK(d.root()["small"].get<int>() == 1);
|
||||
d.read(text);
|
||||
CHECK(d.root()[2]["m7"].get<int>() == 7);
|
||||
CHECK(d.memory_usage() >= with_index / 2);
|
||||
}
|
||||
}
|
||||
@@ -17,6 +17,7 @@
|
||||
#endif
|
||||
using nlohmann::json;
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <fstream>
|
||||
#include <limits>
|
||||
@@ -144,6 +145,32 @@ void check_same(const std::string& text)
|
||||
}
|
||||
}
|
||||
|
||||
// a string value and a key must be accepted or rejected as json::parse does,
|
||||
// and give its value (cheaper than check_same: the options do not matter)
|
||||
void check_string(const std::string& content)
|
||||
{
|
||||
for (const std::string& text :
|
||||
{
|
||||
"[\"" + content + "\"]", "{\"" + content + "\":1}"
|
||||
})
|
||||
{
|
||||
const bool accepted = json::accept(text);
|
||||
for (const bool sentinel :
|
||||
{
|
||||
true, false
|
||||
})
|
||||
{
|
||||
const built b = build(text, false, false, sentinel);
|
||||
if (b.ok != accepted || (accepted && value_of(b) != json::parse(text)))
|
||||
{
|
||||
CAPTURE(text)
|
||||
CHECK(b.ok == accepted);
|
||||
CHECK((b.ok && accepted ? value_of(b) == json::parse(text) : true));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// a small deterministic generator of documents
|
||||
struct generator
|
||||
{
|
||||
@@ -404,6 +431,84 @@ TEST_CASE("json_view builder")
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view builder: strings across vector blocks")
|
||||
{
|
||||
// Strings are scanned 8 or 16 bytes at a time (NEON, SSE2, or SWAR) and
|
||||
// non-ASCII text with the vector UTF-8 check (NEON, SSSE3) or one sequence
|
||||
// at a time. Sequences are placed so that they start at every offset
|
||||
// around the block boundaries of keys (16, 32) and values (8, 24), with
|
||||
// text of several lengths after them.
|
||||
const std::array<std::size_t, 16> prefixes = {{0, 6, 7, 8, 13, 14, 15, 16, 21, 22, 23, 24, 29, 30, 31, 32}};
|
||||
const std::array<std::size_t, 3> suffixes = {{0, 3, 17}};
|
||||
const auto around = [&](const std::string & seq, std::size_t prefix, std::size_t suffix)
|
||||
{
|
||||
return std::string(prefix, 'a') + seq + std::string(suffix, 'b');
|
||||
};
|
||||
|
||||
SECTION("every two-byte sequence")
|
||||
{
|
||||
for (unsigned lead = 0x80; lead <= 0xFF; ++lead)
|
||||
{
|
||||
for (unsigned second = 0; second <= 0xFF; ++second)
|
||||
{
|
||||
if (second == '"' || second == '\\')
|
||||
{
|
||||
continue;
|
||||
}
|
||||
const std::string seq = {static_cast<char>(lead), static_cast<char>(second)};
|
||||
check_string(around(seq, prefixes[(lead + second) % 16], suffixes[second % 3]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("three- and four-byte sequences")
|
||||
{
|
||||
const std::array<unsigned, 8> conts = {{0x7F, 0x80, 0x8F, 0x90, 0x9F, 0xA0, 0xBF, 0xC0}};
|
||||
for (unsigned lead = 0xE0; lead <= 0xF7; ++lead)
|
||||
{
|
||||
for (const unsigned b2 : conts)
|
||||
{
|
||||
for (const unsigned b3 : conts)
|
||||
{
|
||||
for (const std::size_t prefix : prefixes)
|
||||
{
|
||||
std::string seq = {static_cast<char>(lead), static_cast<char>(b2), static_cast<char>(b3)};
|
||||
if (lead >= 0xF0)
|
||||
{
|
||||
seq += static_cast<char>(prefix % 2 == 0 ? 0x80 : 0xBF);
|
||||
}
|
||||
check_string(around(seq, prefix, suffixes[prefix % 3]));
|
||||
// cut short before the end of the string
|
||||
check_string(around(seq.substr(0, seq.size() - 1), prefix, suffixes[prefix % 3]));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("long runs of text with one damaged byte")
|
||||
{
|
||||
const std::array<const char*, 7> chars = {{"a", "\xc3\xa9", "\xe3\x81\x82", "\xf0\x9f\x98\x80", "\xed\x9f\xbf", "\xef\xbf\xbf", "\xf4\x8f\xbf\xbf"}};
|
||||
const std::array<char, 11> damage = {{'\x80', '\xbf', '\xc0', '\xc1', '\xe0', '\xed', '\xf5', '\xff', '\x1f', '"', '\\'}};
|
||||
std::mt19937 rng(5295); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed): reproducible
|
||||
for (int i = 0; i < 4000; ++i)
|
||||
{
|
||||
std::string text;
|
||||
const auto n = rng() % 60;
|
||||
for (unsigned k = 0; k < n; ++k)
|
||||
{
|
||||
text += chars[rng() % chars.size()];
|
||||
}
|
||||
check_string(text);
|
||||
if (!text.empty())
|
||||
{
|
||||
text[rng() % text.size()] = damage[rng() % damage.size()];
|
||||
check_string(text);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view node integer bits")
|
||||
{
|
||||
using nlohmann::detail::view::integer_bits;
|
||||
|
||||
Reference in new issue
Block a user