mirror of
https://github.com/nlohmann/json.git
synced 2026-10-10 08:27:13 +00:00
Bound the probe length of the large-object hash index
The key hash is not seeded, so keys chosen to collide made building the table quadratic (20,000 colliding keys took 470 ms to parse). A key may now sit at most 64 slots from its home slot; if a key would sit further away, the table is dropped and the object is searched linearly. Lookups stop after the same distance. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
2 files changed
+95
-1
No files matched your search
@@ -24,6 +24,12 @@
|
||||
// number of the table (1-based) in `extra`. A slot holds the offset of a key
|
||||
// node from its object node (0: empty). Of duplicate keys, the first is kept,
|
||||
// as for the linear search.
|
||||
//
|
||||
// The hash is not seeded, so keys chosen to collide could make the build
|
||||
// quadratic. A key therefore sits at most index_max_displacement slots away
|
||||
// from its home slot; if a key would sit further away, the table is dropped
|
||||
// and the object is searched linearly (like a small one). For the same
|
||||
// reason, a lookup visits at most index_max_displacement + 1 slots.
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -31,6 +37,11 @@ namespace detail
|
||||
namespace view
|
||||
{
|
||||
|
||||
/// the farthest a key may sit from its home slot (a table with at most half of
|
||||
/// its slots in use gives random keys a distance of about 50 for millions of
|
||||
/// members; and every member costs at most this many steps while building)
|
||||
constexpr std::size_t index_max_displacement = 64;
|
||||
|
||||
/// hash of a key: its bytes, eight at a time, in a fixed byte order
|
||||
inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept
|
||||
{
|
||||
@@ -68,12 +79,14 @@ inline void build_object_index(document_data& d, node* obj)
|
||||
d.index_slots.resize(start + cap, 0);
|
||||
std::uint32_t* const slots = d.index_slots.data() + start;
|
||||
const std::size_t mask = cap - 1;
|
||||
bool degenerate = false;
|
||||
for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1))
|
||||
{
|
||||
const char* const key = d.str(*k);
|
||||
const std::uint64_t hash = key_hash(key, k->len); // (a cast of the call would be useless where std::uint64_t is std::size_t)
|
||||
std::size_t i = static_cast<std::size_t>(hash) & mask;
|
||||
bool duplicate = false;
|
||||
std::size_t distance = 0;
|
||||
while (slots[i] != 0)
|
||||
{
|
||||
const node* const other = obj + slots[i];
|
||||
@@ -82,13 +95,27 @@ inline void build_object_index(document_data& d, node* obj)
|
||||
duplicate = true; // keep the first
|
||||
break;
|
||||
}
|
||||
if (++distance > index_max_displacement)
|
||||
{
|
||||
degenerate = true; // too many keys share a home region
|
||||
break;
|
||||
}
|
||||
i = (i + 1) & mask;
|
||||
}
|
||||
if (degenerate)
|
||||
{
|
||||
break;
|
||||
}
|
||||
if (!duplicate)
|
||||
{
|
||||
slots[i] = static_cast<std::uint32_t>(k - obj);
|
||||
}
|
||||
}
|
||||
if (degenerate)
|
||||
{
|
||||
d.index_slots.resize(start); // no table: the object is searched linearly
|
||||
return;
|
||||
}
|
||||
d.indexes.push_back(document_data::object_index{start, static_cast<std::uint32_t>(mask)});
|
||||
obj->extra = static_cast<std::uint16_t>(d.indexes.size());
|
||||
}
|
||||
@@ -110,7 +137,7 @@ inline const node* find_indexed(const document_data& d, const node* obj, const c
|
||||
const std::uint32_t* const slots = d.index_slots.data() + ix.start;
|
||||
const std::uint64_t hash = key_hash(key, n); // (a cast of the call would be useless where std::uint64_t is std::size_t)
|
||||
std::size_t i = static_cast<std::size_t>(hash) & ix.mask;
|
||||
for (;;)
|
||||
for (std::size_t distance = 0; distance <= index_max_displacement; ++distance)
|
||||
{
|
||||
const std::uint32_t s = slots[i];
|
||||
if (s == 0)
|
||||
@@ -124,6 +151,7 @@ inline const node* find_indexed(const document_data& d, const node* obj, const c
|
||||
}
|
||||
i = (i + 1) & ix.mask;
|
||||
}
|
||||
return nullptr; // (no key sits further from its home slot)
|
||||
}
|
||||
|
||||
} // namespace view
|
||||
|
||||
@@ -1315,6 +1315,72 @@ TEST_CASE("json_view large objects")
|
||||
CHECK(v.materialize() == j);
|
||||
}
|
||||
|
||||
SECTION("colliding keys")
|
||||
{
|
||||
// the hash is not seeded, so keys that all land in one place must not
|
||||
// make the table build quadratic: such an object gets no table and is
|
||||
// searched linearly
|
||||
constexpr std::size_t members = 300;
|
||||
constexpr std::size_t slots = 1024; // the table size for 300 members: the next power of two >= 600
|
||||
std::vector<std::string> colliding;
|
||||
std::vector<std::string> spread;
|
||||
for (std::uint64_t counter = 0; colliding.size() < members || spread.size() < members; ++counter)
|
||||
{
|
||||
std::string key(8, 'a');
|
||||
for (std::uint64_t x = counter, i = 0; i < 8; ++i, x /= 26)
|
||||
{
|
||||
key[i] = static_cast<char>('a' + (x % 26));
|
||||
}
|
||||
const bool lands_in_slot_zero = (nlohmann::detail::view::key_hash(key.data(), key.size()) & (slots - 1)) == 0;
|
||||
if (lands_in_slot_zero && colliding.size() < members)
|
||||
{
|
||||
colliding.push_back(key);
|
||||
}
|
||||
else if (!lands_in_slot_zero && spread.size() < members)
|
||||
{
|
||||
spread.push_back(key);
|
||||
}
|
||||
}
|
||||
|
||||
const auto make_text = [](const std::vector<std::string>& keys)
|
||||
{
|
||||
std::string text = "{";
|
||||
for (std::size_t i = 0; i < keys.size(); ++i)
|
||||
{
|
||||
text += (i != 0 ? ",\"" : "\"") + keys[i] + "\":" + std::to_string(i % 10);
|
||||
}
|
||||
return text + "}";
|
||||
};
|
||||
const std::string colliding_text = make_text(colliding);
|
||||
const std::string spread_text = make_text(spread);
|
||||
|
||||
json_document with_collisions = json_document::parse(colliding_text);
|
||||
json_document without_collisions = json_document::parse(spread_text);
|
||||
for (const auto* pair :
|
||||
{
|
||||
&colliding, &spread
|
||||
})
|
||||
{
|
||||
const json_view v = (pair == &colliding ? with_collisions : without_collisions).root();
|
||||
for (std::size_t i = 0; i < members; ++i)
|
||||
{
|
||||
CAPTURE(i)
|
||||
CHECK(v[(*pair)[i]].get<std::size_t>() == i % 10);
|
||||
CHECK(v.at((*pair)[i]).get<std::size_t>() == i % 10);
|
||||
CHECK(v.find((*pair)[i]).key() == (*pair)[i]);
|
||||
CHECK(!v.contains((*pair)[i] + "x"));
|
||||
}
|
||||
CHECK(!v.contains("missing"));
|
||||
}
|
||||
CHECK(with_collisions.root() == json::parse(colliding_text));
|
||||
|
||||
// only the object with the spread keys got a table (both texts have the
|
||||
// same length, so the tables are the only difference)
|
||||
with_collisions.shrink_to_fit();
|
||||
without_collisions.shrink_to_fit();
|
||||
CHECK(without_collisions.memory_usage() >= with_collisions.memory_usage() + (slots * sizeof(std::uint32_t)));
|
||||
}
|
||||
|
||||
SECTION("nested, reused, and in arrays")
|
||||
{
|
||||
std::string inner = "{";
|
||||
|
||||
Reference in new issue
Block a user