Bound the probe length of the large-object hash index

The key hash is not seeded, so keys chosen to collide made building the table quadratic (20,000 colliding keys took 470 ms to parse). A key may now sit at most 64 slots from its home slot; if a key would sit further away, the table is dropped and the object is searched linearly. Lookups stop after the same distance.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann committed 2026-10-09 16:11:25 +02:00
1 parent 2f400521e1
commit 331615d3de
2 files changed
+95 -1

No files matched your search

+29 -1
View File
@@ -24,6 +24,12 @@
// number of the table (1-based) in `extra`. A slot holds the offset of a key
// node from its object node (0: empty). Of duplicate keys, the first is kept,
// as for the linear search.
//
// The hash is not seeded, so keys chosen to collide could make the build
// quadratic. A key therefore sits at most index_max_displacement slots away
// from its home slot; if a key would sit further away, the table is dropped
// and the object is searched linearly (like a small one). For the same
// reason, a lookup visits at most index_max_displacement + 1 slots.
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -31,6 +37,11 @@ namespace detail
namespace view
{
/// the farthest a key may sit from its home slot (a table with at most half of
/// its slots in use gives random keys a distance of about 50 for millions of
/// members; and every member costs at most this many steps while building)
constexpr std::size_t index_max_displacement = 64;
/// hash of a key: its bytes, eight at a time, in a fixed byte order
inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept
{
@@ -68,12 +79,14 @@ inline void build_object_index(document_data& d, node* obj)
d.index_slots.resize(start + cap, 0);
std::uint32_t* const slots = d.index_slots.data() + start;
const std::size_t mask = cap - 1;
bool degenerate = false;
for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1))
{
const char* const key = d.str(*k);
const std::uint64_t hash = key_hash(key, k->len); // (a cast of the call would be useless where std::uint64_t is std::size_t)
std::size_t i = static_cast<std::size_t>(hash) & mask;
bool duplicate = false;
std::size_t distance = 0;
while (slots[i] != 0)
{
const node* const other = obj + slots[i];
@@ -82,13 +95,27 @@ inline void build_object_index(document_data& d, node* obj)
duplicate = true; // keep the first
break;
}
if (++distance > index_max_displacement)
{
degenerate = true; // too many keys share a home region
break;
}
i = (i + 1) & mask;
}
if (degenerate)
{
break;
}
if (!duplicate)
{
slots[i] = static_cast<std::uint32_t>(k - obj);
}
}
if (degenerate)
{
d.index_slots.resize(start); // no table: the object is searched linearly
return;
}
d.indexes.push_back(document_data::object_index{start, static_cast<std::uint32_t>(mask)});
obj->extra = static_cast<std::uint16_t>(d.indexes.size());
}
@@ -110,7 +137,7 @@ inline const node* find_indexed(const document_data& d, const node* obj, const c
const std::uint32_t* const slots = d.index_slots.data() + ix.start;
const std::uint64_t hash = key_hash(key, n); // (a cast of the call would be useless where std::uint64_t is std::size_t)
std::size_t i = static_cast<std::size_t>(hash) & ix.mask;
for (;;)
for (std::size_t distance = 0; distance <= index_max_displacement; ++distance)
{
const std::uint32_t s = slots[i];
if (s == 0)
@@ -124,6 +151,7 @@ inline const node* find_indexed(const document_data& d, const node* obj, const c
}
i = (i + 1) & ix.mask;
}
return nullptr; // (no key sits further from its home slot)
}
} // namespace view
+66
View File
@@ -1315,6 +1315,72 @@ TEST_CASE("json_view large objects")
CHECK(v.materialize() == j);
}
SECTION("colliding keys")
{
// the hash is not seeded, so keys that all land in one place must not
// make the table build quadratic: such an object gets no table and is
// searched linearly
constexpr std::size_t members = 300;
constexpr std::size_t slots = 1024; // the table size for 300 members: the next power of two >= 600
std::vector<std::string> colliding;
std::vector<std::string> spread;
for (std::uint64_t counter = 0; colliding.size() < members || spread.size() < members; ++counter)
{
std::string key(8, 'a');
for (std::uint64_t x = counter, i = 0; i < 8; ++i, x /= 26)
{
key[i] = static_cast<char>('a' + (x % 26));
}
const bool lands_in_slot_zero = (nlohmann::detail::view::key_hash(key.data(), key.size()) & (slots - 1)) == 0;
if (lands_in_slot_zero && colliding.size() < members)
{
colliding.push_back(key);
}
else if (!lands_in_slot_zero && spread.size() < members)
{
spread.push_back(key);
}
}
const auto make_text = [](const std::vector<std::string>& keys)
{
std::string text = "{";
for (std::size_t i = 0; i < keys.size(); ++i)
{
text += (i != 0 ? ",\"" : "\"") + keys[i] + "\":" + std::to_string(i % 10);
}
return text + "}";
};
const std::string colliding_text = make_text(colliding);
const std::string spread_text = make_text(spread);
json_document with_collisions = json_document::parse(colliding_text);
json_document without_collisions = json_document::parse(spread_text);
for (const auto* pair :
{
&colliding, &spread
})
{
const json_view v = (pair == &colliding ? with_collisions : without_collisions).root();
for (std::size_t i = 0; i < members; ++i)
{
CAPTURE(i)
CHECK(v[(*pair)[i]].get<std::size_t>() == i % 10);
CHECK(v.at((*pair)[i]).get<std::size_t>() == i % 10);
CHECK(v.find((*pair)[i]).key() == (*pair)[i]);
CHECK(!v.contains((*pair)[i] + "x"));
}
CHECK(!v.contains("missing"));
}
CHECK(with_collisions.root() == json::parse(colliding_text));
// only the object with the spread keys got a table (both texts have the
// same length, so the tables are the only difference)
with_collisions.shrink_to_fit();
without_collisions.shrink_to_fit();
CHECK(without_collisions.memory_usage() >= with_collisions.memory_usage() + (slots * sizeof(std::uint32_t)));
}
SECTION("nested, reused, and in arrays")
{
std::string inner = "{";