mirror of
https://github.com/nlohmann/json.git
synced 2026-10-10 08:27:13 +00:00
Keep the last of duplicate keys in the object hash index
Lookups in objects with 128 members or more now return the last member of a repeated key, like the linear search of smaller objects and like materialize() and basic_json::parse(). build_object_index let the first occurrence win, so the same text gave different results depending on the size of the object. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
6 files changed
+115
-16
No files matched your search
@@ -114,8 +114,9 @@ document.
|
||||
the *last* member with that key. This is the member [`materialize()`](materialize.md) (and
|
||||
[`BasicJsonType::parse()`](../basic_json/parse.md)) keeps, so a lookup in the view and in the materialized value
|
||||
agree. [`begin()`](begin.md)/[`end()`](end.md) and [`items()`](items.md) iterate over *all* members, including
|
||||
duplicates, in document order. A lookup scans all members for this: it cannot stop at the first match. See the
|
||||
example below and [`size()`](size.md#notes).
|
||||
duplicates, in document order. A lookup in an object without a hash index scans all members for this: it cannot
|
||||
stop at the first match. The hash index of a larger object (128 members or more) leads to the last member of a key
|
||||
as well. See the example below and [`size()`](size.md#notes).
|
||||
|
||||
!!! info "JSON pointer resolution"
|
||||
|
||||
|
||||
@@ -142,7 +142,8 @@ document: `#!cpp auto v = json_document::parse(text).root();` does not compile.
|
||||
[`contains`](../api/basic_json_view/contains.md), and [`count`](../api/basic_json_view/count.md) resolve to the
|
||||
*last* occurrence -- the one `basic_json::parse()` (and so
|
||||
[`materialize()`](../api/basic_json_view/materialize.md)) keeps for a repeated key -- which makes a lookup scan all
|
||||
members instead of stopping at a match.
|
||||
members instead of stopping at a match (objects with 128 members or more get a hash index that leads to the last
|
||||
occurrence directly).
|
||||
See the [Notes on duplicate keys](../api/basic_json_view/operator%5B%5D.md#notes) of `operator[]`.
|
||||
- **No [`JSON_DIAGNOSTICS`](../api/macros/json_diagnostics.md) path.** Exceptions thrown by `basic_json_view`'s own
|
||||
element access and lookup functions never carry the JSON Pointer path `JSON_DIAGNOSTICS` would otherwise add: the
|
||||
|
||||
@@ -215,8 +215,8 @@ packet-beta
|
||||
- **Large objects** (128 members or more) get a hash index after parsing
|
||||
([`detail/view/object_index.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/view/object_index.hpp)):
|
||||
an open-addressing table whose slots hold the distance from the object's node to a key's node, so that a lookup does
|
||||
not compare every key. The object's `extra` holds the number of its table. Only 65,535 tables fit into `extra`;
|
||||
objects beyond them are searched linearly.
|
||||
not compare every key. Of duplicate keys the table leads to the last, as a linear search does. The object's `extra`
|
||||
holds the number of its table. Only 65,535 tables fit into `extra`; objects beyond them are searched linearly.
|
||||
|
||||
For example, `#!json {"a": [1, 2.5]}` becomes five nodes. Each node's elements follow it, and `next` leads from an
|
||||
array or object past its subtree:
|
||||
|
||||
@@ -22,8 +22,8 @@
|
||||
// large objects). An object with document_data::index_min_members members or
|
||||
// more gets an open-addressing table after parsing; its node stores the
|
||||
// number of the table (1-based) in `extra`. A slot holds the offset of a key
|
||||
// node from its object node (0: empty). Of duplicate keys, the first is kept,
|
||||
// as for the linear search.
|
||||
// node from its object node (0: empty). Of duplicate keys, the last is kept,
|
||||
// as for the linear search, and as basic_json::parse() does.
|
||||
//
|
||||
// The hash is not seeded, so keys chosen to collide could make the build
|
||||
// quadratic. A key therefore sits at most index_max_displacement slots away
|
||||
@@ -92,7 +92,8 @@ inline void build_object_index(document_data& d, node* obj)
|
||||
const node* const other = obj + slots[i];
|
||||
if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0))
|
||||
{
|
||||
duplicate = true; // keep the first
|
||||
duplicate = true; // keep the last: the key's slot now leads to this member
|
||||
slots[i] = static_cast<std::uint32_t>(k - obj);
|
||||
break;
|
||||
}
|
||||
if (++distance > index_max_displacement)
|
||||
@@ -129,7 +130,7 @@ inline void build_object_indexes(document_data& d)
|
||||
}
|
||||
}
|
||||
|
||||
/// the key node of the first member with this key of an indexed object, or
|
||||
/// the key node of the last member with this key of an indexed object, or
|
||||
/// nullptr
|
||||
inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept
|
||||
{
|
||||
|
||||
@@ -2879,8 +2879,8 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
// large objects). An object with document_data::index_min_members members or
|
||||
// more gets an open-addressing table after parsing; its node stores the
|
||||
// number of the table (1-based) in `extra`. A slot holds the offset of a key
|
||||
// node from its object node (0: empty). Of duplicate keys, the first is kept,
|
||||
// as for the linear search.
|
||||
// node from its object node (0: empty). Of duplicate keys, the last is kept,
|
||||
// as for the linear search, and as basic_json::parse() does.
|
||||
//
|
||||
// The hash is not seeded, so keys chosen to collide could make the build
|
||||
// quadratic. A key therefore sits at most index_max_displacement slots away
|
||||
@@ -2949,7 +2949,8 @@ inline void build_object_index(document_data& d, node* obj)
|
||||
const node* const other = obj + slots[i];
|
||||
if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0))
|
||||
{
|
||||
duplicate = true; // keep the first
|
||||
duplicate = true; // keep the last: the key's slot now leads to this member
|
||||
slots[i] = static_cast<std::uint32_t>(k - obj);
|
||||
break;
|
||||
}
|
||||
if (++distance > index_max_displacement)
|
||||
@@ -2986,7 +2987,7 @@ inline void build_object_indexes(document_data& d)
|
||||
}
|
||||
}
|
||||
|
||||
/// the key node of the first member with this key of an indexed object, or
|
||||
/// the key node of the last member with this key of an indexed object, or
|
||||
/// nullptr
|
||||
inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept
|
||||
{
|
||||
|
||||
@@ -1632,20 +1632,115 @@ TEST_CASE("json_view large objects")
|
||||
for (std::size_t i = 0; i < members; ++i)
|
||||
{
|
||||
const std::string key = std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\n" : "");
|
||||
CHECK(v[key].get<std::size_t>() == i);
|
||||
CHECK(v.contains(key));
|
||||
CHECK(v.find(key).key() == key);
|
||||
CHECK(v.at(key).get<std::size_t>() == i);
|
||||
CHECK(!v.contains(key + "x"));
|
||||
if (key == "k1")
|
||||
{
|
||||
continue; // repeated below: the last member wins
|
||||
}
|
||||
CHECK(v[key].get<std::size_t>() == i);
|
||||
CHECK(v.at(key).get<std::size_t>() == i);
|
||||
}
|
||||
CHECK(v[""].get_string() == "empty key");
|
||||
CHECK(v["k1"].get<int>() == 1); // the first of duplicate keys, as for small objects
|
||||
CHECK(v["k1"].get_string() == "a duplicate of an earlier key"); // the last of duplicate keys, as for small objects
|
||||
CHECK(v.at("k1").get_string() == "a duplicate of an earlier key");
|
||||
CHECK(!v.contains("missing"));
|
||||
CHECK_THROWS_WITH_AS(v.at("missing"), "[json.exception.out_of_range.403] key 'missing' not found", json::out_of_range&);
|
||||
CHECK(v == j);
|
||||
CHECK(v.materialize() == j);
|
||||
}
|
||||
|
||||
SECTION("duplicate keys: the last member wins, with and without a table")
|
||||
{
|
||||
// an object of `total` members: the keys "k0".."k<n-1>" in order, then
|
||||
// three keys repeated twice more (one copy in the middle, one at the
|
||||
// end), and two keys repeated once; the value of a member is its
|
||||
// position, so that the last member of a key can be told apart
|
||||
struct member
|
||||
{
|
||||
std::string key;
|
||||
std::size_t position;
|
||||
};
|
||||
const auto make_members = [](std::size_t total)
|
||||
{
|
||||
std::vector<std::string> keys;
|
||||
for (std::size_t i = 0; i + 8 < total; ++i)
|
||||
{
|
||||
keys.push_back("k" + std::to_string(i));
|
||||
}
|
||||
const std::size_t n = keys.size();
|
||||
const std::array<std::size_t, 3> triple = {{3, 17, n - 1}};
|
||||
const std::array<std::size_t, 2> twice = {{5, n / 2}};
|
||||
std::vector<std::string> ordered = keys;
|
||||
for (const std::size_t i : triple)
|
||||
{
|
||||
ordered.insert(ordered.begin() + static_cast<std::ptrdiff_t>(ordered.size() / 2), keys[i]);
|
||||
}
|
||||
for (const std::size_t i : twice)
|
||||
{
|
||||
ordered.insert(ordered.begin() + static_cast<std::ptrdiff_t>(ordered.size() / 3), keys[i]);
|
||||
}
|
||||
for (const std::size_t i : triple)
|
||||
{
|
||||
ordered.push_back(keys[i]);
|
||||
}
|
||||
std::vector<member> result;
|
||||
for (std::size_t i = 0; i < ordered.size(); ++i)
|
||||
{
|
||||
result.push_back({ordered[i], i});
|
||||
}
|
||||
return result;
|
||||
};
|
||||
|
||||
// 100 and 127 members: no table; 128 members and more: a table
|
||||
for (const std::size_t total :
|
||||
{
|
||||
100u, 127u, 128u, 200u, 5000u
|
||||
})
|
||||
{
|
||||
CAPTURE(total)
|
||||
const std::vector<member> members = make_members(total);
|
||||
REQUIRE(members.size() >= total);
|
||||
REQUIRE(members.size() >= 100);
|
||||
std::string text = "{";
|
||||
std::map<std::string, std::size_t> last;
|
||||
for (const member& m : members)
|
||||
{
|
||||
text += (text.size() > 1 ? ",\"" : "\"") + m.key + "\":" + std::to_string(m.position);
|
||||
last[m.key] = m.position;
|
||||
}
|
||||
text += '}';
|
||||
REQUIRE(last.size() < members.size());
|
||||
|
||||
const json_document d = json_document::parse(text);
|
||||
const json_view v = d.root();
|
||||
const json j = json::parse(text);
|
||||
CHECK(v.size() == members.size()); // every occurrence is visited
|
||||
for (const auto& entry : last)
|
||||
{
|
||||
CAPTURE(entry.first)
|
||||
const std::size_t expected = entry.second;
|
||||
CHECK(v[entry.first].get<std::size_t>() == expected);
|
||||
CHECK(v.at(entry.first).get<std::size_t>() == expected);
|
||||
CHECK(v.find(entry.first).value().get<std::size_t>() == expected);
|
||||
CHECK(v.value(entry.first, std::size_t{0}) == expected);
|
||||
CHECK(v.contains(entry.first));
|
||||
CHECK(v.count(entry.first) == 1);
|
||||
const std::string pointer = "/" + entry.first;
|
||||
CHECK(v[json::json_pointer(pointer)].get<std::size_t>() == expected);
|
||||
CHECK(v.at(json::json_pointer(pointer)).get<std::size_t>() == expected);
|
||||
CHECK(v.value(json::json_pointer(pointer), std::size_t{0}) == expected);
|
||||
CHECK(v.contains(json::json_pointer(pointer)));
|
||||
CHECK(j[entry.first].get<std::size_t>() == expected); // as materialize() and parse() keep it
|
||||
}
|
||||
CHECK(!v.contains("k"));
|
||||
CHECK(v["missing"].is_discarded());
|
||||
CHECK(v.materialize() == j);
|
||||
CHECK(v == j);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("colliding keys")
|
||||
{
|
||||
// the hash is not seeded, so keys that all land in one place must not
|
||||
|
||||
Reference in new issue
Block a user