diff --git a/docs/mkdocs/docs/api/basic_json_view/operator[].md b/docs/mkdocs/docs/api/basic_json_view/operator[].md index f97dfea89..4337d95eb 100644 --- a/docs/mkdocs/docs/api/basic_json_view/operator[].md +++ b/docs/mkdocs/docs/api/basic_json_view/operator[].md @@ -114,8 +114,9 @@ document. the *last* member with that key. This is the member [`materialize()`](materialize.md) (and [`BasicJsonType::parse()`](../basic_json/parse.md)) keeps, so a lookup in the view and in the materialized value agree. [`begin()`](begin.md)/[`end()`](end.md) and [`items()`](items.md) iterate over *all* members, including - duplicates, in document order. A lookup scans all members for this: it cannot stop at the first match. See the - example below and [`size()`](size.md#notes). + duplicates, in document order. A lookup in an object without a hash index scans all members for this: it cannot + stop at the first match. The hash index of a larger object (128 members or more) leads to the last member of a key + as well. See the example below and [`size()`](size.md#notes). !!! info "JSON pointer resolution" diff --git a/docs/mkdocs/docs/features/json_view.md b/docs/mkdocs/docs/features/json_view.md index 2ea589724..732846de4 100644 --- a/docs/mkdocs/docs/features/json_view.md +++ b/docs/mkdocs/docs/features/json_view.md @@ -142,7 +142,8 @@ document: `#!cpp auto v = json_document::parse(text).root();` does not compile. [`contains`](../api/basic_json_view/contains.md), and [`count`](../api/basic_json_view/count.md) resolve to the *last* occurrence -- the one `basic_json::parse()` (and so [`materialize()`](../api/basic_json_view/materialize.md)) keeps for a repeated key -- which makes a lookup scan all - members instead of stopping at a match. + members instead of stopping at a match (objects with 128 members or more get a hash index that leads to the last + occurrence directly). See the [Notes on duplicate keys](../api/basic_json_view/operator%5B%5D.md#notes) of `operator[]`. - **No [`JSON_DIAGNOSTICS`](../api/macros/json_diagnostics.md) path.** Exceptions thrown by `basic_json_view`'s own element access and lookup functions never carry the JSON Pointer path `JSON_DIAGNOSTICS` would otherwise add: the diff --git a/docs/mkdocs/docs/home/architecture.md b/docs/mkdocs/docs/home/architecture.md index e851251e4..981f62d35 100644 --- a/docs/mkdocs/docs/home/architecture.md +++ b/docs/mkdocs/docs/home/architecture.md @@ -215,8 +215,8 @@ packet-beta - **Large objects** (128 members or more) get a hash index after parsing ([`detail/view/object_index.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/view/object_index.hpp)): an open-addressing table whose slots hold the distance from the object's node to a key's node, so that a lookup does - not compare every key. The object's `extra` holds the number of its table. Only 65,535 tables fit into `extra`; - objects beyond them are searched linearly. + not compare every key. Of duplicate keys the table leads to the last, as a linear search does. The object's `extra` + holds the number of its table. Only 65,535 tables fit into `extra`; objects beyond them are searched linearly. For example, `#!json {"a": [1, 2.5]}` becomes five nodes. Each node's elements follow it, and `next` leads from an array or object past its subtree: diff --git a/include/nlohmann/detail/view/object_index.hpp b/include/nlohmann/detail/view/object_index.hpp index 8888f2345..721b4d62b 100644 --- a/include/nlohmann/detail/view/object_index.hpp +++ b/include/nlohmann/detail/view/object_index.hpp @@ -22,8 +22,8 @@ // large objects). An object with document_data::index_min_members members or // more gets an open-addressing table after parsing; its node stores the // number of the table (1-based) in `extra`. A slot holds the offset of a key -// node from its object node (0: empty). Of duplicate keys, the first is kept, -// as for the linear search. +// node from its object node (0: empty). Of duplicate keys, the last is kept, +// as for the linear search, and as basic_json::parse() does. // // The hash is not seeded, so keys chosen to collide could make the build // quadratic. A key therefore sits at most index_max_displacement slots away @@ -92,7 +92,8 @@ inline void build_object_index(document_data& d, node* obj) const node* const other = obj + slots[i]; if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0)) { - duplicate = true; // keep the first + duplicate = true; // keep the last: the key's slot now leads to this member + slots[i] = static_cast(k - obj); break; } if (++distance > index_max_displacement) @@ -129,7 +130,7 @@ inline void build_object_indexes(document_data& d) } } -/// the key node of the first member with this key of an indexed object, or +/// the key node of the last member with this key of an indexed object, or /// nullptr inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept { diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index c98297504..78d0820a8 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -2879,8 +2879,8 @@ NLOHMANN_JSON_NAMESPACE_END // large objects). An object with document_data::index_min_members members or // more gets an open-addressing table after parsing; its node stores the // number of the table (1-based) in `extra`. A slot holds the offset of a key -// node from its object node (0: empty). Of duplicate keys, the first is kept, -// as for the linear search. +// node from its object node (0: empty). Of duplicate keys, the last is kept, +// as for the linear search, and as basic_json::parse() does. // // The hash is not seeded, so keys chosen to collide could make the build // quadratic. A key therefore sits at most index_max_displacement slots away @@ -2949,7 +2949,8 @@ inline void build_object_index(document_data& d, node* obj) const node* const other = obj + slots[i]; if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0)) { - duplicate = true; // keep the first + duplicate = true; // keep the last: the key's slot now leads to this member + slots[i] = static_cast(k - obj); break; } if (++distance > index_max_displacement) @@ -2986,7 +2987,7 @@ inline void build_object_indexes(document_data& d) } } -/// the key node of the first member with this key of an indexed object, or +/// the key node of the last member with this key of an indexed object, or /// nullptr inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept { diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 0647821a4..f381c6889 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -1632,20 +1632,115 @@ TEST_CASE("json_view large objects") for (std::size_t i = 0; i < members; ++i) { const std::string key = std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\n" : ""); - CHECK(v[key].get() == i); CHECK(v.contains(key)); CHECK(v.find(key).key() == key); - CHECK(v.at(key).get() == i); CHECK(!v.contains(key + "x")); + if (key == "k1") + { + continue; // repeated below: the last member wins + } + CHECK(v[key].get() == i); + CHECK(v.at(key).get() == i); } CHECK(v[""].get_string() == "empty key"); - CHECK(v["k1"].get() == 1); // the first of duplicate keys, as for small objects + CHECK(v["k1"].get_string() == "a duplicate of an earlier key"); // the last of duplicate keys, as for small objects + CHECK(v.at("k1").get_string() == "a duplicate of an earlier key"); CHECK(!v.contains("missing")); CHECK_THROWS_WITH_AS(v.at("missing"), "[json.exception.out_of_range.403] key 'missing' not found", json::out_of_range&); CHECK(v == j); CHECK(v.materialize() == j); } + SECTION("duplicate keys: the last member wins, with and without a table") + { + // an object of `total` members: the keys "k0".."k" in order, then + // three keys repeated twice more (one copy in the middle, one at the + // end), and two keys repeated once; the value of a member is its + // position, so that the last member of a key can be told apart + struct member + { + std::string key; + std::size_t position; + }; + const auto make_members = [](std::size_t total) + { + std::vector keys; + for (std::size_t i = 0; i + 8 < total; ++i) + { + keys.push_back("k" + std::to_string(i)); + } + const std::size_t n = keys.size(); + const std::array triple = {{3, 17, n - 1}}; + const std::array twice = {{5, n / 2}}; + std::vector ordered = keys; + for (const std::size_t i : triple) + { + ordered.insert(ordered.begin() + static_cast(ordered.size() / 2), keys[i]); + } + for (const std::size_t i : twice) + { + ordered.insert(ordered.begin() + static_cast(ordered.size() / 3), keys[i]); + } + for (const std::size_t i : triple) + { + ordered.push_back(keys[i]); + } + std::vector result; + for (std::size_t i = 0; i < ordered.size(); ++i) + { + result.push_back({ordered[i], i}); + } + return result; + }; + + // 100 and 127 members: no table; 128 members and more: a table + for (const std::size_t total : + { + 100u, 127u, 128u, 200u, 5000u + }) + { + CAPTURE(total) + const std::vector members = make_members(total); + REQUIRE(members.size() >= total); + REQUIRE(members.size() >= 100); + std::string text = "{"; + std::map last; + for (const member& m : members) + { + text += (text.size() > 1 ? ",\"" : "\"") + m.key + "\":" + std::to_string(m.position); + last[m.key] = m.position; + } + text += '}'; + REQUIRE(last.size() < members.size()); + + const json_document d = json_document::parse(text); + const json_view v = d.root(); + const json j = json::parse(text); + CHECK(v.size() == members.size()); // every occurrence is visited + for (const auto& entry : last) + { + CAPTURE(entry.first) + const std::size_t expected = entry.second; + CHECK(v[entry.first].get() == expected); + CHECK(v.at(entry.first).get() == expected); + CHECK(v.find(entry.first).value().get() == expected); + CHECK(v.value(entry.first, std::size_t{0}) == expected); + CHECK(v.contains(entry.first)); + CHECK(v.count(entry.first) == 1); + const std::string pointer = "/" + entry.first; + CHECK(v[json::json_pointer(pointer)].get() == expected); + CHECK(v.at(json::json_pointer(pointer)).get() == expected); + CHECK(v.value(json::json_pointer(pointer), std::size_t{0}) == expected); + CHECK(v.contains(json::json_pointer(pointer))); + CHECK(j[entry.first].get() == expected); // as materialize() and parse() keep it + } + CHECK(!v.contains("k")); + CHECK(v["missing"].is_discarded()); + CHECK(v.materialize() == j); + CHECK(v == j); + } + } + SECTION("colliding keys") { // the hash is not seeded, so keys that all land in one place must not