diff --git a/docs/mkdocs/docs/api/basic_json_view/operator[].md b/docs/mkdocs/docs/api/basic_json_view/operator[].md index 1e40b3093..ec1867c9f 100644 --- a/docs/mkdocs/docs/api/basic_json_view/operator[].md +++ b/docs/mkdocs/docs/api/basic_json_view/operator[].md @@ -116,6 +116,7 @@ document. member in order and so keep the *last* value for a repeated key, so `#!cpp v["a"]` and `#!cpp v.materialize()["a"]` can differ. To get the value `parse()` would give, use [`materialize()`](materialize.md) or iterate the members with [`items()`](items.md) and keep the last match. + The hash index of a larger object (128 members or more) leads to the first member of a key as well. [`begin()`](begin.md)/[`end()`](end.md) and [`items()`](items.md) iterate over *all* members, including duplicates, in document order. See [Duplicate keys](../../features/json_view.md#duplicate-keys) and [`size()`](size.md#notes). diff --git a/docs/mkdocs/docs/features/json_view.md b/docs/mkdocs/docs/features/json_view.md index 668eeb837..2b7d927c0 100644 --- a/docs/mkdocs/docs/features/json_view.md +++ b/docs/mkdocs/docs/features/json_view.md @@ -165,9 +165,10 @@ document: `#!cpp auto v = json_document::parse(text).root();` does not compile. **Why the first?** A lookup can stop as soon as it finds a match. Returning the last member would force every lookup to scan all members of the object, even when the key is found at the very first one: this made lookups in small -objects 1.6 to 3.4 times slower. Other zero-copy parsers that index the source text, such as yyjson and simdjson, also -return the first member. RFC 8259 only says that names within an object SHOULD be unique and that the behavior of a -receiver that sees duplicates is unpredictable, so neither choice is wrong. +objects 1.6 to 3.4 times slower. The hash index of objects with 128 members or more leads to the first member of a +key as well. Other zero-copy parsers that index the source text, such as yyjson and simdjson, also return the first +member. RFC 8259 only says that names within an object SHOULD be unique and that the behavior of a receiver that sees +duplicates is unpredictable, so neither choice is wrong. **What stays the same as `parse()`?** [`materialize()`](../api/basic_json_view/materialize.md) and [`get>()`](../api/basic_json_view/get.md) replay every member in order, so they keep the *last* value diff --git a/docs/mkdocs/docs/home/architecture.md b/docs/mkdocs/docs/home/architecture.md index 981f62d35..9246f7780 100644 --- a/docs/mkdocs/docs/home/architecture.md +++ b/docs/mkdocs/docs/home/architecture.md @@ -215,7 +215,7 @@ packet-beta - **Large objects** (128 members or more) get a hash index after parsing ([`detail/view/object_index.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/view/object_index.hpp)): an open-addressing table whose slots hold the distance from the object's node to a key's node, so that a lookup does - not compare every key. Of duplicate keys the table leads to the last, as a linear search does. The object's `extra` + not compare every key. Of duplicate keys the table leads to the first, as a linear search does. The object's `extra` holds the number of its table. Only 65,535 tables fit into `extra`; objects beyond them are searched linearly. For example, `#!json {"a": [1, 2.5]}` becomes five nodes. Each node's elements follow it, and `next` leads from an diff --git a/include/nlohmann/detail/view/object_index.hpp b/include/nlohmann/detail/view/object_index.hpp index 721b4d62b..8888f2345 100644 --- a/include/nlohmann/detail/view/object_index.hpp +++ b/include/nlohmann/detail/view/object_index.hpp @@ -22,8 +22,8 @@ // large objects). An object with document_data::index_min_members members or // more gets an open-addressing table after parsing; its node stores the // number of the table (1-based) in `extra`. A slot holds the offset of a key -// node from its object node (0: empty). Of duplicate keys, the last is kept, -// as for the linear search, and as basic_json::parse() does. +// node from its object node (0: empty). Of duplicate keys, the first is kept, +// as for the linear search. // // The hash is not seeded, so keys chosen to collide could make the build // quadratic. A key therefore sits at most index_max_displacement slots away @@ -92,8 +92,7 @@ inline void build_object_index(document_data& d, node* obj) const node* const other = obj + slots[i]; if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0)) { - duplicate = true; // keep the last: the key's slot now leads to this member - slots[i] = static_cast(k - obj); + duplicate = true; // keep the first break; } if (++distance > index_max_displacement) @@ -130,7 +129,7 @@ inline void build_object_indexes(document_data& d) } } -/// the key node of the last member with this key of an indexed object, or +/// the key node of the first member with this key of an indexed object, or /// nullptr inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept { diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index a15d27053..2e74e4441 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -2879,8 +2879,8 @@ NLOHMANN_JSON_NAMESPACE_END // large objects). An object with document_data::index_min_members members or // more gets an open-addressing table after parsing; its node stores the // number of the table (1-based) in `extra`. A slot holds the offset of a key -// node from its object node (0: empty). Of duplicate keys, the last is kept, -// as for the linear search, and as basic_json::parse() does. +// node from its object node (0: empty). Of duplicate keys, the first is kept, +// as for the linear search. // // The hash is not seeded, so keys chosen to collide could make the build // quadratic. A key therefore sits at most index_max_displacement slots away @@ -2949,8 +2949,7 @@ inline void build_object_index(document_data& d, node* obj) const node* const other = obj + slots[i]; if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0)) { - duplicate = true; // keep the last: the key's slot now leads to this member - slots[i] = static_cast(k - obj); + duplicate = true; // keep the first break; } if (++distance > index_max_displacement) @@ -2987,7 +2986,7 @@ inline void build_object_indexes(document_data& d) } } -/// the key node of the last member with this key of an indexed object, or +/// the key node of the first member with this key of an indexed object, or /// nullptr inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept { diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 69636345b..19d432650 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -1635,31 +1635,26 @@ TEST_CASE("json_view large objects") for (std::size_t i = 0; i < members; ++i) { const std::string key = std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\n" : ""); + CHECK(v[key].get() == i); CHECK(v.contains(key)); CHECK(v.find(key).key() == key); - CHECK(!v.contains(key + "x")); - if (key == "k1") - { - continue; // repeated below: the last member wins - } - CHECK(v[key].get() == i); CHECK(v.at(key).get() == i); + CHECK(!v.contains(key + "x")); } CHECK(v[""].get_string() == "empty key"); - CHECK(v["k1"].get_string() == "a duplicate of an earlier key"); // the last of duplicate keys, as for small objects - CHECK(v.at("k1").get_string() == "a duplicate of an earlier key"); + CHECK(v["k1"].get() == 1); // the first of duplicate keys, as for small objects CHECK(!v.contains("missing")); CHECK_THROWS_WITH_AS(v.at("missing"), "[json.exception.out_of_range.403] key 'missing' not found", json::out_of_range&); CHECK(v == j); CHECK(v.materialize() == j); } - SECTION("duplicate keys: the last member wins, with and without a table") + SECTION("duplicate keys: the first member wins, with and without a table") { // an object of `total` members: the keys "k0".."k" in order, then // three keys repeated twice more (one copy in the middle, one at the // end), and two keys repeated once; the value of a member is its - // position, so that the last member of a key can be told apart + // position, so that the first member of a key can be told apart struct member { std::string key; @@ -1707,20 +1702,23 @@ TEST_CASE("json_view large objects") REQUIRE(members.size() >= total); REQUIRE(members.size() >= 100); std::string text = "{"; + std::map first; std::map last; for (const member& m : members) { text += (text.size() > 1 ? ",\"" : "\"") + m.key + "\":" + std::to_string(m.position); + first.insert({m.key, m.position}); last[m.key] = m.position; } text += '}'; - REQUIRE(last.size() < members.size()); + REQUIRE(first.size() < members.size()); const json_document d = json_document::parse(text); const json_view v = d.root(); const json j = json::parse(text); CHECK(v.size() == members.size()); // every occurrence is visited - for (const auto& entry : last) + const json m = v.materialize(); + for (const auto& entry : first) { CAPTURE(entry.first) const std::size_t expected = entry.second; @@ -1735,7 +1733,9 @@ TEST_CASE("json_view large objects") CHECK(v.at(json::json_pointer(pointer)).get() == expected); CHECK(v.value(json::json_pointer(pointer), std::size_t{0}) == expected); CHECK(v.contains(json::json_pointer(pointer))); - CHECK(j[entry.first].get() == expected); // as materialize() and parse() keep it + // materialize() and parse() keep the last value instead + CHECK(j[entry.first].get() == last.at(entry.first)); + CHECK(m[entry.first].get() == last.at(entry.first)); } CHECK(!v.contains("k")); CHECK(v["missing"].is_discarded());