diff --git a/docs/mkdocs/docs/api/basic_json_document/index.md b/docs/mkdocs/docs/api/basic_json_document/index.md index 539437616..dd5f2302c 100644 --- a/docs/mkdocs/docs/api/basic_json_document/index.md +++ b/docs/mkdocs/docs/api/basic_json_document/index.md @@ -48,6 +48,7 @@ bookkeeping edits need, and calling any of them on one fails to compile (`#!cpp - **view_type** - the type of view returned by [`root()`](root.md) (`#!cpp basic_json_view`) - **value_t** - the JSON type enumeration, see [`basic_json::value_t`](../basic_json/value_t.md) +- **image_check** - how [`load()`](load.md) validates an image (`full`, `bounds`, `none`), see [`image_check`](load.md#image_check) ## Member functions diff --git a/docs/mkdocs/docs/api/basic_json_document/load.md b/docs/mkdocs/docs/api/basic_json_document/load.md index 7369ff11a..85b191155 100644 --- a/docs/mkdocs/docs/api/basic_json_document/load.md +++ b/docs/mkdocs/docs/api/basic_json_document/load.md @@ -85,9 +85,13 @@ Otherwise throws [`parse_error.116`](../../home/exceptions.md#jsonexceptionparse ## Complexity -Linear in the number of nodes, which are always copied into the document. With `#!cpp check == image_check::full`, -additionally linear in the combined length of the text and the decoded strings; `#!cpp image_check::bounds` and -`#!cpp image_check::none` do not read them. +Linear in the number of nodes, which are always copied into the document. `#!cpp image_check::none` does not read +the text or the decoded strings. `#!cpp image_check::bounds` reads one byte of the text for each float token (its sign, +to check the recorded digits against the token's length), and nothing else of the text or the decoded strings. + +With `#!cpp image_check::full`, linear in the size of the image (the nodes, the text, and the decoded strings), plus +sorting the ranges of the strings and of the float tokens (at most one per node): the contents of each distinct range +are checked once, however many nodes refer to it. ## Notes @@ -108,7 +112,7 @@ How thoroughly `load()` validates `image` before trusting it. | value | checks | guarantees | |----------|--------------------------------------------------------------------------------------------------------------|------------| -| `full` | everything the parser itself guarantees: structure and bounds; that every string is valid UTF-8 (and, for a string still in the source text, that it contains no quote, backslash, or control character); and that every number token is well-formed and matches the value stored for it | reading and serializing a checked image is safe and always produces valid JSON, exactly as for a parsed document | +| `full` | everything the parser itself guarantees: structure and bounds; that every string is valid UTF-8 (and, for a string still in the source text, that it contains no quote, backslash, or control character); and that every number token is well-formed and matches the value stored for it; strings and float tokens may share a range only if the ranges are identical (as nodes that share a value do), and never overlap otherwise | reading and serializing a checked image is safe and always produces valid JSON, exactly as for a parsed document | | `bounds` | structure and bounds only -- that every offset and count in the node index stays inside the image | reading and serializing stay memory-safe, but a crafted image can hold strings that are not valid UTF-8 or that serialize to invalid JSON ([`dump()`](../basic_json_view/dump.md) writes them unchanged or throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316)), and numbers whose values differ from their text | | `none` | nothing | images from a trusted source only -- reading a damaged image is undefined behavior | diff --git a/include/nlohmann/detail/view/image.hpp b/include/nlohmann/detail/view/image.hpp index c1776f699..49ca8dc67 100644 --- a/include/nlohmann/detail/view/image.hpp +++ b/include/nlohmann/detail/view/image.hpp @@ -8,6 +8,7 @@ #pragma once +#include // sort #include // array #include // size_t #include // int64_t, uint8_t, uint16_t, uint32_t, uint64_t @@ -252,17 +253,25 @@ inline std::vector save_image(const document_data& d) return image; } -/// whether a number node matches its token the way the parser records it -/// (after the bounds check) -inline bool check_number(const node& n, const unsigned char* text) +/// the parts of a number token that the checks need +struct number_token +{ + const unsigned char* int_start; + std::size_t int_digits; + std::size_t frac_digits; + std::int64_t exponent; + bool negative; + bool is_float; ///< a fraction or an exponent +}; + +/// whether [s, s + len) is a JSON number (the grammar the parser accepts) +inline bool scan_number_token(const unsigned char* s, std::size_t len, number_token& t) { - const std::size_t len = number_length(n); - const unsigned char* const s = text + n.off; const unsigned char* const e = s + len; const unsigned char* p = s; - const bool negative = *p == '-'; - p += negative ? 1 : 0; - const unsigned char* const int_start = p; + t.negative = p != e && *p == '-'; + p += t.negative ? 1 : 0; + t.int_start = p; if (p == e) { return false; @@ -282,9 +291,9 @@ inline bool check_number(const node& n, const unsigned char* text) { return false; } - const auto int_digits = static_cast(p - int_start); - std::size_t frac_digits = 0; - bool is_float = false; + t.int_digits = static_cast(p - t.int_start); + t.frac_digits = 0; + t.is_float = false; if (p != e && *p == '.') { const unsigned char* const f0 = ++p; @@ -296,10 +305,10 @@ inline bool check_number(const node& n, const unsigned char* text) { return false; } - frac_digits = static_cast(p - f0); - is_float = true; + t.frac_digits = static_cast(p - f0); + t.is_float = true; } - std::int64_t exponent = 0; + t.exponent = 0; if (p != e && (*p | 0x20u) == 'e') { ++p; @@ -311,56 +320,171 @@ inline bool check_number(const node& n, const unsigned char* text) } while (p != e && is_digit(*p)) { - exponent = exponent < 100000 ? (exponent * 10) + (*p - '0') : exponent; + t.exponent = t.exponent < 100000 ? (t.exponent * 10) + (*p - '0') : t.exponent; ++p; } - exponent = exp_negative ? -exponent : exponent; - is_float = true; + t.exponent = exp_negative ? -t.exponent : t.exponent; + t.is_float = true; } - if (p != e) + return p == e; +} + +/// the digit layout the parser records for a float token (compaction +/// writes "many" instead; the caller accepts both) +inline std::uint16_t float_layout(const number_token& t) +{ + return static_cast((t.int_digits < 255 ? t.int_digits : 255) | ((t.frac_digits < 255 ? t.frac_digits : 255) << 8u)); +} + +/// whether a float token (well-formed, per scan_number_token) is finite as +/// double; parse() rejects floats that overflow, and as there, only a number +/// whose magnitude could reach 1e308 needs the conversion +inline bool float_token_finite(const unsigned char* s, std::size_t len, const number_token& t) +{ + if (static_cast(t.int_digits) + t.exponent > 300) + { + node n{}; + n.kind = static_cast(value_t::number_float); + n.len = static_cast(len); + const auto v = float_value(reinterpret_cast(s), n); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return v <= (std::numeric_limits::max)() && v >= -(std::numeric_limits::max)(); + } + return true; +} + +/// whether an integer node matches its token the way the parser records it +/// (after the bounds check; the token has at most 256 characters) +inline bool check_integer(const node& n, const unsigned char* text) +{ + const std::size_t len = number_length(n); + const unsigned char* const s = text + n.off; + number_token t{}; + if (!scan_number_token(s, len, t)) { return false; } - if (n.kind == static_cast(value_t::number_float)) - { - // the digit layout the parser records (or "many", as compaction - // writes it), and a finite value - const auto layout = static_cast((int_digits < 255 ? int_digits : 255) | ((frac_digits < 255 ? frac_digits : 255) << 8u)); - if (n.extra != layout && n.extra != 0xFFFFu) - { - return false; - } - // parse() rejects floats that overflow; as there, only a number whose - // magnitude could reach 1e308 needs the conversion - if (static_cast(int_digits) + exponent > 300) - { - const auto v = float_value(reinterpret_cast(s), n); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) - return v <= (std::numeric_limits::max)() && v >= -(std::numeric_limits::max)(); - } - return true; - } - // integers: the token's value is the stored one; number_integer nodes of - // edits can be non-negative (as basic_json keeps the type of a value) + // the token's value is the stored one; number_integer nodes of edits can + // be non-negative (as basic_json keeps the type of a value) const bool integer = n.kind == static_cast(value_t::number_integer); - if (is_float || int_digits > 20 || (negative && !integer)) + if (t.is_float || t.int_digits > 20 || (t.negative && !integer)) { return false; } // (at most 19 digits cannot overflow; 20 digits are compared with 2^64 - 1) - if (int_digits == 20 && std::memcmp(int_start, "18446744073709551615", 20) > 0) + if (t.int_digits == 20 && std::memcmp(t.int_start, "18446744073709551615", 20) > 0) { return false; } std::uint64_t m = 0; - for (const unsigned char* d = int_start; d != int_start + int_digits; ++d) + for (const unsigned char* d = t.int_start; d != t.int_start + t.int_digits; ++d) { m = (m * 10) + static_cast(*d - '0'); } - if (integer && m > (negative ? std::uint64_t{1} << 63u : (std::uint64_t{1} << 63u) - 1)) + if (integer && m > (t.negative ? std::uint64_t{1} << 63u : (std::uint64_t{1} << 63u) - 1)) { return false; } - return integer_bits(n) == (negative ? 0 - m : m); + return integer_bits(n) == (t.negative ? 0 - m : m); +} + +/// a byte range of the text or of the decoded strings +struct byte_range +{ + std::uint32_t off; + std::uint32_t len; +}; + +/// a float token and the digit layout its node records +struct float_range +{ + std::uint32_t off; + std::uint32_t len; + std::uint16_t extra; +}; + +/// Run check once for each distinct range of ranges (which are within bounds +/// and not empty). Ranges that are not identical must not overlap: save() +/// writes every string and token to a place of its own, and nodes that share +/// a value (copies within a document) share the whole range. This bounds the +/// work by the size of the text, however many nodes point to the same bytes. +template +bool check_distinct_ranges(std::vector& ranges, Check check) +{ + std::sort(ranges.begin(), ranges.end(), [](const byte_range & a, const byte_range & b) + { + return a.off != b.off ? a.off < b.off : a.len < b.len; + }); + std::size_t end = 0; + for (std::size_t i = 0; i < ranges.size(); ++i) + { + const byte_range r = ranges[i]; + if (i != 0 && r.off == ranges[i - 1].off && r.len == ranges[i - 1].len) + { + continue; + } + if (r.off < end || !check(r)) + { + return false; + } + end = static_cast(r.off) + r.len; + } + return true; +} + +/// Check the float nodes: each token once (nodes of the same token must +/// record layouts that match it), tokens must not overlap. +inline bool check_float_ranges(std::vector& ranges, const unsigned char* text) +{ + std::sort(ranges.begin(), ranges.end(), [](const float_range & a, const float_range & b) + { + return a.off != b.off ? a.off < b.off : (a.len != b.len ? a.len < b.len : a.extra < b.extra); + }); + std::size_t end = 0; + std::size_t i = 0; + while (i < ranges.size()) + { + const float_range r = ranges[i]; + if (r.off < end) + { + return false; + } + number_token t{}; + const unsigned char* const s = text + r.off; + if (!scan_number_token(s, r.len, t) || !float_token_finite(s, r.len, t)) + { + return false; + } + // the digit layout the parser records (or "many", as compaction writes it) + const std::uint16_t layout = float_layout(t); + for (; i < ranges.size() && ranges[i].off == r.off && ranges[i].len == r.len; ++i) + { + if (ranges[i].extra != layout && ranges[i].extra != 0xFFFFu) + { + return false; + } + } + end = static_cast(r.off) + r.len; + } + return true; +} + +/// the content checks of check_image: source strings as the parser leaves +/// them (no quotes, backslashes, or control characters), decoded strings +/// (valid UTF-8), and float tokens +inline bool check_contents(const unsigned char* text, const unsigned char* arena, + std::vector& source_strings, std::vector& decoded_strings, + std::vector& floats) +{ + return check_distinct_ranges(source_strings, [text](const byte_range & r) + { + const unsigned char* const b = text + r.off; + return scan_string_run(b, b + r.len) == b + r.len; + }) + && check_distinct_ranges(decoded_strings, [arena](const byte_range & r) + { + return valid_utf8_prefix(arena + r.off, r.len) == r.len; + }) + && check_float_ranges(floats, text); } /// Check the nodes of a loaded image against its text and decoded strings: @@ -368,6 +492,12 @@ inline bool check_number(const node& n, const unsigned char* text) /// objects; keys; bounds; string contents (source strings as the parser /// leaves them: no quotes, backslashes, or control characters; all strings /// valid UTF-8); and number tokens. +/// +/// The structure and the bounds are checked node by node. The contents of +/// strings and of float tokens are checked afterwards, once for each distinct +/// range (see check_distinct_ranges), so that the full check is linear in the +/// size of the image plus the sorting of the ranges, and not in the number of +/// nodes times the size of the text. inline bool check_image(const node* nodes, std::size_t count, const unsigned char* text, std::size_t text_size, const unsigned char* arena, std::size_t arena_size, bool full) { @@ -380,6 +510,9 @@ inline bool check_image(const node* nodes, std::size_t count, const unsigned cha bool expect_key; }; std::vector stack; + std::vector source_strings; + std::vector decoded_strings; + std::vector floats; const auto check_string = [&](const node & n) -> bool { if ((n.flags & ~node_flags::escaped) != 0 || n.extra != 0) @@ -387,18 +520,16 @@ inline bool check_image(const node* nodes, std::size_t count, const unsigned cha return false; } const bool decoded = (n.flags & node_flags::escaped) != 0; - const unsigned char* const base = decoded ? arena : text; const std::size_t limit = decoded ? arena_size : text_size; if (n.off > limit || n.len > limit - n.off) { return false; } - if (!full) + if (full && n.len != 0) { - return true; + (decoded ? decoded_strings : source_strings).push_back(byte_range{n.off, n.len}); } - const unsigned char* const b = base + n.off; - return decoded ? valid_utf8_prefix(b, n.len) == n.len : scan_string_run(b, b + n.len) == b + n.len; + return true; }; // bounds of a number token; the recorded digit layout must lie within it const auto number_in_bounds = [&](const node & n) -> bool @@ -440,7 +571,7 @@ inline bool check_image(const node* nodes, std::size_t count, const unsigned cha } if (i == count) { - return stack.empty(); + return stack.empty() && (!full || check_contents(text, arena, source_strings, decoded_strings, floats)); } if (i != 0 && stack.empty()) { @@ -482,10 +613,23 @@ inline bool check_image(const node* nodes, std::size_t count, const unsigned cha case value_t::number_integer: case value_t::number_unsigned: case value_t::number_float: - if (n.flags != 0 || !number_in_bounds(n) || (full && !check_number(n, text))) + if (n.flags != 0 || !number_in_bounds(n)) { return false; } + if (full) + { + // (the contents of float tokens are checked later; the token of an + // integer has at most 256 characters) + if (n.kind == static_cast(value_t::number_float)) + { + floats.push_back(float_range{n.off, n.len, n.extra}); + } + else if (!check_integer(n, text)) + { + return false; + } + } break; case value_t::array: case value_t::object: diff --git a/tests/src/unit-json_view_image.cpp b/tests/src/unit-json_view_image.cpp index 6a615f7d4..cf39a7bb5 100644 --- a/tests/src/unit-json_view_image.cpp +++ b/tests/src/unit-json_view_image.cpp @@ -712,6 +712,111 @@ TEST_CASE("json_view images: check") } } + SECTION("nodes that share a range") + { + // [big string, then n strings made to point to the big string]: every + // node shares the one range (the check reads it once) + const std::size_t n = 200; + std::string long_text = "[\"" + std::string(5000, 'a') + "\""; + for (std::size_t k = 0; k < n; ++k) + { + long_text += ",\"x\""; + } + long_text += "]"; + const std::vector img = json_document::parse(long_text).save(); + const node big = node_at(img, 1); + std::vector b = img; + for (std::size_t k = 0; k < n; ++k) + { + set_node(b, 2 + k, big); + } + CHECK(load_result(b, image_check::full).empty()); + const json_document d = json_document::load(b); + CHECK(d.root().size() == n + 1); + CHECK(d.root()[n].get() == std::string(5000, 'a')); + CHECK(d.root().dump() == json_document::load(b, image_check::none).root().dump()); + + // the same for decoded strings and float tokens, with a node that + // records another digit layout than its token + const std::vector img2 = json_document::parse(R"(["a\"b", "a\"b", 1.25, 1.25, 1.25e3])").save(); + CHECK(load_result(img2, image_check::full).empty()); + std::vector same = img2; + set_node(same, 2, node_at(img2, 1)); + set_node(same, 4, node_at(img2, 3)); + CHECK(load_result(same, image_check::full).empty()); + CHECK(json_document::load(same).root().dump() == R"(["a\"b","a\"b",1.25,1.25,1250.0])"); + std::vector layout = same; + node f4 = node_at(layout, 4); + f4.extra = 0x0100u; // the layout of "1.", and not that of "1.25" + set_node(layout, 4, f4); + rejected(layout, false); + f4.extra = 0xFFFFu; // "many" digits: fine, as compaction writes it + set_node(layout, 4, f4); + CHECK(load_result(layout, image_check::full).empty()); + } + + SECTION("ranges that overlap") + { + // save() writes every string and every token to a place of its own + // (nodes that share a value share the whole range): ranges that + // overlap without being identical are a damaged image, though each + // range is a valid string or token + // nodes: 0 [ 1 "abcdef" 2 "ghijkl" 3 1.2525 4 9.9 5 "a\"bcd" (decoded) 6 "e\"fgh" (decoded) + const std::vector img = json_document::parse(R"(["abcdef", "ghijkl", 1.2525, 9.9, "a\"bcd", "e\"fgh"])").save(); + REQUIRE(load_result(img, image_check::full).empty()); + // node i with the range (off of node of + shift, length), and extra + const auto aliased = [&](std::size_t i, std::size_t of, std::uint32_t shift, std::uint32_t length, std::uint16_t extra) + { + std::vector b = img; + node n = node_at(b, of); + n.off += shift; + n.len = length; + n.extra = extra; + set_node(b, i, n); + return b; + }; + // source strings + rejected(aliased(2, 1, 0, 4, 0), false); // "abcd": the start of another string + rejected(aliased(2, 1, 1, 4, 0), false); // "bcde": inside + rejected(aliased(2, 1, 2, 4, 0), false); // "cdef": the end + CHECK(load_result(aliased(2, 1, 0, 6, 0), image_check::full).empty()); // the whole range: identical + // decoded strings: inside, and partially overlapping (the arena holds a"bcde"fgh) + rejected(aliased(6, 5, 1, 3, 0), false); + rejected(aliased(6, 5, 3, 5, 0), false); + CHECK(load_result(aliased(6, 5, 0, 5, 0), image_check::full).empty()); + // float tokens: "1.25" and "525" (an integer token as a float) inside "1.2525" + rejected(aliased(4, 3, 0, 4, 0x0201u), false); + rejected(aliased(4, 3, 3, 3, 0x0003u), false); + CHECK(load_result(aliased(4, 3, 0, 6, 0x0401u), image_check::full).empty()); + // a string and a float token may use the same bytes (each is checked by its own kind) + std::vector shared = img; + node as_string = node_at(img, 2); + as_string.off = node_at(img, 3).off; + as_string.len = 6; + set_node(shared, 2, as_string); + CHECK(load_result(shared, image_check::full).empty()); + // empty strings inside others are not ranges + CHECK(load_result(aliased(2, 1, 2, 0, 0), image_check::full).empty()); + } + + SECTION("copies within an edited document") + { + // copies within a document share the value: identical ranges + json_editable_document d = json_editable_document::parse(R"({"s": "ab", "e": "x\"y", "f": 1.5, "g": 1.5e300, "a": [1.5, "ab"]})"); + d.set(d.root(), "c1", d.root()["a"]); + d.set(d.root(), "c2", d.root()["a"]); + d.set(d.root(), "c3", d.root()); + d.push_back(d.root()["a"], d.root()["e"]); + d.push_back(d.root()["a"], d.root()["g"]); + d.push_back(d.root()["a"], d.root()["g"]); + d.set(d.root()["s"], "changed"); + d.set(d.root()["f"], 7.25); + const std::vector saved = d.save(); + CHECK(load_result(saved, image_check::full).empty()); + CHECK(json_document::load(saved).root().dump() == d.root().dump()); + check_round_trip(d); + } + SECTION("integer ranges") { // tokens of many digits, which the parser stores as floats