mirror of
https://github.com/nlohmann/json.git
synced 2026-09-01 05:57:14 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c401d495a3 | ||
|
|
2998f2c731 | ||
|
|
1232fa947c | ||
|
|
ee8b2d1699 | ||
|
|
1bbf39e8b4 | ||
|
|
9b0169d1fb | ||
|
|
4f2bffa445 | ||
|
|
f58dd59903 | ||
|
|
8c4835dcb6 | ||
|
|
00b091ca9c | ||
|
|
53214a49c7 | ||
|
|
455f5bf680 | ||
|
|
1135fabd46 | ||
|
|
0a776d92a2 | ||
|
|
061699735e |
@@ -93,6 +93,58 @@ inline std::size_t find_string_special(const unsigned char* data, std::size_t n)
|
|||||||
return n;
|
return n;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// classify a byte as one the serializer must NOT copy verbatim when
|
||||||
|
// ensure_ascii is requested: the closing quote, an escape, a control character
|
||||||
|
// (< 0x20), DEL (0x7F), or any non-ASCII byte (>= 0x80). Everything else -
|
||||||
|
// printable ASCII except '"' and '\\' - is emitted unchanged. Note this differs
|
||||||
|
// from is_string_special() only in that 0x7F is also a stop (it is escaped as
|
||||||
|
// \u007f under ensure_ascii).
|
||||||
|
inline bool is_ascii_copyable(unsigned char c) noexcept
|
||||||
|
{
|
||||||
|
return c >= 0x20u && c < 0x7Fu && c != '\"' && c != '\\';
|
||||||
|
}
|
||||||
|
|
||||||
|
// return the index of the first byte in [data, data+n) that is NOT
|
||||||
|
// is_ascii_copyable(), or n if every byte can be copied verbatim; scans 8 bytes
|
||||||
|
// at a time. Used by the serializer's ensure_ascii fast path.
|
||||||
|
inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_t n) noexcept
|
||||||
|
{
|
||||||
|
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
||||||
|
constexpr std::uint64_t high = 0x8080808080808080ull;
|
||||||
|
std::size_t i = 0;
|
||||||
|
for (; i + 8 <= n; i += 8)
|
||||||
|
{
|
||||||
|
std::uint64_t v = 0;
|
||||||
|
std::memcpy(&v, data + i, sizeof(v));
|
||||||
|
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
|
||||||
|
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
|
||||||
|
const std::uint64_t d = v ^ 0x7F7F7F7F7F7F7F7Full; // DEL (0x7F)
|
||||||
|
const std::uint64_t stop = ((q - ones) & ~q & high) // == '"'
|
||||||
|
| ((b - ones) & ~b & high) // == '\\'
|
||||||
|
| ((d - ones) & ~d & high) // == 0x7F
|
||||||
|
| ((v - 0x2020202020202020ull) & ~v & high) // < 0x20
|
||||||
|
| (v & high); // >= 0x80
|
||||||
|
if (stop != 0)
|
||||||
|
{
|
||||||
|
for (std::size_t j = 0; j < 8; ++j)
|
||||||
|
{
|
||||||
|
if (!is_ascii_copyable(data[i + j]))
|
||||||
|
{
|
||||||
|
return i + j;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (; i < n; ++i)
|
||||||
|
{
|
||||||
|
if (!is_ascii_copyable(data[i]))
|
||||||
|
{
|
||||||
|
return i;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its
|
// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its
|
||||||
// length (2..4) only when the bytes form a *well-formed* sequence using exactly
|
// length (2..4) only when the bytes form a *well-formed* sequence using exactly
|
||||||
// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts
|
// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -1341,11 +1341,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const error_handler_t error_handler = error_handler_t::strict) const
|
const error_handler_t error_handler = error_handler_t::strict) const
|
||||||
{
|
{
|
||||||
string_t result;
|
string_t result;
|
||||||
serializer s(detail::output_adapter<char, string_t>(result), indent_char, error_handler);
|
detail::output_string_adapter<char, string_t> string_adapter(result);
|
||||||
|
serializer s(string_adapter, indent_char, error_handler);
|
||||||
|
|
||||||
if (indent >= 0)
|
if (indent >= 0)
|
||||||
{
|
{
|
||||||
s.dump(*this, true, ensure_ascii, static_cast<unsigned int>(indent));
|
s.dump(*this, true, ensure_ascii, static_cast<std::size_t>(indent));
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -4055,7 +4056,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
o.width(0);
|
o.width(0);
|
||||||
|
|
||||||
// do the actual serialization
|
// do the actual serialization
|
||||||
serializer s(detail::output_adapter<char>(o), o.fill());
|
detail::output_stream_adapter<char> stream_adapter(o);
|
||||||
|
serializer s(stream_adapter, o.fill());
|
||||||
s.dump(j, pretty_print, false, static_cast<unsigned int>(indentation));
|
s.dump(j, pretty_print, false, static_cast<unsigned int>(indentation));
|
||||||
return o;
|
return o;
|
||||||
}
|
}
|
||||||
|
|||||||
+859
-113
File diff suppressed because it is too large
Load Diff
@@ -98,8 +98,10 @@ void check_escaped(const char* original, const char* escaped = "", bool ensure_a
|
|||||||
void check_escaped(const char* original, const char* escaped, const bool ensure_ascii)
|
void check_escaped(const char* original, const char* escaped, const bool ensure_ascii)
|
||||||
{
|
{
|
||||||
std::stringstream ss;
|
std::stringstream ss;
|
||||||
json::serializer s(nlohmann::detail::output_adapter<char>(ss), ' ');
|
nlohmann::detail::output_stream_adapter<char> adapter(ss);
|
||||||
|
json::serializer s(adapter, ' ');
|
||||||
s.dump_escaped(original, ensure_ascii);
|
s.dump_escaped(original, ensure_ascii);
|
||||||
|
s.flush(); // dump_escaped writes into the serializer's internal buffer
|
||||||
CHECK(ss.str() == escaped);
|
CHECK(ss.str() == escaped);
|
||||||
}
|
}
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|||||||
@@ -382,3 +382,232 @@ TEST_CASE("dump for basic_json with long double number_float_t")
|
|||||||
check_same(100.0L, 100.0);
|
check_same(100.0L, 100.0);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("serialization of strings (bulk fast path)")
|
||||||
|
{
|
||||||
|
// These cases exercise the SWAR bulk-copy fast path in dump_escaped and the
|
||||||
|
// internal write buffer: long runs, escapes interrupting runs, 0x7F/DEL,
|
||||||
|
// multibyte UTF-8 under both ensure_ascii settings, and payloads larger than
|
||||||
|
// the write buffer.
|
||||||
|
|
||||||
|
SECTION("long unescaped ASCII exceeds the write buffer")
|
||||||
|
{
|
||||||
|
const std::string big(3000, 'a');
|
||||||
|
const json j = big;
|
||||||
|
CHECK(j.dump() == '"' + big + '"');
|
||||||
|
CHECK(j.dump(-1, ' ', true) == '"' + big + '"');
|
||||||
|
// round-trips
|
||||||
|
CHECK(json::parse(j.dump()) == j);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("runs interrupted by escapes")
|
||||||
|
{
|
||||||
|
const json j = std::string(500, 'x') + "\n\"\\" + std::string(500, 'y');
|
||||||
|
const std::string out = j.dump();
|
||||||
|
CHECK(out == '"' + std::string(500, 'x') + "\\n\\\"\\\\" + std::string(500, 'y') + '"');
|
||||||
|
CHECK(json::parse(out) == j);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("DEL (0x7F) depends on ensure_ascii")
|
||||||
|
{
|
||||||
|
const json j = std::string("a\x7f" "b");
|
||||||
|
CHECK(j.dump(-1, ' ', false) == "\"a\x7f" "b\""); // copied verbatim
|
||||||
|
CHECK(j.dump(-1, ' ', true) == "\"a\\u007fb\""); // escaped
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("multibyte UTF-8 under both ensure_ascii settings")
|
||||||
|
{
|
||||||
|
const json j = std::string("A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z"); // A é 你 😀 Z
|
||||||
|
// not escaping non-ASCII: bytes are copied through the bulk validator
|
||||||
|
CHECK(j.dump(-1, ' ', false) == "\"A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z\"");
|
||||||
|
// ensure_ascii: escaped (with a surrogate pair for the emoji)
|
||||||
|
CHECK(j.dump(-1, ' ', true) == "\"A\\u00e9\\u4f60\\ud83d\\ude00Z\"");
|
||||||
|
CHECK(json::parse(j.dump(-1, ' ', true)) == j);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("many small structural writes exceed the write buffer")
|
||||||
|
{
|
||||||
|
json arr = json::array();
|
||||||
|
for (int i = 0; i < 2000; ++i)
|
||||||
|
{
|
||||||
|
arr.push_back(i);
|
||||||
|
}
|
||||||
|
const std::string out = arr.dump();
|
||||||
|
CHECK(out.front() == '[');
|
||||||
|
CHECK(out.back() == ']');
|
||||||
|
CHECK(json::parse(out) == arr);
|
||||||
|
|
||||||
|
json obj = json::object();
|
||||||
|
for (int i = 0; i < 500; ++i)
|
||||||
|
{
|
||||||
|
obj["key" + std::to_string(i)] = i;
|
||||||
|
}
|
||||||
|
CHECK(json::parse(obj.dump()) == obj);
|
||||||
|
CHECK(json::parse(obj.dump(2)) == obj);
|
||||||
|
|
||||||
|
// an array of many empty strings emits a long run of single-character
|
||||||
|
// writes ('"', '"', ',') at shallow nesting depth, so the write buffer
|
||||||
|
// fills and flushes mid-run without the deep recursion that would
|
||||||
|
// overflow the stack on some debug builds
|
||||||
|
json many_empty = json::array();
|
||||||
|
for (int i = 0; i < 500; ++i)
|
||||||
|
{
|
||||||
|
many_empty.push_back("");
|
||||||
|
}
|
||||||
|
const std::string out2 = many_empty.dump();
|
||||||
|
CHECK(out2.size() > 1024); // spans multiple write-buffer flushes
|
||||||
|
CHECK(out2.front() == '[');
|
||||||
|
CHECK(out2.back() == ']');
|
||||||
|
CHECK(json::parse(out2) == many_empty);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("invalid UTF-8 handling is unaffected by the fast path")
|
||||||
|
{
|
||||||
|
const json j = std::string("valid\xff" "more");
|
||||||
|
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0xFF", json::type_error&);
|
||||||
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"valid\xef\xbf\xbd" "more\"");
|
||||||
|
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"valid\\ufffdmore\"");
|
||||||
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"validmore\"");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_CASE("indentation is written straight into the write buffer")
|
||||||
|
{
|
||||||
|
// put_indent() memsets the indentation into the write buffer instead of
|
||||||
|
// copying it out of a pre-grown indentation string. These cases cover an
|
||||||
|
// indentation wider than the buffer, a non-space indentation character, and
|
||||||
|
// nesting deep enough that the accumulated indentation spans several
|
||||||
|
// buffer-fulls - the situations the old grow-a-string approach got wrong.
|
||||||
|
|
||||||
|
SECTION("indent_step wider than the write buffer")
|
||||||
|
{
|
||||||
|
const json j = {{"a", 1}};
|
||||||
|
// 2000 > the 1024-byte write buffer, and > the 512 the indentation
|
||||||
|
// string used to start at
|
||||||
|
CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"a\": 1\n}");
|
||||||
|
// several whole buffer-fulls, so the buffer is refilled once and then
|
||||||
|
// flushed repeatedly
|
||||||
|
CHECK(j.dump(5000) == "{\n" + std::string(5000, ' ') + "\"a\": 1\n}");
|
||||||
|
CHECK(j.dump(5000, '\t') == "{\n" + std::string(5000, '\t') + "\"a\": 1\n}");
|
||||||
|
// an exact multiple of the buffer size
|
||||||
|
CHECK(j.dump(4096) == "{\n" + std::string(4096, ' ') + "\"a\": 1\n}");
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("a non-space indentation character is used throughout")
|
||||||
|
{
|
||||||
|
const json j = {{"a", 1}};
|
||||||
|
// 600 is past the point where the indentation used to be grown, which
|
||||||
|
// is where a hard-coded space would have shown up
|
||||||
|
CHECK(j.dump(600, '\t') == "{\n" + std::string(600, '\t') + "\"a\": 1\n}");
|
||||||
|
CHECK(j.dump(3, '.') == "{\n...\"a\": 1\n}");
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("accumulated indentation spans several buffer-fulls")
|
||||||
|
{
|
||||||
|
// five levels deep at 400 per level: the innermost value is indented by
|
||||||
|
// 2000 characters, reached in steps that each straddle the buffer end
|
||||||
|
json j = json::array({1});
|
||||||
|
for (int i = 0; i < 4; ++i)
|
||||||
|
{
|
||||||
|
j = json::array({j});
|
||||||
|
}
|
||||||
|
|
||||||
|
const std::string out = j.dump(400);
|
||||||
|
CHECK(out.find(std::string("\n") + std::string(2000, ' ') + "1\n") != std::string::npos);
|
||||||
|
CHECK(json::parse(out) == j);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("indentation is unchanged for ordinary widths")
|
||||||
|
{
|
||||||
|
const json j = {{"a", {1, 2}}, {"b", nullptr}};
|
||||||
|
CHECK(j.dump(2) == "{\n \"a\": [\n 1,\n 2\n ],\n \"b\": null\n}");
|
||||||
|
CHECK(j.dump(0) == "{\n\"a\": [\n1,\n2\n],\n\"b\": null\n}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_CASE("serialization of deeply nested values")
|
||||||
|
{
|
||||||
|
// dump() descends into a bounded number of levels and writes out whatever
|
||||||
|
// is nested deeper than that without the call stack; see
|
||||||
|
// https://github.com/nlohmann/json/issues/5387
|
||||||
|
|
||||||
|
SECTION("nested deeper than the call stack could follow")
|
||||||
|
{
|
||||||
|
// parsing is iterative, so building these costs little
|
||||||
|
const std::size_t depth = 100000;
|
||||||
|
|
||||||
|
const std::string array_text = std::string(depth, '[') + '0' + std::string(depth, ']');
|
||||||
|
CHECK(json::parse(array_text).dump() == array_text);
|
||||||
|
|
||||||
|
std::string object_text;
|
||||||
|
object_text.reserve((6 * depth) + 1);
|
||||||
|
for (std::size_t i = 0; i < depth; ++i)
|
||||||
|
{
|
||||||
|
object_text += "{\"a\":";
|
||||||
|
}
|
||||||
|
object_text += '1';
|
||||||
|
object_text.append(depth, '}');
|
||||||
|
CHECK(json::parse(object_text).dump() == object_text);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("depths around the bound of the descent")
|
||||||
|
{
|
||||||
|
// Cover every depth around the bound, so that the two ways of writing a
|
||||||
|
// value are known to meet cleanly - wherever the bound is set.
|
||||||
|
for (std::size_t d = 1; d <= 300; ++d)
|
||||||
|
{
|
||||||
|
CAPTURE(d);
|
||||||
|
|
||||||
|
const std::string array_text = std::string(d, '[') + '7' + std::string(d, ']');
|
||||||
|
CHECK(json::parse(array_text).dump() == array_text);
|
||||||
|
|
||||||
|
std::string object_text;
|
||||||
|
for (std::size_t i = 0; i < d; ++i)
|
||||||
|
{
|
||||||
|
object_text += "{\"k\":";
|
||||||
|
}
|
||||||
|
object_text += '7';
|
||||||
|
object_text.append(d, '}');
|
||||||
|
CHECK(json::parse(object_text).dump() == object_text);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("pretty-printing across the bound")
|
||||||
|
{
|
||||||
|
for (std::size_t d = 120; d <= 140; ++d)
|
||||||
|
{
|
||||||
|
CAPTURE(d);
|
||||||
|
|
||||||
|
const json j = json::parse(std::string(d, '[') + '7' + std::string(d, ']'));
|
||||||
|
|
||||||
|
std::string expected;
|
||||||
|
for (std::size_t i = 0; i < d; ++i)
|
||||||
|
{
|
||||||
|
expected += std::string(2 * i, ' ') + "[\n";
|
||||||
|
}
|
||||||
|
expected += std::string(2 * d, ' ') + '7';
|
||||||
|
for (std::size_t i = d; i > 0; --i)
|
||||||
|
{
|
||||||
|
expected += '\n' + std::string(2 * (i - 1), ' ') + ']';
|
||||||
|
}
|
||||||
|
|
||||||
|
CHECK(j.dump(2) == expected);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("an empty container below the bound")
|
||||||
|
{
|
||||||
|
// an empty container is written out in full and never descended into,
|
||||||
|
// so it must not gain a newline when it is reached iteratively
|
||||||
|
for (std::size_t d = 125; d <= 135; ++d)
|
||||||
|
{
|
||||||
|
CAPTURE(d);
|
||||||
|
|
||||||
|
const std::string compact = std::string(d, '[') + "[]" + std::string(d, ']');
|
||||||
|
CHECK(json::parse(compact).dump() == compact);
|
||||||
|
|
||||||
|
const std::string with_object = std::string(d, '[') + "{}" + std::string(d, ']');
|
||||||
|
CHECK(json::parse(with_object).dump() == with_object);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user