Buffer serializer output and add ensure_ascii string fast path

Two further serialization speedups on top of the ensure_ascii=false bulk
copy, both reusing the SWAR primitives in detail/input/string_scan.hpp.

1. Internal write buffer (devirtualization). Every structural character
   ('{', '"', ',', ...) previously went straight to the output adapter
   through a virtual call. Route all writes through put_char/put_chars
   into a 1 KiB buffer that flushes in bulk; the public dump() flushes
   once the top-level value is done (the recursive worker is split out as
   dump_internal). Runs larger than the buffer are written straight
   through, so large payloads are not copied twice. This is the dominant
   cost for object/array-heavy values.

2. ensure_ascii fast path. dump_escaped previously ran the UTF-8 DFA over
   every byte when escaping non-ASCII. Add find_ascii_copyable_run() (a
   SWAR scan stopping at '"', '\\', < 0x20, 0x7F, and >= 0x80) so runs of
   printable ASCII are bulk-copied, with the byte path handling each
   escape/non-ASCII byte exactly as before.

Behavior is unchanged: dump output is byte-for-byte identical to the
previous implementation across ~20k randomized byte strings plus curated
edge cases (all escapes, control chars, 0x7F, valid multibyte,
surrogates, overlong, truncated), for object/array/pretty output, both
ensure_ascii settings, and all three error handlers, in C++11/17/20 at
-O2/-O3. New unit tests cover the buffer flush boundaries, the escape and
0x7F handling, multibyte under both settings, and invalid-UTF-8 handling.

Throughput (g++ -O3, vs the ensure_ascii=false-only baseline):
  long ASCII, ensure_ascii=0   4.2x
  long ASCII, ensure_ascii=1   4.1x
  twitter-like objects         2.7x
  dense CJK                    1.8x

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01XAYM1qhSA2FDaDcGfPW3fG
Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-08-04 08:58:29 +02:00
co-authored by Claude Opus 4.8
parent 4818446af9
commit 7d2bc4a39a
4 changed files with 532 additions and 170 deletions
@@ -87,6 +87,58 @@ inline std::size_t find_string_special(const unsigned char* data, std::size_t n)
return n; return n;
} }
// classify a byte as one the serializer must NOT copy verbatim when
// ensure_ascii is requested: the closing quote, an escape, a control character
// (< 0x20), DEL (0x7F), or any non-ASCII byte (>= 0x80). Everything else -
// printable ASCII except '"' and '\\' - is emitted unchanged. Note this differs
// from is_string_special() only in that 0x7F is also a stop (it is escaped as
// \u007f under ensure_ascii).
inline bool is_ascii_copyable(unsigned char c) noexcept
{
return c >= 0x20u && c < 0x7Fu && c != '\"' && c != '\\';
}
// return the index of the first byte in [data, data+n) that is NOT
// is_ascii_copyable(), or n if every byte can be copied verbatim; scans 8 bytes
// at a time. Used by the serializer's ensure_ascii fast path.
inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_t n) noexcept
{
constexpr std::uint64_t ones = 0x0101010101010101ull;
constexpr std::uint64_t high = 0x8080808080808080ull;
std::size_t i = 0;
for (; i + 8 <= n; i += 8)
{
std::uint64_t v = 0;
std::memcpy(&v, data + i, sizeof(v));
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
const std::uint64_t d = v ^ 0x7F7F7F7F7F7F7F7Full; // DEL (0x7F)
const std::uint64_t stop = ((q - ones) & ~q & high) // == '"'
| ((b - ones) & ~b & high) // == '\\'
| ((d - ones) & ~d & high) // == 0x7F
| ((v - 0x2020202020202020ull) & ~v & high) // < 0x20
| (v & high); // >= 0x80
if (stop != 0)
{
for (std::size_t j = 0; j < 8; ++j)
{
if (!is_ascii_copyable(data[i + j]))
{
return i + j;
}
}
}
}
for (; i < n; ++i)
{
if (!is_ascii_copyable(data[i]))
{
return i;
}
}
return n;
}
// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its // Validate one UTF-8 sequence at the front of [data, data+avail). Returns its
// length (2..4) only when the bytes form a *well-formed* sequence using exactly // length (2..4) only when the bytes form a *well-formed* sequence using exactly
// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts // the same ranges as scan_string()'s per-byte switch, so the bulk path accepts
+172 -85
View File
@@ -16,6 +16,7 @@
#include <cstddef> // size_t, ptrdiff_t #include <cstddef> // size_t, ptrdiff_t
#include <cstdint> // uint8_t #include <cstdint> // uint8_t
#include <cstdio> // snprintf #include <cstdio> // snprintf
#include <cstring> // memcpy
#include <limits> // numeric_limits #include <limits> // numeric_limits
#include <string> // string, char_traits #include <string> // string, char_traits
#include <iomanip> // setfill, setw #include <iomanip> // setfill, setw
@@ -110,6 +111,25 @@ class serializer
const bool ensure_ascii, const bool ensure_ascii,
const unsigned int indent_step, const unsigned int indent_step,
const unsigned int current_indent = 0) const unsigned int current_indent = 0)
{
dump_internal(val, pretty_print, ensure_ascii, indent_step, current_indent);
flush();
}
JSON_PRIVATE_UNLESS_TESTED:
/*!
@brief recursive worker for @ref dump
Identical in behavior to the historical @ref dump, but writes into the
serializer's internal @ref write_buffer instead of issuing a virtual call
per token. The public @ref dump wraps this and flushes the buffer once the
top-level value has been serialized.
*/
void dump_internal(const BasicJsonType& val,
const bool pretty_print,
const bool ensure_ascii,
const unsigned int indent_step,
const unsigned int current_indent = 0)
{ {
switch (val.m_data.m_type) switch (val.m_data.m_type)
{ {
@@ -117,13 +137,13 @@ class serializer
{ {
if (val.m_data.m_value.object->empty()) if (val.m_data.m_value.object->empty())
{ {
o->write_characters("{}", 2); put_chars("{}", 2);
return; return;
} }
if (pretty_print) if (pretty_print)
{ {
o->write_characters("{\n", 2); put_chars("{\n", 2);
// variable to hold indentation for recursive calls // variable to hold indentation for recursive calls
const auto new_indent = current_indent + indent_step; const auto new_indent = current_indent + indent_step;
@@ -136,51 +156,51 @@ class serializer
auto i = val.m_data.m_value.object->cbegin(); auto i = val.m_data.m_value.object->cbegin();
for (std::size_t cnt = 0; cnt < val.m_data.m_value.object->size() - 1; ++cnt, ++i) for (std::size_t cnt = 0; cnt < val.m_data.m_value.object->size() - 1; ++cnt, ++i)
{ {
o->write_characters(indent_string.c_str(), new_indent); put_chars(indent_string.c_str(), new_indent);
o->write_character('\"'); put_char('\"');
dump_escaped(i->first, ensure_ascii); dump_escaped(i->first, ensure_ascii);
o->write_characters("\": ", 3); put_chars("\": ", 3);
dump(i->second, true, ensure_ascii, indent_step, new_indent); dump_internal(i->second, true, ensure_ascii, indent_step, new_indent);
o->write_characters(",\n", 2); put_chars(",\n", 2);
} }
// last element // last element
JSON_ASSERT(i != val.m_data.m_value.object->cend()); JSON_ASSERT(i != val.m_data.m_value.object->cend());
JSON_ASSERT(std::next(i) == val.m_data.m_value.object->cend()); JSON_ASSERT(std::next(i) == val.m_data.m_value.object->cend());
o->write_characters(indent_string.c_str(), new_indent); put_chars(indent_string.c_str(), new_indent);
o->write_character('\"'); put_char('\"');
dump_escaped(i->first, ensure_ascii); dump_escaped(i->first, ensure_ascii);
o->write_characters("\": ", 3); put_chars("\": ", 3);
dump(i->second, true, ensure_ascii, indent_step, new_indent); dump_internal(i->second, true, ensure_ascii, indent_step, new_indent);
o->write_character('\n'); put_char('\n');
o->write_characters(indent_string.c_str(), current_indent); put_chars(indent_string.c_str(), current_indent);
o->write_character('}'); put_char('}');
} }
else else
{ {
o->write_character('{'); put_char('{');
// first n-1 elements // first n-1 elements
auto i = val.m_data.m_value.object->cbegin(); auto i = val.m_data.m_value.object->cbegin();
for (std::size_t cnt = 0; cnt < val.m_data.m_value.object->size() - 1; ++cnt, ++i) for (std::size_t cnt = 0; cnt < val.m_data.m_value.object->size() - 1; ++cnt, ++i)
{ {
o->write_character('\"'); put_char('\"');
dump_escaped(i->first, ensure_ascii); dump_escaped(i->first, ensure_ascii);
o->write_characters("\":", 2); put_chars("\":", 2);
dump(i->second, false, ensure_ascii, indent_step, current_indent); dump_internal(i->second, false, ensure_ascii, indent_step, current_indent);
o->write_character(','); put_char(',');
} }
// last element // last element
JSON_ASSERT(i != val.m_data.m_value.object->cend()); JSON_ASSERT(i != val.m_data.m_value.object->cend());
JSON_ASSERT(std::next(i) == val.m_data.m_value.object->cend()); JSON_ASSERT(std::next(i) == val.m_data.m_value.object->cend());
o->write_character('\"'); put_char('\"');
dump_escaped(i->first, ensure_ascii); dump_escaped(i->first, ensure_ascii);
o->write_characters("\":", 2); put_chars("\":", 2);
dump(i->second, false, ensure_ascii, indent_step, current_indent); dump_internal(i->second, false, ensure_ascii, indent_step, current_indent);
o->write_character('}'); put_char('}');
} }
return; return;
@@ -190,13 +210,13 @@ class serializer
{ {
if (val.m_data.m_value.array->empty()) if (val.m_data.m_value.array->empty())
{ {
o->write_characters("[]", 2); put_chars("[]", 2);
return; return;
} }
if (pretty_print) if (pretty_print)
{ {
o->write_characters("[\n", 2); put_chars("[\n", 2);
// variable to hold indentation for recursive calls // variable to hold indentation for recursive calls
const auto new_indent = current_indent + indent_step; const auto new_indent = current_indent + indent_step;
@@ -209,37 +229,37 @@ class serializer
for (auto i = val.m_data.m_value.array->cbegin(); for (auto i = val.m_data.m_value.array->cbegin();
i != val.m_data.m_value.array->cend() - 1; ++i) i != val.m_data.m_value.array->cend() - 1; ++i)
{ {
o->write_characters(indent_string.c_str(), new_indent); put_chars(indent_string.c_str(), new_indent);
dump(*i, true, ensure_ascii, indent_step, new_indent); dump_internal(*i, true, ensure_ascii, indent_step, new_indent);
o->write_characters(",\n", 2); put_chars(",\n", 2);
} }
// last element // last element
JSON_ASSERT(!val.m_data.m_value.array->empty()); JSON_ASSERT(!val.m_data.m_value.array->empty());
o->write_characters(indent_string.c_str(), new_indent); put_chars(indent_string.c_str(), new_indent);
dump(val.m_data.m_value.array->back(), true, ensure_ascii, indent_step, new_indent); dump_internal(val.m_data.m_value.array->back(), true, ensure_ascii, indent_step, new_indent);
o->write_character('\n'); put_char('\n');
o->write_characters(indent_string.c_str(), current_indent); put_chars(indent_string.c_str(), current_indent);
o->write_character(']'); put_char(']');
} }
else else
{ {
o->write_character('['); put_char('[');
// first n-1 elements // first n-1 elements
for (auto i = val.m_data.m_value.array->cbegin(); for (auto i = val.m_data.m_value.array->cbegin();
i != val.m_data.m_value.array->cend() - 1; ++i) i != val.m_data.m_value.array->cend() - 1; ++i)
{ {
dump(*i, false, ensure_ascii, indent_step, current_indent); dump_internal(*i, false, ensure_ascii, indent_step, current_indent);
o->write_character(','); put_char(',');
} }
// last element // last element
JSON_ASSERT(!val.m_data.m_value.array->empty()); JSON_ASSERT(!val.m_data.m_value.array->empty());
dump(val.m_data.m_value.array->back(), false, ensure_ascii, indent_step, current_indent); dump_internal(val.m_data.m_value.array->back(), false, ensure_ascii, indent_step, current_indent);
o->write_character(']'); put_char(']');
} }
return; return;
@@ -247,9 +267,9 @@ class serializer
case value_t::string: case value_t::string:
{ {
o->write_character('\"'); put_char('\"');
dump_escaped(*val.m_data.m_value.string, ensure_ascii); dump_escaped(*val.m_data.m_value.string, ensure_ascii);
o->write_character('\"'); put_char('\"');
return; return;
} }
@@ -257,7 +277,7 @@ class serializer
{ {
if (pretty_print) if (pretty_print)
{ {
o->write_characters("{\n", 2); put_chars("{\n", 2);
// variable to hold indentation for recursive calls // variable to hold indentation for recursive calls
const auto new_indent = current_indent + indent_step; const auto new_indent = current_indent + indent_step;
@@ -266,9 +286,9 @@ class serializer
indent_string.resize(indent_string.size() * 2, ' '); indent_string.resize(indent_string.size() * 2, ' ');
} }
o->write_characters(indent_string.c_str(), new_indent); put_chars(indent_string.c_str(), new_indent);
o->write_characters("\"bytes\": [", 10); put_chars("\"bytes\": [", 10);
if (!val.m_data.m_value.binary->empty()) if (!val.m_data.m_value.binary->empty())
{ {
@@ -276,30 +296,30 @@ class serializer
i != val.m_data.m_value.binary->cend() - 1; ++i) i != val.m_data.m_value.binary->cend() - 1; ++i)
{ {
dump_integer(*i); dump_integer(*i);
o->write_characters(", ", 2); put_chars(", ", 2);
} }
dump_integer(val.m_data.m_value.binary->back()); dump_integer(val.m_data.m_value.binary->back());
} }
o->write_characters("],\n", 3); put_chars("],\n", 3);
o->write_characters(indent_string.c_str(), new_indent); put_chars(indent_string.c_str(), new_indent);
o->write_characters("\"subtype\": ", 11); put_chars("\"subtype\": ", 11);
if (val.m_data.m_value.binary->has_subtype()) if (val.m_data.m_value.binary->has_subtype())
{ {
dump_integer(val.m_data.m_value.binary->subtype()); dump_integer(val.m_data.m_value.binary->subtype());
} }
else else
{ {
o->write_characters("null", 4); put_chars("null", 4);
} }
o->write_character('\n'); put_char('\n');
o->write_characters(indent_string.c_str(), current_indent); put_chars(indent_string.c_str(), current_indent);
o->write_character('}'); put_char('}');
} }
else else
{ {
o->write_characters("{\"bytes\":[", 10); put_chars("{\"bytes\":[", 10);
if (!val.m_data.m_value.binary->empty()) if (!val.m_data.m_value.binary->empty())
{ {
@@ -307,20 +327,20 @@ class serializer
i != val.m_data.m_value.binary->cend() - 1; ++i) i != val.m_data.m_value.binary->cend() - 1; ++i)
{ {
dump_integer(*i); dump_integer(*i);
o->write_character(','); put_char(',');
} }
dump_integer(val.m_data.m_value.binary->back()); dump_integer(val.m_data.m_value.binary->back());
} }
o->write_characters("],\"subtype\":", 12); put_chars("],\"subtype\":", 12);
if (val.m_data.m_value.binary->has_subtype()) if (val.m_data.m_value.binary->has_subtype())
{ {
dump_integer(val.m_data.m_value.binary->subtype()); dump_integer(val.m_data.m_value.binary->subtype());
o->write_character('}'); put_char('}');
} }
else else
{ {
o->write_characters("null}", 5); put_chars("null}", 5);
} }
} }
return; return;
@@ -330,11 +350,11 @@ class serializer
{ {
if (val.m_data.m_value.boolean) if (val.m_data.m_value.boolean)
{ {
o->write_characters("true", 4); put_chars("true", 4);
} }
else else
{ {
o->write_characters("false", 5); put_chars("false", 5);
} }
return; return;
} }
@@ -359,13 +379,13 @@ class serializer
case value_t::discarded: case value_t::discarded:
{ {
o->write_characters("<discarded>", 11); put_chars("<discarded>", 11);
return; return;
} }
case value_t::null: case value_t::null:
{ {
o->write_characters("null", 4); put_chars("null", 4);
return; return;
} }
@@ -401,28 +421,35 @@ class serializer
for (std::size_t i = 0; i < s.size(); ++i) for (std::size_t i = 0; i < s.size(); ++i)
{ {
// Fast path: when not escaping non-ASCII characters and sitting on a // Fast path: at a character boundary (state == UTF8_ACCEPT),
// character boundary (state == UTF8_ACCEPT), bulk-copy the longest // bulk-copy the longest run of bytes that need no escaping using a
// run of bytes that need no escaping. string_bulk_run() (shared with // SWAR scanner shared with the lexer's contiguous path. The scanner
// the lexer's contiguous scanner) stops exactly at the first byte // stops exactly at the first byte dump_escaped would handle
// that dump_escaped would handle individually - a quote, a backslash, // individually, so that byte is left to the byte-at-a-time path
// a control character (< 0x20), or an ill-formed/truncated UTF-8 // below, keeping escaping output and error diagnostics unchanged.
// sequence - so that byte is left to the byte-at-a-time path below, //
// keeping error handling and diagnostics unchanged. // - ensure_ascii == false: string_bulk_run() copies ordinary bytes
if (!ensure_ascii && state == UTF8_ACCEPT) // and complete well-formed UTF-8, stopping at a quote, backslash,
// control character (< 0x20), or ill-formed/truncated sequence.
// - ensure_ascii == true: only printable ASCII may be copied
// verbatim; find_ascii_copyable_run() additionally stops at 0x7F
// and every non-ASCII byte (>= 0x80), which must be \u-escaped.
if (state == UTF8_ACCEPT)
{ {
const auto* const data = reinterpret_cast<const unsigned char*>(s.data()); const auto* const data = reinterpret_cast<const unsigned char*>(s.data());
const std::size_t run = string_bulk_run(data + i, s.size() - i); const std::size_t run = ensure_ascii
? find_ascii_copyable_run(data + i, s.size() - i)
: string_bulk_run(data + i, s.size() - i);
if (run != 0) if (run != 0)
{ {
// emit any bytes still pending in string_buffer first to // emit any bytes still pending in string_buffer first to
// preserve output order, then write the run directly // preserve output order, then write the run directly
if (bytes != 0) if (bytes != 0)
{ {
o->write_characters(string_buffer.data(), bytes); put_chars(string_buffer.data(), bytes);
bytes = 0; bytes = 0;
} }
o->write_characters(s.data() + i, run); put_chars(s.data() + i, run);
bytes_after_last_accept = 0; bytes_after_last_accept = 0;
undumped_chars = 0; undumped_chars = 0;
i += run; i += run;
@@ -521,7 +548,7 @@ class serializer
// written ("\uxxxx\uxxxx\0") for one code point // written ("\uxxxx\uxxxx\0") for one code point
if (string_buffer.size() - bytes < 13) if (string_buffer.size() - bytes < 13)
{ {
o->write_characters(string_buffer.data(), bytes); put_chars(string_buffer.data(), bytes);
bytes = 0; bytes = 0;
} }
@@ -580,7 +607,7 @@ class serializer
// written ("\uxxxx\uxxxx\0") for one code point // written ("\uxxxx\uxxxx\0") for one code point
if (string_buffer.size() - bytes < 13) if (string_buffer.size() - bytes < 13)
{ {
o->write_characters(string_buffer.data(), bytes); put_chars(string_buffer.data(), bytes);
bytes = 0; bytes = 0;
} }
@@ -619,7 +646,7 @@ class serializer
// write buffer // write buffer
if (bytes > 0) if (bytes > 0)
{ {
o->write_characters(string_buffer.data(), bytes); put_chars(string_buffer.data(), bytes);
} }
} }
else else
@@ -635,22 +662,22 @@ class serializer
case error_handler_t::ignore: case error_handler_t::ignore:
{ {
// write all accepted bytes // write all accepted bytes
o->write_characters(string_buffer.data(), bytes_after_last_accept); put_chars(string_buffer.data(), bytes_after_last_accept);
break; break;
} }
case error_handler_t::replace: case error_handler_t::replace:
{ {
// write all accepted bytes // write all accepted bytes
o->write_characters(string_buffer.data(), bytes_after_last_accept); put_chars(string_buffer.data(), bytes_after_last_accept);
// add a replacement character // add a replacement character
if (ensure_ascii) if (ensure_ascii)
{ {
o->write_characters("\\ufffd", 6); put_chars("\\ufffd", 6);
} }
else else
{ {
o->write_characters("\xEF\xBF\xBD", 3); put_chars("\xEF\xBF\xBD", 3);
} }
break; break;
} }
@@ -662,6 +689,60 @@ class serializer
} }
private: private:
/*!
@brief append a single character to the write buffer
Structural characters ('{', '"', ',', ...) previously went straight to the
output adapter, one virtual call each. Buffering them and flushing in bulk
turns those many indirect calls into a single memcpy plus an occasional
flush, which dominates the cost of serializing object/array-heavy values.
*/
void put_char(char c)
{
if (JSON_HEDLEY_UNLIKELY(write_buffer_pos == write_buffer.size()))
{
flush();
}
write_buffer[write_buffer_pos++] = c;
}
/*!
@brief append @a length characters to the write buffer
Runs that do not fit the buffer are written straight through the output
adapter (after flushing what is pending), so large string/number payloads
are not copied an extra time.
*/
JSON_HEDLEY_NON_NULL(2)
void put_chars(const char* s, std::size_t length)
{
if (JSON_HEDLEY_UNLIKELY(length >= write_buffer.size()))
{
flush();
o->write_characters(s, length);
return;
}
if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + length > write_buffer.size()))
{
flush();
}
std::memcpy(write_buffer.data() + write_buffer_pos, s, length);
write_buffer_pos += length;
}
/*!
@brief flush the write buffer to the output adapter
Writing zero characters is a well-defined no-op for every output adapter, so
the buffered length is passed through unconditionally (no empty-guard branch
to leave uncovered).
*/
void flush()
{
o->write_characters(write_buffer.data(), write_buffer_pos);
write_buffer_pos = 0;
}
/*! /*!
@brief count digits @brief count digits
@@ -785,7 +866,7 @@ class serializer
// special case for "0" // special case for "0"
if (x == 0) if (x == 0)
{ {
o->write_character('0'); put_char('0');
return; return;
} }
@@ -838,7 +919,7 @@ class serializer
*(--buffer_ptr) = static_cast<char>('0' + abs_value); *(--buffer_ptr) = static_cast<char>('0' + abs_value);
} }
o->write_characters(number_buffer.data(), n_chars); put_chars(number_buffer.data(), n_chars);
} }
/*! /*!
@@ -854,7 +935,7 @@ class serializer
// NaN / inf // NaN / inf
if (!std::isfinite(x)) if (!std::isfinite(x))
{ {
o->write_characters("null", 4); put_chars("null", 4);
return; return;
} }
@@ -875,7 +956,7 @@ class serializer
auto* begin = number_buffer.data(); auto* begin = number_buffer.data();
auto* end = ::nlohmann::detail::to_chars(begin, begin + number_buffer.size(), x); auto* end = ::nlohmann::detail::to_chars(begin, begin + number_buffer.size(), x);
o->write_characters(begin, static_cast<size_t>(end - begin)); put_chars(begin, static_cast<size_t>(end - begin));
} }
JSON_HEDLEY_NON_NULL(1) JSON_HEDLEY_NON_NULL(1)
@@ -926,7 +1007,7 @@ class serializer
} }
} }
o->write_characters(number_buffer.data(), static_cast<std::size_t>(len)); put_chars(number_buffer.data(), static_cast<std::size_t>(len));
// determine if we need to append ".0" // determine if we need to append ".0"
const bool value_is_int_like = const bool value_is_int_like =
@@ -938,7 +1019,7 @@ class serializer
if (value_is_int_like) if (value_is_int_like)
{ {
o->write_characters(".0", 2); put_chars(".0", 2);
} }
} }
@@ -1048,6 +1129,12 @@ class serializer
/// error_handler how to react on decoding errors /// error_handler how to react on decoding errors
const error_handler_t error_handler; const error_handler_t error_handler;
/// buffer collecting output before it is flushed to the output adapter, so
/// that the many small structural writes become few bulk writes
std::array<char, 1024> write_buffer{{}};
/// number of valid bytes currently held in @ref write_buffer
std::size_t write_buffer_pos = 0;
}; };
} // namespace detail } // namespace detail
+224 -85
View File
@@ -8193,6 +8193,58 @@ inline std::size_t find_string_special(const unsigned char* data, std::size_t n)
return n; return n;
} }
// classify a byte as one the serializer must NOT copy verbatim when
// ensure_ascii is requested: the closing quote, an escape, a control character
// (< 0x20), DEL (0x7F), or any non-ASCII byte (>= 0x80). Everything else -
// printable ASCII except '"' and '\\' - is emitted unchanged. Note this differs
// from is_string_special() only in that 0x7F is also a stop (it is escaped as
// \u007f under ensure_ascii).
inline bool is_ascii_copyable(unsigned char c) noexcept
{
return c >= 0x20u && c < 0x7Fu && c != '\"' && c != '\\';
}
// return the index of the first byte in [data, data+n) that is NOT
// is_ascii_copyable(), or n if every byte can be copied verbatim; scans 8 bytes
// at a time. Used by the serializer's ensure_ascii fast path.
inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_t n) noexcept
{
constexpr std::uint64_t ones = 0x0101010101010101ull;
constexpr std::uint64_t high = 0x8080808080808080ull;
std::size_t i = 0;
for (; i + 8 <= n; i += 8)
{
std::uint64_t v = 0;
std::memcpy(&v, data + i, sizeof(v));
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
const std::uint64_t d = v ^ 0x7F7F7F7F7F7F7F7Full; // DEL (0x7F)
const std::uint64_t stop = ((q - ones) & ~q & high) // == '"'
| ((b - ones) & ~b & high) // == '\\'
| ((d - ones) & ~d & high) // == 0x7F
| ((v - 0x2020202020202020ull) & ~v & high) // < 0x20
| (v & high); // >= 0x80
if (stop != 0)
{
for (std::size_t j = 0; j < 8; ++j)
{
if (!is_ascii_copyable(data[i + j]))
{
return i + j;
}
}
}
}
for (; i < n; ++i)
{
if (!is_ascii_copyable(data[i]))
{
return i;
}
}
return n;
}
// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its // Validate one UTF-8 sequence at the front of [data, data+avail). Returns its
// length (2..4) only when the bytes form a *well-formed* sequence using exactly // length (2..4) only when the bytes form a *well-formed* sequence using exactly
// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts // the same ranges as scan_string()'s per-byte switch, so the bulk path accepts
@@ -19639,6 +19691,7 @@ NLOHMANN_JSON_NAMESPACE_END
#include <cstddef> // size_t, ptrdiff_t #include <cstddef> // size_t, ptrdiff_t
#include <cstdint> // uint8_t #include <cstdint> // uint8_t
#include <cstdio> // snprintf #include <cstdio> // snprintf
#include <cstring> // memcpy
#include <limits> // numeric_limits #include <limits> // numeric_limits
#include <string> // string, char_traits #include <string> // string, char_traits
#include <iomanip> // setfill, setw #include <iomanip> // setfill, setw
@@ -20861,6 +20914,25 @@ class serializer
const bool ensure_ascii, const bool ensure_ascii,
const unsigned int indent_step, const unsigned int indent_step,
const unsigned int current_indent = 0) const unsigned int current_indent = 0)
{
dump_internal(val, pretty_print, ensure_ascii, indent_step, current_indent);
flush();
}
JSON_PRIVATE_UNLESS_TESTED:
/*!
@brief recursive worker for @ref dump
Identical in behavior to the historical @ref dump, but writes into the
serializer's internal @ref write_buffer instead of issuing a virtual call
per token. The public @ref dump wraps this and flushes the buffer once the
top-level value has been serialized.
*/
void dump_internal(const BasicJsonType& val,
const bool pretty_print,
const bool ensure_ascii,
const unsigned int indent_step,
const unsigned int current_indent = 0)
{ {
switch (val.m_data.m_type) switch (val.m_data.m_type)
{ {
@@ -20868,13 +20940,13 @@ class serializer
{ {
if (val.m_data.m_value.object->empty()) if (val.m_data.m_value.object->empty())
{ {
o->write_characters("{}", 2); put_chars("{}", 2);
return; return;
} }
if (pretty_print) if (pretty_print)
{ {
o->write_characters("{\n", 2); put_chars("{\n", 2);
// variable to hold indentation for recursive calls // variable to hold indentation for recursive calls
const auto new_indent = current_indent + indent_step; const auto new_indent = current_indent + indent_step;
@@ -20887,51 +20959,51 @@ class serializer
auto i = val.m_data.m_value.object->cbegin(); auto i = val.m_data.m_value.object->cbegin();
for (std::size_t cnt = 0; cnt < val.m_data.m_value.object->size() - 1; ++cnt, ++i) for (std::size_t cnt = 0; cnt < val.m_data.m_value.object->size() - 1; ++cnt, ++i)
{ {
o->write_characters(indent_string.c_str(), new_indent); put_chars(indent_string.c_str(), new_indent);
o->write_character('\"'); put_char('\"');
dump_escaped(i->first, ensure_ascii); dump_escaped(i->first, ensure_ascii);
o->write_characters("\": ", 3); put_chars("\": ", 3);
dump(i->second, true, ensure_ascii, indent_step, new_indent); dump_internal(i->second, true, ensure_ascii, indent_step, new_indent);
o->write_characters(",\n", 2); put_chars(",\n", 2);
} }
// last element // last element
JSON_ASSERT(i != val.m_data.m_value.object->cend()); JSON_ASSERT(i != val.m_data.m_value.object->cend());
JSON_ASSERT(std::next(i) == val.m_data.m_value.object->cend()); JSON_ASSERT(std::next(i) == val.m_data.m_value.object->cend());
o->write_characters(indent_string.c_str(), new_indent); put_chars(indent_string.c_str(), new_indent);
o->write_character('\"'); put_char('\"');
dump_escaped(i->first, ensure_ascii); dump_escaped(i->first, ensure_ascii);
o->write_characters("\": ", 3); put_chars("\": ", 3);
dump(i->second, true, ensure_ascii, indent_step, new_indent); dump_internal(i->second, true, ensure_ascii, indent_step, new_indent);
o->write_character('\n'); put_char('\n');
o->write_characters(indent_string.c_str(), current_indent); put_chars(indent_string.c_str(), current_indent);
o->write_character('}'); put_char('}');
} }
else else
{ {
o->write_character('{'); put_char('{');
// first n-1 elements // first n-1 elements
auto i = val.m_data.m_value.object->cbegin(); auto i = val.m_data.m_value.object->cbegin();
for (std::size_t cnt = 0; cnt < val.m_data.m_value.object->size() - 1; ++cnt, ++i) for (std::size_t cnt = 0; cnt < val.m_data.m_value.object->size() - 1; ++cnt, ++i)
{ {
o->write_character('\"'); put_char('\"');
dump_escaped(i->first, ensure_ascii); dump_escaped(i->first, ensure_ascii);
o->write_characters("\":", 2); put_chars("\":", 2);
dump(i->second, false, ensure_ascii, indent_step, current_indent); dump_internal(i->second, false, ensure_ascii, indent_step, current_indent);
o->write_character(','); put_char(',');
} }
// last element // last element
JSON_ASSERT(i != val.m_data.m_value.object->cend()); JSON_ASSERT(i != val.m_data.m_value.object->cend());
JSON_ASSERT(std::next(i) == val.m_data.m_value.object->cend()); JSON_ASSERT(std::next(i) == val.m_data.m_value.object->cend());
o->write_character('\"'); put_char('\"');
dump_escaped(i->first, ensure_ascii); dump_escaped(i->first, ensure_ascii);
o->write_characters("\":", 2); put_chars("\":", 2);
dump(i->second, false, ensure_ascii, indent_step, current_indent); dump_internal(i->second, false, ensure_ascii, indent_step, current_indent);
o->write_character('}'); put_char('}');
} }
return; return;
@@ -20941,13 +21013,13 @@ class serializer
{ {
if (val.m_data.m_value.array->empty()) if (val.m_data.m_value.array->empty())
{ {
o->write_characters("[]", 2); put_chars("[]", 2);
return; return;
} }
if (pretty_print) if (pretty_print)
{ {
o->write_characters("[\n", 2); put_chars("[\n", 2);
// variable to hold indentation for recursive calls // variable to hold indentation for recursive calls
const auto new_indent = current_indent + indent_step; const auto new_indent = current_indent + indent_step;
@@ -20960,37 +21032,37 @@ class serializer
for (auto i = val.m_data.m_value.array->cbegin(); for (auto i = val.m_data.m_value.array->cbegin();
i != val.m_data.m_value.array->cend() - 1; ++i) i != val.m_data.m_value.array->cend() - 1; ++i)
{ {
o->write_characters(indent_string.c_str(), new_indent); put_chars(indent_string.c_str(), new_indent);
dump(*i, true, ensure_ascii, indent_step, new_indent); dump_internal(*i, true, ensure_ascii, indent_step, new_indent);
o->write_characters(",\n", 2); put_chars(",\n", 2);
} }
// last element // last element
JSON_ASSERT(!val.m_data.m_value.array->empty()); JSON_ASSERT(!val.m_data.m_value.array->empty());
o->write_characters(indent_string.c_str(), new_indent); put_chars(indent_string.c_str(), new_indent);
dump(val.m_data.m_value.array->back(), true, ensure_ascii, indent_step, new_indent); dump_internal(val.m_data.m_value.array->back(), true, ensure_ascii, indent_step, new_indent);
o->write_character('\n'); put_char('\n');
o->write_characters(indent_string.c_str(), current_indent); put_chars(indent_string.c_str(), current_indent);
o->write_character(']'); put_char(']');
} }
else else
{ {
o->write_character('['); put_char('[');
// first n-1 elements // first n-1 elements
for (auto i = val.m_data.m_value.array->cbegin(); for (auto i = val.m_data.m_value.array->cbegin();
i != val.m_data.m_value.array->cend() - 1; ++i) i != val.m_data.m_value.array->cend() - 1; ++i)
{ {
dump(*i, false, ensure_ascii, indent_step, current_indent); dump_internal(*i, false, ensure_ascii, indent_step, current_indent);
o->write_character(','); put_char(',');
} }
// last element // last element
JSON_ASSERT(!val.m_data.m_value.array->empty()); JSON_ASSERT(!val.m_data.m_value.array->empty());
dump(val.m_data.m_value.array->back(), false, ensure_ascii, indent_step, current_indent); dump_internal(val.m_data.m_value.array->back(), false, ensure_ascii, indent_step, current_indent);
o->write_character(']'); put_char(']');
} }
return; return;
@@ -20998,9 +21070,9 @@ class serializer
case value_t::string: case value_t::string:
{ {
o->write_character('\"'); put_char('\"');
dump_escaped(*val.m_data.m_value.string, ensure_ascii); dump_escaped(*val.m_data.m_value.string, ensure_ascii);
o->write_character('\"'); put_char('\"');
return; return;
} }
@@ -21008,7 +21080,7 @@ class serializer
{ {
if (pretty_print) if (pretty_print)
{ {
o->write_characters("{\n", 2); put_chars("{\n", 2);
// variable to hold indentation for recursive calls // variable to hold indentation for recursive calls
const auto new_indent = current_indent + indent_step; const auto new_indent = current_indent + indent_step;
@@ -21017,9 +21089,9 @@ class serializer
indent_string.resize(indent_string.size() * 2, ' '); indent_string.resize(indent_string.size() * 2, ' ');
} }
o->write_characters(indent_string.c_str(), new_indent); put_chars(indent_string.c_str(), new_indent);
o->write_characters("\"bytes\": [", 10); put_chars("\"bytes\": [", 10);
if (!val.m_data.m_value.binary->empty()) if (!val.m_data.m_value.binary->empty())
{ {
@@ -21027,30 +21099,30 @@ class serializer
i != val.m_data.m_value.binary->cend() - 1; ++i) i != val.m_data.m_value.binary->cend() - 1; ++i)
{ {
dump_integer(*i); dump_integer(*i);
o->write_characters(", ", 2); put_chars(", ", 2);
} }
dump_integer(val.m_data.m_value.binary->back()); dump_integer(val.m_data.m_value.binary->back());
} }
o->write_characters("],\n", 3); put_chars("],\n", 3);
o->write_characters(indent_string.c_str(), new_indent); put_chars(indent_string.c_str(), new_indent);
o->write_characters("\"subtype\": ", 11); put_chars("\"subtype\": ", 11);
if (val.m_data.m_value.binary->has_subtype()) if (val.m_data.m_value.binary->has_subtype())
{ {
dump_integer(val.m_data.m_value.binary->subtype()); dump_integer(val.m_data.m_value.binary->subtype());
} }
else else
{ {
o->write_characters("null", 4); put_chars("null", 4);
} }
o->write_character('\n'); put_char('\n');
o->write_characters(indent_string.c_str(), current_indent); put_chars(indent_string.c_str(), current_indent);
o->write_character('}'); put_char('}');
} }
else else
{ {
o->write_characters("{\"bytes\":[", 10); put_chars("{\"bytes\":[", 10);
if (!val.m_data.m_value.binary->empty()) if (!val.m_data.m_value.binary->empty())
{ {
@@ -21058,20 +21130,20 @@ class serializer
i != val.m_data.m_value.binary->cend() - 1; ++i) i != val.m_data.m_value.binary->cend() - 1; ++i)
{ {
dump_integer(*i); dump_integer(*i);
o->write_character(','); put_char(',');
} }
dump_integer(val.m_data.m_value.binary->back()); dump_integer(val.m_data.m_value.binary->back());
} }
o->write_characters("],\"subtype\":", 12); put_chars("],\"subtype\":", 12);
if (val.m_data.m_value.binary->has_subtype()) if (val.m_data.m_value.binary->has_subtype())
{ {
dump_integer(val.m_data.m_value.binary->subtype()); dump_integer(val.m_data.m_value.binary->subtype());
o->write_character('}'); put_char('}');
} }
else else
{ {
o->write_characters("null}", 5); put_chars("null}", 5);
} }
} }
return; return;
@@ -21081,11 +21153,11 @@ class serializer
{ {
if (val.m_data.m_value.boolean) if (val.m_data.m_value.boolean)
{ {
o->write_characters("true", 4); put_chars("true", 4);
} }
else else
{ {
o->write_characters("false", 5); put_chars("false", 5);
} }
return; return;
} }
@@ -21110,13 +21182,13 @@ class serializer
case value_t::discarded: case value_t::discarded:
{ {
o->write_characters("<discarded>", 11); put_chars("<discarded>", 11);
return; return;
} }
case value_t::null: case value_t::null:
{ {
o->write_characters("null", 4); put_chars("null", 4);
return; return;
} }
@@ -21152,28 +21224,35 @@ class serializer
for (std::size_t i = 0; i < s.size(); ++i) for (std::size_t i = 0; i < s.size(); ++i)
{ {
// Fast path: when not escaping non-ASCII characters and sitting on a // Fast path: at a character boundary (state == UTF8_ACCEPT),
// character boundary (state == UTF8_ACCEPT), bulk-copy the longest // bulk-copy the longest run of bytes that need no escaping using a
// run of bytes that need no escaping. string_bulk_run() (shared with // SWAR scanner shared with the lexer's contiguous path. The scanner
// the lexer's contiguous scanner) stops exactly at the first byte // stops exactly at the first byte dump_escaped would handle
// that dump_escaped would handle individually - a quote, a backslash, // individually, so that byte is left to the byte-at-a-time path
// a control character (< 0x20), or an ill-formed/truncated UTF-8 // below, keeping escaping output and error diagnostics unchanged.
// sequence - so that byte is left to the byte-at-a-time path below, //
// keeping error handling and diagnostics unchanged. // - ensure_ascii == false: string_bulk_run() copies ordinary bytes
if (!ensure_ascii && state == UTF8_ACCEPT) // and complete well-formed UTF-8, stopping at a quote, backslash,
// control character (< 0x20), or ill-formed/truncated sequence.
// - ensure_ascii == true: only printable ASCII may be copied
// verbatim; find_ascii_copyable_run() additionally stops at 0x7F
// and every non-ASCII byte (>= 0x80), which must be \u-escaped.
if (state == UTF8_ACCEPT)
{ {
const auto* const data = reinterpret_cast<const unsigned char*>(s.data()); const auto* const data = reinterpret_cast<const unsigned char*>(s.data());
const std::size_t run = string_bulk_run(data + i, s.size() - i); const std::size_t run = ensure_ascii
? find_ascii_copyable_run(data + i, s.size() - i)
: string_bulk_run(data + i, s.size() - i);
if (run != 0) if (run != 0)
{ {
// emit any bytes still pending in string_buffer first to // emit any bytes still pending in string_buffer first to
// preserve output order, then write the run directly // preserve output order, then write the run directly
if (bytes != 0) if (bytes != 0)
{ {
o->write_characters(string_buffer.data(), bytes); put_chars(string_buffer.data(), bytes);
bytes = 0; bytes = 0;
} }
o->write_characters(s.data() + i, run); put_chars(s.data() + i, run);
bytes_after_last_accept = 0; bytes_after_last_accept = 0;
undumped_chars = 0; undumped_chars = 0;
i += run; i += run;
@@ -21272,7 +21351,7 @@ class serializer
// written ("\uxxxx\uxxxx\0") for one code point // written ("\uxxxx\uxxxx\0") for one code point
if (string_buffer.size() - bytes < 13) if (string_buffer.size() - bytes < 13)
{ {
o->write_characters(string_buffer.data(), bytes); put_chars(string_buffer.data(), bytes);
bytes = 0; bytes = 0;
} }
@@ -21331,7 +21410,7 @@ class serializer
// written ("\uxxxx\uxxxx\0") for one code point // written ("\uxxxx\uxxxx\0") for one code point
if (string_buffer.size() - bytes < 13) if (string_buffer.size() - bytes < 13)
{ {
o->write_characters(string_buffer.data(), bytes); put_chars(string_buffer.data(), bytes);
bytes = 0; bytes = 0;
} }
@@ -21370,7 +21449,7 @@ class serializer
// write buffer // write buffer
if (bytes > 0) if (bytes > 0)
{ {
o->write_characters(string_buffer.data(), bytes); put_chars(string_buffer.data(), bytes);
} }
} }
else else
@@ -21386,22 +21465,22 @@ class serializer
case error_handler_t::ignore: case error_handler_t::ignore:
{ {
// write all accepted bytes // write all accepted bytes
o->write_characters(string_buffer.data(), bytes_after_last_accept); put_chars(string_buffer.data(), bytes_after_last_accept);
break; break;
} }
case error_handler_t::replace: case error_handler_t::replace:
{ {
// write all accepted bytes // write all accepted bytes
o->write_characters(string_buffer.data(), bytes_after_last_accept); put_chars(string_buffer.data(), bytes_after_last_accept);
// add a replacement character // add a replacement character
if (ensure_ascii) if (ensure_ascii)
{ {
o->write_characters("\\ufffd", 6); put_chars("\\ufffd", 6);
} }
else else
{ {
o->write_characters("\xEF\xBF\xBD", 3); put_chars("\xEF\xBF\xBD", 3);
} }
break; break;
} }
@@ -21413,6 +21492,60 @@ class serializer
} }
private: private:
/*!
@brief append a single character to the write buffer
Structural characters ('{', '"', ',', ...) previously went straight to the
output adapter, one virtual call each. Buffering them and flushing in bulk
turns those many indirect calls into a single memcpy plus an occasional
flush, which dominates the cost of serializing object/array-heavy values.
*/
void put_char(char c)
{
if (JSON_HEDLEY_UNLIKELY(write_buffer_pos == write_buffer.size()))
{
flush();
}
write_buffer[write_buffer_pos++] = c;
}
/*!
@brief append @a length characters to the write buffer
Runs that do not fit the buffer are written straight through the output
adapter (after flushing what is pending), so large string/number payloads
are not copied an extra time.
*/
JSON_HEDLEY_NON_NULL(2)
void put_chars(const char* s, std::size_t length)
{
if (JSON_HEDLEY_UNLIKELY(length >= write_buffer.size()))
{
flush();
o->write_characters(s, length);
return;
}
if (JSON_HEDLEY_UNLIKELY(write_buffer_pos + length > write_buffer.size()))
{
flush();
}
std::memcpy(write_buffer.data() + write_buffer_pos, s, length);
write_buffer_pos += length;
}
/*!
@brief flush the write buffer to the output adapter
Writing zero characters is a well-defined no-op for every output adapter, so
the buffered length is passed through unconditionally (no empty-guard branch
to leave uncovered).
*/
void flush()
{
o->write_characters(write_buffer.data(), write_buffer_pos);
write_buffer_pos = 0;
}
/*! /*!
@brief count digits @brief count digits
@@ -21536,7 +21669,7 @@ class serializer
// special case for "0" // special case for "0"
if (x == 0) if (x == 0)
{ {
o->write_character('0'); put_char('0');
return; return;
} }
@@ -21589,7 +21722,7 @@ class serializer
*(--buffer_ptr) = static_cast<char>('0' + abs_value); *(--buffer_ptr) = static_cast<char>('0' + abs_value);
} }
o->write_characters(number_buffer.data(), n_chars); put_chars(number_buffer.data(), n_chars);
} }
/*! /*!
@@ -21605,7 +21738,7 @@ class serializer
// NaN / inf // NaN / inf
if (!std::isfinite(x)) if (!std::isfinite(x))
{ {
o->write_characters("null", 4); put_chars("null", 4);
return; return;
} }
@@ -21626,7 +21759,7 @@ class serializer
auto* begin = number_buffer.data(); auto* begin = number_buffer.data();
auto* end = ::nlohmann::detail::to_chars(begin, begin + number_buffer.size(), x); auto* end = ::nlohmann::detail::to_chars(begin, begin + number_buffer.size(), x);
o->write_characters(begin, static_cast<size_t>(end - begin)); put_chars(begin, static_cast<size_t>(end - begin));
} }
JSON_HEDLEY_NON_NULL(1) JSON_HEDLEY_NON_NULL(1)
@@ -21677,7 +21810,7 @@ class serializer
} }
} }
o->write_characters(number_buffer.data(), static_cast<std::size_t>(len)); put_chars(number_buffer.data(), static_cast<std::size_t>(len));
// determine if we need to append ".0" // determine if we need to append ".0"
const bool value_is_int_like = const bool value_is_int_like =
@@ -21689,7 +21822,7 @@ class serializer
if (value_is_int_like) if (value_is_int_like)
{ {
o->write_characters(".0", 2); put_chars(".0", 2);
} }
} }
@@ -21799,6 +21932,12 @@ class serializer
/// error_handler how to react on decoding errors /// error_handler how to react on decoding errors
const error_handler_t error_handler; const error_handler_t error_handler;
/// buffer collecting output before it is flushed to the output adapter, so
/// that the many small structural writes become few bulk writes
std::array<char, 1024> write_buffer{{}};
/// number of valid bytes currently held in @ref write_buffer
std::size_t write_buffer_pos = 0;
}; };
} // namespace detail } // namespace detail
+84
View File
@@ -382,3 +382,87 @@ TEST_CASE("dump for basic_json with long double number_float_t")
check_same(100.0L, 100.0); check_same(100.0L, 100.0);
} }
} }
TEST_CASE("serialization of strings (bulk fast path)")
{
// These cases exercise the SWAR bulk-copy fast path in dump_escaped and the
// internal write buffer: long runs, escapes interrupting runs, 0x7F/DEL,
// multibyte UTF-8 under both ensure_ascii settings, and payloads larger than
// the write buffer.
SECTION("long unescaped ASCII exceeds the write buffer")
{
const std::string big(3000, 'a');
const json j = big;
CHECK(j.dump() == '"' + big + '"');
CHECK(j.dump(-1, ' ', true) == '"' + big + '"');
// round-trips
CHECK(json::parse(j.dump()) == j);
}
SECTION("runs interrupted by escapes")
{
const json j = std::string(500, 'x') + "\n\"\\" + std::string(500, 'y');
const std::string out = j.dump();
CHECK(out == '"' + std::string(500, 'x') + "\\n\\\"\\\\" + std::string(500, 'y') + '"');
CHECK(json::parse(out) == j);
}
SECTION("DEL (0x7F) depends on ensure_ascii")
{
const json j = std::string("a\x7f" "b");
CHECK(j.dump(-1, ' ', false) == "\"a\x7f" "b\""); // copied verbatim
CHECK(j.dump(-1, ' ', true) == "\"a\\u007fb\""); // escaped
}
SECTION("multibyte UTF-8 under both ensure_ascii settings")
{
const json j = std::string("A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z"); // A é 你 😀 Z
// not escaping non-ASCII: bytes are copied through the bulk validator
CHECK(j.dump(-1, ' ', false) == "\"A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z\"");
// ensure_ascii: escaped (with a surrogate pair for the emoji)
CHECK(j.dump(-1, ' ', true) == "\"A\\u00e9\\u4f60\\ud83d\\ude00Z\"");
CHECK(json::parse(j.dump(-1, ' ', true)) == j);
}
SECTION("many small structural writes exceed the write buffer")
{
json arr = json::array();
for (int i = 0; i < 2000; ++i)
{
arr.push_back(i);
}
const std::string out = arr.dump();
CHECK(out.front() == '[');
CHECK(out.back() == ']');
CHECK(json::parse(out) == arr);
json obj = json::object();
for (int i = 0; i < 500; ++i)
{
obj["key" + std::to_string(i)] = i;
}
CHECK(json::parse(obj.dump()) == obj);
CHECK(json::parse(obj.dump(2)) == obj);
// deep nesting emits >1024 consecutive single-character writes, forcing
// the write buffer to flush mid-run
json nested = json::array();
for (int i = 0; i < 1100; ++i)
{
nested = json::array({nested});
}
const std::string out2 = nested.dump();
CHECK(out2.substr(0, 1100) == std::string(1100, '['));
CHECK(json::parse(out2) == nested);
}
SECTION("invalid UTF-8 handling is unaffected by the fast path")
{
const json j = std::string("valid\xff" "more");
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0xFF", json::type_error&);
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"valid\xef\xbf\xbd" "more\"");
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"valid\\ufffdmore\"");
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"validmore\"");
}
}