From 4356d0fe50b687638e5f4a0a5325022dfb56cc5b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 18:15:03 +0200 Subject: [PATCH] Share the code point to UTF-8 encoding between the wide-string helpers and the lexer The 1/2/3/4-byte UTF-8 encoding ladder was written out by hand three times: in wide_string_input_helper<..., 4>::fill_buffer() for a UTF-32 code point, in the UTF-16 helper for both a BMP code unit and a valid surrogate pair, and in the lexer's \uXXXX/\uXXXX\uYYYY handling. The copies had drifted: the UTF-32 helper masked the leading bits of each byte (& 0x1Fu, & 0x0Fu, & 0x07u) where the others relied on the shift alone, even though both give the same result for a code point that is already known to be in range. Add detail::encode_utf8(cp, out) in string_utils.hpp, a single encoder that invokes a callable once per output byte, most significant byte first. Use it in the three valid-code-point branches (UTF-32 code points up to U+10FFFF, UTF-16 code units outside the surrogate range, and valid UTF-16 surrogate pairs) and in the lexer's \u handling, where out forwards to add(). The UTF-16 helper's deliberate pass-through of malformed surrogate units and the UTF-32 helper's 0xFF sentinel for code points above U+10FFFF are untouched, since neither reaches the new helper. Behavior-preserving: same bytes in the same order for every valid code point, verified with unit-class_lexer, unit-class_parser, unit-deserialization, unit-wstring and the non-test-data parts of unit-unicode1..5 (ASan/UBSan, C++11/17/20), and an escape-heavy parse microbenchmark that shows no change (about 73 ms either way, median of 3, 1M escape sequences). single_include/ regenerated with make amalgamate; make check-amalgamation leaves a clean tree. Overlaps #5704, which rewrites the wide_string_input_helper specializations touched here. Signed-off-by: Niels Lohmann #5712 item 6 --- .../nlohmann/detail/input/input_adapters.hpp | 69 +++----- include/nlohmann/detail/input/lexer.hpp | 30 +--- include/nlohmann/detail/string_utils.hpp | 55 ++++++ single_include/nlohmann/json.hpp | 156 ++++++++++-------- 4 files changed, 166 insertions(+), 144 deletions(-) diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index 2addf2282..4040fefd7 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -11,6 +11,7 @@ #include // min #include // array #include // size_t +#include // uint32_t #include // strlen #include // begin, end, iterator_traits, random_access_iterator_tag, distance, next #include // streambuf @@ -27,6 +28,7 @@ #include #include #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -448,32 +450,14 @@ struct wide_string_input_helper // get the current character const auto wc = input.get_character(); - // UTF-32 to UTF-8 encoding - if (wc < 0x80) + if (wc <= 0x10FFFF) { - utf8_bytes[0] = static_cast::int_type>(wc); - utf8_bytes_filled = 1; - } - else if (wc <= 0x7FF) - { - utf8_bytes[0] = static_cast::int_type>(0xC0u | ((static_cast(wc) >> 6u) & 0x1Fu)); - utf8_bytes[1] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); - utf8_bytes_filled = 2; - } - else if (wc <= 0xFFFF) - { - utf8_bytes[0] = static_cast::int_type>(0xE0u | ((static_cast(wc) >> 12u) & 0x0Fu)); - utf8_bytes[1] = static_cast::int_type>(0x80u | ((static_cast(wc) >> 6u) & 0x3Fu)); - utf8_bytes[2] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); - utf8_bytes_filled = 3; - } - else if (wc <= 0x10FFFF) - { - utf8_bytes[0] = static_cast::int_type>(0xF0u | ((static_cast(wc) >> 18u) & 0x07u)); - utf8_bytes[1] = static_cast::int_type>(0x80u | ((static_cast(wc) >> 12u) & 0x3Fu)); - utf8_bytes[2] = static_cast::int_type>(0x80u | ((static_cast(wc) >> 6u) & 0x3Fu)); - utf8_bytes[3] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); - utf8_bytes_filled = 4; + // UTF-32 to UTF-8 encoding + utf8_bytes_filled = 0; + encode_utf8(static_cast(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) + { + utf8_bytes[utf8_bytes_filled++] = static_cast::int_type>(byte); + }); } else { @@ -510,24 +494,15 @@ struct wide_string_input_helper // get the current character const auto wc = input.get_character(); - // UTF-16 to UTF-8 encoding - if (wc < 0x80) + if (0xD800 > wc || wc >= 0xE000) { - utf8_bytes[0] = static_cast::int_type>(wc); - utf8_bytes_filled = 1; - } - else if (wc <= 0x7FF) - { - utf8_bytes[0] = static_cast::int_type>(0xC0u | ((static_cast(wc) >> 6u))); - utf8_bytes[1] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); - utf8_bytes_filled = 2; - } - else if (0xD800 > wc || wc >= 0xE000) - { - utf8_bytes[0] = static_cast::int_type>(0xE0u | ((static_cast(wc) >> 12u))); - utf8_bytes[1] = static_cast::int_type>(0x80u | ((static_cast(wc) >> 6u) & 0x3Fu)); - utf8_bytes[2] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); - utf8_bytes_filled = 3; + // a UTF-16 code unit outside the surrogate range is a valid + // code point (at most U+FFFF) on its own + utf8_bytes_filled = 0; + encode_utf8(static_cast(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) + { + utf8_bytes[utf8_bytes_filled++] = static_cast::int_type>(byte); + }); } else { @@ -545,11 +520,11 @@ struct wide_string_input_helper if (0xDC00 <= wc2 && wc2 <= 0xDFFF) { const auto charcode = 0x10000u + (((static_cast(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); - utf8_bytes[0] = static_cast::int_type>(0xF0u | (charcode >> 18u)); - utf8_bytes[1] = static_cast::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu)); - utf8_bytes[2] = static_cast::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu)); - utf8_bytes[3] = static_cast::int_type>(0x80u | (charcode & 0x3Fu)); - utf8_bytes_filled = 4; + utf8_bytes_filled = 0; + encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) + { + utf8_bytes[utf8_bytes_filled++] = static_cast::int_type>(byte); + }); valid_pair = true; } } diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index 20c7c84cc..7d239852b 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -11,6 +11,7 @@ #include // array #include // localeconv #include // size_t +#include // uint32_t #include // snprintf #include // strtof, strtod, strtold, strtoll, strtoull #include // initializer_list @@ -24,6 +25,7 @@ #include #include #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -500,32 +502,10 @@ class lexer : public lexer_base JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF); // translate codepoint into bytes - if (codepoint < 0x80) + encode_utf8(static_cast(codepoint), [this](std::uint32_t byte) { - // 1-byte characters: 0xxxxxxx (ASCII) - add(static_cast(codepoint)); - } - else if (codepoint <= 0x7FF) - { - // 2-byte characters: 110xxxxx 10xxxxxx - add(static_cast(0xC0u | (static_cast(codepoint) >> 6u))); - add(static_cast(0x80u | (static_cast(codepoint) & 0x3Fu))); - } - else if (codepoint <= 0xFFFF) - { - // 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx - add(static_cast(0xE0u | (static_cast(codepoint) >> 12u))); - add(static_cast(0x80u | ((static_cast(codepoint) >> 6u) & 0x3Fu))); - add(static_cast(0x80u | (static_cast(codepoint) & 0x3Fu))); - } - else - { - // 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx - add(static_cast(0xF0u | (static_cast(codepoint) >> 18u))); - add(static_cast(0x80u | ((static_cast(codepoint) >> 12u) & 0x3Fu))); - add(static_cast(0x80u | ((static_cast(codepoint) >> 6u) & 0x3Fu))); - add(static_cast(0x80u | (static_cast(codepoint) & 0x3Fu))); - } + add(static_cast(byte)); + }); break; } diff --git a/include/nlohmann/detail/string_utils.hpp b/include/nlohmann/detail/string_utils.hpp index 8d9956716..542fb6e9a 100644 --- a/include/nlohmann/detail/string_utils.hpp +++ b/include/nlohmann/detail/string_utils.hpp @@ -36,6 +36,61 @@ StringType to_string(std::size_t value) return result; } +/////////////////// +// UTF-8 encoding // +/////////////////// + +/*! +@brief encode a Unicode code point as UTF-8 + +Used to turn a decoded code point back into bytes: by the wide-string input +adapters in input_adapters.hpp (one code point per UTF-32 unit, per UTF-16 +unit outside the surrogate range, and per valid UTF-16 surrogate pair), and +by the lexer's `\uXXXX`/`\uXXXX\uYYYY` handling in lexer.hpp. Passing a +code point above U+10FFFF, or one in the surrogate range U+D800..U+DFFF, is +undefined behavior; callers are expected to have rejected those already +(the wide-string adapters pass malformed units through unencoded instead of +calling this function, and the lexer rejects unpaired surrogates before +reaching it). + +@tparam Out a callable invoked with one byte (as std::uint32_t, 0x00..0xFF) + at a time, most significant byte first +@param[in] cp the code point to encode (at most U+10FFFF) +@param[in] out called once for each byte of the UTF-8 encoding of @a cp +*/ +template +void encode_utf8(std::uint32_t cp, Out&& out) +{ + JSON_ASSERT(cp <= 0x10FFFF); + + if (cp < 0x80) + { + // 1-byte characters: 0xxxxxxx (ASCII) + out(cp); + } + else if (cp <= 0x7FF) + { + // 2-byte characters: 110xxxxx 10xxxxxx + out(0xC0u | (cp >> 6u)); + out(0x80u | (cp & 0x3Fu)); + } + else if (cp <= 0xFFFF) + { + // 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx + out(0xE0u | (cp >> 12u)); + out(0x80u | ((cp >> 6u) & 0x3Fu)); + out(0x80u | (cp & 0x3Fu)); + } + else + { + // 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx + out(0xF0u | (cp >> 18u)); + out(0x80u | ((cp >> 12u) & 0x3Fu)); + out(0x80u | ((cp >> 6u) & 0x3Fu)); + out(0x80u | (cp & 0x3Fu)); + } +} + /////////////////// // UTF-8 decoding // /////////////////// diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 70148f18c..61cbe4ac8 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -6234,6 +6234,61 @@ StringType to_string(std::size_t value) return result; } +/////////////////// +// UTF-8 encoding // +/////////////////// + +/*! +@brief encode a Unicode code point as UTF-8 + +Used to turn a decoded code point back into bytes: by the wide-string input +adapters in input_adapters.hpp (one code point per UTF-32 unit, per UTF-16 +unit outside the surrogate range, and per valid UTF-16 surrogate pair), and +by the lexer's `\uXXXX`/`\uXXXX\uYYYY` handling in lexer.hpp. Passing a +code point above U+10FFFF, or one in the surrogate range U+D800..U+DFFF, is +undefined behavior; callers are expected to have rejected those already +(the wide-string adapters pass malformed units through unencoded instead of +calling this function, and the lexer rejects unpaired surrogates before +reaching it). + +@tparam Out a callable invoked with one byte (as std::uint32_t, 0x00..0xFF) + at a time, most significant byte first +@param[in] cp the code point to encode (at most U+10FFFF) +@param[in] out called once for each byte of the UTF-8 encoding of @a cp +*/ +template +void encode_utf8(std::uint32_t cp, Out&& out) +{ + JSON_ASSERT(cp <= 0x10FFFF); + + if (cp < 0x80) + { + // 1-byte characters: 0xxxxxxx (ASCII) + out(cp); + } + else if (cp <= 0x7FF) + { + // 2-byte characters: 110xxxxx 10xxxxxx + out(0xC0u | (cp >> 6u)); + out(0x80u | (cp & 0x3Fu)); + } + else if (cp <= 0xFFFF) + { + // 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx + out(0xE0u | (cp >> 12u)); + out(0x80u | ((cp >> 6u) & 0x3Fu)); + out(0x80u | (cp & 0x3Fu)); + } + else + { + // 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx + out(0xF0u | (cp >> 18u)); + out(0x80u | ((cp >> 12u) & 0x3Fu)); + out(0x80u | ((cp >> 6u) & 0x3Fu)); + out(0x80u | (cp & 0x3Fu)); + } +} + /////////////////// // UTF-8 decoding // /////////////////// @@ -7563,6 +7618,7 @@ NLOHMANN_JSON_NAMESPACE_END #include // min #include // array #include // size_t +#include // uint32_t #include // strlen #include // begin, end, iterator_traits, random_access_iterator_tag, distance, next #include // streambuf @@ -7583,6 +7639,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -8004,32 +8062,14 @@ struct wide_string_input_helper // get the current character const auto wc = input.get_character(); - // UTF-32 to UTF-8 encoding - if (wc < 0x80) + if (wc <= 0x10FFFF) { - utf8_bytes[0] = static_cast::int_type>(wc); - utf8_bytes_filled = 1; - } - else if (wc <= 0x7FF) - { - utf8_bytes[0] = static_cast::int_type>(0xC0u | ((static_cast(wc) >> 6u) & 0x1Fu)); - utf8_bytes[1] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); - utf8_bytes_filled = 2; - } - else if (wc <= 0xFFFF) - { - utf8_bytes[0] = static_cast::int_type>(0xE0u | ((static_cast(wc) >> 12u) & 0x0Fu)); - utf8_bytes[1] = static_cast::int_type>(0x80u | ((static_cast(wc) >> 6u) & 0x3Fu)); - utf8_bytes[2] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); - utf8_bytes_filled = 3; - } - else if (wc <= 0x10FFFF) - { - utf8_bytes[0] = static_cast::int_type>(0xF0u | ((static_cast(wc) >> 18u) & 0x07u)); - utf8_bytes[1] = static_cast::int_type>(0x80u | ((static_cast(wc) >> 12u) & 0x3Fu)); - utf8_bytes[2] = static_cast::int_type>(0x80u | ((static_cast(wc) >> 6u) & 0x3Fu)); - utf8_bytes[3] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); - utf8_bytes_filled = 4; + // UTF-32 to UTF-8 encoding + utf8_bytes_filled = 0; + encode_utf8(static_cast(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) + { + utf8_bytes[utf8_bytes_filled++] = static_cast::int_type>(byte); + }); } else { @@ -8066,24 +8106,15 @@ struct wide_string_input_helper // get the current character const auto wc = input.get_character(); - // UTF-16 to UTF-8 encoding - if (wc < 0x80) + if (0xD800 > wc || wc >= 0xE000) { - utf8_bytes[0] = static_cast::int_type>(wc); - utf8_bytes_filled = 1; - } - else if (wc <= 0x7FF) - { - utf8_bytes[0] = static_cast::int_type>(0xC0u | ((static_cast(wc) >> 6u))); - utf8_bytes[1] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); - utf8_bytes_filled = 2; - } - else if (0xD800 > wc || wc >= 0xE000) - { - utf8_bytes[0] = static_cast::int_type>(0xE0u | ((static_cast(wc) >> 12u))); - utf8_bytes[1] = static_cast::int_type>(0x80u | ((static_cast(wc) >> 6u) & 0x3Fu)); - utf8_bytes[2] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); - utf8_bytes_filled = 3; + // a UTF-16 code unit outside the surrogate range is a valid + // code point (at most U+FFFF) on its own + utf8_bytes_filled = 0; + encode_utf8(static_cast(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) + { + utf8_bytes[utf8_bytes_filled++] = static_cast::int_type>(byte); + }); } else { @@ -8101,11 +8132,11 @@ struct wide_string_input_helper if (0xDC00 <= wc2 && wc2 <= 0xDFFF) { const auto charcode = 0x10000u + (((static_cast(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); - utf8_bytes[0] = static_cast::int_type>(0xF0u | (charcode >> 18u)); - utf8_bytes[1] = static_cast::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu)); - utf8_bytes[2] = static_cast::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu)); - utf8_bytes[3] = static_cast::int_type>(0x80u | (charcode & 0x3Fu)); - utf8_bytes_filled = 4; + utf8_bytes_filled = 0; + encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte) + { + utf8_bytes[utf8_bytes_filled++] = static_cast::int_type>(byte); + }); valid_pair = true; } } @@ -8487,6 +8518,7 @@ NLOHMANN_JSON_NAMESPACE_END #include // array #include // localeconv #include // size_t +#include // uint32_t #include // snprintf #include // strtof, strtod, strtold, strtoll, strtoull #include // initializer_list @@ -9129,6 +9161,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -9605,32 +9639,10 @@ class lexer : public lexer_base JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF); // translate codepoint into bytes - if (codepoint < 0x80) + encode_utf8(static_cast(codepoint), [this](std::uint32_t byte) { - // 1-byte characters: 0xxxxxxx (ASCII) - add(static_cast(codepoint)); - } - else if (codepoint <= 0x7FF) - { - // 2-byte characters: 110xxxxx 10xxxxxx - add(static_cast(0xC0u | (static_cast(codepoint) >> 6u))); - add(static_cast(0x80u | (static_cast(codepoint) & 0x3Fu))); - } - else if (codepoint <= 0xFFFF) - { - // 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx - add(static_cast(0xE0u | (static_cast(codepoint) >> 12u))); - add(static_cast(0x80u | ((static_cast(codepoint) >> 6u) & 0x3Fu))); - add(static_cast(0x80u | (static_cast(codepoint) & 0x3Fu))); - } - else - { - // 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx - add(static_cast(0xF0u | (static_cast(codepoint) >> 18u))); - add(static_cast(0x80u | ((static_cast(codepoint) >> 12u) & 0x3Fu))); - add(static_cast(0x80u | ((static_cast(codepoint) >> 6u) & 0x3Fu))); - add(static_cast(0x80u | (static_cast(codepoint) & 0x3Fu))); - } + add(static_cast(byte)); + }); break; }