Share the code point to UTF-8 encoding between the wide-string helpers and the lexer

The 1/2/3/4-byte UTF-8 encoding ladder was written out by hand three
times: in wide_string_input_helper<..., 4>::fill_buffer() for a UTF-32
code point, in the UTF-16 helper for both a BMP code unit and a valid
surrogate pair, and in the lexer's \uXXXX/\uXXXX\uYYYY handling. The
copies had drifted: the UTF-32 helper masked the leading bits of each
byte (& 0x1Fu, & 0x0Fu, & 0x07u) where the others relied on the shift
alone, even though both give the same result for a code point that is
already known to be in range.

Add detail::encode_utf8(cp, out) in string_utils.hpp, a single encoder
that invokes a callable once per output byte, most significant byte
first. Use it in the three valid-code-point branches (UTF-32 code
points up to U+10FFFF, UTF-16 code units outside the surrogate range,
and valid UTF-16 surrogate pairs) and in the lexer's \u handling, where
out forwards to add(). The UTF-16 helper's deliberate pass-through of
malformed surrogate units and the UTF-32 helper's 0xFF sentinel for
code points above U+10FFFF are untouched, since neither reaches the new
helper.

Behavior-preserving: same bytes in the same order for every valid code
point, verified with unit-class_lexer, unit-class_parser,
unit-deserialization, unit-wstring and the non-test-data parts of
unit-unicode1..5 (ASan/UBSan, C++11/17/20), and an escape-heavy parse
microbenchmark that shows no change (about 73 ms either way, median of
3, 1M escape sequences). single_include/ regenerated with make
amalgamate; make check-amalgamation leaves a clean tree.

Overlaps #5704, which rewrites the wide_string_input_helper
specializations touched here.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

#5712 item 6
This commit is contained in:
Niels Lohmann
2026-09-30 18:15:03 +02:00
parent 32d5540d37
commit 4356d0fe50
4 changed files with 166 additions and 144 deletions
@@ -11,6 +11,7 @@
#include <algorithm> // min #include <algorithm> // min
#include <array> // array #include <array> // array
#include <cstddef> // size_t #include <cstddef> // size_t
#include <cstdint> // uint32_t
#include <cstring> // strlen #include <cstring> // strlen
#include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next #include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next
#include <streambuf> // streambuf #include <streambuf> // streambuf
@@ -27,6 +28,7 @@
#include <nlohmann/detail/iterators/iterator_traits.hpp> #include <nlohmann/detail/iterators/iterator_traits.hpp>
#include <nlohmann/detail/macro_scope.hpp> #include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/type_traits.hpp> #include <nlohmann/detail/meta/type_traits.hpp>
#include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail namespace detail
@@ -448,32 +450,14 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
// get the current character // get the current character
const auto wc = input.get_character(); const auto wc = input.get_character();
// UTF-32 to UTF-8 encoding if (wc <= 0x10FFFF)
if (wc < 0x80)
{ {
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc); // UTF-32 to UTF-8 encoding
utf8_bytes_filled = 1; utf8_bytes_filled = 0;
} encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
else if (wc <= 0x7FF) {
{ utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u) & 0x1Fu)); });
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 2;
}
else if (wc <= 0xFFFF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u) & 0x0Fu));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 3;
}
else if (wc <= 0x10FFFF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | ((static_cast<unsigned int>(wc) >> 18u) & 0x07u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 12u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 4;
} }
else else
{ {
@@ -510,24 +494,15 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
// get the current character // get the current character
const auto wc = input.get_character(); const auto wc = input.get_character();
// UTF-16 to UTF-8 encoding if (0xD800 > wc || wc >= 0xE000)
if (wc < 0x80)
{ {
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc); // a UTF-16 code unit outside the surrogate range is a valid
utf8_bytes_filled = 1; // code point (at most U+FFFF) on its own
} utf8_bytes_filled = 0;
else if (wc <= 0x7FF) encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
{ {
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u))); utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu)); });
utf8_bytes_filled = 2;
}
else if (0xD800 > wc || wc >= 0xE000)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u)));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 3;
} }
else else
{ {
@@ -545,11 +520,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (0xDC00 <= wc2 && wc2 <= 0xDFFF) if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{ {
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u)); utf8_bytes_filled = 0;
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu)); encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu)); {
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (charcode & 0x3Fu)); utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
utf8_bytes_filled = 4; });
valid_pair = true; valid_pair = true;
} }
} }
+5 -25
View File
@@ -11,6 +11,7 @@
#include <array> // array #include <array> // array
#include <clocale> // localeconv #include <clocale> // localeconv
#include <cstddef> // size_t #include <cstddef> // size_t
#include <cstdint> // uint32_t
#include <cstdio> // snprintf #include <cstdio> // snprintf
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull #include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
#include <initializer_list> // initializer_list #include <initializer_list> // initializer_list
@@ -24,6 +25,7 @@
#include <nlohmann/detail/input/string_scan.hpp> #include <nlohmann/detail/input/string_scan.hpp>
#include <nlohmann/detail/macro_scope.hpp> #include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/type_traits.hpp> #include <nlohmann/detail/meta/type_traits.hpp>
#include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail namespace detail
@@ -500,32 +502,10 @@ class lexer : public lexer_base<BasicJsonType>
JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF); JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF);
// translate codepoint into bytes // translate codepoint into bytes
if (codepoint < 0x80) encode_utf8(static_cast<std::uint32_t>(codepoint), [this](std::uint32_t byte)
{ {
// 1-byte characters: 0xxxxxxx (ASCII) add(static_cast<char_int_type>(byte));
add(static_cast<char_int_type>(codepoint)); });
}
else if (codepoint <= 0x7FF)
{
// 2-byte characters: 110xxxxx 10xxxxxx
add(static_cast<char_int_type>(0xC0u | (static_cast<unsigned int>(codepoint) >> 6u)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
else if (codepoint <= 0xFFFF)
{
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
add(static_cast<char_int_type>(0xE0u | (static_cast<unsigned int>(codepoint) >> 12u)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
else
{
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
add(static_cast<char_int_type>(0xF0u | (static_cast<unsigned int>(codepoint) >> 18u)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 12u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
break; break;
} }
+55
View File
@@ -36,6 +36,61 @@ StringType to_string(std::size_t value)
return result; return result;
} }
///////////////////
// UTF-8 encoding //
///////////////////
/*!
@brief encode a Unicode code point as UTF-8
Used to turn a decoded code point back into bytes: by the wide-string input
adapters in input_adapters.hpp (one code point per UTF-32 unit, per UTF-16
unit outside the surrogate range, and per valid UTF-16 surrogate pair), and
by the lexer's `\uXXXX`/`\uXXXX\uYYYY` handling in lexer.hpp. Passing a
code point above U+10FFFF, or one in the surrogate range U+D800..U+DFFF, is
undefined behavior; callers are expected to have rejected those already
(the wide-string adapters pass malformed units through unencoded instead of
calling this function, and the lexer rejects unpaired surrogates before
reaching it).
@tparam Out a callable invoked with one byte (as std::uint32_t, 0x00..0xFF)
at a time, most significant byte first
@param[in] cp the code point to encode (at most U+10FFFF)
@param[in] out called once for each byte of the UTF-8 encoding of @a cp
*/
template<typename Out>
void encode_utf8(std::uint32_t cp, Out&& out)
{
JSON_ASSERT(cp <= 0x10FFFF);
if (cp < 0x80)
{
// 1-byte characters: 0xxxxxxx (ASCII)
out(cp);
}
else if (cp <= 0x7FF)
{
// 2-byte characters: 110xxxxx 10xxxxxx
out(0xC0u | (cp >> 6u));
out(0x80u | (cp & 0x3Fu));
}
else if (cp <= 0xFFFF)
{
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
out(0xE0u | (cp >> 12u));
out(0x80u | ((cp >> 6u) & 0x3Fu));
out(0x80u | (cp & 0x3Fu));
}
else
{
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
out(0xF0u | (cp >> 18u));
out(0x80u | ((cp >> 12u) & 0x3Fu));
out(0x80u | ((cp >> 6u) & 0x3Fu));
out(0x80u | (cp & 0x3Fu));
}
}
/////////////////// ///////////////////
// UTF-8 decoding // // UTF-8 decoding //
/////////////////// ///////////////////
+84 -72
View File
@@ -6234,6 +6234,61 @@ StringType to_string(std::size_t value)
return result; return result;
} }
///////////////////
// UTF-8 encoding //
///////////////////
/*!
@brief encode a Unicode code point as UTF-8
Used to turn a decoded code point back into bytes: by the wide-string input
adapters in input_adapters.hpp (one code point per UTF-32 unit, per UTF-16
unit outside the surrogate range, and per valid UTF-16 surrogate pair), and
by the lexer's `\uXXXX`/`\uXXXX\uYYYY` handling in lexer.hpp. Passing a
code point above U+10FFFF, or one in the surrogate range U+D800..U+DFFF, is
undefined behavior; callers are expected to have rejected those already
(the wide-string adapters pass malformed units through unencoded instead of
calling this function, and the lexer rejects unpaired surrogates before
reaching it).
@tparam Out a callable invoked with one byte (as std::uint32_t, 0x00..0xFF)
at a time, most significant byte first
@param[in] cp the code point to encode (at most U+10FFFF)
@param[in] out called once for each byte of the UTF-8 encoding of @a cp
*/
template<typename Out>
void encode_utf8(std::uint32_t cp, Out&& out)
{
JSON_ASSERT(cp <= 0x10FFFF);
if (cp < 0x80)
{
// 1-byte characters: 0xxxxxxx (ASCII)
out(cp);
}
else if (cp <= 0x7FF)
{
// 2-byte characters: 110xxxxx 10xxxxxx
out(0xC0u | (cp >> 6u));
out(0x80u | (cp & 0x3Fu));
}
else if (cp <= 0xFFFF)
{
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
out(0xE0u | (cp >> 12u));
out(0x80u | ((cp >> 6u) & 0x3Fu));
out(0x80u | (cp & 0x3Fu));
}
else
{
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
out(0xF0u | (cp >> 18u));
out(0x80u | ((cp >> 12u) & 0x3Fu));
out(0x80u | ((cp >> 6u) & 0x3Fu));
out(0x80u | (cp & 0x3Fu));
}
}
/////////////////// ///////////////////
// UTF-8 decoding // // UTF-8 decoding //
/////////////////// ///////////////////
@@ -7563,6 +7618,7 @@ NLOHMANN_JSON_NAMESPACE_END
#include <algorithm> // min #include <algorithm> // min
#include <array> // array #include <array> // array
#include <cstddef> // size_t #include <cstddef> // size_t
#include <cstdint> // uint32_t
#include <cstring> // strlen #include <cstring> // strlen
#include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next #include <iterator> // begin, end, iterator_traits, random_access_iterator_tag, distance, next
#include <streambuf> // streambuf #include <streambuf> // streambuf
@@ -7583,6 +7639,8 @@ NLOHMANN_JSON_NAMESPACE_END
// #include <nlohmann/detail/meta/type_traits.hpp> // #include <nlohmann/detail/meta/type_traits.hpp>
// #include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail namespace detail
@@ -8004,32 +8062,14 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
// get the current character // get the current character
const auto wc = input.get_character(); const auto wc = input.get_character();
// UTF-32 to UTF-8 encoding if (wc <= 0x10FFFF)
if (wc < 0x80)
{ {
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc); // UTF-32 to UTF-8 encoding
utf8_bytes_filled = 1; utf8_bytes_filled = 0;
} encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
else if (wc <= 0x7FF) {
{ utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u) & 0x1Fu)); });
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 2;
}
else if (wc <= 0xFFFF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u) & 0x0Fu));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 3;
}
else if (wc <= 0x10FFFF)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | ((static_cast<unsigned int>(wc) >> 18u) & 0x07u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 12u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 4;
} }
else else
{ {
@@ -8066,24 +8106,15 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
// get the current character // get the current character
const auto wc = input.get_character(); const auto wc = input.get_character();
// UTF-16 to UTF-8 encoding if (0xD800 > wc || wc >= 0xE000)
if (wc < 0x80)
{ {
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc); // a UTF-16 code unit outside the surrogate range is a valid
utf8_bytes_filled = 1; // code point (at most U+FFFF) on its own
} utf8_bytes_filled = 0;
else if (wc <= 0x7FF) encode_utf8(static_cast<std::uint32_t>(wc), [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
{ {
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xC0u | ((static_cast<unsigned int>(wc) >> 6u))); utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu)); });
utf8_bytes_filled = 2;
}
else if (0xD800 > wc || wc >= 0xE000)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xE0u | ((static_cast<unsigned int>(wc) >> 12u)));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((static_cast<unsigned int>(wc) >> 6u) & 0x3Fu));
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | (static_cast<unsigned int>(wc) & 0x3Fu));
utf8_bytes_filled = 3;
} }
else else
{ {
@@ -8101,11 +8132,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (0xDC00 <= wc2 && wc2 <= 0xDFFF) if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{ {
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u)); utf8_bytes_filled = 0;
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu)); encode_utf8(charcode, [&utf8_bytes, &utf8_bytes_filled](std::uint32_t byte)
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu)); {
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (charcode & 0x3Fu)); utf8_bytes[utf8_bytes_filled++] = static_cast<std::char_traits<char>::int_type>(byte);
utf8_bytes_filled = 4; });
valid_pair = true; valid_pair = true;
} }
} }
@@ -8487,6 +8518,7 @@ NLOHMANN_JSON_NAMESPACE_END
#include <array> // array #include <array> // array
#include <clocale> // localeconv #include <clocale> // localeconv
#include <cstddef> // size_t #include <cstddef> // size_t
#include <cstdint> // uint32_t
#include <cstdio> // snprintf #include <cstdio> // snprintf
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull #include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
#include <initializer_list> // initializer_list #include <initializer_list> // initializer_list
@@ -9129,6 +9161,8 @@ NLOHMANN_JSON_NAMESPACE_END
// #include <nlohmann/detail/meta/type_traits.hpp> // #include <nlohmann/detail/meta/type_traits.hpp>
// #include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail namespace detail
@@ -9605,32 +9639,10 @@ class lexer : public lexer_base<BasicJsonType>
JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF); JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF);
// translate codepoint into bytes // translate codepoint into bytes
if (codepoint < 0x80) encode_utf8(static_cast<std::uint32_t>(codepoint), [this](std::uint32_t byte)
{ {
// 1-byte characters: 0xxxxxxx (ASCII) add(static_cast<char_int_type>(byte));
add(static_cast<char_int_type>(codepoint)); });
}
else if (codepoint <= 0x7FF)
{
// 2-byte characters: 110xxxxx 10xxxxxx
add(static_cast<char_int_type>(0xC0u | (static_cast<unsigned int>(codepoint) >> 6u)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
else if (codepoint <= 0xFFFF)
{
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
add(static_cast<char_int_type>(0xE0u | (static_cast<unsigned int>(codepoint) >> 12u)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
else
{
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
add(static_cast<char_int_type>(0xF0u | (static_cast<unsigned int>(codepoint) >> 18u)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 12u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | ((static_cast<unsigned int>(codepoint) >> 6u) & 0x3Fu)));
add(static_cast<char_int_type>(0x80u | (static_cast<unsigned int>(codepoint) & 0x3Fu)));
}
break; break;
} }