Merge remote-tracking branch 'origin/develop' into claude/fix-issue-3989-db7e45

Signed-off-by: Niels Lohmann <mail@nlohmann.me>

# Conflicts:
#	include/nlohmann/detail/input/lexer.hpp
#	include/nlohmann/detail/input/parser.hpp
#	single_include/nlohmann/json.hpp
#	tests/src/fuzzer-parse_json.cpp
This commit is contained in:
Niels Lohmann
2026-10-01 07:36:30 +02:00
242 changed files with 2975 additions and 12803 deletions
+83 -5
View File
@@ -3,6 +3,7 @@
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
@@ -37,6 +38,71 @@ StringType to_string(std::size_t value)
return result;
}
/// @return a byte as two uppercase hexadecimal digits
inline std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
///////////////////
// UTF-8 encoding //
///////////////////
/*!
@brief encode a Unicode code point as UTF-8
Used to turn a decoded code point back into bytes: by the wide-string input
adapters in input_adapters.hpp (one code point per UTF-32 unit, per UTF-16
unit outside the surrogate range, and per valid UTF-16 surrogate pair), and
by the lexer's `\uXXXX`/`\uXXXX\uYYYY` handling in lexer.hpp. Passing a
code point above U+10FFFF, or one in the surrogate range U+D800..U+DFFF, is
undefined behavior; callers are expected to have rejected those already
(the wide-string adapters pass malformed units through unencoded instead of
calling this function, and the lexer rejects unpaired surrogates before
reaching it).
@tparam Out a callable invoked with one byte (as std::uint32_t, 0x00..0xFF)
at a time, most significant byte first
@param[in] cp the code point to encode (at most U+10FFFF)
@param[in] out called once for each byte of the UTF-8 encoding of @a cp
*/
template<typename Out>
void encode_utf8(std::uint32_t cp, Out&& out)
{
JSON_ASSERT(cp <= 0x10FFFF);
if (cp < 0x80)
{
// 1-byte characters: 0xxxxxxx (ASCII)
out(cp);
}
else if (cp <= 0x7FF)
{
// 2-byte characters: 110xxxxx 10xxxxxx
out(0xC0u | (cp >> 6u));
out(0x80u | (cp & 0x3Fu));
}
else if (cp <= 0xFFFF)
{
// 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx
out(0xE0u | (cp >> 12u));
out(0x80u | ((cp >> 6u) & 0x3Fu));
out(0x80u | (cp & 0x3Fu));
}
else
{
// 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
out(0xF0u | (cp >> 18u));
out(0x80u | ((cp >> 12u) & 0x3Fu));
out(0x80u | ((cp >> 6u) & 0x3Fu));
out(0x80u | (cp & 0x3Fu));
}
}
///////////////////
// UTF-8 decoding //
///////////////////
@@ -52,11 +118,23 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally
written by Björn Hoehrmann. See
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
This decoder is the single source of truth for UTF-8 validation in this
library: it is used both by the serializer (to escape and, in strict mode,
reject ill-formed UTF-8 when dumping a string) and by the binary readers
(to reject ill-formed UTF-8 in CBOR/MessagePack/BSON/UBJSON text strings at
decode time; see @ref is_valid_utf8 below).
The library checks UTF-8 well-formedness (RFC 3629, section 4) in four
places, which differ in speed, diagnostics, and how they read the input:
- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in
strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR,
MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in
text strings at decode time).
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
diagnostic for each kind of error.
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
bulk string scan, the bulk path of the BON8 reader, and the BON8 writer.
They must accept exactly what the lexer's switch accepts.
- the byte path of binary_reader::get_bon8_string(): BON8 input without bulk
access, and the bytes the bulk path leaves to it.
All four must accept the same set of sequences, so a change to one needs a
matching change to the others.
@param[in,out] state the current decoder state
@param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT)