Compare commits

..
Author SHA1 Message Date
Niels Lohmann 2add64396a Keep a NUL byte ending a // comment as the end of input
With the default NUL handling (JSON_STRICT_NUL_HANDLING not set), a NUL
byte in the input is treated as the real end of input everywhere -
except when it immediately ends a `//` comment: scan_comment() matched
'\0' as a comment terminator like '\n', so the NUL was consumed as
part of the comment and scan() never saw it as end of input; the next
get() then kept reading past it. Multi-line comments and
JSON_STRICT_NUL_HANDLING=1 were unaffected, since there the NUL is
just part of the comment text.

Fix scan_comment() to leave the NUL unconsumed (unget()) instead of
returning it as part of the comment, so the following scan() reports
it as end of input, exactly as for a NUL anywhere else.

Fixes #5659.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 23:38:59 +02:00
8 changed files with 436 additions and 269 deletions
+6 -1
View File
@@ -975,10 +975,15 @@ class lexer : public lexer_base<BasicJsonType>
case '\n':
case '\r':
case char_traits<char_type>::eof():
return true;
#if !JSON_STRICT_NUL_HANDLING
case '\0':
#endif
// a NUL byte is the end of the input (see scan()),
// so leave it for scan() to see
unget();
return true;
#endif
default:
break;
@@ -29,7 +29,6 @@
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/output/output_adapters.hpp>
#include <nlohmann/detail/string_concat.hpp>
#include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -2279,10 +2278,20 @@ class binary_writer
const std::size_t valid = valid_utf8_prefix(data, s.size());
if (JSON_HEDLEY_UNLIKELY(valid != s.size()))
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", detail::hex_byte(data[valid])), &context));
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", hex_byte(data[valid])), &context));
}
}
/// @return a byte as two uppercase hexadecimal digits
static std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
/*!
@brief write an integer in the shortest encoding
@@ -8,7 +8,9 @@
#pragma once
#include <algorithm> // copy
#include <cstddef> // size_t
#include <iterator> // back_inserter
#include <memory> // shared_ptr, make_shared
#include <string> // basic_string
#include <utility> // move
@@ -29,10 +31,6 @@ namespace detail
template<typename CharType> struct output_adapter_protocol
{
virtual void write_character(CharType c) = 0;
/// @param[in] s pointer to the characters to write; binary_writer legitimately
/// passes a null pointer together with length 0 for an empty
/// string or binary value, so implementations must tolerate that
/// @param[in] length number of characters at @a s
virtual void write_characters(const CharType* s, std::size_t length) = 0;
virtual ~output_adapter_protocol() = default;
@@ -99,6 +97,7 @@ class output_vector_adapter : public output_adapter_protocol<CharType>
sink.write_character(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
sink.write_characters(s, length);
@@ -123,6 +122,7 @@ class output_stream_adapter : public output_adapter_protocol<CharType>
stream.put(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
stream.write(s, static_cast<std::streamsize>(length));
@@ -147,6 +147,7 @@ class output_string_adapter : public output_adapter_protocol<CharType>
str.push_back(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
str.append(s, length);
+176 -70
View File
@@ -3,23 +3,24 @@
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#pragma once
#include <algorithm> // remove, fill, find, none_of, min
#include <algorithm> // reverse, remove, fill, find, none_of, min
#include <array> // array
#include <clocale> // localeconv, lconv
#include <cmath> // isfinite
#include <cmath> // labs, isfinite, isnan, signbit
#include <cstddef> // size_t, ptrdiff_t
#include <cstdint> // uint8_t
#include <cstdio> // snprintf
#include <cstring> // memcpy, memset
#include <iterator> // next
#include <limits> // numeric_limits
#include <string> // string, char_traits
#include <type_traits> // is_same
#include <utility> // move
#include <vector> // vector
#include <nlohmann/detail/conversions/to_chars.hpp>
@@ -27,6 +28,7 @@
#include <nlohmann/detail/input/string_scan.hpp>
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/cpp_future.hpp>
#include <nlohmann/detail/output/binary_writer.hpp>
#include <nlohmann/detail/output/output_adapters.hpp>
#include <nlohmann/detail/recursion_depth_limit.hpp>
#include <nlohmann/detail/string_concat.hpp>
@@ -81,6 +83,7 @@ class serializer
const std::size_t indent_step_ = 0,
error_handler_t error_handler_ = error_handler_t::strict)
: o(&s)
, locale(std::localeconv())
, indent_char(ichar)
, pretty_print(pretty_print_)
, ensure_ascii(ensure_ascii_)
@@ -103,10 +106,9 @@ class serializer
additional parameter. Arrays and objects are serialized without recursion,
however deeply they are nested.
- strings and object keys are escaped using @ref dump_escaped
- integer numbers are converted using a digit-pair lookup table (@ref dump_integer)
- floating-point numbers are converted to a string using @ref dump_float, which
uses `to_chars` for IEEE-754 types and `snprintf` otherwise
- strings and object keys are escaped using `escape_string()`
- integer numbers are converted implicitly via `operator<<`
- floating-point numbers are converted to a string using `"%g"` format
- binary values are serialized as objects containing the subtype and the
byte array
@@ -281,16 +283,127 @@ class serializer
}
case value_t::string:
case value_t::binary:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::discarded:
case value_t::null:
default:
dump_scalar(val, current_indent);
{
put_char('"');
dump_escaped(*val.m_data.m_value.string);
put_char('"');
return;
}
case value_t::binary:
{
if (pretty_print)
{
put_literal("{\n");
// variable to hold indentation for recursive calls
const auto new_indent = next_indent(current_indent, indent_step);
put_indent(new_indent);
put_literal("\"bytes\": [");
if (!val.m_data.m_value.binary->empty())
{
for (auto i = val.m_data.m_value.binary->cbegin();
i != val.m_data.m_value.binary->cend() - 1; ++i)
{
dump_byte(*i);
put_literal(", ");
}
dump_byte(val.m_data.m_value.binary->back());
}
put_literal("],\n");
put_indent(new_indent);
put_literal("\"subtype\": ");
if (val.m_data.m_value.binary->has_subtype())
{
dump_integer(val.m_data.m_value.binary->subtype());
}
else
{
put_literal("null");
}
put_char('\n');
put_indent(current_indent);
put_char('}');
}
else
{
put_literal("{\"bytes\":[");
if (!val.m_data.m_value.binary->empty())
{
for (auto i = val.m_data.m_value.binary->cbegin();
i != val.m_data.m_value.binary->cend() - 1; ++i)
{
dump_byte(*i);
put_char(',');
}
dump_byte(val.m_data.m_value.binary->back());
}
put_literal("],\"subtype\":");
if (val.m_data.m_value.binary->has_subtype())
{
dump_integer(val.m_data.m_value.binary->subtype());
put_char('}');
}
else
{
put_literal("null}");
}
}
return;
}
case value_t::boolean:
{
if (val.m_data.m_value.boolean)
{
put_literal("true");
}
else
{
put_literal("false");
}
return;
}
case value_t::number_integer:
{
dump_integer(val.m_data.m_value.number_integer);
return;
}
case value_t::number_unsigned:
{
dump_integer(val.m_data.m_value.number_unsigned);
return;
}
case value_t::number_float:
{
dump_float(val.m_data.m_value.number_float);
return;
}
case value_t::discarded:
{
put_literal("<discarded>");
return;
}
case value_t::null:
{
put_literal("null");
return;
}
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
}
@@ -447,9 +560,9 @@ class serializer
@brief serialize the value @a val, but not the elements of a container
An object or array with elements is opened and pushed onto @a stack for
@ref dump_iteratively to walk; everything else - including a binary value,
@ref dump_internal to walk; everything else - including a binary value,
which looks like an object but has no elements to descend into - is written
out in full by @ref dump_scalar.
out here in full.
*/
void dump_value(const BasicJsonType& val,
const std::size_t current_indent,
@@ -507,35 +620,6 @@ class serializer
return;
}
case value_t::string:
case value_t::binary:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::discarded:
case value_t::null:
default:
dump_scalar(val, current_indent);
return;
}
}
/*!
@brief serialize the value @a val, which is neither an object nor an array
Shared by @ref dump_internal and @ref dump_value, so that a value is written
the same way however deeply it is nested. A binary value is written out here
in full: it looks like an object, but has no elements to descend into.
@param[in] val value to serialize; not an object or array
@param[in] current_indent the indentation of @a val, used for a
pretty-printed binary value
*/
void dump_scalar(const BasicJsonType& val, const std::size_t current_indent)
{
switch (val.m_data.m_type)
{
case value_t::string:
{
put_char('"');
@@ -656,8 +740,6 @@ class serializer
return;
}
case value_t::object: // LCOV_EXCL_LINE
case value_t::array: // LCOV_EXCL_LINE
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
@@ -686,7 +768,7 @@ class serializer
Escape a string by replacing certain special characters by a sequence of an
escape character (backslash) and another character and other control
characters by a sequence of "\u" followed by a four-digit hex
representation. The escaped string is appended to @ref write_buffer.
representation. The escaped string is written to output stream @a o.
@param[in] s the string to escape
@@ -880,7 +962,7 @@ class serializer
{
case error_handler_t::strict:
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", detail::hex_byte(byte)), nullptr));
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", hex_bytes(byte | 0)), nullptr));
}
case error_handler_t::ignore:
@@ -913,9 +995,9 @@ class serializer
}
else
{
string_buffer[bytes++] = '\xEF';
string_buffer[bytes++] = '\xBF';
string_buffer[bytes++] = '\xBD';
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xEF');
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBF');
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBD');
}
// write buffer and reset index; there must be 13 bytes
@@ -972,7 +1054,7 @@ class serializer
{
case error_handler_t::strict:
{
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", detail::hex_byte(static_cast<std::uint8_t>(s[s.size() - 1]))), nullptr));
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast<std::uint8_t>(s[s.size() - 1] | 0))), nullptr));
}
case error_handler_t::ignore:
@@ -1193,6 +1275,20 @@ class serializer
}
}
/*!
* @brief convert a byte to a uppercase hex representation
* @param[in] byte byte to represent
* @return representation ("00".."FF")
*/
static std::string hex_bytes(std::uint8_t byte)
{
std::string result = "FF";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
/*!
* @brief write a lowercase "\uXXXX" escape sequence into @a string_buffer
*
@@ -1306,7 +1402,7 @@ class serializer
/*!
@brief dump an integer
Dump a given integer, appending it to @ref write_buffer. Works internally with
Dump a given integer to output stream @a o. Works internally with
@a number_buffer.
@param[in] x integer number (signed or unsigned) to dump
@@ -1343,7 +1439,7 @@ class serializer
}
// use a pointer to fill the buffer
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto)
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
number_unsigned_t abs_value;
@@ -1397,7 +1493,7 @@ class serializer
/*!
@brief dump a floating-point number
Dump a given floating-point number, appending it to @ref write_buffer. Works internally
Dump a given floating-point number to output stream @a o. Works internally
with @a number_buffer.
@param[in] x floating-point number to dump
@@ -1458,28 +1554,21 @@ class serializer
// check if the buffer was large enough
JSON_ASSERT(static_cast<std::size_t>(len) < number_buffer.size());
// look up the locale's thousands separator and decimal point now,
// matching what snprintf_float() just used (see lexer::get_decimal_point())
const auto* loc = std::localeconv();
JSON_ASSERT(loc != nullptr);
const char thousands_sep = (loc->thousands_sep == nullptr) ? '\0' : *loc->thousands_sep;
const char decimal_point = (loc->decimal_point == nullptr) ? '\0' : *loc->decimal_point;
// erase thousands separators
if (thousands_sep != '\0')
if (locale.thousands_sep != '\0')
{
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::remove returns an iterator, see https://github.com/nlohmann/json/issues/3081
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, thousands_sep);
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, locale.thousands_sep);
std::fill(end, number_buffer.end(), '\0');
JSON_ASSERT((end - number_buffer.begin()) <= len);
len = (end - number_buffer.begin());
}
// convert decimal point to '.'
if (decimal_point != '\0' && decimal_point != '.')
if (locale.decimal_point != '\0' && locale.decimal_point != '.')
{
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::find returns an iterator, see https://github.com/nlohmann/json/issues/3081
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), decimal_point);
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), locale.decimal_point);
if (dec_pos != number_buffer.end())
{
*dec_pos = '.';
@@ -1524,17 +1613,34 @@ class serializer
*/
number_unsigned_t remove_sign(number_integer_t x) noexcept
{
JSON_ASSERT(x < 0);
JSON_ASSERT(x < 0 && x < (std::numeric_limits<number_integer_t>::max)()); // NOLINT(misc-redundant-expression)
return static_cast<number_unsigned_t>(-(x + 1)) + 1;
}
private:
/// the locale's thousand separator and decimal point characters
struct locale_chars
{
explicit locale_chars(const std::lconv* loc) noexcept
: thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->thousands_sep)))
, decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->decimal_point)))
{}
const char thousands_sep;
const char decimal_point;
};
/// the output of the serializer (non-owning; the adapter lives at the call site)
output_adapter_protocol<char>* o = nullptr;
/// a (hopefully) large enough character buffer
std::array<char, 64> number_buffer{{}};
/// computed once from std::localeconv() at construction; @ref
/// locale_chars keeps std::localeconv()'s pointer from having to be held
/// past the constructor, while still letting these stay const
const locale_chars locale;
/// string buffer
std::array<char, 512> string_buffer{{}};
-11
View File
@@ -3,7 +3,6 @@
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
@@ -37,16 +36,6 @@ StringType to_string(std::size_t value)
return result;
}
/// @return a byte as two uppercase hexadecimal digits
inline std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
///////////////////
// UTF-8 decoding //
///////////////////
+199 -89
View File
@@ -6199,7 +6199,6 @@ NLOHMANN_JSON_NAMESPACE_END
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
@@ -6235,16 +6234,6 @@ StringType to_string(std::size_t value)
return result;
}
/// @return a byte as two uppercase hexadecimal digits
inline std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
///////////////////
// UTF-8 decoding //
///////////////////
@@ -10078,10 +10067,15 @@ class lexer : public lexer_base<BasicJsonType>
case '\n':
case '\r':
case char_traits<char_type>::eof():
return true;
#if !JSON_STRICT_NUL_HANDLING
case '\0':
#endif
// a NUL byte is the end of the input (see scan()),
// so leave it for scan() to see
unget();
return true;
#endif
default:
break;
@@ -20157,7 +20151,9 @@ NLOHMANN_JSON_NAMESPACE_END
#include <algorithm> // copy
#include <cstddef> // size_t
#include <iterator> // back_inserter
#include <memory> // shared_ptr, make_shared
#include <string> // basic_string
#include <utility> // move
@@ -20179,10 +20175,6 @@ namespace detail
template<typename CharType> struct output_adapter_protocol
{
virtual void write_character(CharType c) = 0;
/// @param[in] s pointer to the characters to write; binary_writer legitimately
/// passes a null pointer together with length 0 for an empty
/// string or binary value, so implementations must tolerate that
/// @param[in] length number of characters at @a s
virtual void write_characters(const CharType* s, std::size_t length) = 0;
virtual ~output_adapter_protocol() = default;
@@ -20249,6 +20241,7 @@ class output_vector_adapter : public output_adapter_protocol<CharType>
sink.write_character(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
sink.write_characters(s, length);
@@ -20273,6 +20266,7 @@ class output_stream_adapter : public output_adapter_protocol<CharType>
stream.put(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
stream.write(s, static_cast<std::streamsize>(length));
@@ -20297,6 +20291,7 @@ class output_string_adapter : public output_adapter_protocol<CharType>
str.push_back(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
str.append(s, length);
@@ -20369,8 +20364,6 @@ NLOHMANN_JSON_NAMESPACE_END
// #include <nlohmann/detail/string_concat.hpp>
// #include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -22620,10 +22613,20 @@ class binary_writer
const std::size_t valid = valid_utf8_prefix(data, s.size());
if (JSON_HEDLEY_UNLIKELY(valid != s.size()))
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", detail::hex_byte(data[valid])), &context));
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", hex_byte(data[valid])), &context));
}
}
/// @return a byte as two uppercase hexadecimal digits
static std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
/*!
@brief write an integer in the shortest encoding
@@ -22956,23 +22959,24 @@ NLOHMANN_JSON_NAMESPACE_END
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include <algorithm> // remove, fill, find, none_of, min
#include <algorithm> // reverse, remove, fill, find, none_of, min
#include <array> // array
#include <clocale> // localeconv, lconv
#include <cmath> // isfinite
#include <cmath> // labs, isfinite, isnan, signbit
#include <cstddef> // size_t, ptrdiff_t
#include <cstdint> // uint8_t
#include <cstdio> // snprintf
#include <cstring> // memcpy, memset
#include <iterator> // next
#include <limits> // numeric_limits
#include <string> // string, char_traits
#include <type_traits> // is_same
#include <utility> // move
#include <vector> // vector
// #include <nlohmann/detail/conversions/to_chars.hpp>
@@ -24104,6 +24108,8 @@ NLOHMANN_JSON_NAMESPACE_END
// #include <nlohmann/detail/meta/cpp_future.hpp>
// #include <nlohmann/detail/output/binary_writer.hpp>
// #include <nlohmann/detail/output/output_adapters.hpp>
// #include <nlohmann/detail/recursion_depth_limit.hpp>
@@ -24163,6 +24169,7 @@ class serializer
const std::size_t indent_step_ = 0,
error_handler_t error_handler_ = error_handler_t::strict)
: o(&s)
, locale(std::localeconv())
, indent_char(ichar)
, pretty_print(pretty_print_)
, ensure_ascii(ensure_ascii_)
@@ -24185,10 +24192,9 @@ class serializer
additional parameter. Arrays and objects are serialized without recursion,
however deeply they are nested.
- strings and object keys are escaped using @ref dump_escaped
- integer numbers are converted using a digit-pair lookup table (@ref dump_integer)
- floating-point numbers are converted to a string using @ref dump_float, which
uses `to_chars` for IEEE-754 types and `snprintf` otherwise
- strings and object keys are escaped using `escape_string()`
- integer numbers are converted implicitly via `operator<<`
- floating-point numbers are converted to a string using `"%g"` format
- binary values are serialized as objects containing the subtype and the
byte array
@@ -24363,16 +24369,127 @@ class serializer
}
case value_t::string:
case value_t::binary:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::discarded:
case value_t::null:
default:
dump_scalar(val, current_indent);
{
put_char('"');
dump_escaped(*val.m_data.m_value.string);
put_char('"');
return;
}
case value_t::binary:
{
if (pretty_print)
{
put_literal("{\n");
// variable to hold indentation for recursive calls
const auto new_indent = next_indent(current_indent, indent_step);
put_indent(new_indent);
put_literal("\"bytes\": [");
if (!val.m_data.m_value.binary->empty())
{
for (auto i = val.m_data.m_value.binary->cbegin();
i != val.m_data.m_value.binary->cend() - 1; ++i)
{
dump_byte(*i);
put_literal(", ");
}
dump_byte(val.m_data.m_value.binary->back());
}
put_literal("],\n");
put_indent(new_indent);
put_literal("\"subtype\": ");
if (val.m_data.m_value.binary->has_subtype())
{
dump_integer(val.m_data.m_value.binary->subtype());
}
else
{
put_literal("null");
}
put_char('\n');
put_indent(current_indent);
put_char('}');
}
else
{
put_literal("{\"bytes\":[");
if (!val.m_data.m_value.binary->empty())
{
for (auto i = val.m_data.m_value.binary->cbegin();
i != val.m_data.m_value.binary->cend() - 1; ++i)
{
dump_byte(*i);
put_char(',');
}
dump_byte(val.m_data.m_value.binary->back());
}
put_literal("],\"subtype\":");
if (val.m_data.m_value.binary->has_subtype())
{
dump_integer(val.m_data.m_value.binary->subtype());
put_char('}');
}
else
{
put_literal("null}");
}
}
return;
}
case value_t::boolean:
{
if (val.m_data.m_value.boolean)
{
put_literal("true");
}
else
{
put_literal("false");
}
return;
}
case value_t::number_integer:
{
dump_integer(val.m_data.m_value.number_integer);
return;
}
case value_t::number_unsigned:
{
dump_integer(val.m_data.m_value.number_unsigned);
return;
}
case value_t::number_float:
{
dump_float(val.m_data.m_value.number_float);
return;
}
case value_t::discarded:
{
put_literal("<discarded>");
return;
}
case value_t::null:
{
put_literal("null");
return;
}
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
}
@@ -24529,9 +24646,9 @@ class serializer
@brief serialize the value @a val, but not the elements of a container
An object or array with elements is opened and pushed onto @a stack for
@ref dump_iteratively to walk; everything else - including a binary value,
@ref dump_internal to walk; everything else - including a binary value,
which looks like an object but has no elements to descend into - is written
out in full by @ref dump_scalar.
out here in full.
*/
void dump_value(const BasicJsonType& val,
const std::size_t current_indent,
@@ -24589,35 +24706,6 @@ class serializer
return;
}
case value_t::string:
case value_t::binary:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::discarded:
case value_t::null:
default:
dump_scalar(val, current_indent);
return;
}
}
/*!
@brief serialize the value @a val, which is neither an object nor an array
Shared by @ref dump_internal and @ref dump_value, so that a value is written
the same way however deeply it is nested. A binary value is written out here
in full: it looks like an object, but has no elements to descend into.
@param[in] val value to serialize; not an object or array
@param[in] current_indent the indentation of @a val, used for a
pretty-printed binary value
*/
void dump_scalar(const BasicJsonType& val, const std::size_t current_indent)
{
switch (val.m_data.m_type)
{
case value_t::string:
{
put_char('"');
@@ -24738,8 +24826,6 @@ class serializer
return;
}
case value_t::object: // LCOV_EXCL_LINE
case value_t::array: // LCOV_EXCL_LINE
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
@@ -24768,7 +24854,7 @@ class serializer
Escape a string by replacing certain special characters by a sequence of an
escape character (backslash) and another character and other control
characters by a sequence of "\u" followed by a four-digit hex
representation. The escaped string is appended to @ref write_buffer.
representation. The escaped string is written to output stream @a o.
@param[in] s the string to escape
@@ -24962,7 +25048,7 @@ class serializer
{
case error_handler_t::strict:
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", detail::hex_byte(byte)), nullptr));
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", hex_bytes(byte | 0)), nullptr));
}
case error_handler_t::ignore:
@@ -24995,9 +25081,9 @@ class serializer
}
else
{
string_buffer[bytes++] = '\xEF';
string_buffer[bytes++] = '\xBF';
string_buffer[bytes++] = '\xBD';
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xEF');
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBF');
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBD');
}
// write buffer and reset index; there must be 13 bytes
@@ -25054,7 +25140,7 @@ class serializer
{
case error_handler_t::strict:
{
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", detail::hex_byte(static_cast<std::uint8_t>(s[s.size() - 1]))), nullptr));
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast<std::uint8_t>(s[s.size() - 1] | 0))), nullptr));
}
case error_handler_t::ignore:
@@ -25275,6 +25361,20 @@ class serializer
}
}
/*!
* @brief convert a byte to a uppercase hex representation
* @param[in] byte byte to represent
* @return representation ("00".."FF")
*/
static std::string hex_bytes(std::uint8_t byte)
{
std::string result = "FF";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
/*!
* @brief write a lowercase "\uXXXX" escape sequence into @a string_buffer
*
@@ -25388,7 +25488,7 @@ class serializer
/*!
@brief dump an integer
Dump a given integer, appending it to @ref write_buffer. Works internally with
Dump a given integer to output stream @a o. Works internally with
@a number_buffer.
@param[in] x integer number (signed or unsigned) to dump
@@ -25425,7 +25525,7 @@ class serializer
}
// use a pointer to fill the buffer
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto)
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
number_unsigned_t abs_value;
@@ -25479,7 +25579,7 @@ class serializer
/*!
@brief dump a floating-point number
Dump a given floating-point number, appending it to @ref write_buffer. Works internally
Dump a given floating-point number to output stream @a o. Works internally
with @a number_buffer.
@param[in] x floating-point number to dump
@@ -25540,28 +25640,21 @@ class serializer
// check if the buffer was large enough
JSON_ASSERT(static_cast<std::size_t>(len) < number_buffer.size());
// look up the locale's thousands separator and decimal point now,
// matching what snprintf_float() just used (see lexer::get_decimal_point())
const auto* loc = std::localeconv();
JSON_ASSERT(loc != nullptr);
const char thousands_sep = (loc->thousands_sep == nullptr) ? '\0' : *loc->thousands_sep;
const char decimal_point = (loc->decimal_point == nullptr) ? '\0' : *loc->decimal_point;
// erase thousands separators
if (thousands_sep != '\0')
if (locale.thousands_sep != '\0')
{
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::remove returns an iterator, see https://github.com/nlohmann/json/issues/3081
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, thousands_sep);
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, locale.thousands_sep);
std::fill(end, number_buffer.end(), '\0');
JSON_ASSERT((end - number_buffer.begin()) <= len);
len = (end - number_buffer.begin());
}
// convert decimal point to '.'
if (decimal_point != '\0' && decimal_point != '.')
if (locale.decimal_point != '\0' && locale.decimal_point != '.')
{
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::find returns an iterator, see https://github.com/nlohmann/json/issues/3081
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), decimal_point);
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), locale.decimal_point);
if (dec_pos != number_buffer.end())
{
*dec_pos = '.';
@@ -25606,17 +25699,34 @@ class serializer
*/
number_unsigned_t remove_sign(number_integer_t x) noexcept
{
JSON_ASSERT(x < 0);
JSON_ASSERT(x < 0 && x < (std::numeric_limits<number_integer_t>::max)()); // NOLINT(misc-redundant-expression)
return static_cast<number_unsigned_t>(-(x + 1)) + 1;
}
private:
/// the locale's thousand separator and decimal point characters
struct locale_chars
{
explicit locale_chars(const std::lconv* loc) noexcept
: thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->thousands_sep)))
, decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->decimal_point)))
{}
const char thousands_sep;
const char decimal_point;
};
/// the output of the serializer (non-owning; the adapter lives at the call site)
output_adapter_protocol<char>* o = nullptr;
/// a (hopefully) large enough character buffer
std::array<char, 64> number_buffer{{}};
/// computed once from std::localeconv() at construction; @ref
/// locale_chars keeps std::localeconv()'s pointer from having to be held
/// past the constructor, while still letting these stay const
const locale_chars locale;
/// string buffer
std::array<char, 512> string_buffer{{}};
+39
View File
@@ -592,6 +592,45 @@ TEST_CASE("parser class")
// parsing from a string literal is unaffected either way
CHECK(json::parse("123") == json(123));
// a NUL byte that ends a // comment ends the input just
// like a NUL byte anywhere else (issue #5659); before the
// fix, the NUL was consumed as part of the comment, and
// scanning continued with whatever followed it
{
// same as "//c" alone (real end of input after the
// comment), rather than continuing with "[1]"
std::string s1 = "//c";
s1.push_back('\0');
s1 += "[1]";
json _; // NOLINT(readability-identifier-naming)
CHECK_THROWS_WITH_AS(_ = json::parse(s1, nullptr, true, true),
"[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - unexpected end of input; expected '[', '{', or a literal",
json::parse_error&);
CHECK_FALSE(json::accept(s1, true, true));
}
{
// same as "[1, //c" alone, rather than continuing with " 2]"
std::string s2 = "[1, //c";
s2.push_back('\0');
s2 += " 2]";
json _; // NOLINT(readability-identifier-naming)
CHECK_THROWS_WITH_AS(_ = json::parse(s2, nullptr, true, true),
"[json.exception.parse_error.101] parse error at line 1, column 8: syntax error while parsing value - unexpected end of input; expected '[', '{', or a literal",
json::parse_error&);
CHECK_FALSE(json::accept(s2, true, true));
}
{
// same as "1 //c" alone: the comment (and the NUL that
// ends it) is ignored, and "x" is never reached
std::string s3 = "1 //c";
s3.push_back('\0');
s3 += "x";
CHECK(json::parse(s3, nullptr, true, true) == json(1));
CHECK(json::accept(s3, true, true));
}
}
#endif
-92
View File
@@ -14,10 +14,7 @@ using nlohmann::json;
#include <array>
#include <clocale>
#include <limits>
#include <map>
#include <ostream>
#include <streambuf>
#include <string>
#include <utility>
#include <vector>
@@ -388,92 +385,3 @@ TEST_CASE("locale with a multi-byte decimal point")
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
}
namespace
{
// a streambuf that switches LC_NUMERIC the first time anything is written to
// it, so a dump() in progress can be made to change locale mid-flight: after
// the serializer was constructed (and, before #5709 item 3, after it had
// cached std::localeconv() for the whole call) but before a later float is
// converted
struct LocaleSwitchingStreambuf final : std::streambuf
{
explicit LocaleSwitchingStreambuf(const char* switch_to)
: locale_after_first_write(switch_to)
{}
std::string data {}; // NOLINT(readability-redundant-member-init)
std::string locale_after_first_write;
bool switched = false;
std::streamsize xsputn(const char* s, std::streamsize n) override
{
if (!switched)
{
switched = std::setlocale(LC_NUMERIC, locale_after_first_write.c_str()) != nullptr;
}
data.append(s, static_cast<std::size_t>(n));
return n;
}
};
} // namespace
TEST_CASE("locale changes during a single dump() (#5709 item 3)")
{
// dump_float() only reads the locale on the snprintf path, taken for a
// number_float_t that is not an IEEE-754 single or double, i.e. not
// (is_iec559 && digits == 24 && max_exponent == 128) and not (is_iec559
// && digits == 53 && max_exponent == 1024) - see dump_float(). Checking
// is_iec559 alone is not enough: on x86_64, long double is a 64-bit
// (80-bit extended) format for which is_iec559 is also true, so it still
// takes the snprintf path this test means to exercise. Only a
// number_float_t whose digits/max_exponent match float or double (e.g.
// long double on 64-bit Arm, where it is IEEE-754 double) takes the
// locale-independent to_chars() path instead, and this test is a no-op
// there.
using long_double_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, long double>;
using ld_limits = std::numeric_limits<long_double_json::number_float_t>;
const bool is_ieee_single_or_double =
(ld_limits::is_iec559 && ld_limits::digits == 24 && ld_limits::max_exponent == 128) ||
(ld_limits::is_iec559 && ld_limits::digits == 53 && ld_limits::max_exponent == 1024);
if (is_ieee_single_or_double)
{
MESSAGE("long double is IEEE-754 single or double on this platform; dump_float()'s snprintf/locale path is not exercised here");
}
const char* de_DE_name = "de_DE.UTF-8";
if (std::setlocale(LC_NUMERIC, de_DE_name) == nullptr)
{
de_DE_name = "de_DE";
if (std::setlocale(LC_NUMERIC, de_DE_name) == nullptr)
{
MESSAGE("locale de_DE is not usable");
return;
}
}
const std::string decimal_point = std::localeconv()->decimal_point;
REQUIRE(std::setlocale(LC_NUMERIC, "C") != nullptr);
if (decimal_point != ",")
{
MESSAGE("de_DE's decimal point is not ',' on this platform, skipping");
return;
}
// a string long enough to overflow the serializer's internal write
// buffer, so that it is flushed to the output adapter - and the locale
// switched - before the number after it is converted
const std::string padding(5000, 'a');
const long_double_json j = { padding, 1234.5L };
LocaleSwitchingStreambuf buf(de_DE_name);
std::ostream os(&buf);
os << j;
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
REQUIRE(buf.switched);
// whatever locale was in effect when the float was actually converted,
// the output is normalized to use '.' as the decimal point: it must be
// looked up at conversion time, not once for the whole dump() - the same
// fix #5597 made on the parser side
CHECK(buf.data == "[\"" + padding + "\",1234.5]");
}