Compare commits

..
Author SHA1 Message Date
Niels Lohmann cc6d74feb0 Copy a pair-shaped array value under JSON_BRACE_INIT_COPY_SEMANTICS
With JSON_BRACE_INIT_COPY_SEMANTICS enabled, single-element brace
initialization from a JSON value decided whether to copy the value or
build an object by inspecting the value's runtime shape: a two-element
array whose first element is a string, such as ["key", 42], was turned
into an object instead of being copied. This made the behavior depend
on the element's content, and it did not distinguish an existing value
of this shape from a nested braced pair written in the source, such as
the inner {"key", "value"} of {{"key", "value"}}.

json_ref now records whether it was constructed from a braced list
(true only for the std::initializer_list<json_ref> constructor used
for nested braced lists) or from a value. The initializer-list
constructor uses this to copy or move a single non-braced-list element
before deciding whether the list describes an object, so a JSON value
is always copied regardless of its shape, while a braced pair written
in the source still creates an object.

Fixes #5662.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 23:41:48 +02:00
10 changed files with 454 additions and 270 deletions
@@ -50,9 +50,11 @@ The default value is `0` (disabled — existing behavior is preserved).
```
Code that relies on these producing arrays must use `json::array()` instead (see below). Lists with more than one
element, and a single `[string, value]` pair such as `{{"key", "value"}}`, which still creates an object, are not
affected. The library's own conversions are not affected either: for example, `std::tuple<int>{5}` still becomes
`[5]`.
element, and a single `[string, value]` pair *written as a braced list*, such as `{{"key", "value"}}`, which still
creates an object, are not affected. This exception is based on how the pair is written, not on the shape of its
value: an existing JSON value that happens to be a two-element array with a string as its first element, such as
`json arr = {"key", 42};`, is still copied by `json j{arr};` rather than turned into an object. The library's own
conversions are not affected either: for example, `std::tuple<int>{5}` still becomes `[5]`.
!!! note "ABI compatibility"
+9
View File
@@ -34,6 +34,7 @@ class json_ref
json_ref(std::initializer_list<json_ref> init)
: owned_value(init)
, braced_list(true)
{}
template <
@@ -69,9 +70,17 @@ class json_ref
return &** this;
}
/// whether the value was written as a braced list, such as {"key", 1},
/// rather than given as a value
bool is_braced_list() const noexcept
{
return braced_list;
}
private:
mutable value_type owned_value = nullptr;
value_type const* value_ref = nullptr;
bool braced_list = false;
};
} // namespace detail
@@ -29,7 +29,6 @@
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/output/output_adapters.hpp>
#include <nlohmann/detail/string_concat.hpp>
#include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -2279,10 +2278,20 @@ class binary_writer
const std::size_t valid = valid_utf8_prefix(data, s.size());
if (JSON_HEDLEY_UNLIKELY(valid != s.size()))
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", detail::hex_byte(data[valid])), &context));
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", hex_byte(data[valid])), &context));
}
}
/// @return a byte as two uppercase hexadecimal digits
static std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
/*!
@brief write an integer in the shortest encoding
@@ -8,7 +8,9 @@
#pragma once
#include <algorithm> // copy
#include <cstddef> // size_t
#include <iterator> // back_inserter
#include <memory> // shared_ptr, make_shared
#include <string> // basic_string
#include <utility> // move
@@ -29,10 +31,6 @@ namespace detail
template<typename CharType> struct output_adapter_protocol
{
virtual void write_character(CharType c) = 0;
/// @param[in] s pointer to the characters to write; binary_writer legitimately
/// passes a null pointer together with length 0 for an empty
/// string or binary value, so implementations must tolerate that
/// @param[in] length number of characters at @a s
virtual void write_characters(const CharType* s, std::size_t length) = 0;
virtual ~output_adapter_protocol() = default;
@@ -99,6 +97,7 @@ class output_vector_adapter : public output_adapter_protocol<CharType>
sink.write_character(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
sink.write_characters(s, length);
@@ -123,6 +122,7 @@ class output_stream_adapter : public output_adapter_protocol<CharType>
stream.put(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
stream.write(s, static_cast<std::streamsize>(length));
@@ -147,6 +147,7 @@ class output_string_adapter : public output_adapter_protocol<CharType>
str.push_back(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
str.append(s, length);
+176 -70
View File
@@ -3,23 +3,24 @@
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#pragma once
#include <algorithm> // remove, fill, find, none_of, min
#include <algorithm> // reverse, remove, fill, find, none_of, min
#include <array> // array
#include <clocale> // localeconv, lconv
#include <cmath> // isfinite
#include <cmath> // labs, isfinite, isnan, signbit
#include <cstddef> // size_t, ptrdiff_t
#include <cstdint> // uint8_t
#include <cstdio> // snprintf
#include <cstring> // memcpy, memset
#include <iterator> // next
#include <limits> // numeric_limits
#include <string> // string, char_traits
#include <type_traits> // is_same
#include <utility> // move
#include <vector> // vector
#include <nlohmann/detail/conversions/to_chars.hpp>
@@ -27,6 +28,7 @@
#include <nlohmann/detail/input/string_scan.hpp>
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/cpp_future.hpp>
#include <nlohmann/detail/output/binary_writer.hpp>
#include <nlohmann/detail/output/output_adapters.hpp>
#include <nlohmann/detail/recursion_depth_limit.hpp>
#include <nlohmann/detail/string_concat.hpp>
@@ -81,6 +83,7 @@ class serializer
const std::size_t indent_step_ = 0,
error_handler_t error_handler_ = error_handler_t::strict)
: o(&s)
, locale(std::localeconv())
, indent_char(ichar)
, pretty_print(pretty_print_)
, ensure_ascii(ensure_ascii_)
@@ -103,10 +106,9 @@ class serializer
additional parameter. Arrays and objects are serialized without recursion,
however deeply they are nested.
- strings and object keys are escaped using @ref dump_escaped
- integer numbers are converted using a digit-pair lookup table (@ref dump_integer)
- floating-point numbers are converted to a string using @ref dump_float, which
uses `to_chars` for IEEE-754 types and `snprintf` otherwise
- strings and object keys are escaped using `escape_string()`
- integer numbers are converted implicitly via `operator<<`
- floating-point numbers are converted to a string using `"%g"` format
- binary values are serialized as objects containing the subtype and the
byte array
@@ -281,16 +283,127 @@ class serializer
}
case value_t::string:
case value_t::binary:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::discarded:
case value_t::null:
default:
dump_scalar(val, current_indent);
{
put_char('"');
dump_escaped(*val.m_data.m_value.string);
put_char('"');
return;
}
case value_t::binary:
{
if (pretty_print)
{
put_literal("{\n");
// variable to hold indentation for recursive calls
const auto new_indent = next_indent(current_indent, indent_step);
put_indent(new_indent);
put_literal("\"bytes\": [");
if (!val.m_data.m_value.binary->empty())
{
for (auto i = val.m_data.m_value.binary->cbegin();
i != val.m_data.m_value.binary->cend() - 1; ++i)
{
dump_byte(*i);
put_literal(", ");
}
dump_byte(val.m_data.m_value.binary->back());
}
put_literal("],\n");
put_indent(new_indent);
put_literal("\"subtype\": ");
if (val.m_data.m_value.binary->has_subtype())
{
dump_integer(val.m_data.m_value.binary->subtype());
}
else
{
put_literal("null");
}
put_char('\n');
put_indent(current_indent);
put_char('}');
}
else
{
put_literal("{\"bytes\":[");
if (!val.m_data.m_value.binary->empty())
{
for (auto i = val.m_data.m_value.binary->cbegin();
i != val.m_data.m_value.binary->cend() - 1; ++i)
{
dump_byte(*i);
put_char(',');
}
dump_byte(val.m_data.m_value.binary->back());
}
put_literal("],\"subtype\":");
if (val.m_data.m_value.binary->has_subtype())
{
dump_integer(val.m_data.m_value.binary->subtype());
put_char('}');
}
else
{
put_literal("null}");
}
}
return;
}
case value_t::boolean:
{
if (val.m_data.m_value.boolean)
{
put_literal("true");
}
else
{
put_literal("false");
}
return;
}
case value_t::number_integer:
{
dump_integer(val.m_data.m_value.number_integer);
return;
}
case value_t::number_unsigned:
{
dump_integer(val.m_data.m_value.number_unsigned);
return;
}
case value_t::number_float:
{
dump_float(val.m_data.m_value.number_float);
return;
}
case value_t::discarded:
{
put_literal("<discarded>");
return;
}
case value_t::null:
{
put_literal("null");
return;
}
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
}
@@ -447,9 +560,9 @@ class serializer
@brief serialize the value @a val, but not the elements of a container
An object or array with elements is opened and pushed onto @a stack for
@ref dump_iteratively to walk; everything else - including a binary value,
@ref dump_internal to walk; everything else - including a binary value,
which looks like an object but has no elements to descend into - is written
out in full by @ref dump_scalar.
out here in full.
*/
void dump_value(const BasicJsonType& val,
const std::size_t current_indent,
@@ -507,35 +620,6 @@ class serializer
return;
}
case value_t::string:
case value_t::binary:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::discarded:
case value_t::null:
default:
dump_scalar(val, current_indent);
return;
}
}
/*!
@brief serialize the value @a val, which is neither an object nor an array
Shared by @ref dump_internal and @ref dump_value, so that a value is written
the same way however deeply it is nested. A binary value is written out here
in full: it looks like an object, but has no elements to descend into.
@param[in] val value to serialize; not an object or array
@param[in] current_indent the indentation of @a val, used for a
pretty-printed binary value
*/
void dump_scalar(const BasicJsonType& val, const std::size_t current_indent)
{
switch (val.m_data.m_type)
{
case value_t::string:
{
put_char('"');
@@ -656,8 +740,6 @@ class serializer
return;
}
case value_t::object: // LCOV_EXCL_LINE
case value_t::array: // LCOV_EXCL_LINE
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
@@ -686,7 +768,7 @@ class serializer
Escape a string by replacing certain special characters by a sequence of an
escape character (backslash) and another character and other control
characters by a sequence of "\u" followed by a four-digit hex
representation. The escaped string is appended to @ref write_buffer.
representation. The escaped string is written to output stream @a o.
@param[in] s the string to escape
@@ -880,7 +962,7 @@ class serializer
{
case error_handler_t::strict:
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", detail::hex_byte(byte)), nullptr));
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", hex_bytes(byte | 0)), nullptr));
}
case error_handler_t::ignore:
@@ -913,9 +995,9 @@ class serializer
}
else
{
string_buffer[bytes++] = '\xEF';
string_buffer[bytes++] = '\xBF';
string_buffer[bytes++] = '\xBD';
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xEF');
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBF');
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBD');
}
// write buffer and reset index; there must be 13 bytes
@@ -972,7 +1054,7 @@ class serializer
{
case error_handler_t::strict:
{
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", detail::hex_byte(static_cast<std::uint8_t>(s[s.size() - 1]))), nullptr));
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast<std::uint8_t>(s[s.size() - 1] | 0))), nullptr));
}
case error_handler_t::ignore:
@@ -1193,6 +1275,20 @@ class serializer
}
}
/*!
* @brief convert a byte to a uppercase hex representation
* @param[in] byte byte to represent
* @return representation ("00".."FF")
*/
static std::string hex_bytes(std::uint8_t byte)
{
std::string result = "FF";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
/*!
* @brief write a lowercase "\uXXXX" escape sequence into @a string_buffer
*
@@ -1306,7 +1402,7 @@ class serializer
/*!
@brief dump an integer
Dump a given integer, appending it to @ref write_buffer. Works internally with
Dump a given integer to output stream @a o. Works internally with
@a number_buffer.
@param[in] x integer number (signed or unsigned) to dump
@@ -1343,7 +1439,7 @@ class serializer
}
// use a pointer to fill the buffer
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto)
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
number_unsigned_t abs_value;
@@ -1397,7 +1493,7 @@ class serializer
/*!
@brief dump a floating-point number
Dump a given floating-point number, appending it to @ref write_buffer. Works internally
Dump a given floating-point number to output stream @a o. Works internally
with @a number_buffer.
@param[in] x floating-point number to dump
@@ -1458,28 +1554,21 @@ class serializer
// check if the buffer was large enough
JSON_ASSERT(static_cast<std::size_t>(len) < number_buffer.size());
// look up the locale's thousands separator and decimal point now,
// matching what snprintf_float() just used (see lexer::get_decimal_point())
const auto* loc = std::localeconv();
JSON_ASSERT(loc != nullptr);
const char thousands_sep = (loc->thousands_sep == nullptr) ? '\0' : *loc->thousands_sep;
const char decimal_point = (loc->decimal_point == nullptr) ? '\0' : *loc->decimal_point;
// erase thousands separators
if (thousands_sep != '\0')
if (locale.thousands_sep != '\0')
{
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::remove returns an iterator, see https://github.com/nlohmann/json/issues/3081
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, thousands_sep);
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, locale.thousands_sep);
std::fill(end, number_buffer.end(), '\0');
JSON_ASSERT((end - number_buffer.begin()) <= len);
len = (end - number_buffer.begin());
}
// convert decimal point to '.'
if (decimal_point != '\0' && decimal_point != '.')
if (locale.decimal_point != '\0' && locale.decimal_point != '.')
{
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::find returns an iterator, see https://github.com/nlohmann/json/issues/3081
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), decimal_point);
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), locale.decimal_point);
if (dec_pos != number_buffer.end())
{
*dec_pos = '.';
@@ -1524,17 +1613,34 @@ class serializer
*/
number_unsigned_t remove_sign(number_integer_t x) noexcept
{
JSON_ASSERT(x < 0);
JSON_ASSERT(x < 0 && x < (std::numeric_limits<number_integer_t>::max)()); // NOLINT(misc-redundant-expression)
return static_cast<number_unsigned_t>(-(x + 1)) + 1;
}
private:
/// the locale's thousand separator and decimal point characters
struct locale_chars
{
explicit locale_chars(const std::lconv* loc) noexcept
: thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->thousands_sep)))
, decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->decimal_point)))
{}
const char thousands_sep;
const char decimal_point;
};
/// the output of the serializer (non-owning; the adapter lives at the call site)
output_adapter_protocol<char>* o = nullptr;
/// a (hopefully) large enough character buffer
std::array<char, 64> number_buffer{{}};
/// computed once from std::localeconv() at construction; @ref
/// locale_chars keeps std::localeconv()'s pointer from having to be held
/// past the constructor, while still letting these stay const
const locale_chars locale;
/// string buffer
std::array<char, 512> string_buffer{{}};
-11
View File
@@ -3,7 +3,6 @@
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
@@ -37,16 +36,6 @@ StringType to_string(std::size_t value)
return result;
}
/// @return a byte as two uppercase hexadecimal digits
inline std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
///////////////////
// UTF-8 decoding //
///////////////////
+12
View File
@@ -1672,6 +1672,18 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
bool type_deduction = true,
value_t manual_type = value_t::array)
{
#if JSON_BRACE_INIT_COPY_SEMANTICS
// a single element that is a value rather than a braced list is
// copied or moved as is, whatever its content looks like
if (type_deduction && init.size() == 1 && !init.begin()->is_braced_list())
{
*this = init.begin()->moved_or_copied();
set_parents();
assert_invariant();
return;
}
#endif
// check if each element is an array with two elements whose first
// element is a string
bool is_an_object = std::all_of(init.begin(), init.end(),
+214 -88
View File
@@ -6199,7 +6199,6 @@ NLOHMANN_JSON_NAMESPACE_END
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
@@ -6235,16 +6234,6 @@ StringType to_string(std::size_t value)
return result;
}
/// @return a byte as two uppercase hexadecimal digits
inline std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
///////////////////
// UTF-8 decoding //
///////////////////
@@ -20058,6 +20047,7 @@ class json_ref
json_ref(std::initializer_list<json_ref> init)
: owned_value(init)
, braced_list(true)
{}
template <
@@ -20093,9 +20083,17 @@ class json_ref
return &** this;
}
/// whether the value was written as a braced list, such as {"key", 1},
/// rather than given as a value
bool is_braced_list() const noexcept
{
return braced_list;
}
private:
mutable value_type owned_value = nullptr;
value_type const* value_ref = nullptr;
bool braced_list = false;
};
} // namespace detail
@@ -20157,7 +20155,9 @@ NLOHMANN_JSON_NAMESPACE_END
#include <algorithm> // copy
#include <cstddef> // size_t
#include <iterator> // back_inserter
#include <memory> // shared_ptr, make_shared
#include <string> // basic_string
#include <utility> // move
@@ -20179,10 +20179,6 @@ namespace detail
template<typename CharType> struct output_adapter_protocol
{
virtual void write_character(CharType c) = 0;
/// @param[in] s pointer to the characters to write; binary_writer legitimately
/// passes a null pointer together with length 0 for an empty
/// string or binary value, so implementations must tolerate that
/// @param[in] length number of characters at @a s
virtual void write_characters(const CharType* s, std::size_t length) = 0;
virtual ~output_adapter_protocol() = default;
@@ -20249,6 +20245,7 @@ class output_vector_adapter : public output_adapter_protocol<CharType>
sink.write_character(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
sink.write_characters(s, length);
@@ -20273,6 +20270,7 @@ class output_stream_adapter : public output_adapter_protocol<CharType>
stream.put(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
stream.write(s, static_cast<std::streamsize>(length));
@@ -20297,6 +20295,7 @@ class output_string_adapter : public output_adapter_protocol<CharType>
str.push_back(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
str.append(s, length);
@@ -20369,8 +20368,6 @@ NLOHMANN_JSON_NAMESPACE_END
// #include <nlohmann/detail/string_concat.hpp>
// #include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -22620,10 +22617,20 @@ class binary_writer
const std::size_t valid = valid_utf8_prefix(data, s.size());
if (JSON_HEDLEY_UNLIKELY(valid != s.size()))
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", detail::hex_byte(data[valid])), &context));
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", hex_byte(data[valid])), &context));
}
}
/// @return a byte as two uppercase hexadecimal digits
static std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
/*!
@brief write an integer in the shortest encoding
@@ -22956,23 +22963,24 @@ NLOHMANN_JSON_NAMESPACE_END
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include <algorithm> // remove, fill, find, none_of, min
#include <algorithm> // reverse, remove, fill, find, none_of, min
#include <array> // array
#include <clocale> // localeconv, lconv
#include <cmath> // isfinite
#include <cmath> // labs, isfinite, isnan, signbit
#include <cstddef> // size_t, ptrdiff_t
#include <cstdint> // uint8_t
#include <cstdio> // snprintf
#include <cstring> // memcpy, memset
#include <iterator> // next
#include <limits> // numeric_limits
#include <string> // string, char_traits
#include <type_traits> // is_same
#include <utility> // move
#include <vector> // vector
// #include <nlohmann/detail/conversions/to_chars.hpp>
@@ -24104,6 +24112,8 @@ NLOHMANN_JSON_NAMESPACE_END
// #include <nlohmann/detail/meta/cpp_future.hpp>
// #include <nlohmann/detail/output/binary_writer.hpp>
// #include <nlohmann/detail/output/output_adapters.hpp>
// #include <nlohmann/detail/recursion_depth_limit.hpp>
@@ -24163,6 +24173,7 @@ class serializer
const std::size_t indent_step_ = 0,
error_handler_t error_handler_ = error_handler_t::strict)
: o(&s)
, locale(std::localeconv())
, indent_char(ichar)
, pretty_print(pretty_print_)
, ensure_ascii(ensure_ascii_)
@@ -24185,10 +24196,9 @@ class serializer
additional parameter. Arrays and objects are serialized without recursion,
however deeply they are nested.
- strings and object keys are escaped using @ref dump_escaped
- integer numbers are converted using a digit-pair lookup table (@ref dump_integer)
- floating-point numbers are converted to a string using @ref dump_float, which
uses `to_chars` for IEEE-754 types and `snprintf` otherwise
- strings and object keys are escaped using `escape_string()`
- integer numbers are converted implicitly via `operator<<`
- floating-point numbers are converted to a string using `"%g"` format
- binary values are serialized as objects containing the subtype and the
byte array
@@ -24363,16 +24373,127 @@ class serializer
}
case value_t::string:
case value_t::binary:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::discarded:
case value_t::null:
default:
dump_scalar(val, current_indent);
{
put_char('"');
dump_escaped(*val.m_data.m_value.string);
put_char('"');
return;
}
case value_t::binary:
{
if (pretty_print)
{
put_literal("{\n");
// variable to hold indentation for recursive calls
const auto new_indent = next_indent(current_indent, indent_step);
put_indent(new_indent);
put_literal("\"bytes\": [");
if (!val.m_data.m_value.binary->empty())
{
for (auto i = val.m_data.m_value.binary->cbegin();
i != val.m_data.m_value.binary->cend() - 1; ++i)
{
dump_byte(*i);
put_literal(", ");
}
dump_byte(val.m_data.m_value.binary->back());
}
put_literal("],\n");
put_indent(new_indent);
put_literal("\"subtype\": ");
if (val.m_data.m_value.binary->has_subtype())
{
dump_integer(val.m_data.m_value.binary->subtype());
}
else
{
put_literal("null");
}
put_char('\n');
put_indent(current_indent);
put_char('}');
}
else
{
put_literal("{\"bytes\":[");
if (!val.m_data.m_value.binary->empty())
{
for (auto i = val.m_data.m_value.binary->cbegin();
i != val.m_data.m_value.binary->cend() - 1; ++i)
{
dump_byte(*i);
put_char(',');
}
dump_byte(val.m_data.m_value.binary->back());
}
put_literal("],\"subtype\":");
if (val.m_data.m_value.binary->has_subtype())
{
dump_integer(val.m_data.m_value.binary->subtype());
put_char('}');
}
else
{
put_literal("null}");
}
}
return;
}
case value_t::boolean:
{
if (val.m_data.m_value.boolean)
{
put_literal("true");
}
else
{
put_literal("false");
}
return;
}
case value_t::number_integer:
{
dump_integer(val.m_data.m_value.number_integer);
return;
}
case value_t::number_unsigned:
{
dump_integer(val.m_data.m_value.number_unsigned);
return;
}
case value_t::number_float:
{
dump_float(val.m_data.m_value.number_float);
return;
}
case value_t::discarded:
{
put_literal("<discarded>");
return;
}
case value_t::null:
{
put_literal("null");
return;
}
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
}
@@ -24529,9 +24650,9 @@ class serializer
@brief serialize the value @a val, but not the elements of a container
An object or array with elements is opened and pushed onto @a stack for
@ref dump_iteratively to walk; everything else - including a binary value,
@ref dump_internal to walk; everything else - including a binary value,
which looks like an object but has no elements to descend into - is written
out in full by @ref dump_scalar.
out here in full.
*/
void dump_value(const BasicJsonType& val,
const std::size_t current_indent,
@@ -24589,35 +24710,6 @@ class serializer
return;
}
case value_t::string:
case value_t::binary:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::discarded:
case value_t::null:
default:
dump_scalar(val, current_indent);
return;
}
}
/*!
@brief serialize the value @a val, which is neither an object nor an array
Shared by @ref dump_internal and @ref dump_value, so that a value is written
the same way however deeply it is nested. A binary value is written out here
in full: it looks like an object, but has no elements to descend into.
@param[in] val value to serialize; not an object or array
@param[in] current_indent the indentation of @a val, used for a
pretty-printed binary value
*/
void dump_scalar(const BasicJsonType& val, const std::size_t current_indent)
{
switch (val.m_data.m_type)
{
case value_t::string:
{
put_char('"');
@@ -24738,8 +24830,6 @@ class serializer
return;
}
case value_t::object: // LCOV_EXCL_LINE
case value_t::array: // LCOV_EXCL_LINE
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
@@ -24768,7 +24858,7 @@ class serializer
Escape a string by replacing certain special characters by a sequence of an
escape character (backslash) and another character and other control
characters by a sequence of "\u" followed by a four-digit hex
representation. The escaped string is appended to @ref write_buffer.
representation. The escaped string is written to output stream @a o.
@param[in] s the string to escape
@@ -24962,7 +25052,7 @@ class serializer
{
case error_handler_t::strict:
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", detail::hex_byte(byte)), nullptr));
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", hex_bytes(byte | 0)), nullptr));
}
case error_handler_t::ignore:
@@ -24995,9 +25085,9 @@ class serializer
}
else
{
string_buffer[bytes++] = '\xEF';
string_buffer[bytes++] = '\xBF';
string_buffer[bytes++] = '\xBD';
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xEF');
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBF');
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBD');
}
// write buffer and reset index; there must be 13 bytes
@@ -25054,7 +25144,7 @@ class serializer
{
case error_handler_t::strict:
{
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", detail::hex_byte(static_cast<std::uint8_t>(s[s.size() - 1]))), nullptr));
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast<std::uint8_t>(s[s.size() - 1] | 0))), nullptr));
}
case error_handler_t::ignore:
@@ -25275,6 +25365,20 @@ class serializer
}
}
/*!
* @brief convert a byte to a uppercase hex representation
* @param[in] byte byte to represent
* @return representation ("00".."FF")
*/
static std::string hex_bytes(std::uint8_t byte)
{
std::string result = "FF";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
/*!
* @brief write a lowercase "\uXXXX" escape sequence into @a string_buffer
*
@@ -25388,7 +25492,7 @@ class serializer
/*!
@brief dump an integer
Dump a given integer, appending it to @ref write_buffer. Works internally with
Dump a given integer to output stream @a o. Works internally with
@a number_buffer.
@param[in] x integer number (signed or unsigned) to dump
@@ -25425,7 +25529,7 @@ class serializer
}
// use a pointer to fill the buffer
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto)
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
number_unsigned_t abs_value;
@@ -25479,7 +25583,7 @@ class serializer
/*!
@brief dump a floating-point number
Dump a given floating-point number, appending it to @ref write_buffer. Works internally
Dump a given floating-point number to output stream @a o. Works internally
with @a number_buffer.
@param[in] x floating-point number to dump
@@ -25540,28 +25644,21 @@ class serializer
// check if the buffer was large enough
JSON_ASSERT(static_cast<std::size_t>(len) < number_buffer.size());
// look up the locale's thousands separator and decimal point now,
// matching what snprintf_float() just used (see lexer::get_decimal_point())
const auto* loc = std::localeconv();
JSON_ASSERT(loc != nullptr);
const char thousands_sep = (loc->thousands_sep == nullptr) ? '\0' : *loc->thousands_sep;
const char decimal_point = (loc->decimal_point == nullptr) ? '\0' : *loc->decimal_point;
// erase thousands separators
if (thousands_sep != '\0')
if (locale.thousands_sep != '\0')
{
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::remove returns an iterator, see https://github.com/nlohmann/json/issues/3081
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, thousands_sep);
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, locale.thousands_sep);
std::fill(end, number_buffer.end(), '\0');
JSON_ASSERT((end - number_buffer.begin()) <= len);
len = (end - number_buffer.begin());
}
// convert decimal point to '.'
if (decimal_point != '\0' && decimal_point != '.')
if (locale.decimal_point != '\0' && locale.decimal_point != '.')
{
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::find returns an iterator, see https://github.com/nlohmann/json/issues/3081
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), decimal_point);
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), locale.decimal_point);
if (dec_pos != number_buffer.end())
{
*dec_pos = '.';
@@ -25606,17 +25703,34 @@ class serializer
*/
number_unsigned_t remove_sign(number_integer_t x) noexcept
{
JSON_ASSERT(x < 0);
JSON_ASSERT(x < 0 && x < (std::numeric_limits<number_integer_t>::max)()); // NOLINT(misc-redundant-expression)
return static_cast<number_unsigned_t>(-(x + 1)) + 1;
}
private:
/// the locale's thousand separator and decimal point characters
struct locale_chars
{
explicit locale_chars(const std::lconv* loc) noexcept
: thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->thousands_sep)))
, decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->decimal_point)))
{}
const char thousands_sep;
const char decimal_point;
};
/// the output of the serializer (non-owning; the adapter lives at the call site)
output_adapter_protocol<char>* o = nullptr;
/// a (hopefully) large enough character buffer
std::array<char, 64> number_buffer{{}};
/// computed once from std::localeconv() at construction; @ref
/// locale_chars keeps std::localeconv()'s pointer from having to be held
/// past the constructor, while still letting these stay const
const locale_chars locale;
/// string buffer
std::array<char, 512> string_buffer{{}};
@@ -27648,6 +27762,18 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
bool type_deduction = true,
value_t manual_type = value_t::array)
{
#if JSON_BRACE_INIT_COPY_SEMANTICS
// a single element that is a value rather than a braced list is
// copied or moved as is, whatever its content looks like
if (type_deduction && init.size() == 1 && !init.begin()->is_braced_list())
{
*this = init.begin()->moved_or_copied();
set_parents();
assert_invariant();
return;
}
#endif
// check if each element is an array with two elements whose first
// element is a string
bool is_an_object = std::all_of(init.begin(), init.end(),
@@ -72,6 +72,28 @@ TEST_CASE("JSON_BRACE_INIT_COPY_SEMANTICS")
CHECK(j7 == json::array({1, 2}));
}
SECTION("single-element brace initialization copies a pair-shaped array value (#5662)")
{
// a JSON value that happens to be a 2-element array whose first
// element is a string must still be copied, not turned into an
// object; only a braced list written in the source, such as the
// inner {"key", "value"} of {{"key", "value"}}, describes an object
json const pair_shaped = json::array({"key", 42});
json const j1{pair_shaped};
CHECK(j1.is_array());
CHECK(j1 == pair_shaped);
json const j2 = {pair_shaped};
CHECK(j2.is_array());
CHECK(j2 == pair_shaped);
// the same holds for an rvalue of the same shape
json const j3{json::array({"key", 42})};
CHECK(j3.is_array());
CHECK(j3 == pair_shaped);
}
SECTION("what the macro does not change")
{
// lists with more than one element are unaffected
-92
View File
@@ -14,10 +14,7 @@ using nlohmann::json;
#include <array>
#include <clocale>
#include <limits>
#include <map>
#include <ostream>
#include <streambuf>
#include <string>
#include <utility>
#include <vector>
@@ -388,92 +385,3 @@ TEST_CASE("locale with a multi-byte decimal point")
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
}
namespace
{
// a streambuf that switches LC_NUMERIC the first time anything is written to
// it, so a dump() in progress can be made to change locale mid-flight: after
// the serializer was constructed (and, before #5709 item 3, after it had
// cached std::localeconv() for the whole call) but before a later float is
// converted
struct LocaleSwitchingStreambuf final : std::streambuf
{
explicit LocaleSwitchingStreambuf(const char* switch_to)
: locale_after_first_write(switch_to)
{}
std::string data {}; // NOLINT(readability-redundant-member-init)
std::string locale_after_first_write;
bool switched = false;
std::streamsize xsputn(const char* s, std::streamsize n) override
{
if (!switched)
{
switched = std::setlocale(LC_NUMERIC, locale_after_first_write.c_str()) != nullptr;
}
data.append(s, static_cast<std::size_t>(n));
return n;
}
};
} // namespace
TEST_CASE("locale changes during a single dump() (#5709 item 3)")
{
// dump_float() only reads the locale on the snprintf path, taken for a
// number_float_t that is not an IEEE-754 single or double, i.e. not
// (is_iec559 && digits == 24 && max_exponent == 128) and not (is_iec559
// && digits == 53 && max_exponent == 1024) - see dump_float(). Checking
// is_iec559 alone is not enough: on x86_64, long double is a 64-bit
// (80-bit extended) format for which is_iec559 is also true, so it still
// takes the snprintf path this test means to exercise. Only a
// number_float_t whose digits/max_exponent match float or double (e.g.
// long double on 64-bit Arm, where it is IEEE-754 double) takes the
// locale-independent to_chars() path instead, and this test is a no-op
// there.
using long_double_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, long double>;
using ld_limits = std::numeric_limits<long_double_json::number_float_t>;
const bool is_ieee_single_or_double =
(ld_limits::is_iec559 && ld_limits::digits == 24 && ld_limits::max_exponent == 128) ||
(ld_limits::is_iec559 && ld_limits::digits == 53 && ld_limits::max_exponent == 1024);
if (is_ieee_single_or_double)
{
MESSAGE("long double is IEEE-754 single or double on this platform; dump_float()'s snprintf/locale path is not exercised here");
}
const char* de_DE_name = "de_DE.UTF-8";
if (std::setlocale(LC_NUMERIC, de_DE_name) == nullptr)
{
de_DE_name = "de_DE";
if (std::setlocale(LC_NUMERIC, de_DE_name) == nullptr)
{
MESSAGE("locale de_DE is not usable");
return;
}
}
const std::string decimal_point = std::localeconv()->decimal_point;
REQUIRE(std::setlocale(LC_NUMERIC, "C") != nullptr);
if (decimal_point != ",")
{
MESSAGE("de_DE's decimal point is not ',' on this platform, skipping");
return;
}
// a string long enough to overflow the serializer's internal write
// buffer, so that it is flushed to the output adapter - and the locale
// switched - before the number after it is converted
const std::string padding(5000, 'a');
const long_double_json j = { padding, 1234.5L };
LocaleSwitchingStreambuf buf(de_DE_name);
std::ostream os(&buf);
os << j;
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
REQUIRE(buf.switched);
// whatever locale was in effect when the float was actually converted,
// the output is normalized to use '.' as the decimal point: it must be
// looked up at conversion time, not once for the whole dump() - the same
// fix #5597 made on the parser side
CHECK(buf.data == "[\"" + padding + "\",1234.5]");
}