Compare commits

..
Author SHA1 Message Date
Niels Lohmann 8ab63556aa Drop the version history note for a bug that was never released
The regression came from #5390, which is not in any release. Addresses review comment by @gregmarr.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-30 07:33:14 +02:00
Niels Lohmann 3d3b90b05d Classify leaves with operator<=> itself past the nesting bound
In C++20, an ordered comparison past the nesting bound classified a pair of
leaves by asking == first and then order_leaves(), which calls < and > -
both derived from <=>. For a pair of binary values with the same bytes but a
different subtype, == reports them unequal, while <=> (through
std::vector<std::uint8_t>::operator<=>) reports them equivalent, so the pair
ended the comparison as unordered instead of letting the next element
decide - unlike an array or object within the bound, which compares such a
pair with its own operator<=> and gets equivalent. So operator<=>, and the
<, <=, >, >= derived from it, could give a different result for the same two
values depending on how deeply the values were nested, or unordered at every
depth with JSON_NO_THREAD_LOCAL defined.

compare_leaves() now classifies such a pair in C++20 with operator<=> itself
instead, matching how a value within the bound is compared; the equality-only
and pre-C++20 ordered cases are unchanged. Which of the three runs is chosen
by overloading on std::integral_constant<bool, Ordered>, the same tag
dispatch order_leaves() already uses, rather than a runtime "if (Ordered)" on
a template parameter, which MSVC would flag as a constant condition (C4127).

Added a regression test to unit-comparison.cpp that nests such a pair 0, 127,
128 and 200 levels deep (127 stays within the 128-level bound, 128 and 200
do not) and checks that operator<=> and operator< agree at every depth.

Fixes #5654.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 23:20:26 +02:00
8 changed files with 535 additions and 269 deletions
@@ -29,7 +29,6 @@
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/output/output_adapters.hpp>
#include <nlohmann/detail/string_concat.hpp>
#include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -2279,10 +2278,20 @@ class binary_writer
const std::size_t valid = valid_utf8_prefix(data, s.size());
if (JSON_HEDLEY_UNLIKELY(valid != s.size()))
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", detail::hex_byte(data[valid])), &context));
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", hex_byte(data[valid])), &context));
}
}
/// @return a byte as two uppercase hexadecimal digits
static std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
/*!
@brief write an integer in the shortest encoding
@@ -8,7 +8,9 @@
#pragma once
#include <algorithm> // copy
#include <cstddef> // size_t
#include <iterator> // back_inserter
#include <memory> // shared_ptr, make_shared
#include <string> // basic_string
#include <utility> // move
@@ -29,10 +31,6 @@ namespace detail
template<typename CharType> struct output_adapter_protocol
{
virtual void write_character(CharType c) = 0;
/// @param[in] s pointer to the characters to write; binary_writer legitimately
/// passes a null pointer together with length 0 for an empty
/// string or binary value, so implementations must tolerate that
/// @param[in] length number of characters at @a s
virtual void write_characters(const CharType* s, std::size_t length) = 0;
virtual ~output_adapter_protocol() = default;
@@ -99,6 +97,7 @@ class output_vector_adapter : public output_adapter_protocol<CharType>
sink.write_character(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
sink.write_characters(s, length);
@@ -123,6 +122,7 @@ class output_stream_adapter : public output_adapter_protocol<CharType>
stream.put(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
stream.write(s, static_cast<std::streamsize>(length));
@@ -147,6 +147,7 @@ class output_string_adapter : public output_adapter_protocol<CharType>
str.push_back(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
str.append(s, length);
+176 -70
View File
@@ -3,23 +3,24 @@
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#pragma once
#include <algorithm> // remove, fill, find, none_of, min
#include <algorithm> // reverse, remove, fill, find, none_of, min
#include <array> // array
#include <clocale> // localeconv, lconv
#include <cmath> // isfinite
#include <cmath> // labs, isfinite, isnan, signbit
#include <cstddef> // size_t, ptrdiff_t
#include <cstdint> // uint8_t
#include <cstdio> // snprintf
#include <cstring> // memcpy, memset
#include <iterator> // next
#include <limits> // numeric_limits
#include <string> // string, char_traits
#include <type_traits> // is_same
#include <utility> // move
#include <vector> // vector
#include <nlohmann/detail/conversions/to_chars.hpp>
@@ -27,6 +28,7 @@
#include <nlohmann/detail/input/string_scan.hpp>
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/cpp_future.hpp>
#include <nlohmann/detail/output/binary_writer.hpp>
#include <nlohmann/detail/output/output_adapters.hpp>
#include <nlohmann/detail/recursion_depth_limit.hpp>
#include <nlohmann/detail/string_concat.hpp>
@@ -81,6 +83,7 @@ class serializer
const std::size_t indent_step_ = 0,
error_handler_t error_handler_ = error_handler_t::strict)
: o(&s)
, locale(std::localeconv())
, indent_char(ichar)
, pretty_print(pretty_print_)
, ensure_ascii(ensure_ascii_)
@@ -103,10 +106,9 @@ class serializer
additional parameter. Arrays and objects are serialized without recursion,
however deeply they are nested.
- strings and object keys are escaped using @ref dump_escaped
- integer numbers are converted using a digit-pair lookup table (@ref dump_integer)
- floating-point numbers are converted to a string using @ref dump_float, which
uses `to_chars` for IEEE-754 types and `snprintf` otherwise
- strings and object keys are escaped using `escape_string()`
- integer numbers are converted implicitly via `operator<<`
- floating-point numbers are converted to a string using `"%g"` format
- binary values are serialized as objects containing the subtype and the
byte array
@@ -281,16 +283,127 @@ class serializer
}
case value_t::string:
case value_t::binary:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::discarded:
case value_t::null:
default:
dump_scalar(val, current_indent);
{
put_char('"');
dump_escaped(*val.m_data.m_value.string);
put_char('"');
return;
}
case value_t::binary:
{
if (pretty_print)
{
put_literal("{\n");
// variable to hold indentation for recursive calls
const auto new_indent = next_indent(current_indent, indent_step);
put_indent(new_indent);
put_literal("\"bytes\": [");
if (!val.m_data.m_value.binary->empty())
{
for (auto i = val.m_data.m_value.binary->cbegin();
i != val.m_data.m_value.binary->cend() - 1; ++i)
{
dump_byte(*i);
put_literal(", ");
}
dump_byte(val.m_data.m_value.binary->back());
}
put_literal("],\n");
put_indent(new_indent);
put_literal("\"subtype\": ");
if (val.m_data.m_value.binary->has_subtype())
{
dump_integer(val.m_data.m_value.binary->subtype());
}
else
{
put_literal("null");
}
put_char('\n');
put_indent(current_indent);
put_char('}');
}
else
{
put_literal("{\"bytes\":[");
if (!val.m_data.m_value.binary->empty())
{
for (auto i = val.m_data.m_value.binary->cbegin();
i != val.m_data.m_value.binary->cend() - 1; ++i)
{
dump_byte(*i);
put_char(',');
}
dump_byte(val.m_data.m_value.binary->back());
}
put_literal("],\"subtype\":");
if (val.m_data.m_value.binary->has_subtype())
{
dump_integer(val.m_data.m_value.binary->subtype());
put_char('}');
}
else
{
put_literal("null}");
}
}
return;
}
case value_t::boolean:
{
if (val.m_data.m_value.boolean)
{
put_literal("true");
}
else
{
put_literal("false");
}
return;
}
case value_t::number_integer:
{
dump_integer(val.m_data.m_value.number_integer);
return;
}
case value_t::number_unsigned:
{
dump_integer(val.m_data.m_value.number_unsigned);
return;
}
case value_t::number_float:
{
dump_float(val.m_data.m_value.number_float);
return;
}
case value_t::discarded:
{
put_literal("<discarded>");
return;
}
case value_t::null:
{
put_literal("null");
return;
}
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
}
@@ -447,9 +560,9 @@ class serializer
@brief serialize the value @a val, but not the elements of a container
An object or array with elements is opened and pushed onto @a stack for
@ref dump_iteratively to walk; everything else - including a binary value,
@ref dump_internal to walk; everything else - including a binary value,
which looks like an object but has no elements to descend into - is written
out in full by @ref dump_scalar.
out here in full.
*/
void dump_value(const BasicJsonType& val,
const std::size_t current_indent,
@@ -507,35 +620,6 @@ class serializer
return;
}
case value_t::string:
case value_t::binary:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::discarded:
case value_t::null:
default:
dump_scalar(val, current_indent);
return;
}
}
/*!
@brief serialize the value @a val, which is neither an object nor an array
Shared by @ref dump_internal and @ref dump_value, so that a value is written
the same way however deeply it is nested. A binary value is written out here
in full: it looks like an object, but has no elements to descend into.
@param[in] val value to serialize; not an object or array
@param[in] current_indent the indentation of @a val, used for a
pretty-printed binary value
*/
void dump_scalar(const BasicJsonType& val, const std::size_t current_indent)
{
switch (val.m_data.m_type)
{
case value_t::string:
{
put_char('"');
@@ -656,8 +740,6 @@ class serializer
return;
}
case value_t::object: // LCOV_EXCL_LINE
case value_t::array: // LCOV_EXCL_LINE
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
@@ -686,7 +768,7 @@ class serializer
Escape a string by replacing certain special characters by a sequence of an
escape character (backslash) and another character and other control
characters by a sequence of "\u" followed by a four-digit hex
representation. The escaped string is appended to @ref write_buffer.
representation. The escaped string is written to output stream @a o.
@param[in] s the string to escape
@@ -880,7 +962,7 @@ class serializer
{
case error_handler_t::strict:
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", detail::hex_byte(byte)), nullptr));
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", hex_bytes(byte | 0)), nullptr));
}
case error_handler_t::ignore:
@@ -913,9 +995,9 @@ class serializer
}
else
{
string_buffer[bytes++] = '\xEF';
string_buffer[bytes++] = '\xBF';
string_buffer[bytes++] = '\xBD';
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xEF');
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBF');
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBD');
}
// write buffer and reset index; there must be 13 bytes
@@ -972,7 +1054,7 @@ class serializer
{
case error_handler_t::strict:
{
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", detail::hex_byte(static_cast<std::uint8_t>(s[s.size() - 1]))), nullptr));
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast<std::uint8_t>(s[s.size() - 1] | 0))), nullptr));
}
case error_handler_t::ignore:
@@ -1193,6 +1275,20 @@ class serializer
}
}
/*!
* @brief convert a byte to a uppercase hex representation
* @param[in] byte byte to represent
* @return representation ("00".."FF")
*/
static std::string hex_bytes(std::uint8_t byte)
{
std::string result = "FF";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
/*!
* @brief write a lowercase "\uXXXX" escape sequence into @a string_buffer
*
@@ -1306,7 +1402,7 @@ class serializer
/*!
@brief dump an integer
Dump a given integer, appending it to @ref write_buffer. Works internally with
Dump a given integer to output stream @a o. Works internally with
@a number_buffer.
@param[in] x integer number (signed or unsigned) to dump
@@ -1343,7 +1439,7 @@ class serializer
}
// use a pointer to fill the buffer
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto)
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
number_unsigned_t abs_value;
@@ -1397,7 +1493,7 @@ class serializer
/*!
@brief dump a floating-point number
Dump a given floating-point number, appending it to @ref write_buffer. Works internally
Dump a given floating-point number to output stream @a o. Works internally
with @a number_buffer.
@param[in] x floating-point number to dump
@@ -1458,28 +1554,21 @@ class serializer
// check if the buffer was large enough
JSON_ASSERT(static_cast<std::size_t>(len) < number_buffer.size());
// look up the locale's thousands separator and decimal point now,
// matching what snprintf_float() just used (see lexer::get_decimal_point())
const auto* loc = std::localeconv();
JSON_ASSERT(loc != nullptr);
const char thousands_sep = (loc->thousands_sep == nullptr) ? '\0' : *loc->thousands_sep;
const char decimal_point = (loc->decimal_point == nullptr) ? '\0' : *loc->decimal_point;
// erase thousands separators
if (thousands_sep != '\0')
if (locale.thousands_sep != '\0')
{
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::remove returns an iterator, see https://github.com/nlohmann/json/issues/3081
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, thousands_sep);
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, locale.thousands_sep);
std::fill(end, number_buffer.end(), '\0');
JSON_ASSERT((end - number_buffer.begin()) <= len);
len = (end - number_buffer.begin());
}
// convert decimal point to '.'
if (decimal_point != '\0' && decimal_point != '.')
if (locale.decimal_point != '\0' && locale.decimal_point != '.')
{
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::find returns an iterator, see https://github.com/nlohmann/json/issues/3081
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), decimal_point);
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), locale.decimal_point);
if (dec_pos != number_buffer.end())
{
*dec_pos = '.';
@@ -1524,17 +1613,34 @@ class serializer
*/
number_unsigned_t remove_sign(number_integer_t x) noexcept
{
JSON_ASSERT(x < 0);
JSON_ASSERT(x < 0 && x < (std::numeric_limits<number_integer_t>::max)()); // NOLINT(misc-redundant-expression)
return static_cast<number_unsigned_t>(-(x + 1)) + 1;
}
private:
/// the locale's thousand separator and decimal point characters
struct locale_chars
{
explicit locale_chars(const std::lconv* loc) noexcept
: thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->thousands_sep)))
, decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->decimal_point)))
{}
const char thousands_sep;
const char decimal_point;
};
/// the output of the serializer (non-owning; the adapter lives at the call site)
output_adapter_protocol<char>* o = nullptr;
/// a (hopefully) large enough character buffer
std::array<char, 64> number_buffer{{}};
/// computed once from std::localeconv() at construction; @ref
/// locale_chars keeps std::localeconv()'s pointer from having to be held
/// past the constructor, while still letting these stay const
const locale_chars locale;
/// string buffer
std::array<char, 512> string_buffer{{}};
-11
View File
@@ -3,7 +3,6 @@
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
@@ -37,16 +36,6 @@ StringType to_string(std::size_t value)
return result;
}
/// @return a byte as two uppercase hexadecimal digits
inline std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
///////////////////
// UTF-8 decoding //
///////////////////
+51 -1
View File
@@ -1307,15 +1307,65 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
*/
template<bool Ordered>
static compare_result compare_leaves(const_reference lhs, const_reference rhs) noexcept
{
return compare_leaves(lhs, rhs, std::integral_constant<bool, Ordered> {});
}
/// @brief compare two leaves that are only being checked for equality
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::false_type /*ordered*/) noexcept
{
if (lhs == rhs)
{
return compare_result::equal;
}
return order_leaves(lhs, rhs, std::integral_constant<bool, Ordered> {});
return order_leaves(lhs, rhs, std::false_type {});
}
#if JSON_HAS_THREE_WAY_COMPARISON
/*!
@brief compare two leaves that are being ordered, for operator<=>
Reached only from operator<=>, so the leaves must be classified exactly
as operator<=> classifies them - which is not the same as asking
== and then order_leaves(), the way the other overload does it. The two
disagree on a binary value: == also compares the subtype, but <=> compares
only the bytes, through std::vector<std::uint8_t>::operator<=>. Using <=>
itself here keeps a leaf pair classified the same way regardless of how
deep it is nested - == first would again call operator<=> a level down
through order_leaves(), but call it after a mismatching == already ended
the comparison for a pair that <=> alone would still call equivalent.
*/
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept
{
const std::partial_ordering order = lhs <=> rhs; // *NOPAD*
if (order == 0)
{
return compare_result::equal;
}
if (order < 0)
{
return compare_result::less;
}
if (order > 0)
{
return compare_result::greater;
}
return compare_result::unordered;
}
#else
/// @brief compare two leaves that are being ordered, for operator<
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept
{
if (lhs == rhs)
{
return compare_result::equal;
}
return order_leaves(lhs, rhs, std::true_type {});
}
#endif
/*!
@brief compare two object keys
+244 -89
View File
@@ -6199,7 +6199,6 @@ NLOHMANN_JSON_NAMESPACE_END
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
@@ -6235,16 +6234,6 @@ StringType to_string(std::size_t value)
return result;
}
/// @return a byte as two uppercase hexadecimal digits
inline std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
///////////////////
// UTF-8 decoding //
///////////////////
@@ -20157,7 +20146,9 @@ NLOHMANN_JSON_NAMESPACE_END
#include <algorithm> // copy
#include <cstddef> // size_t
#include <iterator> // back_inserter
#include <memory> // shared_ptr, make_shared
#include <string> // basic_string
#include <utility> // move
@@ -20179,10 +20170,6 @@ namespace detail
template<typename CharType> struct output_adapter_protocol
{
virtual void write_character(CharType c) = 0;
/// @param[in] s pointer to the characters to write; binary_writer legitimately
/// passes a null pointer together with length 0 for an empty
/// string or binary value, so implementations must tolerate that
/// @param[in] length number of characters at @a s
virtual void write_characters(const CharType* s, std::size_t length) = 0;
virtual ~output_adapter_protocol() = default;
@@ -20249,6 +20236,7 @@ class output_vector_adapter : public output_adapter_protocol<CharType>
sink.write_character(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
sink.write_characters(s, length);
@@ -20273,6 +20261,7 @@ class output_stream_adapter : public output_adapter_protocol<CharType>
stream.put(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
stream.write(s, static_cast<std::streamsize>(length));
@@ -20297,6 +20286,7 @@ class output_string_adapter : public output_adapter_protocol<CharType>
str.push_back(c);
}
JSON_HEDLEY_NON_NULL(2)
void write_characters(const CharType* s, std::size_t length) override
{
str.append(s, length);
@@ -20369,8 +20359,6 @@ NLOHMANN_JSON_NAMESPACE_END
// #include <nlohmann/detail/string_concat.hpp>
// #include <nlohmann/detail/string_utils.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -22620,10 +22608,20 @@ class binary_writer
const std::size_t valid = valid_utf8_prefix(data, s.size());
if (JSON_HEDLEY_UNLIKELY(valid != s.size()))
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", detail::hex_byte(data[valid])), &context));
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", hex_byte(data[valid])), &context));
}
}
/// @return a byte as two uppercase hexadecimal digits
static std::string hex_byte(const std::uint8_t byte)
{
std::string result = "00";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
/*!
@brief write an integer in the shortest encoding
@@ -22956,23 +22954,24 @@ NLOHMANN_JSON_NAMESPACE_END
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann <bjoern@hoehrmann.de>
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#include <algorithm> // remove, fill, find, none_of, min
#include <algorithm> // reverse, remove, fill, find, none_of, min
#include <array> // array
#include <clocale> // localeconv, lconv
#include <cmath> // isfinite
#include <cmath> // labs, isfinite, isnan, signbit
#include <cstddef> // size_t, ptrdiff_t
#include <cstdint> // uint8_t
#include <cstdio> // snprintf
#include <cstring> // memcpy, memset
#include <iterator> // next
#include <limits> // numeric_limits
#include <string> // string, char_traits
#include <type_traits> // is_same
#include <utility> // move
#include <vector> // vector
// #include <nlohmann/detail/conversions/to_chars.hpp>
@@ -24104,6 +24103,8 @@ NLOHMANN_JSON_NAMESPACE_END
// #include <nlohmann/detail/meta/cpp_future.hpp>
// #include <nlohmann/detail/output/binary_writer.hpp>
// #include <nlohmann/detail/output/output_adapters.hpp>
// #include <nlohmann/detail/recursion_depth_limit.hpp>
@@ -24163,6 +24164,7 @@ class serializer
const std::size_t indent_step_ = 0,
error_handler_t error_handler_ = error_handler_t::strict)
: o(&s)
, locale(std::localeconv())
, indent_char(ichar)
, pretty_print(pretty_print_)
, ensure_ascii(ensure_ascii_)
@@ -24185,10 +24187,9 @@ class serializer
additional parameter. Arrays and objects are serialized without recursion,
however deeply they are nested.
- strings and object keys are escaped using @ref dump_escaped
- integer numbers are converted using a digit-pair lookup table (@ref dump_integer)
- floating-point numbers are converted to a string using @ref dump_float, which
uses `to_chars` for IEEE-754 types and `snprintf` otherwise
- strings and object keys are escaped using `escape_string()`
- integer numbers are converted implicitly via `operator<<`
- floating-point numbers are converted to a string using `"%g"` format
- binary values are serialized as objects containing the subtype and the
byte array
@@ -24363,16 +24364,127 @@ class serializer
}
case value_t::string:
case value_t::binary:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::discarded:
case value_t::null:
default:
dump_scalar(val, current_indent);
{
put_char('"');
dump_escaped(*val.m_data.m_value.string);
put_char('"');
return;
}
case value_t::binary:
{
if (pretty_print)
{
put_literal("{\n");
// variable to hold indentation for recursive calls
const auto new_indent = next_indent(current_indent, indent_step);
put_indent(new_indent);
put_literal("\"bytes\": [");
if (!val.m_data.m_value.binary->empty())
{
for (auto i = val.m_data.m_value.binary->cbegin();
i != val.m_data.m_value.binary->cend() - 1; ++i)
{
dump_byte(*i);
put_literal(", ");
}
dump_byte(val.m_data.m_value.binary->back());
}
put_literal("],\n");
put_indent(new_indent);
put_literal("\"subtype\": ");
if (val.m_data.m_value.binary->has_subtype())
{
dump_integer(val.m_data.m_value.binary->subtype());
}
else
{
put_literal("null");
}
put_char('\n');
put_indent(current_indent);
put_char('}');
}
else
{
put_literal("{\"bytes\":[");
if (!val.m_data.m_value.binary->empty())
{
for (auto i = val.m_data.m_value.binary->cbegin();
i != val.m_data.m_value.binary->cend() - 1; ++i)
{
dump_byte(*i);
put_char(',');
}
dump_byte(val.m_data.m_value.binary->back());
}
put_literal("],\"subtype\":");
if (val.m_data.m_value.binary->has_subtype())
{
dump_integer(val.m_data.m_value.binary->subtype());
put_char('}');
}
else
{
put_literal("null}");
}
}
return;
}
case value_t::boolean:
{
if (val.m_data.m_value.boolean)
{
put_literal("true");
}
else
{
put_literal("false");
}
return;
}
case value_t::number_integer:
{
dump_integer(val.m_data.m_value.number_integer);
return;
}
case value_t::number_unsigned:
{
dump_integer(val.m_data.m_value.number_unsigned);
return;
}
case value_t::number_float:
{
dump_float(val.m_data.m_value.number_float);
return;
}
case value_t::discarded:
{
put_literal("<discarded>");
return;
}
case value_t::null:
{
put_literal("null");
return;
}
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
}
@@ -24529,9 +24641,9 @@ class serializer
@brief serialize the value @a val, but not the elements of a container
An object or array with elements is opened and pushed onto @a stack for
@ref dump_iteratively to walk; everything else - including a binary value,
@ref dump_internal to walk; everything else - including a binary value,
which looks like an object but has no elements to descend into - is written
out in full by @ref dump_scalar.
out here in full.
*/
void dump_value(const BasicJsonType& val,
const std::size_t current_indent,
@@ -24589,35 +24701,6 @@ class serializer
return;
}
case value_t::string:
case value_t::binary:
case value_t::boolean:
case value_t::number_integer:
case value_t::number_unsigned:
case value_t::number_float:
case value_t::discarded:
case value_t::null:
default:
dump_scalar(val, current_indent);
return;
}
}
/*!
@brief serialize the value @a val, which is neither an object nor an array
Shared by @ref dump_internal and @ref dump_value, so that a value is written
the same way however deeply it is nested. A binary value is written out here
in full: it looks like an object, but has no elements to descend into.
@param[in] val value to serialize; not an object or array
@param[in] current_indent the indentation of @a val, used for a
pretty-printed binary value
*/
void dump_scalar(const BasicJsonType& val, const std::size_t current_indent)
{
switch (val.m_data.m_type)
{
case value_t::string:
{
put_char('"');
@@ -24738,8 +24821,6 @@ class serializer
return;
}
case value_t::object: // LCOV_EXCL_LINE
case value_t::array: // LCOV_EXCL_LINE
default: // LCOV_EXCL_LINE
JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE
}
@@ -24768,7 +24849,7 @@ class serializer
Escape a string by replacing certain special characters by a sequence of an
escape character (backslash) and another character and other control
characters by a sequence of "\u" followed by a four-digit hex
representation. The escaped string is appended to @ref write_buffer.
representation. The escaped string is written to output stream @a o.
@param[in] s the string to escape
@@ -24962,7 +25043,7 @@ class serializer
{
case error_handler_t::strict:
{
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", detail::hex_byte(byte)), nullptr));
JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", hex_bytes(byte | 0)), nullptr));
}
case error_handler_t::ignore:
@@ -24995,9 +25076,9 @@ class serializer
}
else
{
string_buffer[bytes++] = '\xEF';
string_buffer[bytes++] = '\xBF';
string_buffer[bytes++] = '\xBD';
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xEF');
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBF');
string_buffer[bytes++] = detail::binary_writer<BasicJsonType, char>::to_char_type('\xBD');
}
// write buffer and reset index; there must be 13 bytes
@@ -25054,7 +25135,7 @@ class serializer
{
case error_handler_t::strict:
{
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", detail::hex_byte(static_cast<std::uint8_t>(s[s.size() - 1]))), nullptr));
JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast<std::uint8_t>(s[s.size() - 1] | 0))), nullptr));
}
case error_handler_t::ignore:
@@ -25275,6 +25356,20 @@ class serializer
}
}
/*!
* @brief convert a byte to a uppercase hex representation
* @param[in] byte byte to represent
* @return representation ("00".."FF")
*/
static std::string hex_bytes(std::uint8_t byte)
{
std::string result = "FF";
constexpr const char* nibble_to_hex = "0123456789ABCDEF";
result[0] = nibble_to_hex[byte / 16];
result[1] = nibble_to_hex[byte % 16];
return result;
}
/*!
* @brief write a lowercase "\uXXXX" escape sequence into @a string_buffer
*
@@ -25388,7 +25483,7 @@ class serializer
/*!
@brief dump an integer
Dump a given integer, appending it to @ref write_buffer. Works internally with
Dump a given integer to output stream @a o. Works internally with
@a number_buffer.
@param[in] x integer number (signed or unsigned) to dump
@@ -25425,7 +25520,7 @@ class serializer
}
// use a pointer to fill the buffer
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto)
auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
number_unsigned_t abs_value;
@@ -25479,7 +25574,7 @@ class serializer
/*!
@brief dump a floating-point number
Dump a given floating-point number, appending it to @ref write_buffer. Works internally
Dump a given floating-point number to output stream @a o. Works internally
with @a number_buffer.
@param[in] x floating-point number to dump
@@ -25540,28 +25635,21 @@ class serializer
// check if the buffer was large enough
JSON_ASSERT(static_cast<std::size_t>(len) < number_buffer.size());
// look up the locale's thousands separator and decimal point now,
// matching what snprintf_float() just used (see lexer::get_decimal_point())
const auto* loc = std::localeconv();
JSON_ASSERT(loc != nullptr);
const char thousands_sep = (loc->thousands_sep == nullptr) ? '\0' : *loc->thousands_sep;
const char decimal_point = (loc->decimal_point == nullptr) ? '\0' : *loc->decimal_point;
// erase thousands separators
if (thousands_sep != '\0')
if (locale.thousands_sep != '\0')
{
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::remove returns an iterator, see https://github.com/nlohmann/json/issues/3081
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, thousands_sep);
const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, locale.thousands_sep);
std::fill(end, number_buffer.end(), '\0');
JSON_ASSERT((end - number_buffer.begin()) <= len);
len = (end - number_buffer.begin());
}
// convert decimal point to '.'
if (decimal_point != '\0' && decimal_point != '.')
if (locale.decimal_point != '\0' && locale.decimal_point != '.')
{
// NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::find returns an iterator, see https://github.com/nlohmann/json/issues/3081
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), decimal_point);
const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), locale.decimal_point);
if (dec_pos != number_buffer.end())
{
*dec_pos = '.';
@@ -25606,17 +25694,34 @@ class serializer
*/
number_unsigned_t remove_sign(number_integer_t x) noexcept
{
JSON_ASSERT(x < 0);
JSON_ASSERT(x < 0 && x < (std::numeric_limits<number_integer_t>::max)()); // NOLINT(misc-redundant-expression)
return static_cast<number_unsigned_t>(-(x + 1)) + 1;
}
private:
/// the locale's thousand separator and decimal point characters
struct locale_chars
{
explicit locale_chars(const std::lconv* loc) noexcept
: thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->thousands_sep)))
, decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->decimal_point)))
{}
const char thousands_sep;
const char decimal_point;
};
/// the output of the serializer (non-owning; the adapter lives at the call site)
output_adapter_protocol<char>* o = nullptr;
/// a (hopefully) large enough character buffer
std::array<char, 64> number_buffer{{}};
/// computed once from std::localeconv() at construction; @ref
/// locale_chars keeps std::localeconv()'s pointer from having to be held
/// past the constructor, while still letting these stay const
const locale_chars locale;
/// string buffer
std::array<char, 512> string_buffer{{}};
@@ -27283,15 +27388,65 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
*/
template<bool Ordered>
static compare_result compare_leaves(const_reference lhs, const_reference rhs) noexcept
{
return compare_leaves(lhs, rhs, std::integral_constant<bool, Ordered> {});
}
/// @brief compare two leaves that are only being checked for equality
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::false_type /*ordered*/) noexcept
{
if (lhs == rhs)
{
return compare_result::equal;
}
return order_leaves(lhs, rhs, std::integral_constant<bool, Ordered> {});
return order_leaves(lhs, rhs, std::false_type {});
}
#if JSON_HAS_THREE_WAY_COMPARISON
/*!
@brief compare two leaves that are being ordered, for operator<=>
Reached only from operator<=>, so the leaves must be classified exactly
as operator<=> classifies them - which is not the same as asking
== and then order_leaves(), the way the other overload does it. The two
disagree on a binary value: == also compares the subtype, but <=> compares
only the bytes, through std::vector<std::uint8_t>::operator<=>. Using <=>
itself here keeps a leaf pair classified the same way regardless of how
deep it is nested - == first would again call operator<=> a level down
through order_leaves(), but call it after a mismatching == already ended
the comparison for a pair that <=> alone would still call equivalent.
*/
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept
{
const std::partial_ordering order = lhs <=> rhs; // *NOPAD*
if (order == 0)
{
return compare_result::equal;
}
if (order < 0)
{
return compare_result::less;
}
if (order > 0)
{
return compare_result::greater;
}
return compare_result::unordered;
}
#else
/// @brief compare two leaves that are being ordered, for operator<
static compare_result compare_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept
{
if (lhs == rhs)
{
return compare_result::equal;
}
return order_leaves(lhs, rhs, std::true_type {});
}
#endif
/*!
@brief compare two object keys
+48
View File
@@ -952,3 +952,51 @@ TEST_CASE("containers are compared element by element")
}
}
}
#if JSON_HAS_THREE_WAY_COMPARISON
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
TEST_CASE("operator<=> of binary values with a different subtype does not depend on nesting depth")
{
// #5654: std::vector<std::uint8_t>::operator<=>, which the binary type's
// own operator<=> uses, ignores the subtype that operator== checks. So a
// pair of binary values with the same bytes but a different subtype is
// unequal, yet <=>-equivalent - the same inconsistency between == and <=>
// that a NaN has. Within the nesting bound, an array compares itself
// with std::vector's own operator<=>, which treats an equivalent pair as
// undecided and lets the next element decide, same as
// std::lexicographical_compare_three_way does. Past the bound,
// compare_iteratively<true>() takes over and must classify the pair the
// same way, or the result of operator<=> - and of <, which C++20 derives
// from it - depends on how deeply the values are nested.
const json a = json::array({json::binary({1}, 1), 1});
const json b = json::array({json::binary({1}, 2), 2});
// the root inconsistency: unequal, yet <=>-equivalent
CHECK_FALSE(a[0] == b[0]);
CHECK((a[0] <=> b[0]) == std::partial_ordering::equivalent); // *NOPAD*
const auto deep = [](const json & j, const std::size_t depth)
{
json result = j;
for (std::size_t i = 0; i < depth; ++i)
{
result = json::array({std::move(result)});
}
return result;
};
// 127 levels stay within nesting_depth_limit() (128); 128 and 200 do not,
// and must still agree with the levels that do
for (const std::size_t depth : std::vector<std::size_t> {0, 127, 128, 200})
{
CAPTURE(depth);
const json x = deep(a, depth);
const json y = deep(b, depth);
CHECK((x <=> y) == std::partial_ordering::less); // *NOPAD*
CHECK((y <=> x) == std::partial_ordering::greater); // *NOPAD*
CHECK(x < y);
CHECK(y > x);
CHECK_FALSE(y < x);
}
}
#endif
-92
View File
@@ -14,10 +14,7 @@ using nlohmann::json;
#include <array>
#include <clocale>
#include <limits>
#include <map>
#include <ostream>
#include <streambuf>
#include <string>
#include <utility>
#include <vector>
@@ -388,92 +385,3 @@ TEST_CASE("locale with a multi-byte decimal point")
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
}
namespace
{
// a streambuf that switches LC_NUMERIC the first time anything is written to
// it, so a dump() in progress can be made to change locale mid-flight: after
// the serializer was constructed (and, before #5709 item 3, after it had
// cached std::localeconv() for the whole call) but before a later float is
// converted
struct LocaleSwitchingStreambuf final : std::streambuf
{
explicit LocaleSwitchingStreambuf(const char* switch_to)
: locale_after_first_write(switch_to)
{}
std::string data {}; // NOLINT(readability-redundant-member-init)
std::string locale_after_first_write;
bool switched = false;
std::streamsize xsputn(const char* s, std::streamsize n) override
{
if (!switched)
{
switched = std::setlocale(LC_NUMERIC, locale_after_first_write.c_str()) != nullptr;
}
data.append(s, static_cast<std::size_t>(n));
return n;
}
};
} // namespace
TEST_CASE("locale changes during a single dump() (#5709 item 3)")
{
// dump_float() only reads the locale on the snprintf path, taken for a
// number_float_t that is not an IEEE-754 single or double, i.e. not
// (is_iec559 && digits == 24 && max_exponent == 128) and not (is_iec559
// && digits == 53 && max_exponent == 1024) - see dump_float(). Checking
// is_iec559 alone is not enough: on x86_64, long double is a 64-bit
// (80-bit extended) format for which is_iec559 is also true, so it still
// takes the snprintf path this test means to exercise. Only a
// number_float_t whose digits/max_exponent match float or double (e.g.
// long double on 64-bit Arm, where it is IEEE-754 double) takes the
// locale-independent to_chars() path instead, and this test is a no-op
// there.
using long_double_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, long double>;
using ld_limits = std::numeric_limits<long_double_json::number_float_t>;
const bool is_ieee_single_or_double =
(ld_limits::is_iec559 && ld_limits::digits == 24 && ld_limits::max_exponent == 128) ||
(ld_limits::is_iec559 && ld_limits::digits == 53 && ld_limits::max_exponent == 1024);
if (is_ieee_single_or_double)
{
MESSAGE("long double is IEEE-754 single or double on this platform; dump_float()'s snprintf/locale path is not exercised here");
}
const char* de_DE_name = "de_DE.UTF-8";
if (std::setlocale(LC_NUMERIC, de_DE_name) == nullptr)
{
de_DE_name = "de_DE";
if (std::setlocale(LC_NUMERIC, de_DE_name) == nullptr)
{
MESSAGE("locale de_DE is not usable");
return;
}
}
const std::string decimal_point = std::localeconv()->decimal_point;
REQUIRE(std::setlocale(LC_NUMERIC, "C") != nullptr);
if (decimal_point != ",")
{
MESSAGE("de_DE's decimal point is not ',' on this platform, skipping");
return;
}
// a string long enough to overflow the serializer's internal write
// buffer, so that it is flushed to the output adapter - and the locale
// switched - before the number after it is converted
const std::string padding(5000, 'a');
const long_double_json j = { padding, 1234.5L };
LocaleSwitchingStreambuf buf(de_DE_name);
std::ostream os(&buf);
os << j;
CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr);
REQUIRE(buf.switched);
// whatever locale was in effect when the float was actually converted,
// the output is normalized to use '.' as the decimal point: it must be
// looked up at conversion time, not once for the whole dump() - the same
// fix #5597 made on the parser side
CHECK(buf.data == "[\"" + padding + "\",1234.5]");
}