diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index f20379436..9d844d85e 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -28,6 +28,7 @@ #include #include #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -2128,20 +2129,10 @@ class binary_writer const std::size_t valid = valid_utf8_prefix(data, s.size()); if (JSON_HEDLEY_UNLIKELY(valid != s.size())) { - JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", hex_byte(data[valid])), &context)); + JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", detail::hex_byte(data[valid])), &context)); } } - /// @return a byte as two uppercase hexadecimal digits - static std::string hex_byte(const std::uint8_t byte) - { - std::string result = "00"; - constexpr const char* nibble_to_hex = "0123456789ABCDEF"; - result[0] = nibble_to_hex[byte / 16]; - result[1] = nibble_to_hex[byte % 16]; - return result; - } - /*! @brief write an integer in the shortest encoding diff --git a/include/nlohmann/detail/output/output_adapters.hpp b/include/nlohmann/detail/output/output_adapters.hpp index 7eb73121c..231cf53ce 100644 --- a/include/nlohmann/detail/output/output_adapters.hpp +++ b/include/nlohmann/detail/output/output_adapters.hpp @@ -8,9 +8,7 @@ #pragma once -#include // copy #include // size_t -#include // back_inserter #include // shared_ptr, make_shared #include // basic_string #include // move @@ -31,6 +29,10 @@ namespace detail template struct output_adapter_protocol { virtual void write_character(CharType c) = 0; + /// @param[in] s pointer to the characters to write; binary_writer legitimately + /// passes a null pointer together with length 0 for an empty + /// string or binary value, so implementations must tolerate that + /// @param[in] length number of characters at @a s virtual void write_characters(const CharType* s, std::size_t length) = 0; virtual ~output_adapter_protocol() = default; @@ -97,7 +99,6 @@ class output_vector_adapter : public output_adapter_protocol sink.write_character(c); } - JSON_HEDLEY_NON_NULL(2) void write_characters(const CharType* s, std::size_t length) override { sink.write_characters(s, length); @@ -122,7 +123,6 @@ class output_stream_adapter : public output_adapter_protocol stream.put(c); } - JSON_HEDLEY_NON_NULL(2) void write_characters(const CharType* s, std::size_t length) override { stream.write(s, static_cast(length)); @@ -147,7 +147,6 @@ class output_string_adapter : public output_adapter_protocol str.push_back(c); } - JSON_HEDLEY_NON_NULL(2) void write_characters(const CharType* s, std::size_t length) override { str.append(s, length); diff --git a/include/nlohmann/detail/output/serializer.hpp b/include/nlohmann/detail/output/serializer.hpp index f968b6001..9715a7896 100644 --- a/include/nlohmann/detail/output/serializer.hpp +++ b/include/nlohmann/detail/output/serializer.hpp @@ -3,24 +3,23 @@ // | | |__ | | | | | | version 3.12.0 // |_____|_____|_____|_|___| https://github.com/nlohmann/json // -// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT #pragma once -#include // reverse, remove, fill, find, none_of, min +#include // remove, fill, find, none_of, min #include // array #include // localeconv, lconv -#include // labs, isfinite, isnan, signbit +#include // isfinite #include // size_t, ptrdiff_t #include // uint8_t #include // snprintf #include // memcpy, memset +#include // next #include // numeric_limits #include // string, char_traits #include // is_same -#include // move #include // vector #include @@ -28,7 +27,6 @@ #include #include #include -#include #include #include #include @@ -83,7 +81,6 @@ class serializer const std::size_t indent_step_ = 0, error_handler_t error_handler_ = error_handler_t::strict) : o(&s) - , locale(std::localeconv()) , indent_char(ichar) , pretty_print(pretty_print_) , ensure_ascii(ensure_ascii_) @@ -106,9 +103,10 @@ class serializer additional parameter. Arrays and objects are serialized without recursion, however deeply they are nested. - - strings and object keys are escaped using `escape_string()` - - integer numbers are converted implicitly via `operator<<` - - floating-point numbers are converted to a string using `"%g"` format + - strings and object keys are escaped using @ref dump_escaped + - integer numbers are converted using a digit-pair lookup table (@ref dump_integer) + - floating-point numbers are converted to a string using @ref dump_float, which + uses `to_chars` for IEEE-754 types and `snprintf` otherwise - binary values are serialized as objects containing the subtype and the byte array @@ -283,127 +281,16 @@ class serializer } case value_t::string: - { - put_char('"'); - dump_escaped(*val.m_data.m_value.string); - put_char('"'); - return; - } - case value_t::binary: - { - if (pretty_print) - { - put_literal("{\n"); - - // variable to hold indentation for recursive calls - const auto new_indent = next_indent(current_indent, indent_step); - - put_indent(new_indent); - - put_literal("\"bytes\": ["); - - if (!val.m_data.m_value.binary->empty()) - { - for (auto i = val.m_data.m_value.binary->cbegin(); - i != val.m_data.m_value.binary->cend() - 1; ++i) - { - dump_byte(*i); - put_literal(", "); - } - dump_byte(val.m_data.m_value.binary->back()); - } - - put_literal("],\n"); - put_indent(new_indent); - - put_literal("\"subtype\": "); - if (val.m_data.m_value.binary->has_subtype()) - { - dump_integer(val.m_data.m_value.binary->subtype()); - } - else - { - put_literal("null"); - } - put_char('\n'); - put_indent(current_indent); - put_char('}'); - } - else - { - put_literal("{\"bytes\":["); - - if (!val.m_data.m_value.binary->empty()) - { - for (auto i = val.m_data.m_value.binary->cbegin(); - i != val.m_data.m_value.binary->cend() - 1; ++i) - { - dump_byte(*i); - put_char(','); - } - dump_byte(val.m_data.m_value.binary->back()); - } - - put_literal("],\"subtype\":"); - if (val.m_data.m_value.binary->has_subtype()) - { - dump_integer(val.m_data.m_value.binary->subtype()); - put_char('}'); - } - else - { - put_literal("null}"); - } - } - return; - } - case value_t::boolean: - { - if (val.m_data.m_value.boolean) - { - put_literal("true"); - } - else - { - put_literal("false"); - } - return; - } - case value_t::number_integer: - { - dump_integer(val.m_data.m_value.number_integer); - return; - } - case value_t::number_unsigned: - { - dump_integer(val.m_data.m_value.number_unsigned); - return; - } - case value_t::number_float: - { - dump_float(val.m_data.m_value.number_float); - return; - } - case value_t::discarded: - { - put_literal(""); - return; - } - case value_t::null: - { - put_literal("null"); + default: + dump_scalar(val, current_indent); return; - } - - default: // LCOV_EXCL_LINE - JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } } @@ -560,9 +447,9 @@ class serializer @brief serialize the value @a val, but not the elements of a container An object or array with elements is opened and pushed onto @a stack for - @ref dump_internal to walk; everything else - including a binary value, + @ref dump_iteratively to walk; everything else - including a binary value, which looks like an object but has no elements to descend into - is written - out here in full. + out in full by @ref dump_scalar. */ void dump_value(const BasicJsonType& val, const std::size_t current_indent, @@ -620,6 +507,35 @@ class serializer return; } + case value_t::string: + case value_t::binary: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::discarded: + case value_t::null: + default: + dump_scalar(val, current_indent); + return; + } + } + + /*! + @brief serialize the value @a val, which is neither an object nor an array + + Shared by @ref dump_internal and @ref dump_value, so that a value is written + the same way however deeply it is nested. A binary value is written out here + in full: it looks like an object, but has no elements to descend into. + + @param[in] val value to serialize; not an object or array + @param[in] current_indent the indentation of @a val, used for a + pretty-printed binary value + */ + void dump_scalar(const BasicJsonType& val, const std::size_t current_indent) + { + switch (val.m_data.m_type) + { case value_t::string: { put_char('"'); @@ -740,6 +656,8 @@ class serializer return; } + case value_t::object: // LCOV_EXCL_LINE + case value_t::array: // LCOV_EXCL_LINE default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -768,7 +686,7 @@ class serializer Escape a string by replacing certain special characters by a sequence of an escape character (backslash) and another character and other control characters by a sequence of "\u" followed by a four-digit hex - representation. The escaped string is written to output stream @a o. + representation. The escaped string is appended to @ref write_buffer. @param[in] s the string to escape @@ -962,7 +880,7 @@ class serializer { case error_handler_t::strict: { - JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", hex_bytes(byte | 0)), nullptr)); + JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", detail::hex_byte(byte)), nullptr)); } case error_handler_t::ignore: @@ -995,9 +913,9 @@ class serializer } else { - string_buffer[bytes++] = detail::binary_writer::to_char_type('\xEF'); - string_buffer[bytes++] = detail::binary_writer::to_char_type('\xBF'); - string_buffer[bytes++] = detail::binary_writer::to_char_type('\xBD'); + string_buffer[bytes++] = '\xEF'; + string_buffer[bytes++] = '\xBF'; + string_buffer[bytes++] = '\xBD'; } // write buffer and reset index; there must be 13 bytes @@ -1054,7 +972,7 @@ class serializer { case error_handler_t::strict: { - JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast(s[s.size() - 1] | 0))), nullptr)); + JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", detail::hex_byte(static_cast(s[s.size() - 1]))), nullptr)); } case error_handler_t::ignore: @@ -1275,20 +1193,6 @@ class serializer } } - /*! - * @brief convert a byte to a uppercase hex representation - * @param[in] byte byte to represent - * @return representation ("00".."FF") - */ - static std::string hex_bytes(std::uint8_t byte) - { - std::string result = "FF"; - constexpr const char* nibble_to_hex = "0123456789ABCDEF"; - result[0] = nibble_to_hex[byte / 16]; - result[1] = nibble_to_hex[byte % 16]; - return result; - } - /*! * @brief write a lowercase "\uXXXX" escape sequence into @a string_buffer * @@ -1402,7 +1306,7 @@ class serializer /*! @brief dump an integer - Dump a given integer to output stream @a o. Works internally with + Dump a given integer, appending it to @ref write_buffer. Works internally with @a number_buffer. @param[in] x integer number (signed or unsigned) to dump @@ -1439,7 +1343,7 @@ class serializer } // use a pointer to fill the buffer - auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto,cppcoreguidelines-pro-type-vararg,hicpp-vararg) + auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto) number_unsigned_t abs_value; @@ -1493,7 +1397,7 @@ class serializer /*! @brief dump a floating-point number - Dump a given floating-point number to output stream @a o. Works internally + Dump a given floating-point number, appending it to @ref write_buffer. Works internally with @a number_buffer. @param[in] x floating-point number to dump @@ -1554,21 +1458,28 @@ class serializer // check if the buffer was large enough JSON_ASSERT(static_cast(len) < number_buffer.size()); + // look up the locale's thousands separator and decimal point now, + // matching what snprintf_float() just used (see lexer::get_decimal_point()) + const auto* loc = std::localeconv(); + JSON_ASSERT(loc != nullptr); + const char thousands_sep = (loc->thousands_sep == nullptr) ? '\0' : *loc->thousands_sep; + const char decimal_point = (loc->decimal_point == nullptr) ? '\0' : *loc->decimal_point; + // erase thousands separators - if (locale.thousands_sep != '\0') + if (thousands_sep != '\0') { // NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::remove returns an iterator, see https://github.com/nlohmann/json/issues/3081 - const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, locale.thousands_sep); + const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, thousands_sep); std::fill(end, number_buffer.end(), '\0'); JSON_ASSERT((end - number_buffer.begin()) <= len); len = (end - number_buffer.begin()); } // convert decimal point to '.' - if (locale.decimal_point != '\0' && locale.decimal_point != '.') + if (decimal_point != '\0' && decimal_point != '.') { // NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::find returns an iterator, see https://github.com/nlohmann/json/issues/3081 - const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), locale.decimal_point); + const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), decimal_point); if (dec_pos != number_buffer.end()) { *dec_pos = '.'; @@ -1613,34 +1524,17 @@ class serializer */ number_unsigned_t remove_sign(number_integer_t x) noexcept { - JSON_ASSERT(x < 0 && x < (std::numeric_limits::max)()); // NOLINT(misc-redundant-expression) + JSON_ASSERT(x < 0); return static_cast(-(x + 1)) + 1; } private: - /// the locale's thousand separator and decimal point characters - struct locale_chars - { - explicit locale_chars(const std::lconv* loc) noexcept - : thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits::to_char_type(* (loc->thousands_sep))) - , decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits::to_char_type(* (loc->decimal_point))) - {} - - const char thousands_sep; - const char decimal_point; - }; - /// the output of the serializer (non-owning; the adapter lives at the call site) output_adapter_protocol* o = nullptr; /// a (hopefully) large enough character buffer std::array number_buffer{{}}; - /// computed once from std::localeconv() at construction; @ref - /// locale_chars keeps std::localeconv()'s pointer from having to be held - /// past the constructor, while still letting these stay const - const locale_chars locale; - /// string buffer std::array string_buffer{{}}; diff --git a/include/nlohmann/detail/string_utils.hpp b/include/nlohmann/detail/string_utils.hpp index 142943cd6..2bfbc8a04 100644 --- a/include/nlohmann/detail/string_utils.hpp +++ b/include/nlohmann/detail/string_utils.hpp @@ -3,6 +3,7 @@ // | | |__ | | | | | | version 3.12.0 // |_____|_____|_____|_|___| https://github.com/nlohmann/json // +// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT @@ -36,6 +37,16 @@ StringType to_string(std::size_t value) return result; } +/// @return a byte as two uppercase hexadecimal digits +inline std::string hex_byte(const std::uint8_t byte) +{ + std::string result = "00"; + constexpr const char* nibble_to_hex = "0123456789ABCDEF"; + result[0] = nibble_to_hex[byte / 16]; + result[1] = nibble_to_hex[byte % 16]; + return result; +} + /////////////////// // UTF-8 decoding // /////////////////// diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 44e5201df..283b1bb76 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -6223,6 +6223,7 @@ NLOHMANN_JSON_NAMESPACE_END // | | |__ | | | | | | version 3.12.0 // |_____|_____|_____|_|___| https://github.com/nlohmann/json // +// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT @@ -6258,6 +6259,16 @@ StringType to_string(std::size_t value) return result; } +/// @return a byte as two uppercase hexadecimal digits +inline std::string hex_byte(const std::uint8_t byte) +{ + std::string result = "00"; + constexpr const char* nibble_to_hex = "0123456789ABCDEF"; + result[0] = nibble_to_hex[byte / 16]; + result[1] = nibble_to_hex[byte % 16]; + return result; +} + /////////////////// // UTF-8 decoding // /////////////////// @@ -20903,9 +20914,7 @@ NLOHMANN_JSON_NAMESPACE_END -#include // copy #include // size_t -#include // back_inserter #include // shared_ptr, make_shared #include // basic_string #include // move @@ -20927,6 +20936,10 @@ namespace detail template struct output_adapter_protocol { virtual void write_character(CharType c) = 0; + /// @param[in] s pointer to the characters to write; binary_writer legitimately + /// passes a null pointer together with length 0 for an empty + /// string or binary value, so implementations must tolerate that + /// @param[in] length number of characters at @a s virtual void write_characters(const CharType* s, std::size_t length) = 0; virtual ~output_adapter_protocol() = default; @@ -20993,7 +21006,6 @@ class output_vector_adapter : public output_adapter_protocol sink.write_character(c); } - JSON_HEDLEY_NON_NULL(2) void write_characters(const CharType* s, std::size_t length) override { sink.write_characters(s, length); @@ -21018,7 +21030,6 @@ class output_stream_adapter : public output_adapter_protocol stream.put(c); } - JSON_HEDLEY_NON_NULL(2) void write_characters(const CharType* s, std::size_t length) override { stream.write(s, static_cast(length)); @@ -21043,7 +21054,6 @@ class output_string_adapter : public output_adapter_protocol str.push_back(c); } - JSON_HEDLEY_NON_NULL(2) void write_characters(const CharType* s, std::size_t length) override { str.append(s, length); @@ -21116,6 +21126,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -23216,20 +23228,10 @@ class binary_writer const std::size_t valid = valid_utf8_prefix(data, s.size()); if (JSON_HEDLEY_UNLIKELY(valid != s.size())) { - JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", hex_byte(data[valid])), &context)); + JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", detail::hex_byte(data[valid])), &context)); } } - /// @return a byte as two uppercase hexadecimal digits - static std::string hex_byte(const std::uint8_t byte) - { - std::string result = "00"; - constexpr const char* nibble_to_hex = "0123456789ABCDEF"; - result[0] = nibble_to_hex[byte / 16]; - result[1] = nibble_to_hex[byte % 16]; - return result; - } - /*! @brief write an integer in the shortest encoding @@ -23571,24 +23573,23 @@ NLOHMANN_JSON_NAMESPACE_END // | | |__ | | | | | | version 3.12.0 // |_____|_____|_____|_|___| https://github.com/nlohmann/json // -// SPDX-FileCopyrightText: 2008, 2009 Björn Hoehrmann // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann // SPDX-License-Identifier: MIT -#include // reverse, remove, fill, find, none_of, min +#include // remove, fill, find, none_of, min #include // array #include // localeconv, lconv -#include // labs, isfinite, isnan, signbit +#include // isfinite #include // size_t, ptrdiff_t #include // uint8_t #include // snprintf #include // memcpy, memset +#include // next #include // numeric_limits #include // string, char_traits #include // is_same -#include // move #include // vector // #include @@ -24720,8 +24721,6 @@ NLOHMANN_JSON_NAMESPACE_END // #include -// #include - // #include // #include @@ -24781,7 +24780,6 @@ class serializer const std::size_t indent_step_ = 0, error_handler_t error_handler_ = error_handler_t::strict) : o(&s) - , locale(std::localeconv()) , indent_char(ichar) , pretty_print(pretty_print_) , ensure_ascii(ensure_ascii_) @@ -24804,9 +24802,10 @@ class serializer additional parameter. Arrays and objects are serialized without recursion, however deeply they are nested. - - strings and object keys are escaped using `escape_string()` - - integer numbers are converted implicitly via `operator<<` - - floating-point numbers are converted to a string using `"%g"` format + - strings and object keys are escaped using @ref dump_escaped + - integer numbers are converted using a digit-pair lookup table (@ref dump_integer) + - floating-point numbers are converted to a string using @ref dump_float, which + uses `to_chars` for IEEE-754 types and `snprintf` otherwise - binary values are serialized as objects containing the subtype and the byte array @@ -24981,127 +24980,16 @@ class serializer } case value_t::string: - { - put_char('"'); - dump_escaped(*val.m_data.m_value.string); - put_char('"'); - return; - } - case value_t::binary: - { - if (pretty_print) - { - put_literal("{\n"); - - // variable to hold indentation for recursive calls - const auto new_indent = next_indent(current_indent, indent_step); - - put_indent(new_indent); - - put_literal("\"bytes\": ["); - - if (!val.m_data.m_value.binary->empty()) - { - for (auto i = val.m_data.m_value.binary->cbegin(); - i != val.m_data.m_value.binary->cend() - 1; ++i) - { - dump_byte(*i); - put_literal(", "); - } - dump_byte(val.m_data.m_value.binary->back()); - } - - put_literal("],\n"); - put_indent(new_indent); - - put_literal("\"subtype\": "); - if (val.m_data.m_value.binary->has_subtype()) - { - dump_integer(val.m_data.m_value.binary->subtype()); - } - else - { - put_literal("null"); - } - put_char('\n'); - put_indent(current_indent); - put_char('}'); - } - else - { - put_literal("{\"bytes\":["); - - if (!val.m_data.m_value.binary->empty()) - { - for (auto i = val.m_data.m_value.binary->cbegin(); - i != val.m_data.m_value.binary->cend() - 1; ++i) - { - dump_byte(*i); - put_char(','); - } - dump_byte(val.m_data.m_value.binary->back()); - } - - put_literal("],\"subtype\":"); - if (val.m_data.m_value.binary->has_subtype()) - { - dump_integer(val.m_data.m_value.binary->subtype()); - put_char('}'); - } - else - { - put_literal("null}"); - } - } - return; - } - case value_t::boolean: - { - if (val.m_data.m_value.boolean) - { - put_literal("true"); - } - else - { - put_literal("false"); - } - return; - } - case value_t::number_integer: - { - dump_integer(val.m_data.m_value.number_integer); - return; - } - case value_t::number_unsigned: - { - dump_integer(val.m_data.m_value.number_unsigned); - return; - } - case value_t::number_float: - { - dump_float(val.m_data.m_value.number_float); - return; - } - case value_t::discarded: - { - put_literal(""); - return; - } - case value_t::null: - { - put_literal("null"); + default: + dump_scalar(val, current_indent); return; - } - - default: // LCOV_EXCL_LINE - JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } } @@ -25258,9 +25146,9 @@ class serializer @brief serialize the value @a val, but not the elements of a container An object or array with elements is opened and pushed onto @a stack for - @ref dump_internal to walk; everything else - including a binary value, + @ref dump_iteratively to walk; everything else - including a binary value, which looks like an object but has no elements to descend into - is written - out here in full. + out in full by @ref dump_scalar. */ void dump_value(const BasicJsonType& val, const std::size_t current_indent, @@ -25318,6 +25206,35 @@ class serializer return; } + case value_t::string: + case value_t::binary: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::discarded: + case value_t::null: + default: + dump_scalar(val, current_indent); + return; + } + } + + /*! + @brief serialize the value @a val, which is neither an object nor an array + + Shared by @ref dump_internal and @ref dump_value, so that a value is written + the same way however deeply it is nested. A binary value is written out here + in full: it looks like an object, but has no elements to descend into. + + @param[in] val value to serialize; not an object or array + @param[in] current_indent the indentation of @a val, used for a + pretty-printed binary value + */ + void dump_scalar(const BasicJsonType& val, const std::size_t current_indent) + { + switch (val.m_data.m_type) + { case value_t::string: { put_char('"'); @@ -25438,6 +25355,8 @@ class serializer return; } + case value_t::object: // LCOV_EXCL_LINE + case value_t::array: // LCOV_EXCL_LINE default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -25466,7 +25385,7 @@ class serializer Escape a string by replacing certain special characters by a sequence of an escape character (backslash) and another character and other control characters by a sequence of "\u" followed by a four-digit hex - representation. The escaped string is written to output stream @a o. + representation. The escaped string is appended to @ref write_buffer. @param[in] s the string to escape @@ -25660,7 +25579,7 @@ class serializer { case error_handler_t::strict: { - JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", hex_bytes(byte | 0)), nullptr)); + JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(i), ": 0x", detail::hex_byte(byte)), nullptr)); } case error_handler_t::ignore: @@ -25693,9 +25612,9 @@ class serializer } else { - string_buffer[bytes++] = detail::binary_writer::to_char_type('\xEF'); - string_buffer[bytes++] = detail::binary_writer::to_char_type('\xBF'); - string_buffer[bytes++] = detail::binary_writer::to_char_type('\xBD'); + string_buffer[bytes++] = '\xEF'; + string_buffer[bytes++] = '\xBF'; + string_buffer[bytes++] = '\xBD'; } // write buffer and reset index; there must be 13 bytes @@ -25752,7 +25671,7 @@ class serializer { case error_handler_t::strict: { - JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast(s[s.size() - 1] | 0))), nullptr)); + JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", detail::hex_byte(static_cast(s[s.size() - 1]))), nullptr)); } case error_handler_t::ignore: @@ -25973,20 +25892,6 @@ class serializer } } - /*! - * @brief convert a byte to a uppercase hex representation - * @param[in] byte byte to represent - * @return representation ("00".."FF") - */ - static std::string hex_bytes(std::uint8_t byte) - { - std::string result = "FF"; - constexpr const char* nibble_to_hex = "0123456789ABCDEF"; - result[0] = nibble_to_hex[byte / 16]; - result[1] = nibble_to_hex[byte % 16]; - return result; - } - /*! * @brief write a lowercase "\uXXXX" escape sequence into @a string_buffer * @@ -26100,7 +26005,7 @@ class serializer /*! @brief dump an integer - Dump a given integer to output stream @a o. Works internally with + Dump a given integer, appending it to @ref write_buffer. Works internally with @a number_buffer. @param[in] x integer number (signed or unsigned) to dump @@ -26137,7 +26042,7 @@ class serializer } // use a pointer to fill the buffer - auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto,cppcoreguidelines-pro-type-vararg,hicpp-vararg) + auto buffer_ptr = number_buffer.begin(); // NOLINT(llvm-qualified-auto,readability-qualified-auto) number_unsigned_t abs_value; @@ -26191,7 +26096,7 @@ class serializer /*! @brief dump a floating-point number - Dump a given floating-point number to output stream @a o. Works internally + Dump a given floating-point number, appending it to @ref write_buffer. Works internally with @a number_buffer. @param[in] x floating-point number to dump @@ -26252,21 +26157,28 @@ class serializer // check if the buffer was large enough JSON_ASSERT(static_cast(len) < number_buffer.size()); + // look up the locale's thousands separator and decimal point now, + // matching what snprintf_float() just used (see lexer::get_decimal_point()) + const auto* loc = std::localeconv(); + JSON_ASSERT(loc != nullptr); + const char thousands_sep = (loc->thousands_sep == nullptr) ? '\0' : *loc->thousands_sep; + const char decimal_point = (loc->decimal_point == nullptr) ? '\0' : *loc->decimal_point; + // erase thousands separators - if (locale.thousands_sep != '\0') + if (thousands_sep != '\0') { // NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::remove returns an iterator, see https://github.com/nlohmann/json/issues/3081 - const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, locale.thousands_sep); + const auto end = std::remove(number_buffer.begin(), number_buffer.begin() + len, thousands_sep); std::fill(end, number_buffer.end(), '\0'); JSON_ASSERT((end - number_buffer.begin()) <= len); len = (end - number_buffer.begin()); } // convert decimal point to '.' - if (locale.decimal_point != '\0' && locale.decimal_point != '.') + if (decimal_point != '\0' && decimal_point != '.') { // NOLINTNEXTLINE(readability-qualified-auto,llvm-qualified-auto): std::find returns an iterator, see https://github.com/nlohmann/json/issues/3081 - const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), locale.decimal_point); + const auto dec_pos = std::find(number_buffer.begin(), number_buffer.end(), decimal_point); if (dec_pos != number_buffer.end()) { *dec_pos = '.'; @@ -26311,34 +26223,17 @@ class serializer */ number_unsigned_t remove_sign(number_integer_t x) noexcept { - JSON_ASSERT(x < 0 && x < (std::numeric_limits::max)()); // NOLINT(misc-redundant-expression) + JSON_ASSERT(x < 0); return static_cast(-(x + 1)) + 1; } private: - /// the locale's thousand separator and decimal point characters - struct locale_chars - { - explicit locale_chars(const std::lconv* loc) noexcept - : thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits::to_char_type(* (loc->thousands_sep))) - , decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits::to_char_type(* (loc->decimal_point))) - {} - - const char thousands_sep; - const char decimal_point; - }; - /// the output of the serializer (non-owning; the adapter lives at the call site) output_adapter_protocol* o = nullptr; /// a (hopefully) large enough character buffer std::array number_buffer{{}}; - /// computed once from std::localeconv() at construction; @ref - /// locale_chars keeps std::localeconv()'s pointer from having to be held - /// past the constructor, while still letting these stay const - const locale_chars locale; - /// string buffer std::array string_buffer{{}}; diff --git a/tests/src/unit-locale-cpp.cpp b/tests/src/unit-locale-cpp.cpp index 14f743a66..1a1f77c5d 100644 --- a/tests/src/unit-locale-cpp.cpp +++ b/tests/src/unit-locale-cpp.cpp @@ -14,7 +14,10 @@ using nlohmann::json; #include #include +#include #include +#include +#include #include #include #include @@ -385,3 +388,92 @@ TEST_CASE("locale with a multi-byte decimal point") CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr); } + +namespace +{ +// a streambuf that switches LC_NUMERIC the first time anything is written to +// it, so a dump() in progress can be made to change locale mid-flight: after +// the serializer was constructed (and, before #5709 item 3, after it had +// cached std::localeconv() for the whole call) but before a later float is +// converted +struct LocaleSwitchingStreambuf final : std::streambuf +{ + explicit LocaleSwitchingStreambuf(const char* switch_to) + : locale_after_first_write(switch_to) + {} + + std::string data {}; // NOLINT(readability-redundant-member-init) + std::string locale_after_first_write; + bool switched = false; + + std::streamsize xsputn(const char* s, std::streamsize n) override + { + if (!switched) + { + switched = std::setlocale(LC_NUMERIC, locale_after_first_write.c_str()) != nullptr; + } + data.append(s, static_cast(n)); + return n; + } +}; +} // namespace + +TEST_CASE("locale changes during a single dump() (#5709 item 3)") +{ + // dump_float() only reads the locale on the snprintf path, taken for a + // number_float_t that is not an IEEE-754 single or double, i.e. not + // (is_iec559 && digits == 24 && max_exponent == 128) and not (is_iec559 + // && digits == 53 && max_exponent == 1024) - see dump_float(). Checking + // is_iec559 alone is not enough: on x86_64, long double is a 64-bit + // (80-bit extended) format for which is_iec559 is also true, so it still + // takes the snprintf path this test means to exercise. Only a + // number_float_t whose digits/max_exponent match float or double (e.g. + // long double on 64-bit Arm, where it is IEEE-754 double) takes the + // locale-independent to_chars() path instead, and this test is a no-op + // there. + using long_double_json = nlohmann::basic_json; + using ld_limits = std::numeric_limits; + const bool is_ieee_single_or_double = + (ld_limits::is_iec559 && ld_limits::digits == 24 && ld_limits::max_exponent == 128) || + (ld_limits::is_iec559 && ld_limits::digits == 53 && ld_limits::max_exponent == 1024); + if (is_ieee_single_or_double) + { + MESSAGE("long double is IEEE-754 single or double on this platform; dump_float()'s snprintf/locale path is not exercised here"); + } + + const char* de_DE_name = "de_DE.UTF-8"; + if (std::setlocale(LC_NUMERIC, de_DE_name) == nullptr) + { + de_DE_name = "de_DE"; + if (std::setlocale(LC_NUMERIC, de_DE_name) == nullptr) + { + MESSAGE("locale de_DE is not usable"); + return; + } + } + const std::string decimal_point = std::localeconv()->decimal_point; + REQUIRE(std::setlocale(LC_NUMERIC, "C") != nullptr); + if (decimal_point != ",") + { + MESSAGE("de_DE's decimal point is not ',' on this platform, skipping"); + return; + } + + // a string long enough to overflow the serializer's internal write + // buffer, so that it is flushed to the output adapter - and the locale + // switched - before the number after it is converted + const std::string padding(5000, 'a'); + const long_double_json j = { padding, 1234.5L }; + + LocaleSwitchingStreambuf buf(de_DE_name); + std::ostream os(&buf); + os << j; + CHECK(std::setlocale(LC_NUMERIC, "C") != nullptr); + + REQUIRE(buf.switched); + // whatever locale was in effect when the float was actually converted, + // the output is normalized to use '.' as the decimal point: it must be + // looked up at conversion time, not once for the whole dump() - the same + // fix #5597 made on the parser side + CHECK(buf.data == "[\"" + padding + "\",1234.5]"); +}