Compare commits

..
Author SHA1 Message Date
Niels Lohmann cc6d74feb0 Copy a pair-shaped array value under JSON_BRACE_INIT_COPY_SEMANTICS
With JSON_BRACE_INIT_COPY_SEMANTICS enabled, single-element brace
initialization from a JSON value decided whether to copy the value or
build an object by inspecting the value's runtime shape: a two-element
array whose first element is a string, such as ["key", 42], was turned
into an object instead of being copied. This made the behavior depend
on the element's content, and it did not distinguish an existing value
of this shape from a nested braced pair written in the source, such as
the inner {"key", "value"} of {{"key", "value"}}.

json_ref now records whether it was constructed from a braced list
(true only for the std::initializer_list<json_ref> constructor used
for nested braced lists) or from a value. The initializer-list
constructor uses this to copy or move a single non-braced-list element
before deciding whether the list describes an object, so a JSON value
is always copied regardless of its shape, while a braced pair written
in the source still creates an object.

Fixes #5662.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 23:41:48 +02:00
7 changed files with 81 additions and 77 deletions
@@ -50,9 +50,11 @@ The default value is `0` (disabled — existing behavior is preserved).
```
Code that relies on these producing arrays must use `json::array()` instead (see below). Lists with more than one
element, and a single `[string, value]` pair such as `{{"key", "value"}}`, which still creates an object, are not
affected. The library's own conversions are not affected either: for example, `std::tuple<int>{5}` still becomes
`[5]`.
element, and a single `[string, value]` pair *written as a braced list*, such as `{{"key", "value"}}`, which still
creates an object, are not affected. This exception is based on how the pair is written, not on the shape of its
value: an existing JSON value that happens to be a two-element array with a string as its first element, such as
`json arr = {"key", 42};`, is still copied by `json j{arr};` rather than turned into an object. The library's own
conversions are not affected either: for example, `std::tuple<int>{5}` still becomes `[5]`.
!!! note "ABI compatibility"
@@ -445,10 +445,8 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
}
else
{
// get the current character; converted to an unsigned type so that
// a negative unit (wint_t is signed on some platforms) is not
// mistaken for an ASCII character or for EOF
const auto wc = static_cast<std::uint32_t>(input.get_character());
// get the current character
const auto wc = input.get_character();
// UTF-32 to UTF-8 encoding
if (wc < 0x80)
@@ -543,11 +541,9 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
bool valid_pair = false;
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
{
// only consume the next unit if it completes the pair
const auto wc2 = static_cast<unsigned int>(*input.current);
const auto wc2 = static_cast<unsigned int>(input.get_character());
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{
input.get_character();
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
@@ -560,8 +556,7 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (!valid_pair)
{
// emit a byte that is never valid UTF-8 (see the UTF-32 case)
utf8_bytes[0] = 0xFF;
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
utf8_bytes_filled = 1;
}
}
+9
View File
@@ -34,6 +34,7 @@ class json_ref
json_ref(std::initializer_list<json_ref> init)
: owned_value(init)
, braced_list(true)
{}
template <
@@ -69,9 +70,17 @@ class json_ref
return &** this;
}
/// whether the value was written as a braced list, such as {"key", 1},
/// rather than given as a value
bool is_braced_list() const noexcept
{
return braced_list;
}
private:
mutable value_type owned_value = nullptr;
value_type const* value_ref = nullptr;
bool braced_list = false;
};
} // namespace detail
+12
View File
@@ -1672,6 +1672,18 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
bool type_deduction = true,
value_t manual_type = value_t::array)
{
#if JSON_BRACE_INIT_COPY_SEMANTICS
// a single element that is a value rather than a braced list is
// copied or moved as is, whatever its content looks like
if (type_deduction && init.size() == 1 && !init.begin()->is_braced_list())
{
*this = init.begin()->moved_or_copied();
set_parents();
assert_invariant();
return;
}
#endif
// check if each element is an array with two elements whose first
// element is a string
bool is_an_object = std::all_of(init.begin(), init.end(),
+25 -9
View File
@@ -7989,10 +7989,8 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
}
else
{
// get the current character; converted to an unsigned type so that
// a negative unit (wint_t is signed on some platforms) is not
// mistaken for an ASCII character or for EOF
const auto wc = static_cast<std::uint32_t>(input.get_character());
// get the current character
const auto wc = input.get_character();
// UTF-32 to UTF-8 encoding
if (wc < 0x80)
@@ -8087,11 +8085,9 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
bool valid_pair = false;
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
{
// only consume the next unit if it completes the pair
const auto wc2 = static_cast<unsigned int>(*input.current);
const auto wc2 = static_cast<unsigned int>(input.get_character());
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{
input.get_character();
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
@@ -8104,8 +8100,7 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (!valid_pair)
{
// emit a byte that is never valid UTF-8 (see the UTF-32 case)
utf8_bytes[0] = 0xFF;
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
utf8_bytes_filled = 1;
}
}
@@ -20052,6 +20047,7 @@ class json_ref
json_ref(std::initializer_list<json_ref> init)
: owned_value(init)
, braced_list(true)
{}
template <
@@ -20087,9 +20083,17 @@ class json_ref
return &** this;
}
/// whether the value was written as a braced list, such as {"key", 1},
/// rather than given as a value
bool is_braced_list() const noexcept
{
return braced_list;
}
private:
mutable value_type owned_value = nullptr;
value_type const* value_ref = nullptr;
bool braced_list = false;
};
} // namespace detail
@@ -27758,6 +27762,18 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
bool type_deduction = true,
value_t manual_type = value_t::array)
{
#if JSON_BRACE_INIT_COPY_SEMANTICS
// a single element that is a value rather than a braced list is
// copied or moved as is, whatever its content looks like
if (type_deduction && init.size() == 1 && !init.begin()->is_braced_list())
{
*this = init.begin()->moved_or_copied();
set_parents();
assert_invariant();
return;
}
#endif
// check if each element is an array with two elements whose first
// element is a string
bool is_an_object = std::all_of(init.begin(), init.end(),
@@ -72,6 +72,28 @@ TEST_CASE("JSON_BRACE_INIT_COPY_SEMANTICS")
CHECK(j7 == json::array({1, 2}));
}
SECTION("single-element brace initialization copies a pair-shaped array value (#5662)")
{
// a JSON value that happens to be a 2-element array whose first
// element is a string must still be copied, not turned into an
// object; only a braced list written in the source, such as the
// inner {"key", "value"} of {{"key", "value"}}, describes an object
json const pair_shaped = json::array({"key", 42});
json const j1{pair_shaped};
CHECK(j1.is_array());
CHECK(j1 == pair_shaped);
json const j2 = {pair_shaped};
CHECK(j2.is_array());
CHECK(j2 == pair_shaped);
// the same holds for an rvalue of the same shape
json const j3{json::array({"key", 42})};
CHECK(j3.is_array());
CHECK(j3 == pair_shaped);
}
SECTION("what the macro does not change")
{
// lists with more than one element are unaffected
+4 -56
View File
@@ -8,7 +8,6 @@
#include "doctest_compatibility.h"
#include <cwchar>
#include <nlohmann/json.hpp>
using nlohmann::json;
@@ -99,15 +98,15 @@ TEST_CASE("wide strings")
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
// a lone low surrogate cannot start a pair
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a high surrogate followed by a non-low-surrogate unit is invalid
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// ... also when the unit is above the low surrogates
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a lone low surrogate must not swallow the following unit: pairing
// it with any second unit would produce valid UTF-8, so the error
// has to report an ill-formed byte at the surrogate's own position
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a valid surrogate pair is still decoded (U+1F600)
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
}
@@ -142,56 +141,5 @@ TEST_CASE("wide strings")
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
}
}
SECTION("malformed wide-string input outside strings (#5645)")
{
json _;
// a lone low surrogate inside a literal must not be truncated to its
// low byte and mistaken for the letter the literal expects next
// (0xDC72 truncates to 'r', which is what "true" expects after 't')
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast<char16_t>(0xDC72), u'u', u'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
// ... also when the lone surrogate is the last unit of the input
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'f', u'a', u'l', u's', static_cast<char16_t>(0xDD65)}),
"[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'fals\xFF'", json::parse_error&);
// a high surrogate followed by a unit that is not its low surrogate
// must not silently swallow that unit
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast<char16_t>(0xD872), u'X', u'u', u'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
// ... in particular, if the swallowed unit is the newline that ends a
// // comment, the comment must not extend over the following line
CHECK(json::parse(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast<char16_t>(0xD800), u'\n',
u',', u'2', u' ', u'/', u'/', u'\n', u']'},
nullptr, true, /*ignore_comments*/true) == json::parse("[1,2]"));
CHECK(json::accept(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast<char16_t>(0xD800), u'\n',
u',', u'2', u' ', u'/', u'/', u'\n', u']'}, /*ignore_comments*/true));
// cases 5 and 6 use a 32-bit wchar_t (Linux, macOS, the BSDs) to reach
// the UTF-32 helper tested above via u32string; the 16-bit wchar_t of
// Windows goes through the UTF-16 helper instead, already covered by
// the u16string cases above
#if WCHAR_MAX > 0xFFFFu
// a negative wchar_t must not be mistaken for
// char_traits<char>::eof() and silently end the input, letting
// trailing garbage pass the strict end-of-input check (only observable
// where wint_t is signed, e.g. macOS/the BSDs; on Linux wint_t is
// unsigned and this was already handled by #5348)
std::wstring w = L"[1]";
w.push_back(static_cast<wchar_t>(-1));
w += L"garbage";
CHECK(!json::accept(w));
CHECK_THROWS_WITH_AS(_ = json::parse(w),
"[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
// other negative wchar_t units must not be truncated to their low
// byte (0xFFFFFF72 truncates to 'r', as in the u16string case above)
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L't', static_cast<wchar_t>(0xFFFFFF72), L'u', L'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
#endif
}
}
#endif