Compare commits

..
Author SHA1 Message Date
Niels Lohmann cc749ac699 Keep full byte2 x byte3 combinatorics in the wrong-fourth-byte sections
The maintainer wants exhaustive coverage of every byte combination here
rather than the representative-prefix reduction, matching the style of
the sibling "wrong second/third byte" sections in the same files.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-05 23:18:51 +02:00
Niels Lohmann 1c6ca81b8a Fix dead ill-formed-fourth-byte UTF-8 test sections (byte3/byte4 typo)
The "ill-formed: wrong fourth byte" SECTIONs in unit-unicode3.cpp,
unit-unicode4.cpp, and unit-unicode5.cpp guarded their loop with a check
on byte3 instead of byte4. Since the enclosing loop already restricts
byte3 to its valid range, the guard was always true and the section's
"continue" fired unconditionally, so check_utf8string()/check_utf8dump()
were never actually invoked for a malformed fourth byte.

Fixing the guard naively (byte3 -> byte4) would also have swept the full
byte2 x byte3 combinatorics for every byte4 value, adding millions of
redundant iterations: the lexer validates continuation bytes strictly in
sequence with early exit (see next_byte_in_range() in lexer.hpp), so once
byte2/byte3 are within their valid range, the byte4 outcome does not
depend on which valid byte2/byte3 values were chosen. Instead, byte2 and
byte3 are now held to a small hedge of representative valid prefixes
(range corners plus a midpoint) while byte4 is still swept exhaustively
over its full 0x00-0xFF range, since that is the actual property under
test. Also fixed the garbled "skip fourth second byte" comment in
unit-unicode3.cpp.

Verified offline: before the fix, the "wrong fourth byte" subcase
executes 0 assertions in all three files (proving it was dead code);
after the fix, it executes 11520 (unicode3), 34560 (unicode4), and 11520
(unicode5) assertions, and a deliberately reintroduced bug in the
lexer's byte4 range check causes it to fail (proving it is now
meaningful). Total per-file assertion counts grow by the same small
amounts, not by millions, and all other sections in these files still
pass unchanged.

Fixes #5416

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-05 23:05:42 +02:00
6 changed files with 18 additions and 93 deletions
@@ -398,34 +398,6 @@ inline void from_json(const BasicJsonType& j, CompatibleArrayType& bin)
}
}
template<typename BasicJsonType, typename ConstructibleObjectType>
auto from_json_object_impl(const BasicJsonType& j, ConstructibleObjectType& obj, priority_tag<1> /*unused*/)
-> decltype(
obj.reserve(std::declval<typename ConstructibleObjectType::size_type>()),
void())
{
ConstructibleObjectType ret;
const auto* inner_object = j.template get_ptr<const typename BasicJsonType::object_t*>();
ret.reserve(inner_object->size());
for (const auto& p : *inner_object)
{
ret.emplace(p.first, p.second.template get<typename ConstructibleObjectType::mapped_type>());
}
obj = std::move(ret);
}
template<typename BasicJsonType, typename ConstructibleObjectType>
inline void from_json_object_impl(const BasicJsonType& j, ConstructibleObjectType& obj, priority_tag<0> /*unused*/)
{
ConstructibleObjectType ret;
const auto* inner_object = j.template get_ptr<const typename BasicJsonType::object_t*>();
for (const auto& p : *inner_object)
{
ret.emplace(p.first, p.second.template get<typename ConstructibleObjectType::mapped_type>());
}
obj = std::move(ret);
}
template<typename BasicJsonType, typename ConstructibleObjectType,
enable_if_t<is_constructible_object_type<BasicJsonType, ConstructibleObjectType>::value, int> = 0>
inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj)
@@ -435,7 +407,13 @@ inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj)
JSON_THROW(type_error::create(302, concat("type must be object, but is ", j.type_name()), &j));
}
from_json_object_impl(j, obj, priority_tag<1> {});
ConstructibleObjectType ret;
const auto* inner_object = j.template get_ptr<const typename BasicJsonType::object_t*>();
for (const auto& p : *inner_object)
{
ret.emplace(p.first, p.second.template get<typename ConstructibleObjectType::mapped_type>());
}
obj = std::move(ret);
}
// overload for arithmetic types, not chosen for basic_json template arguments
+7 -29
View File
@@ -5693,34 +5693,6 @@ inline void from_json(const BasicJsonType& j, CompatibleArrayType& bin)
}
}
template<typename BasicJsonType, typename ConstructibleObjectType>
auto from_json_object_impl(const BasicJsonType& j, ConstructibleObjectType& obj, priority_tag<1> /*unused*/)
-> decltype(
obj.reserve(std::declval<typename ConstructibleObjectType::size_type>()),
void())
{
ConstructibleObjectType ret;
const auto* inner_object = j.template get_ptr<const typename BasicJsonType::object_t*>();
ret.reserve(inner_object->size());
for (const auto& p : *inner_object)
{
ret.emplace(p.first, p.second.template get<typename ConstructibleObjectType::mapped_type>());
}
obj = std::move(ret);
}
template<typename BasicJsonType, typename ConstructibleObjectType>
inline void from_json_object_impl(const BasicJsonType& j, ConstructibleObjectType& obj, priority_tag<0> /*unused*/)
{
ConstructibleObjectType ret;
const auto* inner_object = j.template get_ptr<const typename BasicJsonType::object_t*>();
for (const auto& p : *inner_object)
{
ret.emplace(p.first, p.second.template get<typename ConstructibleObjectType::mapped_type>());
}
obj = std::move(ret);
}
template<typename BasicJsonType, typename ConstructibleObjectType,
enable_if_t<is_constructible_object_type<BasicJsonType, ConstructibleObjectType>::value, int> = 0>
inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj)
@@ -5730,7 +5702,13 @@ inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj)
JSON_THROW(type_error::create(302, concat("type must be object, but is ", j.type_name()), &j));
}
from_json_object_impl(j, obj, priority_tag<1> {});
ConstructibleObjectType ret;
const auto* inner_object = j.template get_ptr<const typename BasicJsonType::object_t*>();
for (const auto& p : *inner_object)
{
ret.emplace(p.first, p.second.template get<typename ConstructibleObjectType::mapped_type>());
}
obj = std::move(ret);
}
// overload for arithmetic types, not chosen for basic_json template arguments
-31
View File
@@ -1389,37 +1389,6 @@ TEST_CASE("value conversion")
// CHECK(m5["one"] == "eins");
}
SECTION("reserve is called on containers that support it (#5406)")
{
// build a larger object so that a missing/incorrect reserve()
// call would be more likely to corrupt or drop elements
json j_large;
for (int i = 0; i < 100; ++i)
{
j_large[std::to_string(i)] = i;
}
SECTION("std::unordered_map (supports reserve)")
{
const auto m = j_large.get<std::unordered_map<std::string, int>>();
CHECK(m.size() == 100);
for (int i = 0; i < 100; ++i)
{
CHECK(m.at(std::to_string(i)) == i);
}
}
SECTION("std::map (no reserve, fallback path)")
{
const auto m = j_large.get<std::map<std::string, int>>();
CHECK(m.size() == 100);
for (int i = 0; i < 100; ++i)
{
CHECK(m.at(std::to_string(i)) == i);
}
}
}
SECTION("std::multimap")
{
j1.get<std::multimap<std::string, int>>();
+2 -2
View File
@@ -304,8 +304,8 @@ TEST_CASE("Unicode (3/5)" * doctest::skip())
{
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip fourth second byte
if (0x80 <= byte3 && byte3 <= 0xBF)
// skip correct fourth byte
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
+1 -1
View File
@@ -305,7 +305,7 @@ TEST_CASE("Unicode (4/5)" * doctest::skip())
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip correct fourth byte
if (0x80 <= byte3 && byte3 <= 0xBF)
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}
+1 -1
View File
@@ -305,7 +305,7 @@ TEST_CASE("Unicode (5/5)" * doctest::skip())
for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4)
{
// skip correct fourth byte
if (0x80 <= byte3 && byte3 <= 0xBF)
if (0x80 <= byte4 && byte4 <= 0xBF)
{
continue;
}