Compare commits

..
Author SHA1 Message Date
Niels Lohmann 1e6257bf49 Add deprecated from_bon8/from_bjdata(ptr, len) overloads
from_bon8(ptr, len) and from_bjdata(ptr, len) had no overload for a
pointer and a length, unlike from_cbor/from_msgpack/from_ubjson/
from_bson. The call instead bound to from_*(InputType&&, bool strict),
which read ptr as a NUL-terminated C string via strlen and silently
converted len to the strict flag. Data containing a 0x00 byte was cut
off there; data without one was read past the end of the buffer.

Add a deprecated (ptr, len, strict, allow_exceptions) overload for
each function that forwards to (ptr, ptr + len, ...), matching the
existing deprecated overloads of the other four binary readers. Since
neither function ever had this overload, the deprecation is declared
as of version 3.13.0, the next unreleased version, rather than the
version each function was originally added in.

Fixes #5648.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 23:26:44 +02:00
8 changed files with 126 additions and 74 deletions
@@ -111,3 +111,12 @@ Linear in the size of the input.
- Added in version 3.11.0.
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
!!! warning "Deprecation"
- Overload (2) replaces calls to `from_bjdata` with a pointer and a length as first two parameters, which has been
deprecated in version 3.13.0. This overload will be removed in version 4.0.0. Please replace all calls like
`#!cpp from_bjdata(ptr, len, ...);` with `#!cpp from_bjdata(ptr, ptr+len, ...);`.
You should be warned by your compiler with a `-Wdeprecated-declarations` warning if you are using a deprecated
function.
@@ -106,3 +106,12 @@ Linear in the size of the input.
## Version history
- Added in version 3.13.0.
!!! warning "Deprecation"
- Overload (2) replaces calls to `from_bon8` with a pointer and a length as first two parameters, which has been
deprecated in version 3.13.0. This overload will be removed in version 4.0.0. Please replace all calls like
`#!cpp from_bon8(ptr, len, ...);` with `#!cpp from_bon8(ptr, ptr+len, ...);`.
You should be warned by your compiler with a `-Wdeprecated-declarations` warning if you are using a deprecated
function.
@@ -445,10 +445,8 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
}
else
{
// get the current character; converted to an unsigned type so that
// a negative unit (wint_t is signed on some platforms) is not
// mistaken for an ASCII character or for EOF
const auto wc = static_cast<std::uint32_t>(input.get_character());
// get the current character
const auto wc = input.get_character();
// UTF-32 to UTF-8 encoding
if (wc < 0x80)
@@ -543,11 +541,9 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
bool valid_pair = false;
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
{
// only consume the next unit if it completes the pair
const auto wc2 = static_cast<unsigned int>(*input.current);
const auto wc2 = static_cast<unsigned int>(input.get_character());
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{
input.get_character();
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
@@ -560,8 +556,7 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (!valid_pair)
{
// emit a byte that is never valid UTF-8 (see the UTF-32 case)
utf8_bytes[0] = 0xFF;
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
utf8_bytes_filled = 1;
}
}
+20
View File
@@ -5547,6 +5547,16 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
return result;
}
template<typename T>
JSON_HEDLEY_WARN_UNUSED_RESULT
JSON_HEDLEY_DEPRECATED_FOR(3.13.0, from_bjdata(ptr, ptr + len))
static basic_json from_bjdata(const T* ptr, std::size_t len,
const bool strict = true,
const bool allow_exceptions = true)
{
return from_bjdata(ptr, ptr + len, strict, allow_exceptions);
}
/// @brief create a JSON value from an input in BON8 format
/// @sa https://json.nlohmann.me/api/basic_json/from_bon8/
template<typename InputType>
@@ -5584,6 +5594,16 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
return result;
}
template<typename T>
JSON_HEDLEY_WARN_UNUSED_RESULT
JSON_HEDLEY_DEPRECATED_FOR(3.13.0, from_bon8(ptr, ptr + len))
static basic_json from_bon8(const T* ptr, std::size_t len,
const bool strict = true,
const bool allow_exceptions = true)
{
return from_bon8(ptr, ptr + len, strict, allow_exceptions);
}
/// @brief create a JSON value from an input in BSON format
/// @sa https://json.nlohmann.me/api/basic_json/from_bson/
template<typename InputType>
+24 -9
View File
@@ -7989,10 +7989,8 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
}
else
{
// get the current character; converted to an unsigned type so that
// a negative unit (wint_t is signed on some platforms) is not
// mistaken for an ASCII character or for EOF
const auto wc = static_cast<std::uint32_t>(input.get_character());
// get the current character
const auto wc = input.get_character();
// UTF-32 to UTF-8 encoding
if (wc < 0x80)
@@ -8087,11 +8085,9 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
bool valid_pair = false;
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
{
// only consume the next unit if it completes the pair
const auto wc2 = static_cast<unsigned int>(*input.current);
const auto wc2 = static_cast<unsigned int>(input.get_character());
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{
input.get_character();
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
@@ -8104,8 +8100,7 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (!valid_pair)
{
// emit a byte that is never valid UTF-8 (see the UTF-32 case)
utf8_bytes[0] = 0xFF;
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
utf8_bytes_filled = 1;
}
}
@@ -31633,6 +31628,16 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
return result;
}
template<typename T>
JSON_HEDLEY_WARN_UNUSED_RESULT
JSON_HEDLEY_DEPRECATED_FOR(3.13.0, from_bjdata(ptr, ptr + len))
static basic_json from_bjdata(const T* ptr, std::size_t len,
const bool strict = true,
const bool allow_exceptions = true)
{
return from_bjdata(ptr, ptr + len, strict, allow_exceptions);
}
/// @brief create a JSON value from an input in BON8 format
/// @sa https://json.nlohmann.me/api/basic_json/from_bon8/
template<typename InputType>
@@ -31670,6 +31675,16 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
return result;
}
template<typename T>
JSON_HEDLEY_WARN_UNUSED_RESULT
JSON_HEDLEY_DEPRECATED_FOR(3.13.0, from_bon8(ptr, ptr + len))
static basic_json from_bon8(const T* ptr, std::size_t len,
const bool strict = true,
const bool allow_exceptions = true)
{
return from_bon8(ptr, ptr + len, strict, allow_exceptions);
}
/// @brief create a JSON value from an input in BSON format
/// @sa https://json.nlohmann.me/api/basic_json/from_bson/
template<typename InputType>
+28
View File
@@ -4572,3 +4572,31 @@ TEST_CASE("BJData roundtrips" * doctest::skip())
}
}
}
TEST_CASE("issue #5648 - from_bjdata(ptr, len) must read len bytes, not treat ptr as a C string")
{
// to_bjdata() encodes the integer 0 as the two bytes 'i' 0x00 (a BJData
// type marker followed by the value byte 0x00), so the packed data
// below contains a 0x00 byte before its end.
const json j = {{"a", 0}};
const std::vector<std::uint8_t> packed = json::to_bjdata(j);
bool contains_nul = false;
for (const auto byte : packed)
{
contains_nul |= (byte == 0x00);
}
REQUIRE(contains_nul);
// before the fix, from_bjdata had no (ptr, len) overload, so this call
// bound to from_bjdata(InputType&&, bool strict) instead: ptr was read
// as a NUL-terminated C string (stopping at the embedded 0x00 byte), and
// len was silently converted to the strict flag. The deprecated
// overload added for this issue forwards to from_bjdata(ptr, ptr + len,
// ...) instead, like from_ubjson's deprecated (ptr, len) overload does.
json result;
CHECK_NOTHROW(result = json::from_bjdata(packed.data(), packed.size()));
CHECK(result == j);
// len must not collapse into the strict flag either
CHECK(json::from_bjdata(packed.data(), packed.size(), false) == j);
}
+28
View File
@@ -1010,6 +1010,34 @@ TEST_CASE("BON8 roundtrips" * doctest::skip())
}
}
TEST_CASE("issue #5648 - from_bon8(ptr, len) must read len bytes, not treat ptr as a C string")
{
// to_bon8() encodes the integer 40 as the two bytes 0xC2 0x00 (a BON8
// lead byte followed by a continuation byte of 0x00), so the packed data
// below contains a 0x00 byte before its end.
const json j = {{"a", 40}, {"b", "x"}};
const std::vector<std::uint8_t> packed = json::to_bon8(j);
bool contains_nul = false;
for (const auto byte : packed)
{
contains_nul |= (byte == 0x00);
}
REQUIRE(contains_nul);
// before the fix, from_bon8 had no (ptr, len) overload, so this call
// bound to from_bon8(InputType&&, bool strict) instead: ptr was read as
// a NUL-terminated C string (stopping at the embedded 0x00 byte), and
// len was silently converted to the strict flag. The deprecated
// overload added for this issue forwards to from_bon8(ptr, ptr + len,
// ...) instead, like from_cbor's deprecated (ptr, len) overload does.
json result;
CHECK_NOTHROW(result = json::from_bon8(packed.data(), packed.size()));
CHECK(result == j);
// len must not collapse into the strict flag either
CHECK(json::from_bon8(packed.data(), packed.size(), false) == j);
}
#ifdef JSON_HAS_CPP_17
TEST_CASE("BON8 with std::byte")
{
+4 -56
View File
@@ -8,7 +8,6 @@
#include "doctest_compatibility.h"
#include <cwchar>
#include <nlohmann/json.hpp>
using nlohmann::json;
@@ -99,15 +98,15 @@ TEST_CASE("wide strings")
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
// a lone low surrogate cannot start a pair
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a high surrogate followed by a non-low-surrogate unit is invalid
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// ... also when the unit is above the low surrogates
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a lone low surrogate must not swallow the following unit: pairing
// it with any second unit would produce valid UTF-8, so the error
// has to report an ill-formed byte at the surrogate's own position
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
// a valid surrogate pair is still decoded (U+1F600)
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
}
@@ -142,56 +141,5 @@ TEST_CASE("wide strings")
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
}
}
SECTION("malformed wide-string input outside strings (#5645)")
{
json _;
// a lone low surrogate inside a literal must not be truncated to its
// low byte and mistaken for the letter the literal expects next
// (0xDC72 truncates to 'r', which is what "true" expects after 't')
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast<char16_t>(0xDC72), u'u', u'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
// ... also when the lone surrogate is the last unit of the input
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'f', u'a', u'l', u's', static_cast<char16_t>(0xDD65)}),
"[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'fals\xFF'", json::parse_error&);
// a high surrogate followed by a unit that is not its low surrogate
// must not silently swallow that unit
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast<char16_t>(0xD872), u'X', u'u', u'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
// ... in particular, if the swallowed unit is the newline that ends a
// // comment, the comment must not extend over the following line
CHECK(json::parse(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast<char16_t>(0xD800), u'\n',
u',', u'2', u' ', u'/', u'/', u'\n', u']'},
nullptr, true, /*ignore_comments*/true) == json::parse("[1,2]"));
CHECK(json::accept(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast<char16_t>(0xD800), u'\n',
u',', u'2', u' ', u'/', u'/', u'\n', u']'}, /*ignore_comments*/true));
// cases 5 and 6 use a 32-bit wchar_t (Linux, macOS, the BSDs) to reach
// the UTF-32 helper tested above via u32string; the 16-bit wchar_t of
// Windows goes through the UTF-16 helper instead, already covered by
// the u16string cases above
#if WCHAR_MAX > 0xFFFFu
// a negative wchar_t must not be mistaken for
// char_traits<char>::eof() and silently end the input, letting
// trailing garbage pass the strict end-of-input check (only observable
// where wint_t is signed, e.g. macOS/the BSDs; on Linux wint_t is
// unsigned and this was already handled by #5348)
std::wstring w = L"[1]";
w.push_back(static_cast<wchar_t>(-1));
w += L"garbage";
CHECK(!json::accept(w));
CHECK_THROWS_WITH_AS(_ = json::parse(w),
"[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
// other negative wchar_t units must not be truncated to their low
// byte (0xFFFFFF72 truncates to 'r', as in the u16string case above)
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L't', static_cast<wchar_t>(0xFFFFFF72), L'u', L'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
#endif
}
}
#endif