Compare commits

..
Author SHA1 Message Date
Niels Lohmann 2add64396a Keep a NUL byte ending a // comment as the end of input
With the default NUL handling (JSON_STRICT_NUL_HANDLING not set), a NUL
byte in the input is treated as the real end of input everywhere -
except when it immediately ends a `//` comment: scan_comment() matched
'\0' as a comment terminator like '\n', so the NUL was consumed as
part of the comment and scan() never saw it as end of input; the next
get() then kept reading past it. Multi-line comments and
JSON_STRICT_NUL_HANDLING=1 were unaffected, since there the NUL is
just part of the comment text.

Fix scan_comment() to leave the NUL unconsumed (unget()) instead of
returning it as part of the comment, so the following scan() reports
it as end of input, exactly as for a NUL anywhere else.

Fixes #5659.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 23:38:59 +02:00
7 changed files with 103 additions and 164 deletions
@@ -132,14 +132,8 @@ The library uses the following mapping from JSON values types to BJData types ac
parsed back as a regular array,
- every entry of `"_ArraySize_"` is a positive integer, and their product is representable as a `std::size_t`,
- `"_ArrayData_"` is an array holding exactly that many elements, and
- every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"`: for the integer types, a
value that fits the named width; for `double`, any value; for `single`, a value that survives narrowing to
`float` and back without change (for instance, `0.1` does not, since it is not exactly representable as
`float`).
An annotated object is always read back with its keys in the order shown above, `"_ArrayType_"`, `"_ArraySize_"`,
`"_ArrayData_"`, regardless of the order the ND-array's header stores them in on the wire. This matters for
`ordered_json`, whose comparison takes key order into account.
- every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"` (a floating-point number for
`single` and `double`, an integer otherwise).
The current version of this library does not yet support automatic detection of and conversion from a nested JSON
array input to a BJData ND-array.
+21 -42
View File
@@ -2728,15 +2728,10 @@ class binary_reader
is_ndarray can only return `true` when its initial value
is `false`
@param[in] prefix type marker if already read, otherwise set to 0
@param[in] ndarray_dtype the element type marker of the enclosing bjdata ndarray if
already known (it precedes the dimension vector read here),
otherwise 0; used to emit the "_ArrayType_" annotation key
before "_ArraySize_" if a dimension vector turns out to
describe an ndarray
@return whether size determination completed
*/
bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0, char_int_type ndarray_dtype = 0)
bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0)
{
if (prefix == 0)
{
@@ -2906,37 +2901,8 @@ class binary_reader
}
}
if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3)))
{
return false;
}
// the element type precedes the dimension vector (see get_ubjson_size_type)
// and is passed down as ndarray_dtype; emit it here so the annotation keys
// follow the documented _ArrayType_, _ArraySize_, _ArrayData_ order
if (ndarray_dtype != 0)
{
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), ndarray_dtype, [](const bjd_type & p, char_int_type t)
{
return p.first < t;
});
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != ndarray_dtype))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
}
string_t type_key = "_ArrayType_";
string_t type = it->second; // sax->string() takes a reference
if (JSON_HEDLEY_UNLIKELY(!sax->key(type_key) || !sax->string(type)))
{
return false;
}
}
string_t key = "_ArraySize_";
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(dim.size())))
if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3) || !sax->key(key) || !sax->start_array(dim.size())))
{
return false;
}
@@ -3037,7 +3003,7 @@ class binary_reader
exception_message(input_format, concat("expected '#' after type information; last byte: 0x", last_token), "size"), nullptr));
}
const bool is_error = get_ubjson_size_value(result.first, is_ndarray, 0, result.second);
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
// an ndarray was read here only if the flag flipped; when it was
// seeded true, get_ubjson_size_value() already rejected the nested
// dimension vector
@@ -3273,17 +3239,30 @@ class binary_reader
if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0)
{
size_and_type.second &= ~(static_cast<char_int_type>(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t)
{
return p.first < t;
});
string_t key = "_ArrayType_";
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
}
string_t type = it->second; // sax->string() takes a reference
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type)))
{
return false;
}
// the "_ArrayType_" and "_ArraySize_" annotation keys were already emitted by
// get_ubjson_size_value() (the type marker is known before the dimension vector
// that determines size_and_type.first is read, so it is emitted first there to
// match the documented _ArrayType_, _ArraySize_, _ArrayData_ key order)
if (size_and_type.second == 'C' || size_and_type.second == 'B')
{
size_and_type.second = 'U';
}
string_t key = "_ArrayData_";
key = "_ArrayData_";
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(size_and_type.first) ))
{
return false;
+6 -1
View File
@@ -975,10 +975,15 @@ class lexer : public lexer_base<BasicJsonType>
case '\n':
case '\r':
case char_traits<char_type>::eof():
return true;
#if !JSON_STRICT_NUL_HANDLING
case '\0':
#endif
// a NUL byte is the end of the input (see scan()),
// so leave it for scan() to see
unget();
return true;
#endif
default:
break;
@@ -1991,21 +1991,9 @@ class binary_writer
case 'd':
{
const auto dval = el.template get<double>();
#ifdef __GNUC__
JSON_HEDLEY_DIAGNOSTIC_PUSH
JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal")
#endif
// a value that would be rounded (rather than exactly represented) by the
// narrowing to float is treated like an out-of-range integer element above;
// this is the same criterion write_compact_float() uses for CBOR/MessagePack
in_range = std::isnan(dval) ||
in_range = !std::isfinite(dval) ||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
dval <= static_cast<double>((std::numeric_limits<float>::max)()) &&
static_cast<double>(static_cast<float>(dval)) == dval) ||
std::isinf(dval);
#ifdef __GNUC__
JSON_HEDLEY_DIAGNOSTIC_POP
#endif
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
break;
}
default:
+29 -57
View File
@@ -10067,10 +10067,15 @@ class lexer : public lexer_base<BasicJsonType>
case '\n':
case '\r':
case char_traits<char_type>::eof():
return true;
#if !JSON_STRICT_NUL_HANDLING
case '\0':
#endif
// a NUL byte is the end of the input (see scan()),
// so leave it for scan() to see
unget();
return true;
#endif
default:
break;
@@ -15495,15 +15500,10 @@ class binary_reader
is_ndarray can only return `true` when its initial value
is `false`
@param[in] prefix type marker if already read, otherwise set to 0
@param[in] ndarray_dtype the element type marker of the enclosing bjdata ndarray if
already known (it precedes the dimension vector read here),
otherwise 0; used to emit the "_ArrayType_" annotation key
before "_ArraySize_" if a dimension vector turns out to
describe an ndarray
@return whether size determination completed
*/
bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0, char_int_type ndarray_dtype = 0)
bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0)
{
if (prefix == 0)
{
@@ -15673,37 +15673,8 @@ class binary_reader
}
}
if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3)))
{
return false;
}
// the element type precedes the dimension vector (see get_ubjson_size_type)
// and is passed down as ndarray_dtype; emit it here so the annotation keys
// follow the documented _ArrayType_, _ArraySize_, _ArrayData_ order
if (ndarray_dtype != 0)
{
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), ndarray_dtype, [](const bjd_type & p, char_int_type t)
{
return p.first < t;
});
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != ndarray_dtype))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
}
string_t type_key = "_ArrayType_";
string_t type = it->second; // sax->string() takes a reference
if (JSON_HEDLEY_UNLIKELY(!sax->key(type_key) || !sax->string(type)))
{
return false;
}
}
string_t key = "_ArraySize_";
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(dim.size())))
if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3) || !sax->key(key) || !sax->start_array(dim.size())))
{
return false;
}
@@ -15804,7 +15775,7 @@ class binary_reader
exception_message(input_format, concat("expected '#' after type information; last byte: 0x", last_token), "size"), nullptr));
}
const bool is_error = get_ubjson_size_value(result.first, is_ndarray, 0, result.second);
const bool is_error = get_ubjson_size_value(result.first, is_ndarray);
// an ndarray was read here only if the flag flipped; when it was
// seeded true, get_ubjson_size_value() already rejected the nested
// dimension vector
@@ -16040,17 +16011,30 @@ class binary_reader
if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0)
{
size_and_type.second &= ~(static_cast<char_int_type>(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker
auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t)
{
return p.first < t;
});
string_t key = "_ArrayType_";
if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second))
{
auto last_token = get_token_string();
return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read,
exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr));
}
string_t type = it->second; // sax->string() takes a reference
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type)))
{
return false;
}
// the "_ArrayType_" and "_ArraySize_" annotation keys were already emitted by
// get_ubjson_size_value() (the type marker is known before the dimension vector
// that determines size_and_type.first is read, so it is emitted first there to
// match the documented _ArrayType_, _ArraySize_, _ArrayData_ key order)
if (size_and_type.second == 'C' || size_and_type.second == 'B')
{
size_and_type.second = 'U';
}
string_t key = "_ArrayData_";
key = "_ArrayData_";
if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(size_and_type.first) ))
{
return false;
@@ -22342,21 +22326,9 @@ class binary_writer
case 'd':
{
const auto dval = el.template get<double>();
#ifdef __GNUC__
JSON_HEDLEY_DIAGNOSTIC_PUSH
JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal")
#endif
// a value that would be rounded (rather than exactly represented) by the
// narrowing to float is treated like an out-of-range integer element above;
// this is the same criterion write_compact_float() uses for CBOR/MessagePack
in_range = std::isnan(dval) ||
in_range = !std::isfinite(dval) ||
(dval >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
dval <= static_cast<double>((std::numeric_limits<float>::max)()) &&
static_cast<double>(static_cast<float>(dval)) == dval) ||
std::isinf(dval);
#ifdef __GNUC__
JSON_HEDLEY_DIAGNOSTIC_POP
#endif
dval <= static_cast<double>((std::numeric_limits<float>::max)()));
break;
}
default:
+4 -42
View File
@@ -11,7 +11,6 @@
#define JSON_TESTS_PRIVATE
#include <nlohmann/json.hpp>
using nlohmann::json;
using ordered_json = nlohmann::ordered_json;
#include <algorithm>
#include <climits>
@@ -2295,33 +2294,29 @@ TEST_CASE("BJData")
SECTION("start_array() in ndarray _ArraySize_")
{
// _ArrayType_ (2 events: key + string) is now emitted before
// _ArraySize_ (see GitHub issue #5661), which shifts the events
// below later by the same 2 events
std::vector<uint8_t> const v = {'[', '$', 'i', '#', '[', '$', 'i', '#', 'i', 2, 2, 1, 1, 2};
SaxCountdown scp(4);
SaxCountdown scp(2);
CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata));
}
SECTION("number_integer() in ndarray _ArraySize_")
{
std::vector<uint8_t> const v = {'[', '$', 'U', '#', '[', '$', 'i', '#', 'i', 2, 2, 1, 1, 2};
SaxCountdown scp(5);
SaxCountdown scp(3);
CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata));
}
SECTION("key() in ndarray _ArrayType_")
{
// _ArrayType_ is emitted right after start_object(), before _ArraySize_
std::vector<uint8_t> const v = {'[', '$', 'U', '#', '[', '$', 'U', '#', 'i', 2, 2, 2, 1, 2, 3, 4};
SaxCountdown scp(1);
SaxCountdown scp(6);
CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata));
}
SECTION("string() in ndarray _ArrayType_")
{
std::vector<uint8_t> const v = {'[', '$', 'U', '#', '[', '$', 'U', '#', 'i', 2, 2, 2, 1, 2, 3, 4};
SaxCountdown scp(2);
SaxCountdown scp(7);
CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata));
}
@@ -2924,22 +2919,6 @@ TEST_CASE("BJData")
CHECK(out_single.at(0) == '{');
CHECK(json::from_bjdata(out_single) == j_single);
// a double element that is finite and within the range of "single"
// but is not exactly representable as a float, so narrowing it would
// silently round it (0.1 is read back as 0.10000000149011612); this,
// like the overflow case above, falls back to a plain object (see
// GitHub issue #5661)
json const j_single_rounded = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 0.1}}});
const auto out_single_rounded = json::to_bjdata(j_single_rounded);
CHECK(out_single_rounded.at(0) == '{');
CHECK(json::from_bjdata(out_single_rounded) == j_single_rounded);
// a double element that underflows to 0 when narrowed to "single"
json const j_single_underflow = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 1e-300}}});
const auto out_single_underflow = json::to_bjdata(j_single_underflow);
CHECK(out_single_underflow.at(0) == '{');
CHECK(json::from_bjdata(out_single_underflow) == j_single_underflow);
// in-range boundary values still use the compact ndarray encoding
json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {0, 255}}});
CHECK(json::to_bjdata(j_uint8_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, 'i', 1, ']', 0, 255}));
@@ -2953,23 +2932,6 @@ TEST_CASE("BJData")
CHECK(json::from_bjdata(out_single_ok) == json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5f, -1.5f}}}));
}
SECTION("ndarray annotation keys are read back in the documented order")
{
// from_bjdata() must emit the annotation object's keys in the order
// used throughout the documentation, _ArrayType_, _ArraySize_,
// _ArrayData_: the type marker precedes the dimension vector on the
// wire (see get_ubjson_size_type()), so it is known, and emitted,
// before _ArraySize_. For a plain json this key order is invisible
// (its comparison ignores it), but for an ordered_json it is not (see
// GitHub issue #5661).
const ordered_json o = ordered_json::parse(R"({"_ArrayType_":"uint8","_ArraySize_":[2,2],"_ArrayData_":[1,2,3,4]})");
const auto packed = ordered_json::to_bjdata(o);
CHECK(packed.at(0) == '[');
const ordered_json o_back = ordered_json::from_bjdata(packed);
CHECK(o_back == o);
CHECK(o_back.dump() == o.dump());
}
SECTION("ndarray that would not be read back as an annotated object stays as object")
{
// the reader only restores an annotated object from an ND-array
+39
View File
@@ -592,6 +592,45 @@ TEST_CASE("parser class")
// parsing from a string literal is unaffected either way
CHECK(json::parse("123") == json(123));
// a NUL byte that ends a // comment ends the input just
// like a NUL byte anywhere else (issue #5659); before the
// fix, the NUL was consumed as part of the comment, and
// scanning continued with whatever followed it
{
// same as "//c" alone (real end of input after the
// comment), rather than continuing with "[1]"
std::string s1 = "//c";
s1.push_back('\0');
s1 += "[1]";
json _; // NOLINT(readability-identifier-naming)
CHECK_THROWS_WITH_AS(_ = json::parse(s1, nullptr, true, true),
"[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - unexpected end of input; expected '[', '{', or a literal",
json::parse_error&);
CHECK_FALSE(json::accept(s1, true, true));
}
{
// same as "[1, //c" alone, rather than continuing with " 2]"
std::string s2 = "[1, //c";
s2.push_back('\0');
s2 += " 2]";
json _; // NOLINT(readability-identifier-naming)
CHECK_THROWS_WITH_AS(_ = json::parse(s2, nullptr, true, true),
"[json.exception.parse_error.101] parse error at line 1, column 8: syntax error while parsing value - unexpected end of input; expected '[', '{', or a literal",
json::parse_error&);
CHECK_FALSE(json::accept(s2, true, true));
}
{
// same as "1 //c" alone: the comment (and the NUL that
// ends it) is ignored, and "x" is never reached
std::string s3 = "1 //c";
s3.push_back('\0');
s3 += "x";
CHECK(json::parse(s3, nullptr, true, true) == json(1));
CHECK(json::accept(s3, true, true));
}
}
#endif