Handle numbers that do not fit narrow number types in the binary readers

With custom number types narrower than the values in a binary document,
for example basic_json<..., std::int32_t, std::uint32_t, float>, every
binary reader (CBOR, MessagePack, UBJSON, BJData, BSON, BON8) passed the
decoded number to the SAX interface with an implicit conversion: the
integer 5000000000 silently became 705032704, and a finite double such as
1e300 became infinity. The lexer handles the same values in JSON text: an
integer that fits neither integer type is stored as number_float_t, and a
finite number that overflows number_float_t is rejected with
out_of_range.406.

Pass every number read from binary input through three helpers that
apply the lexer's rules:
- emit_signed(): number_integer_t, else number_unsigned_t for a
  non-negative value, else number_float_t
- emit_unsigned(): number_unsigned_t, else number_float_t
- emit_float(): out_of_range.406 if a finite value overflows
  number_float_t; infinity and NaN are passed on

For consistency, a CBOR negative integer below the range of
number_integer_t is now stored as number_float_t, like a too small
integer in JSON text, instead of being rejected with parse_error.112.
With the default number types, this is the only change in behavior.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-27 23:23:16 +02:00
parent 373005f7ac
commit 04ed4f593c
9 changed files with 396 additions and 116 deletions
+116
View File
@@ -11,7 +11,12 @@
#include <nlohmann/json.hpp>
using nlohmann::json;
#include <cmath>
#include <fstream>
#include <limits>
#include <map>
#include <string>
#include <vector>
#include "make_test_data_available.hpp"
TEST_CASE("Binary Formats" * doctest::skip())
@@ -224,3 +229,114 @@ TEST_CASE("Binary Formats" * doctest::skip())
CHECK((100.0 * double(ubjson_3_size) / double(json_size)) == Approx(89.450));
}
}
TEST_CASE("Binary formats with narrow number types")
{
// Numbers that do not fit the number types are handled like the lexer
// handles them in JSON text: an integer that fits neither integer type is
// stored as a floating-point number, and a finite floating-point number
// that overflows number_float_t is rejected with out_of_range.406.
using narrow_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int32_t, std::uint32_t, float>;
using bytes = std::vector<std::uint8_t>;
struct binary_format
{
const char* name;
bytes (*encode)(const json&);
narrow_json (*decode)(const bytes&, bool);
};
const std::vector<binary_format> formats =
{
{
"CBOR", [](const json & j) { return json::to_cbor(j); },
[](const bytes & v, bool allow_exceptions)
{
return narrow_json::from_cbor(v, true, allow_exceptions);
}
},
{
"MessagePack", [](const json & j) { return json::to_msgpack(j); },
[](const bytes & v, bool allow_exceptions)
{
return narrow_json::from_msgpack(v, true, allow_exceptions);
}
},
{
"UBJSON", [](const json & j) { return json::to_ubjson(j); },
[](const bytes & v, bool allow_exceptions)
{
return narrow_json::from_ubjson(v, true, allow_exceptions);
}
},
{
"BJData", [](const json & j) { return json::to_bjdata(j); },
[](const bytes & v, bool allow_exceptions)
{
return narrow_json::from_bjdata(v, true, allow_exceptions);
}
},
{
// BSON can only store numbers as object members
"BSON", [](const json & j) { return json::to_bson(json{{"a", j}}); },
[](const bytes & v, bool allow_exceptions)
{
const auto result = narrow_json::from_bson(v, true, allow_exceptions);
return result.is_discarded() ? result : result.at("a");
}
},
{
"BON8", [](const json & j) { return json::to_bon8(j); },
[](const bytes & v, bool allow_exceptions)
{
return narrow_json::from_bon8(v, true, allow_exceptions);
}
},
};
for (const auto& format : formats)
{
const std::string name = format.name;
INFO("format := ", name);
const auto roundtrip = [&format](const json & j)
{
return format.decode(format.encode(j), true);
};
// integers that fit keep their type
CHECK(roundtrip(json(-5)).is_number_integer());
CHECK(roundtrip(json(-5)).get<std::int32_t>() == -5);
CHECK(roundtrip(json(3000000000u)).is_number_unsigned());
CHECK(roundtrip(json(3000000000u)).get<std::uint32_t>() == 3000000000u);
// integers that fit neither integer type are stored as float
CHECK(roundtrip(json(5000000000u)).is_number_float());
CHECK(roundtrip(json(5000000000u)).get<float>() == 5000000000.0f);
if (name != "BON8") // BON8 cannot encode integers above INT64_MAX
{
CHECK(roundtrip(json(10000000000000000000u)).is_number_float());
CHECK(roundtrip(json(10000000000000000000u)).get<float>() == 10000000000000000000.0f);
}
CHECK(roundtrip(json(-3000000000)).is_number_float());
CHECK(roundtrip(json(-3000000000)).get<float>() == -3000000000.0f);
CHECK(roundtrip(json(-5000000000)).is_number_float());
CHECK(roundtrip(json(-5000000000)).get<float>() == -5000000000.0f);
// floating-point numbers that fit
CHECK(roundtrip(json(1.5)).get<float>() == 1.5f);
const auto just_above_max = std::nextafter(static_cast<double>((std::numeric_limits<float>::max)()),
std::numeric_limits<double>::infinity());
CHECK(roundtrip(json(just_above_max)).get<float>() == (std::numeric_limits<float>::max)());
// infinity and NaN are passed on
CHECK(std::isinf(roundtrip(json(std::numeric_limits<double>::infinity())).get<float>()));
CHECK(std::isnan(roundtrip(json(std::numeric_limits<double>::quiet_NaN())).get<float>()));
// finite floating-point numbers that overflow number_float_t are rejected
const std::string message = "[json.exception.out_of_range.406] syntax error while parsing " + name
+ " value: number overflow";
CHECK_THROWS_WITH_AS(roundtrip(json(1e300)), message.c_str(), narrow_json::out_of_range&);
CHECK_THROWS_WITH_AS(roundtrip(json(-1e300)), message.c_str(), narrow_json::out_of_range&);
CHECK(format.decode(format.encode(json(1e300)), false).is_discarded());
}
}
+16 -14
View File
@@ -3146,7 +3146,8 @@ TEST_CASE("Tagged values")
// CBOR encodes negative integers as: result = -1 - n
// For type 0x3B, n is an 8-byte uint64_t. Valid range for n with
// the default int64_t is [0, INT64_MAX], producing results in [INT64_MIN, -1].
// When n > INT64_MAX, the result exceeds int64_t range and is rejected.
// When n > INT64_MAX, the result exceeds int64_t range and is stored
// as a floating-point number, as the lexer does for JSON text.
SECTION("n = 0 is valid (result = -1)")
{
@@ -3167,33 +3168,34 @@ TEST_CASE("Tagged values")
CHECK(result.get<int64_t>() == (std::numeric_limits<int64_t>::min)());
}
SECTION("n = INT64_MAX + 1 is rejected (overflow)")
SECTION("n = INT64_MAX + 1 is stored as float")
{
// n = INT64_MAX + 1 (0x8000000000000000)
// result = -1 - n = -9223372036854775809, which exceeds int64_t range
// result = -1 - n = -9223372036854775809, which exceeds int64_t range;
// the nearest double is -9223372036854775808.0
const std::vector<uint8_t> input = {0x3B, 0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00};
json _;
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input),
"[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow",
json::parse_error);
const auto result = json::from_cbor(input);
CHECK(result.is_number_float());
CHECK(result.get<double>() == -9223372036854775808.0);
CHECK(result == json::parse("-9223372036854775809"));
}
SECTION("n = UINT64_MAX is rejected (overflow)")
SECTION("n = UINT64_MAX is stored as float")
{
// n = UINT64_MAX (0xFFFFFFFFFFFFFFFF)
// result = -1 - n = -18446744073709551616, which exceeds int64_t range
const std::vector<uint8_t> input = {0x3B, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF};
json _;
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input),
"[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow",
json::parse_error);
const auto result = json::from_cbor(input);
CHECK(result.is_number_float());
CHECK(result.get<double>() == -18446744073709551616.0);
CHECK(result == json::parse("-18446744073709551616"));
}
SECTION("overflow with allow_exceptions=false returns discarded")
SECTION("overflow with allow_exceptions=false is not an error")
{
const std::vector<uint8_t> input = {0x3B, 0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00};
const auto result = json::from_cbor(input, true, false);
CHECK(result.is_discarded());
CHECK(result.is_number_float());
}
}