Fix silent wrong results for custom number types

Two bugs for custom number types that compile on develop (found while
analysing #3578):

- compare_integer_with_float() took the signedness of the integer type
  from std::is_signed, which is false for class types such as
  absl::int128 or boost::multiprecision::cpp_int. Any float below zero
  then compared less than every integer, e.g. json(int128(-5)) <
  json(-2.5) was false. The signedness now comes from
  std::numeric_limits, like the digits used for the range bound.

- With number_integer_t/number_unsigned_t wider than 64 bits (e.g.
  __int128), CBOR, MessagePack, and BSON silently truncated integers
  beyond 64 bits (to_cbor of 2^100 read back as 0), BJData truncated
  unsigned ones with the 'M' marker and could encode truncated ND-array
  elements and dimensions, and BON8 did not compile (std::to_string is
  ambiguous for __int128). The writers now throw out_of_range.407 when
  an integer does not fit the format's range ([-2^64, 2^64-1] for CBOR,
  [-2^63, 2^64-1] for MessagePack, int64/uint64 for BSON, int64 for
  BON8); BJData writes such unsigned values as high-precision numbers,
  as it already did for signed ones and UBJSON does for both, and falls
  back to a plain object for the ND-array. The checks use
  std::numeric_limits digits and compile away when the number types are
  at most 64 bits wide, so the default types pay nothing (the CBOR,
  MessagePack, and BSON writers compile to identical code).

The new unit-custom-number-types.cpp tests the comparison with a small
class-type integer and the writers with __int128 where the standard
library supports it as an integral type.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann committed 2026-10-10 16:55:26 +02:00
1 parent f8b47ff6f6
commit 4001fe68b9
10 files changed
+680 -47

No files matched your search

+135 -15
View File
@@ -207,6 +207,10 @@ class binary_writer
{
if (j.m_data.m_value.number_integer >= 0)
{
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::uint64_t>(j.m_data.m_value.number_integer)))
{
throw_integer_out_of_range(j, "CBOR", "[-2^64, 2^64-1]");
}
// CBOR does not differentiate between positive signed
// integers and unsigned integers
write_cbor_head(0x00, static_cast<std::uint64_t>(j.m_data.m_value.number_integer));
@@ -214,6 +218,10 @@ class binary_writer
else
{
// a negative integer n is encoded as -1 - n
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::uint64_t>(-1 - j.m_data.m_value.number_integer)))
{
throw_integer_out_of_range(j, "CBOR", "[-2^64, 2^64-1]");
}
write_cbor_head(0x20, static_cast<std::uint64_t>(-1 - j.m_data.m_value.number_integer));
}
break;
@@ -221,7 +229,11 @@ class binary_writer
case value_t::number_unsigned:
{
write_cbor_head(0x00, j.m_data.m_value.number_unsigned);
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::uint64_t>(j.m_data.m_value.number_unsigned)))
{
throw_integer_out_of_range(j, "CBOR", "uint64");
}
write_cbor_head(0x00, static_cast<std::uint64_t>(j.m_data.m_value.number_unsigned));
break;
}
@@ -429,12 +441,20 @@ class binary_writer
{
if (j.m_data.m_value.number_integer >= 0)
{
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::uint64_t>(j.m_data.m_value.number_integer)))
{
throw_integer_out_of_range(j, "MessagePack", "[-2^63, 2^64-1]");
}
// MessagePack does not differentiate between positive
// signed integers and unsigned integers.
write_msgpack_unsigned(static_cast<std::uint64_t>(j.m_data.m_value.number_integer));
}
else
{
if (JSON_HEDLEY_UNLIKELY(!integer_fits_min<std::int64_t>(j.m_data.m_value.number_integer)))
{
throw_integer_out_of_range(j, "MessagePack", "[-2^63, 2^64-1]");
}
if (j.m_data.m_value.number_integer >= -32)
{
// negative fixnum
@@ -473,6 +493,10 @@ class binary_writer
case value_t::number_unsigned:
{
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::uint64_t>(j.m_data.m_value.number_unsigned)))
{
throw_integer_out_of_range(j, "MessagePack", "uint64");
}
write_msgpack_unsigned(static_cast<std::uint64_t>(j.m_data.m_value.number_unsigned));
break;
}
@@ -834,6 +858,84 @@ class binary_writer
JSON_THROW(type_error::create(321, concat("cannot serialize discarded value to ", format_name), &j));
}
/*!
@brief whether integer @a n does not exceed the maximum of @a TargetType
The binary formats store integers in at most 64 bits, but number_integer_t
and number_unsigned_t may be wider (e.g., __int128). The range of the
integer is taken from std::numeric_limits, so a number type whose range
fits into @a TargetType is never checked: the function is then constant
true, and the default int64_t/uint64_t types pay nothing.
*/
template<typename TargetType, typename NumberType>
static constexpr bool integer_fits_max(const NumberType n) noexcept
{
return integer_fits_max<TargetType>(n, std::integral_constant < bool,
(std::numeric_limits<NumberType>::digits > std::numeric_limits<TargetType>::digits) > ());
}
template<typename TargetType, typename NumberType>
static constexpr bool integer_fits_max(const NumberType /*unused*/, std::false_type /*may_exceed*/) noexcept
{
return true;
}
template<typename TargetType, typename NumberType>
static constexpr bool integer_fits_max(const NumberType n, std::true_type /*may_exceed*/) noexcept
{
// NumberType has more digits than TargetType, so it can hold its maximum
return n <= static_cast<NumberType>((std::numeric_limits<TargetType>::max)());
}
/*!
@brief whether integer @a n is not below the minimum of @a TargetType;
constant true if NumberType cannot hold such a value
*/
template<typename TargetType, typename NumberType>
static constexpr bool integer_fits_min(const NumberType n) noexcept
{
return integer_fits_min<TargetType>(n, std::integral_constant < bool, std::numeric_limits<NumberType>::is_signed &&
(!std::numeric_limits<TargetType>::is_signed || std::numeric_limits<NumberType>::digits > std::numeric_limits<TargetType>::digits) > ());
}
template<typename TargetType, typename NumberType>
static constexpr bool integer_fits_min(const NumberType /*unused*/, std::false_type /*may_fall_below*/) noexcept
{
return true;
}
template<typename TargetType, typename NumberType>
static constexpr bool integer_fits_min(const NumberType n, std::true_type /*may_fall_below*/) noexcept
{
// NumberType is signed and either TargetType is unsigned (minimum 0)
// or NumberType has more digits, so it can hold the minimum
return n >= static_cast<NumberType>((std::numeric_limits<TargetType>::min)());
}
/*!
@brief whether integer @a n lies in the range of @a TargetType
*/
template<typename TargetType, typename NumberType>
static constexpr bool integer_fits(const NumberType n) noexcept
{
return integer_fits_min<TargetType>(n) && integer_fits_max<TargetType>(n);
}
/*!
@brief throws because the integer @a j is too large for @a format_name
@throw out_of_range.407 always
*/
JSON_HEDLEY_NO_RETURN static void throw_integer_out_of_range(const BasicJsonType& j, const char* format_name, const char* range)
{
// dump() rather than std::to_string(), which has no overload for
// integer types wider than 64 bits
const auto number = j.dump();
static_cast<void>(number); // unused when JSON_NOEXCEPTION is defined
static_cast<void>(format_name);
static_cast<void>(range);
JSON_THROW(out_of_range::create(407, concat("integer number ", std::string(number.begin(), number.end()), " cannot be represented by ", format_name, " as it does not fit ", range), &j));
}
void write_msgpack_array_prefix(const std::size_t N, const BasicJsonType& j)
{
const auto n = to_msgpack_length(N, j);
@@ -1590,6 +1692,8 @@ class binary_writer
is neither an object nor an array
@throw out_of_range.415 if @a j is binary with a subtype that does not fit
into a byte, before anything is written
@throw out_of_range.407 if @a j is an integer that does not fit the 64 bits
of a BSON integer, before anything is written
@throw type_error.316 if @a j is a string that is not valid UTF-8, before
anything is written
@throw type_error.321 if @a j is discarded
@@ -1608,10 +1712,18 @@ class binary_writer
return 8ul;
case value_t::number_integer:
return calc_bson_integer_size(j.m_data.m_value.number_integer);
if (JSON_HEDLEY_UNLIKELY(!integer_fits<std::int64_t>(j.m_data.m_value.number_integer)))
{
throw_integer_out_of_range(j, "BSON", "int64");
}
return calc_bson_integer_size(static_cast<std::int64_t>(j.m_data.m_value.number_integer));
case value_t::number_unsigned:
return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned);
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::uint64_t>(j.m_data.m_value.number_unsigned)))
{
throw_integer_out_of_range(j, "BSON", "uint64");
}
return calc_bson_unsigned_size(static_cast<std::uint64_t>(j.m_data.m_value.number_unsigned));
case value_t::string:
return calc_bson_string_size(*j.m_data.m_value.string, j);
@@ -1649,11 +1761,12 @@ class binary_writer
case value_t::number_float:
return write_bson_double(name, j.m_data.m_value.number_float);
// calc_bson_value_size() checked that integers fit 64 bits
case value_t::number_integer:
return write_bson_integer(name, j.m_data.m_value.number_integer);
return write_bson_integer(name, static_cast<std::int64_t>(j.m_data.m_value.number_integer));
case value_t::number_unsigned:
return write_bson_unsigned(name, j.m_data.m_value.number_unsigned);
return write_bson_unsigned(name, static_cast<std::uint64_t>(j.m_data.m_value.number_unsigned));
case value_t::string:
return write_bson_string(name, *j.m_data.m_value.string);
@@ -2108,7 +2221,7 @@ class binary_writer
{
return 'L';
}
if (use_bjdata && std::is_unsigned<NumberType>::value)
if (use_bjdata && std::is_unsigned<NumberType>::value && integer_fits_max<std::uint64_t>(n))
{
return 'M';
}
@@ -2233,14 +2346,16 @@ class binary_writer
@brief checks whether a JSON number fits into @a TargetType
@param[in] el a JSON number of either the signed or unsigned integer kind
@return whether @a el's value can be represented by @a TargetType without
wrapping, regardless of which of the two kinds it is stored as
wrapping, regardless of which of the two kinds it is stored as;
false for a value that does not even fit the 64-bit type it is
read as (possible for number types wider than 64 bits)
*/
template<typename TargetType>
static bool bjdata_ndarray_value_in_range(const BasicJsonType& el)
{
return el.is_number_unsigned()
? value_in_range_of<TargetType>(el.template get<std::uint64_t>())
: value_in_range_of<TargetType>(el.template get<std::int64_t>());
? integer_fits_max<std::uint64_t>(el.m_data.m_value.number_unsigned) && value_in_range_of<TargetType>(el.template get<std::uint64_t>())
: integer_fits<std::int64_t>(el.m_data.m_value.number_integer) && value_in_range_of<TargetType>(el.template get<std::int64_t>());
}
/*!
@@ -2472,10 +2587,11 @@ class binary_writer
for (const auto& el : dims)
{
// a dimension is read as an unsigned value below, so anything that
// is not a non-negative integer is rejected: a non-integer entry
// would pun unrelated bytes as the dimension, and a negative one
// would wrap into a nonsensical length
if (!el.is_number_integer() || (!el.is_number_unsigned() && el.template get<std::int64_t>() < 0))
// is not a non-negative integer in the range of std::uint64_t is
// rejected: a non-integer entry would pun unrelated bytes as the
// dimension, and a negative or wider one would wrap into a
// nonsensical length
if (!el.is_number_integer() || !bjdata_ndarray_value_in_range<std::uint64_t>(el))
{
return true;
}
@@ -2589,9 +2705,9 @@ class binary_writer
case value_t::number_unsigned:
{
if (j.m_data.m_value.number_unsigned > static_cast<typename BasicJsonType::number_unsigned_t>((std::numeric_limits<std::int64_t>::max)()))
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::int64_t>(j.m_data.m_value.number_unsigned)))
{
JSON_THROW(out_of_range::create(407, concat("integer number ", std::to_string(j.m_data.m_value.number_unsigned), " cannot be represented by BON8 as it does not fit int64"), &j));
throw_integer_out_of_range(j, "BON8", "int64");
}
write_bon8_integer(static_cast<std::int64_t>(j.m_data.m_value.number_unsigned));
string_open = false;
@@ -2600,6 +2716,10 @@ class binary_writer
case value_t::number_integer:
{
if (JSON_HEDLEY_UNLIKELY(!integer_fits<std::int64_t>(j.m_data.m_value.number_integer)))
{
throw_integer_out_of_range(j, "BON8", "int64");
}
write_bon8_integer(static_cast<std::int64_t>(j.m_data.m_value.number_integer));
string_open = false;
break;
+6 -4
View File
@@ -14,7 +14,6 @@
#include <cstdint> // uint8_t
#include <limits> // numeric_limits
#include <string> // string
#include <type_traits> // is_signed
#include <nlohmann/detail/macro_scope.hpp>
#if JSON_HAS_THREE_WAY_COMPARISON
@@ -147,13 +146,16 @@ FloatType compare_integer_with_float(const IntegerType i, const FloatType f) noe
// values of IntegerType lie in [-bound, bound) when signed and in
// [0, bound) when unsigned; digits excludes the sign bit, so bound is a
// power of two that the float represents exactly
const FloatType bound = std::ldexp(static_cast<FloatType>(1), std::numeric_limits<IntegerType>::digits);
// power of two that the float represents exactly; the signedness comes
// from numeric_limits as well, because std::is_signed is false for class
// types such as 128-bit or multiprecision integers
using limits = std::numeric_limits<IntegerType>;
const FloatType bound = std::ldexp(static_cast<FloatType>(1), limits::digits);
if (f >= bound)
{
return ordered(-1);
}
if (std::is_signed<IntegerType>::value ? (f < -bound) : (f < static_cast<FloatType>(0)))
if (limits::is_signed ? (f < -bound) : (f < static_cast<FloatType>(0)))
{
return ordered(1);
}