Merge remote-tracking branch 'origin/develop' into claude/issue-5387-duplicate-check-bd7853

The nodiscard-safe dump() wrapping and test_utils.hpp include that
develop added to unit-regression2.cpp landed in the "issue #2067"
section, which this branch's test-file split had already relocated to
unit-regression3.cpp; ported both there.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-09 13:12:12 +02:00
57 changed files with 1905 additions and 163 deletions
@@ -398,6 +398,17 @@ inline void from_json(const BasicJsonType& j, CompatibleArrayType& bin)
}
}
template<typename ConstructibleObjectType>
auto from_json_object_reserve(ConstructibleObjectType& obj, typename ConstructibleObjectType::size_type size, priority_tag<1> /*unused*/)
-> decltype(obj.reserve(size), void())
{
obj.reserve(size);
}
template<typename ConstructibleObjectType>
inline void from_json_object_reserve(ConstructibleObjectType& /*obj*/, std::size_t /*size*/, priority_tag<0> /*unused*/)
{}
template<typename BasicJsonType, typename ConstructibleObjectType,
enable_if_t<is_constructible_object_type<BasicJsonType, ConstructibleObjectType>::value, int> = 0>
inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj)
@@ -409,6 +420,7 @@ inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj)
ConstructibleObjectType ret;
const auto* inner_object = j.template get_ptr<const typename BasicJsonType::object_t*>();
from_json_object_reserve(ret, inner_object->size(), priority_tag<1> {});
for (const auto& p : *inner_object)
{
ret.emplace(p.first, p.second.template get<typename ConstructibleObjectType::mapped_type>());
@@ -1075,8 +1075,8 @@ char* to_chars(char* first, const char* last, FloatType value)
}
#ifdef __GNUC__
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wfloat-equal"
JSON_HEDLEY_DIAGNOSTIC_PUSH
JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal")
#endif
if (value == 0) // +-0
{
@@ -1087,7 +1087,7 @@ char* to_chars(char* first, const char* last, FloatType value)
return first;
}
#ifdef __GNUC__
#pragma GCC diagnostic pop
JSON_HEDLEY_DIAGNOSTIC_POP
#endif
JSON_ASSERT(last - first >= std::numeric_limits<FloatType>::max_digits10);
+3 -3
View File
@@ -33,8 +33,8 @@
// code stumbling over this. See https://github.com/nlohmann/json/issues/4087
// for a discussion.
#if defined(__clang__)
#pragma clang diagnostic push
#pragma clang diagnostic ignored "-Wweak-vtables"
JSON_HEDLEY_DIAGNOSTIC_PUSH
JSON_HEDLEY_PRAGMA(clang diagnostic ignored "-Wweak-vtables")
#endif
NLOHMANN_JSON_NAMESPACE_BEGIN
@@ -287,5 +287,5 @@ class other_error : public exception
NLOHMANN_JSON_NAMESPACE_END
#if defined(__clang__)
#pragma clang diagnostic pop
JSON_HEDLEY_DIAGNOSTIC_POP
#endif
+128 -5
View File
@@ -149,10 +149,11 @@ class lexer : public lexer_base<BasicJsonType>
public:
using token_type = typename lexer_base<BasicJsonType>::token_type;
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false, bool discard_number_values_ = false) noexcept
: ia(std::move(adapter))
, ignore_comments(ignore_comments_)
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
, discard_number_values(discard_number_values_)
{}
// deleted because of pointer members
@@ -1279,6 +1280,58 @@ scan_number_done:
// we are done scanning a number)
unget();
// If the caller does not need the converted value (only whether the
// input is syntactically valid; see json_sax_acceptor/accept()), an
// unsigned/integer token can be reported without calling
// strtoull()/strtoll() at all, *provided* we can already tell from
// the digit count alone that the conversion cannot overflow 64 bits.
// Such tokens are always finite and are accepted unconditionally by
// the parser regardless of their actual value (parser::sax_parse_internal()
// never checks finiteness for value_unsigned/value_integer), so the
// classification below is all that is needed.
//
// A decimal number with up to 18 digits is always representable in
// both std::uint64_t and std::int64_t (18 nines is ~1e18, well below
// both UINT64_MAX ~1.8e19 and INT64_MAX ~9.2e18), so strtoull()/strtoll()
// could not have set errno to ERANGE for it. Numbers with more digits
// (rare in practice) fall through to the exact code below, unchanged,
// so their handling -- including reclassification to value_float when
// the value overflows 64 bits, and rejection when it is not even
// finite as a double -- is bit-for-bit identical to before this
// optimization.
//
// Note this reasons about std::uint64_t/std::int64_t, not about
// number_unsigned_t/number_integer_t (BasicJsonType's own, possibly
// narrower, template parameters -- e.g. std::uint32_t). That is fine
// *only* because discard_number_values is exclusively set by
// accept() (see json.hpp), and accept() always parses through the
// library's own json_sax_acceptor -- never a user-supplied SAX
// consumer -- whose number_unsigned()/number_integer()/number_float()
// callbacks unconditionally discard their argument and return true.
// So for every caller that can reach this branch, neither the token
// classification below nor the eventual (possibly narrowed, and on
// this fast path left stale/unset) value_unsigned/value_integer is
// ever consulted -- an unsigned/integer token is accepted outright,
// and even a >18-digit token that this fast path deliberately falls
// through for is, once reclassified to value_float, still finite
// (and thus accepted) for any digit count that fits in number_unsigned_t
// or number_integer_t regardless of that type's width. If this
// function is ever taught to run with discard_number_values true for
// a caller that *does* read the converted value, this reasoning (and
// the fast path below) would need to be revisited.
if (discard_number_values)
{
constexpr std::size_t safe_digit_count = 18;
if (number_type == token_type::value_unsigned && token_buffer.size() <= safe_digit_count)
{
return token_type::value_unsigned;
}
if (number_type == token_type::value_integer && token_buffer.size() - 1 <= safe_digit_count)
{
return token_type::value_integer;
}
}
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
errno = 0;
@@ -1393,8 +1446,7 @@ scan_number_done:
*/
char_int_type get()
{
++position.chars_read_total;
++position.chars_read_current_line;
advance_position();
if (next_unget)
{
@@ -1406,6 +1458,23 @@ scan_number_done:
current = ia.get_character();
}
return track_after_read();
}
/// shared head of get() / get_ignoring_pending_unget(): bump the
/// per-character position counters (line-count-on-'\n' bookkeeping is
/// handled afterwards, in track_after_read(), once `current` is known)
void advance_position() noexcept
{
++position.chars_read_total;
++position.chars_read_current_line;
}
/// shared tail of get() / get_ignoring_pending_unget(): capture the
/// character for error messages (if needed) and update line/column
/// bookkeeping for the character now in `current`
char_int_type track_after_read()
{
// seekable adapters reconstruct the token lazily on error (see
// get_token_string), so the eager per-character copy is skipped
capture_char(std::integral_constant<bool, lazy_token_string> {});
@@ -1419,6 +1488,29 @@ scan_number_done:
return current;
}
/*!
@brief like get(), but for call sites that can prove no unget() is pending
get() has to check the `next_unget` flag on every call, because a
previous token may have ended with unget() (e.g. scan_number() always
ungets the character that terminated the number, so the next call to
scan() can see it again). skip_whitespace() reads that first,
possibly-ungotten character via a plain get(), but every further
character it reads is guaranteed to be a fresh read: nothing between
those calls invokes unget(). This variant skips the (otherwise always
false) next_unget branch for those calls; it is not a general
replacement for get().
*/
char_int_type get_ignoring_pending_unget()
{
JSON_ASSERT(!next_unget);
advance_position();
current = ia.get_character();
return track_after_read();
}
/// seekable adapter: nothing to capture, the token is rebuilt on error
void capture_char(std::true_type /*lazy*/) const noexcept {}
@@ -1612,13 +1704,37 @@ scan_number_done:
return true;
}
/// whether `current` is one of the four JSON whitespace characters
bool current_is_whitespace() const noexcept
{
return current == ' ' || current == '\t' || current == '\n' || current == '\r';
}
void skip_whitespace()
{
// the first character may be a pending unget() left over from the
// previous token (see get_ignoring_pending_unget()); every
// subsequent character read by this loop is guaranteed fresh, since
// nothing below calls unget()
get();
if (!current_is_whitespace())
{
return;
}
// this is written as an if-guarded do-while (rather than a plain
// while loop) because that shape is what lets both GCC and Clang
// keep the input adapter's read pointer in a register across
// iterations; the equivalent while-loop measurably defeated that
// optimization in testing, turning long whitespace runs (e.g. the
// indentation of pretty-printed JSON) from a register-only loop
// into one that reloads the pointer from memory every character
do
{
get();
get_ignoring_pending_unget();
}
while (current == ' ' || current == '\t' || current == '\n' || current == '\r');
while (current_is_whitespace());
}
token_type scan()
@@ -1754,6 +1870,13 @@ scan_number_done:
const char_int_type decimal_point_char = '.';
/// the position of the decimal point in the input
std::size_t decimal_point_position = std::string::npos;
/// whether the caller (e.g. accept()/json_sax_acceptor) only needs the
/// token classification and never looks at the converted numeric value;
/// when set, scan_number() may skip strtoull()/strtoll() for
/// value_unsigned/value_integer tokens whose digit count guarantees they
/// fit into 64 bits (see scan_number())
const bool discard_number_values = false;
};
} // namespace detail
+3 -2
View File
@@ -72,9 +72,10 @@ class parser
parser_callback_t<BasicJsonType> cb = nullptr,
const bool allow_exceptions_ = true,
const bool ignore_comments = false,
const bool ignore_trailing_commas_ = false)
const bool ignore_trailing_commas_ = false,
const bool discard_number_values_ = false)
: callback(std::move(cb))
, m_lexer(std::move(adapter), ignore_comments)
, m_lexer(std::move(adapter), ignore_comments, discard_number_values_)
, allow_exceptions(allow_exceptions_)
, ignore_trailing_commas(ignore_trailing_commas_)
{
@@ -18,6 +18,7 @@
#endif
#include <nlohmann/detail/abi_macros.hpp>
#include <nlohmann/detail/macro_scope.hpp>
#include <nlohmann/detail/meta/type_traits.hpp>
#include <nlohmann/detail/string_utils.hpp>
#include <nlohmann/detail/value_t.hpp>
@@ -206,10 +207,10 @@ NLOHMANN_JSON_NAMESPACE_END
namespace std
{
// Fix: https://github.com/nlohmann/json/issues/1401
#if defined(__clang__)
// Fix: https://github.com/nlohmann/json/issues/1401
#pragma clang diagnostic push
#pragma clang diagnostic ignored "-Wmismatched-tags"
JSON_HEDLEY_DIAGNOSTIC_PUSH
JSON_HEDLEY_PRAGMA(clang diagnostic ignored "-Wmismatched-tags")
#endif
template<typename IteratorType>
class tuple_size<::nlohmann::detail::iteration_proxy_value<IteratorType>> // NOLINT(cert-dcl58-cpp)
@@ -224,7 +225,7 @@ class tuple_element<N, ::nlohmann::detail::iteration_proxy_value<IteratorType >>
::nlohmann::detail::iteration_proxy_value<IteratorType >> ()));
};
#if defined(__clang__)
#pragma clang diagnostic pop
JSON_HEDLEY_DIAGNOSTIC_POP
#endif
} // namespace std
+14
View File
@@ -748,6 +748,20 @@ class json_pointer
}
}
// the reference token consists only of digits at this point (cf. checks
// above); however, its numeric value might not be representable, in which
// case array_index() would throw out_of_range.404/410 -- contains() must
// not throw (see #5395), so such a reference token is treated as "not found"
errno = 0; // strtoull() does not reset errno on success
char* p_end = nullptr; // NOLINT(misc-const-correctness)
const unsigned long long magnitude = std::strtoull(reference_token.c_str(), &p_end, 10); // NOLINT(runtime/int)
if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX
|| magnitude >= static_cast<unsigned long long>((std::numeric_limits<typename BasicJsonType::size_type>::max)()))) // NOLINT(runtime/int)
{
// the array index cannot be represented as size_type
return false;
}
const auto idx = array_index<BasicJsonType>(reference_token);
if (idx >= ptr->size())
{
@@ -1668,6 +1668,15 @@ class binary_writer
CharType dtype = it->second;
key = "_ArraySize_";
// the dimensions are written verbatim as the header length below, so a
// value that is not an array cannot produce a valid one: null emits 'Z'
// and an object emits '{', neither of which a reader accepts after '#'.
// Such an object is not a valid ndarray and falls back to a plain object.
if (!value.at(key).is_array())
{
return true;
}
std::size_t len = (value.at(key).empty() ? 0 : 1);
for (const auto& el : value.at(key))
{
@@ -1841,8 +1850,8 @@ class binary_writer
void write_compact_float(const number_float_t n, detail::input_format_t format)
{
#ifdef __GNUC__
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wfloat-equal"
JSON_HEDLEY_DIAGNOSTIC_PUSH
JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal")
#endif
if (!std::isfinite(n) || ((static_cast<double>(n) >= static_cast<double>(std::numeric_limits<float>::lowest()) &&
static_cast<double>(n) <= static_cast<double>((std::numeric_limits<float>::max)()) &&
@@ -1861,7 +1870,7 @@ class binary_writer
write_number(n);
}
#ifdef __GNUC__
#pragma GCC diagnostic pop
JSON_HEDLEY_DIAGNOSTIC_POP
#endif
}
+65
View File
@@ -9,9 +9,12 @@
#pragma once
#include <array> // array
#include <cmath> // isnan, ldexp, trunc
#include <cstddef> // size_t
#include <cstdint> // uint8_t
#include <limits> // numeric_limits
#include <string> // string
#include <type_traits> // is_signed
#include <nlohmann/detail/macro_scope.hpp>
#if JSON_HAS_THREE_WAY_COMPARISON
@@ -114,5 +117,67 @@ inline bool operator<(const value_t lhs, const value_t rhs) noexcept
}
#endif
/*!
@brief compare an integer with a floating point number without precision loss
Widening the integer to the floating point type loses precision beyond the
float's mantissa, which makes equality intransitive: both 2^63-2 and 2^63-1
round to 2^63, so each compares equal to that float while differing from each
other. Ordering built on that is not a strict weak ordering, so sorting such
values, or using them as keys in an ordered container, is undefined behavior.
Returns a value to be compared against zero with the original operator, which
reproduces the exact ordering. A NaN operand is returned as is, so comparing it
against zero keeps NaN's semantics: false for the relational operators and
unordered for `<=>`.
*/
template<typename IntegerType, typename FloatType>
FloatType compare_integer_with_float(const IntegerType i, const FloatType f) noexcept
{
const auto ordered = [](int c) noexcept
{
return static_cast<FloatType>(c);
};
if (std::isnan(f))
{
return f;
}
// values of IntegerType lie in [-bound, bound) when signed and in
// [0, bound) when unsigned; digits excludes the sign bit, so bound is a
// power of two that the float represents exactly
const FloatType bound = std::ldexp(static_cast<FloatType>(1), std::numeric_limits<IntegerType>::digits);
if (f >= bound)
{
return ordered(-1);
}
if (std::is_signed<IntegerType>::value ? (f < -bound) : (f < static_cast<FloatType>(0)))
{
return ordered(1);
}
// f is now within the integer's range, so truncating it is exact
const FloatType truncated = std::trunc(f);
const auto as_integer = static_cast<IntegerType>(truncated);
if (i != as_integer)
{
return ordered(i < as_integer ? -1 : 1);
}
// the integer parts agree, so any fractional part decides
const FloatType fraction = f - truncated;
if (fraction > static_cast<FloatType>(0))
{
return ordered(-1);
}
if (fraction < static_cast<FloatType>(0))
{
return ordered(1);
}
return ordered(0);
}
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END