Merge branch 'develop' into claude/abi-tag-strict-nul-handling

Resolve the conflict with #5344, which added the _psp ABI tag: _snul now
follows _psp, NLOHMANN_JSON_ABI_TAGS_CONCAT takes six tags, and
nlohmann_json.natvis and the amalgamation are regenerated.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-25 18:09:46 +02:00
55 changed files with 5380 additions and 185 deletions
+14 -3
View File
@@ -38,6 +38,10 @@
#define JSON_BRACE_INIT_COPY_SEMANTICS 0
#endif
#ifndef JSON_PRECISE_STREAM_POSITION
#define JSON_PRECISE_STREAM_POSITION 0
#endif
#ifndef JSON_STRICT_NUL_HANDLING
#define JSON_STRICT_NUL_HANDLING 0
#endif
@@ -66,6 +70,12 @@
#define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS
#endif
#if JSON_PRECISE_STREAM_POSITION
#define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION _psp
#else
#define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION
#endif
#if JSON_STRICT_NUL_HANDLING
#define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING _snul
#else
@@ -77,9 +87,9 @@
#endif
// Construct the namespace ABI tags component
#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e
#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \
NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e)
#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f
#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \
NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f)
#define NLOHMANN_JSON_ABI_TAGS \
NLOHMANN_JSON_ABI_TAGS_CONCAT( \
@@ -87,6 +97,7 @@
NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \
NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \
NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \
NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \
NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING)
// Construct the namespace version component
+99 -3
View File
@@ -11,8 +11,10 @@
#include <cstdint> // uint8_t
#include <cstddef> // size_t
#include <functional> // hash
#include <vector> // vector
#include <nlohmann/detail/abi_macros.hpp>
#include <nlohmann/detail/recursion_depth_limit.hpp>
#include <nlohmann/detail/value_t.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
@@ -26,6 +28,9 @@ inline std::size_t combine(std::size_t seed, std::size_t h) noexcept
return seed;
}
template<typename BasicJsonType>
std::size_t hash_iteratively(const BasicJsonType& j);
/*!
@brief hash a JSON value
@@ -33,12 +38,21 @@ The hash function tries to rely on std::hash where possible. Furthermore, the
type of the JSON value is taken into account to have different hash values for
null, 0, 0U, and false, etc.
Hashing an array or an object hashes its elements, which used to call this
function again once per nesting level, so a value nested deeply enough
exhausted the call stack and terminated the process. The descent is bounded
here: once @ref recursion_depth_limit levels have been entered, @ref
hash_iteratively hashes what is left without the call stack. A value nested
less deeply than that - all but a vanishing minority - is hashed exactly as
before, without allocating.
@tparam BasicJsonType basic_json specialization
@param j JSON value to hash
@param depth nesting level of @a j, counted from the value passed by the caller
@return hash value of j
*/
template<typename BasicJsonType>
std::size_t hash(const BasicJsonType& j)
std::size_t hash(const BasicJsonType& j, const std::size_t depth = 0)
{
using string_t = typename BasicJsonType::string_t;
using number_integer_t = typename BasicJsonType::number_integer_t;
@@ -56,22 +70,32 @@ std::size_t hash(const BasicJsonType& j)
case BasicJsonType::value_t::object:
{
if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit()))
{
return hash_iteratively(j);
}
auto seed = combine(type, j.size());
for (const auto& element : j.items())
{
const auto h = std::hash<string_t> {}(element.key());
seed = combine(seed, h);
seed = combine(seed, hash(element.value()));
seed = combine(seed, hash(element.value(), depth + 1));
}
return seed;
}
case BasicJsonType::value_t::array:
{
if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit()))
{
return hash_iteratively(j);
}
auto seed = combine(type, j.size());
for (const auto& element : j)
{
seed = combine(seed, hash(element));
seed = combine(seed, hash(element, depth + 1));
}
return seed;
}
@@ -127,5 +151,77 @@ std::size_t hash(const BasicJsonType& j)
}
}
/// an array or object whose elements @ref hash_iteratively is hashing
template<typename BasicJsonType>
struct hash_frame
{
hash_frame(const BasicJsonType* value_, std::size_t seed_) noexcept
: value(value_), position(value_->cbegin()), seed(seed_)
{}
const BasicJsonType* value;
typename BasicJsonType::const_iterator position;
std::size_t seed;
};
/*!
@brief hash the array or object @a j without the call stack
Computes the same value as @ref hash, keeping the arrays and objects it has
entered on an explicit stack instead of descending into them. Only reached for
values nested deeper than @ref recursion_depth_limit.
@tparam BasicJsonType basic_json specialization
@param j array or object to hash
@return hash value of j
*/
template<typename BasicJsonType>
std::size_t hash_iteratively(const BasicJsonType& j)
{
using string_t = typename BasicJsonType::string_t;
std::vector<hash_frame<BasicJsonType>> stack;
stack.emplace_back(&j, combine(static_cast<std::size_t>(j.type()), j.size()));
while (true)
{
// a copy, as entering an element below can reallocate the stack; the
// frame itself is only changed through stack.back()
const hash_frame<BasicJsonType> frame = stack.back();
if (frame.position == frame.value->cend())
{
// all elements are hashed: fold this value's hash into its parent's
// seed, exactly where the recursive version returns it
const std::size_t h = frame.seed;
stack.pop_back();
if (stack.empty())
{
return h;
}
stack.back().seed = combine(stack.back().seed, h);
continue;
}
if (frame.value->is_object())
{
stack.back().seed = combine(stack.back().seed, std::hash<string_t> {}(frame.position.key()));
}
// advance before entering the element, which pushes onto the stack
const BasicJsonType& element = *frame.position;
++stack.back().position;
if (element.is_structured())
{
stack.emplace_back(&element, combine(static_cast<std::size_t>(element.type()), element.size()));
}
else
{
stack.back().seed = combine(stack.back().seed, hash(element));
}
}
}
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
@@ -101,6 +101,11 @@ class input_stream_adapter
// maintain ifstream flags, except eof
if (is != nullptr)
{
#if JSON_PRECISE_STREAM_POSITION
// consume the character last returned by get_character() unless it
// was given back with release_lookahead()
commit_lookahead();
#endif
is->clear(is->rdstate() & std::ios::eofbit);
}
}
@@ -114,6 +119,58 @@ class input_stream_adapter
input_stream_adapter& operator=(input_stream_adapter&) = delete;
input_stream_adapter& operator=(input_stream_adapter&&) = delete;
#if JSON_PRECISE_STREAM_POSITION
input_stream_adapter(input_stream_adapter&& rhs) noexcept
: is(rhs.is), sb(rhs.sb), lookahead(rhs.lookahead)
{
rhs.is = nullptr;
rhs.sb = nullptr;
rhs.lookahead = false;
}
// Whether the character last returned by get_character() can be given back
// to the input with release_lookahead().
static constexpr bool supports_lookahead = true;
// std::istream/std::streambuf use std::char_traits<char>::to_int_type, to
// ensure that std::char_traits<char>::eof() and the character 0xFF do not
// end up as the same value, e.g., 0xFFFFFFFF.
//
// The character is peeked rather than consumed: it is only stepped over
// once the next character is requested, or when the adapter is destroyed.
// Until then, release_lookahead() can leave it in the input.
std::char_traits<char>::int_type get_character()
{
if (lookahead)
{
// step over the character returned by the previous call
sb->sbumpc();
}
auto res = sb->sgetc();
// set eof manually, as we don't use the istream interface.
if (JSON_HEDLEY_UNLIKELY(res == std::char_traits<char>::eof()))
{
// there is nothing to step over next time
lookahead = false;
is->clear(is->rdstate() | std::ios::eofbit);
}
else
{
lookahead = true;
}
return res;
}
// Leave the character last returned by get_character() in the input, so
// that the next read from the stream - by this adapter or by the caller
// once parsing is done - sees it again. Unlike putting a consumed
// character back, this cannot fail.
void release_lookahead() noexcept
{
lookahead = false;
}
#else
input_stream_adapter(input_stream_adapter&& rhs) noexcept
: is(rhs.is), sb(rhs.sb)
{
@@ -124,6 +181,9 @@ class input_stream_adapter
// std::istream/std::streambuf use std::char_traits<char>::to_int_type, to
// ensure that std::char_traits<char>::eof() and the character 0xFF do not
// end up as the same value, e.g., 0xFFFFFFFF.
//
// The character is consumed, so the character that terminates a number
// stays consumed after parsing; see JSON_PRECISE_STREAM_POSITION.
std::char_traits<char>::int_type get_character()
{
auto res = sb->sbumpc();
@@ -134,10 +194,14 @@ class input_stream_adapter
}
return res;
}
#endif
template<class T>
std::size_t get_elements(T* dest, std::size_t count = 1)
{
#if JSON_PRECISE_STREAM_POSITION
commit_lookahead();
#endif
auto res = static_cast<std::size_t>(sb->sgetn(reinterpret_cast<char*>(dest), static_cast<std::streamsize>(count * sizeof(T))));
if (JSON_HEDLEY_UNLIKELY(res < count * sizeof(T)))
{
@@ -147,9 +211,27 @@ class input_stream_adapter
}
private:
#if JSON_PRECISE_STREAM_POSITION
// Step over the character last returned by get_character(). The character
// has already been peeked successfully, so for every streambuf with a get
// area this is a pointer increment that cannot fail.
void commit_lookahead()
{
if (lookahead)
{
lookahead = false;
sb->sbumpc();
}
}
#endif
/// the associated input stream
std::istream* is = nullptr;
std::streambuf* sb = nullptr;
#if JSON_PRECISE_STREAM_POSITION
/// whether get_character() peeked a character that is not consumed yet
bool lookahead = false;
#endif
};
#endif // JSON_NO_IO
+65
View File
@@ -127,6 +127,25 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/)
return false;
}
// Detect whether an input adapter reads with one character of lookahead that
// can be left in the input (see input_stream_adapter::supports_lookahead,
// which is only defined with JSON_PRECISE_STREAM_POSITION), detected like
// supports_seek above.
template<typename InputAdapterType>
using detect_supports_lookahead = decltype(InputAdapterType::supports_lookahead);
template<typename InputAdapterType>
constexpr bool input_adapter_supports_lookahead(std::true_type /*detected*/)
{
return InputAdapterType::supports_lookahead;
}
template<typename InputAdapterType>
constexpr bool input_adapter_supports_lookahead(std::false_type /*detected*/)
{
return false;
}
// Detect whether an input adapter exposes a contiguous byte block that the
// lexer can scan directly (see iterator_input_adapter::supports_bulk_scan).
// Adapters without the flag - file, stream, wide-string, user-defined - fall
@@ -167,6 +186,12 @@ class lexer : public lexer_base<BasicJsonType>
static constexpr bool lazy_token_string =
input_adapter_supports_seek<InputAdapterType>(is_detected<detect_supports_seek, InputAdapterType> {});
/// whether a simulated unget can be passed on to the input adapter, which
/// then leaves the character in the input; see
/// input_adapter_supports_lookahead
static constexpr bool can_release_lookahead =
input_adapter_supports_lookahead<InputAdapterType>(is_detected<detect_supports_lookahead, InputAdapterType> {});
/// whether string scanning may bulk-consume runs of ordinary characters
/// directly from a contiguous input buffer (SWAR fast path). This requires
/// the token to be reconstructible lazily (lazy_token_string), so bypassing
@@ -1898,6 +1923,21 @@ scan_number_done:
uncapture_char(std::integral_constant<bool, lazy_token_string> {});
}
/// adapter without lookahead: nothing to do (see release_lookahead)
void release_lookahead_impl(std::false_type /*can_release*/) const noexcept {}
/// adapter with lookahead: leave the character in the input instead
void release_lookahead_impl(std::true_type /*can_release*/)
{
if (next_unget)
{
// the character is read from the input again rather than replayed
// from current, so the adapter must not step over it
next_unget = false;
ia.release_lookahead();
}
}
/// seekable adapter: nothing was captured, so nothing to undo
void uncapture_char(std::true_type /*lazy*/) const noexcept {}
@@ -1961,6 +2001,31 @@ scan_number_done:
return position;
}
/*!
@brief pass a pending simulated unget on to the input
unget() only rewinds the lexer's own bookkeeping, so the character that
terminated the last token (e.g. the character after a number) would still
be stepped over when the input adapter is done. Callers that hand the
input back to the user afterwards - operator>> and non-strict sax_parse -
call this once when scanning is done, so that the input is positioned
right after the value.
Adapters without lookahead (see input_adapter_supports_lookahead) are not
handed back to the user, so this is a no-op for them. Without
JSON_PRECISE_STREAM_POSITION, no adapter has lookahead, so this is always a
no-op and the terminating character stays consumed.
Scanning may continue after this call: @a next_unget is cleared, and the
character is read from the input again instead of being replayed from
@a current. A pending unget of EOF needs no special case, because reaching
EOF leaves no lookahead to release.
*/
void release_lookahead()
{
release_lookahead_impl(std::integral_constant<bool, can_release_lookahead> {});
}
#if JSON_DIAGNOSTIC_POSITIONS
/// return the offset of the first character of the last read token; unlike
/// the token's parsed value, this accounts for escape sequences
+45 -16
View File
@@ -100,13 +100,22 @@ class parser
json_sax_dom_callback_parser<BasicJsonType, InputAdapterType> sdp(result, callback, allow_exceptions, &m_lexer);
sax_parse_internal(&sdp);
// in strict mode, input must be completely read
if (strict && (get_token() != token_type::end_of_input))
if (strict)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(),
exception_message(token_type::end_of_input, "value"), nullptr));
// in strict mode, input must be completely read
if (get_token() != token_type::end_of_input)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(),
exception_message(token_type::end_of_input, "value"), nullptr));
}
}
else
{
// the caller keeps using the input: position it right after
// the value by leaving the character that terminated it
m_lexer.release_lookahead();
}
// in case of an error, return a discarded value
@@ -128,12 +137,20 @@ class parser
json_sax_dom_parser<BasicJsonType, InputAdapterType> sdp(result, allow_exceptions, &m_lexer);
sax_parse_internal(&sdp);
// in strict mode, input must be completely read
if (strict && (get_token() != token_type::end_of_input))
if (strict)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
// in strict mode, input must be completely read
if (get_token() != token_type::end_of_input)
{
sdp.parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
}
}
else
{
// see above
m_lexer.release_lookahead();
}
// in case of an error, return a discarded value
@@ -166,12 +183,24 @@ class parser
(void)detail::is_sax_static_asserts<SAX, BasicJsonType> {};
const bool result = sax_parse_internal(sax);
// strict mode: next byte must be EOF
if (result && strict && (get_token() != token_type::end_of_input))
if (result)
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
if (strict)
{
// strict mode: next byte must be EOF
if (get_token() != token_type::end_of_input)
{
return sax->parse_error(m_lexer.get_position(),
m_lexer.get_token_string(),
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
}
}
else
{
// the caller keeps using the input: position it right after
// the value by leaving the character that terminated it
m_lexer.release_lookahead();
}
}
return result;
@@ -44,6 +44,7 @@
#undef JSON_HAS_STATIC_RTTI
#undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON
#undef JSON_BRACE_INIT_COPY_SEMANTICS
#undef JSON_PRECISE_STREAM_POSITION
#undef JSON_STRICT_NUL_HANDLING
#endif
+5 -11
View File
@@ -30,6 +30,7 @@
#include <nlohmann/detail/meta/cpp_future.hpp>
#include <nlohmann/detail/output/binary_writer.hpp>
#include <nlohmann/detail/output/output_adapters.hpp>
#include <nlohmann/detail/recursion_depth_limit.hpp>
#include <nlohmann/detail/string_concat.hpp>
#include <nlohmann/detail/value_t.hpp>
@@ -133,7 +134,7 @@ class serializer
Serializing a container descends into its elements, so a value nested deeply
enough used to exhaust the call stack and terminate the process with no
exception to catch. The descent is bounded here: once @ref dump_depth_limit
exception to catch. The descent is bounded here: once @ref recursion_depth_limit
levels have been entered, @ref dump_iteratively writes out what is left
without the call stack. A value nested less deeply than that - all but a
vanishing minority - is written by exactly the code that always wrote it.
@@ -148,7 +149,7 @@ class serializer
{
case value_t::object:
{
if (JSON_HEDLEY_UNLIKELY(depth >= dump_depth_limit()))
if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit()))
{
dump_iteratively(val, current_indent);
return;
@@ -223,7 +224,7 @@ class serializer
case value_t::array:
{
if (JSON_HEDLEY_UNLIKELY(depth >= dump_depth_limit()))
if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit()))
{
dump_iteratively(val, current_indent);
return;
@@ -408,19 +409,12 @@ class serializer
}
private:
/// the number of levels @ref dump_internal descends into before it hands
/// over to @ref dump_iteratively
static constexpr std::size_t dump_depth_limit()
{
return 128;
}
/*!
@brief write out @a val and everything below it without the call stack
Emits the same bytes as @ref dump_internal, keeping the containers it has
entered on an explicit stack instead of descending into them. Only reached
for values nested deeper than @ref dump_depth_limit, which is why it is not
for values nested deeper than @ref recursion_depth_limit, which is why it is not
written for speed: walking every value this way measured up to 20% slower on
object-heavy documents than letting the compiler drive the descent.
*/
@@ -0,0 +1,35 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
#pragma once
#include <cstddef> // size_t
#include <nlohmann/detail/abi_macros.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
{
/*!
@brief the number of nesting levels an operation recurses into
Operations that walk a value (serializing, hashing, merging, ...) recurse once
per nesting level, which is fastest, but a value nested deeply enough would
exhaust the call stack. So they recurse only this many levels deep and finish
whatever lies below with an explicit stack. All of them share this limit.
@sa https://github.com/nlohmann/json/issues/5387
*/
constexpr std::size_t recursion_depth_limit() noexcept
{
return 128;
}
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END