Merge remote-tracking branch 'origin/develop' into claude/parser-performance-research-4hkdf8

Resolves the number-parsing conflict between the discard_number_values
fast path (an early return for accept()-only calls that skips value
conversion for short digit runs) and the contiguous number scanner's
convert_number()/convert_integer() split: the discard check now runs
as the first step of convert_number(), so both the byte-at-a-time
scanner and the bulk contiguous path benefit from it.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-09 13:05:02 +02:00
54 changed files with 1621 additions and 139 deletions
+53
View File
@@ -128,6 +128,51 @@ json_test_set_test_options(test-unicode4 TEST_PROPERTIES TIMEOUT 3000)
# add unit tests
#############################################################################
# Generate the leak checks for every JSON_HEDLEY_* macro defined in
# hedley.hpp; tests/src/unit-no-macro-leak.cpp #include-s the result after
# nlohmann/json.hpp (see issue #5408). Using the shared
# cmake/scripts/gen_hedley_undef_check.cmake script (also used by `make
# update_hedley_undef`) instead of a hand-maintained list of macro names
# means this test can never go stale after a future `make update_hedley`.
set(hedley_hpp "${PROJECT_SOURCE_DIR}/include/nlohmann/thirdparty/hedley/hedley.hpp")
set(hedley_undef_check_script "${PROJECT_SOURCE_DIR}/cmake/scripts/gen_hedley_undef_check.cmake")
set(hedley_undef_checks "${PROJECT_BINARY_DIR}/include/hedley_undef_checks.inc")
# Reconfigure whenever the vendored header or the generator script changes,
# so a `cmake --build` after `make update_hedley` does not silently keep a
# stale generated file around.
set_property(DIRECTORY APPEND PROPERTY CMAKE_CONFIGURE_DEPENDS
"${hedley_hpp}"
"${hedley_undef_check_script}")
# Generate once at configure time, so the very first build (before any
# custom-command build step has run) already has an up-to-date file.
execute_process(
COMMAND ${CMAKE_COMMAND}
"-DHEDLEY_HPP=${hedley_hpp}"
"-DOUTPUT=${hedley_undef_checks}"
-DMODE=checks
-P "${hedley_undef_check_script}"
RESULT_VARIABLE hedley_undef_check_result
)
if(NOT hedley_undef_check_result EQUAL 0)
message(FATAL_ERROR "Failed to generate ${hedley_undef_checks}")
endif()
# Also (re)generate as a build step, so an incremental build after editing
# hedley.hpp without a full reconfigure still picks up the change.
add_custom_command(
OUTPUT "${hedley_undef_checks}"
COMMAND ${CMAKE_COMMAND}
"-DHEDLEY_HPP=${hedley_hpp}"
"-DOUTPUT=${hedley_undef_checks}"
-DMODE=checks
-P "${hedley_undef_check_script}"
DEPENDS "${hedley_hpp}" "${hedley_undef_check_script}"
COMMENT "Generating Hedley undef leak checks"
VERBATIM)
add_custom_target(generate_hedley_undef_checks DEPENDS "${hedley_undef_checks}")
if("${JSON_TestStandards}" STREQUAL "")
set(test_cxx_standards 11 14 17 20 23)
unset(test_force)
@@ -231,6 +276,14 @@ foreach(file ${files})
json_test_add_test_for(${file} MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force})
endforeach()
# tests/src/unit-no-macro-leak.cpp #include-s the generated leak-check file,
# so its test targets must be built after generate_hedley_undef_checks.
foreach(cxx_standard ${test_cxx_standards})
if(TARGET test-no-macro-leak_cpp${cxx_standard})
add_dependencies(test-no-macro-leak_cpp${cxx_standard} generate_hedley_undef_checks)
endif()
endforeach()
if(json_32bit_test_only)
# Skip all other tests in this file
return()
+9
View File
@@ -15,6 +15,15 @@
namespace utils
{
// Some tests intentionally discard the [[nodiscard]]/JSON_HEDLEY_WARN_UNUSED_RESULT
// return value of a call they only make to exercise its side effects (e.g. checking
// that it does not throw). A plain (void) cast on the call expression does not
// suppress GCC's warning for functions using the GNU __attribute__((warn_unused_result))
// form (as opposed to the C++17 [[nodiscard]] attribute) -- passing the value into an
// ordinary function call does.
template<typename T>
inline void ignore_return_value(T&& /*unused*/) noexcept {}
inline std::vector<std::uint8_t> read_binary_file(const std::string& filename)
{
std::ifstream file(filename, std::ios::binary);
+25
View File
@@ -2751,6 +2751,31 @@ TEST_CASE("BJData")
CHECK(json::to_bjdata(j_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, 'i', 3, ']', 1, 2, 3, 4, 5, 6}));
CHECK(json::from_bjdata(json::to_bjdata(j_ok), true, true) == j_ok);
}
SECTION("ndarray whose _ArraySize_ is not an array stays as object")
{
// the shape is written verbatim as the header length, so a
// value that is not an array cannot produce a valid one: null
// would emit 'Z' and an object '{', neither of which a reader
// accepts after '#'. Both have to stay plain objects.
json const j_null = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", nullptr}, {"_ArrayData_", json::array()}});
const auto out_null = json::to_bjdata(j_null);
CHECK(out_null.at(0) == '{');
CHECK(json::from_bjdata(out_null) == j_null);
// an object shape passes the per-entry check by iterating its
// values rather than dimensions, so it needs rejecting too
json const j_obj = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {{"a", 1}}}, {"_ArrayData_", {1}}});
const auto out_obj = json::to_bjdata(j_obj);
CHECK(out_obj.at(0) == '{');
CHECK(json::from_bjdata(out_obj) == j_obj);
// a scalar shape is not a dimension list either
json const j_num = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", 1}, {"_ArrayData_", {1}}});
const auto out_num = json::to_bjdata(j_num);
CHECK(out_num.at(0) == '{');
CHECK(json::from_bjdata(out_num) == j_num);
}
}
}
+142 -3
View File
@@ -38,6 +38,54 @@ class huge_binary_t : public std::vector<std::uint8_t>
using huge_binary_json = nlohmann::basic_json <
std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t,
double, std::allocator, nlohmann::adl_serializer, huge_binary_t, void >;
// a string type that can be made to report a size beyond INT32_MAX without
// allocating that much memory, so BSON length overflow can be tested for
// strings and (embedded) documents as well, following the same idea as
// huge_binary_t.
//
// Unlike huge_binary_t (which is only ever used as the BSON *value* type),
// this type doubles as basic_json's StringType and is therefore also used
// for *object keys* (e.g. "s" or "nested" below). Only the designated test
// value is meant to lie about its size - if every huge_string_t (including
// keys) reported a huge size, the running totals computed while walking the
// BSON document (see calc_bson_object_size & friends in binary_writer.hpp)
// would need more than 32 bits, and on platforms where std::size_t is only
// 32 bits wide that arithmetic would silently wrap around, producing wrong
// (or even unguarded) lengths. The fake size is therefore opt-in via
// as_huge(), and plain strings - in particular object keys - keep reporting
// their real, small size.
class huge_string_t : public std::string
{
public:
using std::string::string;
huge_string_t(const std::string& s) : std::string(s) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions)
// returns a copy of @a s whose size() pretends to be huge
static huge_string_t as_huge(const std::string& s)
{
huge_string_t result(s);
result.pretend_huge = true;
return result;
}
size_type size() const noexcept
{
if (pretend_huge)
{
// one byte more than the BSON length field can represent
return static_cast<size_type>((std::numeric_limits<std::int32_t>::max)()) + 1;
}
return std::string::size();
}
private:
bool pretend_huge = false;
};
using huge_string_json = nlohmann::basic_json <
std::map, std::vector, huge_string_t, bool, std::int64_t, std::uint64_t,
double, std::allocator, nlohmann::adl_serializer, std::vector<std::uint8_t>, void >;
} // namespace
TEST_CASE("BSON")
@@ -105,10 +153,36 @@ TEST_CASE("BSON")
SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON")
{
huge_binary_json j;
j["b"] = huge_binary_json::binary(huge_binary_t{});
// out_of_range.412 is thrown from a single shared helper
// (to_bson_length) that guards the BSON length fields of binary
// values, strings, and (embedded) documents alike
SECTION("binary")
{
huge_binary_json j;
j["b"] = huge_binary_json::binary(huge_binary_t{});
CHECK_THROWS_WITH_AS(huge_binary_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483661 exceeds maximum of 2147483647", huge_binary_json::out_of_range&);
CHECK_THROWS_WITH_AS(huge_binary_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483661 exceeds maximum of 2147483647", huge_binary_json::out_of_range&);
}
SECTION("string")
{
huge_string_json j;
j["s"] = huge_string_t::as_huge("value");
CHECK_THROWS_WITH_AS(huge_string_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483661 exceeds maximum of 2147483647", huge_string_json::out_of_range&);
}
SECTION("document")
{
// an oversized string nested one level deep makes the
// *embedded* document's own length exceed INT32_MAX as well
huge_string_json nested;
nested["s"] = huge_string_t::as_huge("value");
huge_string_json j;
j["nested"] = nested;
CHECK_THROWS_WITH_AS(huge_string_json::to_bson(j), "[json.exception.out_of_range.412] BSON length 2147483674 exceeds maximum of 2147483647", huge_string_json::out_of_range&);
}
}
SECTION("string length must be at least 1")
@@ -193,6 +267,23 @@ TEST_CASE("BSON")
CHECK(json::from_bson(result, true, false) == j);
}
SECTION("non-empty object with bool from a non-0/1 byte (lenient parsing)")
{
// documented lenient behavior (see gh-5333): any non-zero byte
// is accepted as `true`, not just 0x01
std::vector<std::uint8_t> const input =
{
0x0D, 0x00, 0x00, 0x00, // size (little endian)
0x08, // entry: boolean
'e', 'n', 't', 'r', 'y', '\x00',
0x02, // value = 0x02 (neither 0x00 nor 0x01)
0x00 // end marker
};
const json expected = { { "entry", true } };
CHECK(json::from_bson(input) == expected);
}
SECTION("non-empty object with double")
{
json const j =
@@ -499,6 +590,29 @@ TEST_CASE("BSON")
CHECK(json::from_bson(result, true, false) == j);
}
SECTION("array elements with non-conforming keys (lenient parsing)")
{
// documented lenient behavior (see gh-5333): BSON array element
// keys are not checked against the required decimal sequence
// "0", "1", "2", ... - elements are taken in encoded order
std::vector<std::uint8_t> const input =
{
0x26, 0x00, 0x00, 0x00, // size (little endian)
0x04, 'e', 'n', 't', 'r', 'y', '\x00', // entry: embedded array
0x1A, 0x00, 0x00, 0x00, // size (little endian)
0x10, '5', 0x00, 0x0A, 0x00, 0x00, 0x00, // key "5" (bogus) -> 10
0x10, 'x', 0x00, 0x14, 0x00, 0x00, 0x00, // key "x" (non-numeric) -> 20
0x10, '1', 0x00, 0x1E, 0x00, 0x00, 0x00, // key "1" (out of order) -> 30
0x00, // end marker (embedded array)
0x00 // end marker
};
const json expected = { { "entry", json::array({10, 20, 30}) } };
CHECK(json::from_bson(input) == expected);
}
SECTION("non-empty object with binary member")
{
const size_t N = 10;
@@ -594,6 +708,31 @@ TEST_CASE("BSON")
CHECK(json::from_bson(result, true, false) == j);
}
SECTION("binary member with subtype 0x02 (old binary) keeps its inner length prefix (lenient parsing)")
{
// documented lenient behavior (see gh-5333): the payload for
// binary subtype 0x02 ("old binary") is returned as-is,
// including its own inner 4-byte length prefix; it is not
// stripped or reinterpreted
std::vector<std::uint8_t> const input =
{
0x17, 0x00, 0x00, 0x00, // size (little endian)
0x05, 'e', 'n', 't', 'r', 'y', '\x00', // entry: binary
0x06, 0x00, 0x00, 0x00, // size of binary (little endian)
0x02, // "old binary" subtype
0x02, 0x00, 0x00, 0x00, // inner length prefix (part of the old-binary payload)
0x68, 0x69, // payload ('h', 'i')
0x00 // end marker
};
// the inner length prefix is part of the (unmodified) payload
const std::vector<std::uint8_t> expected_payload = {0x02, 0x00, 0x00, 0x00, 0x68, 0x69};
const json expected = { { "entry", json::binary(expected_payload, 0x02) } };
CHECK(json::from_bson(input) == expected);
}
SECTION("Some more complex document")
{
json const j =
+161 -1
View File
@@ -23,6 +23,8 @@ using nlohmann::json;
#include <utility>
#include <vector>
#include "test_utils.hpp"
namespace
{
class SaxEventLogger
@@ -624,7 +626,8 @@ TEST_CASE("parser class")
SECTION("overflow")
{
// overflows during parsing yield an exception
CHECK_THROWS_WITH_AS(parser_helper("1.18973e+4932").empty(), "[json.exception.out_of_range.406] number overflow parsing '1.18973e+4932'", json::out_of_range&);
// empty() is nodiscard; the exception is thrown by parser_helper() itself, before empty() would run
CHECK_THROWS_WITH_AS(utils::ignore_return_value(parser_helper("1.18973e+4932").empty()), "[json.exception.out_of_range.406] number overflow parsing '1.18973e+4932'", json::out_of_range&);
}
SECTION("invalid numbers")
@@ -930,6 +933,98 @@ TEST_CASE("parser class")
CHECK(accept_helper("+1") == false);
CHECK(accept_helper("+0") == false);
}
SECTION("issue #5411 - skip conversion when accept() does not need the numeric value")
{
// lexer::scan_number() may skip strtoull()/strtoll() for
// value_unsigned/value_integer tokens when the caller (e.g.
// json::accept()) does not need the converted value, as long
// as the digit count alone guarantees no 64-bit overflow (see
// the "safe_digit_count" fast path in scan_number()). This
// differential test checks that json::accept() (which enables
// the fast path) and json::parse() (which never does) always
// agree, over a corpus that exercises both the fast path
// (<=18 digits) and the untouched, exact fallback path (>=19
// digits) -- including reclassification of huge digit-only
// integers to a (possibly non-finite) floating-point value.
const std::vector<std::pair<std::string, bool>> cases =
{
// normal small/large integers, both signs
{"0", true}, {"1", true}, {"-1", true}, {"42", true}, {"-42", true},
{"123456789", true}, {"-123456789", true},
// digit-count boundary around the 18-digit safe cutoff (both signs)
{std::string(17, '9'), true},
{std::string(18, '9'), true},
{std::string(19, '9'), true},
{std::string(20, '9'), true},
{"-" + std::string(17, '9'), true},
{"-" + std::string(18, '9'), true},
{"-" + std::string(19, '9'), true},
{"-" + std::string(20, '9'), true},
// 64-bit boundaries
{"9223372036854775807", true}, // INT64_MAX
{"-9223372036854775808", true}, // INT64_MIN
{"18446744073709551615", true}, // UINT64_MAX
{"18446744073709551616", true}, // UINT64_MAX + 1 (overflows uint64_t, finite double)
// the 28-digit example from the issue: overflows uint64_t
// but is finite as a double, so the scanner reclassifies
// it to value_float and it is accepted
{"9999999999999999999999999999", true},
// huge digit-only integers that overflow even a double -> rejected
{std::string(309, '9'), false},
{std::string(400, '9'), false},
{"1" + std::string(400, '0'), false},
// 1e999 / 1e400 style overflow -> rejected
{"1e999", false},
{"1e400", false},
{"-1e999", false},
{"1E999", false},
// values straddling DBL_MAX
{"1.7976931348623157e308", true}, // <= DBL_MAX, finite
{"1.7976931348623159e308", false}, // > DBL_MAX, overflows to inf
// a mix of other valid/invalid numeric syntax
{"3.14159", true},
{"-0.0", true},
{"1.0e10", true},
{"01", false},
{"-", false},
{"1.", false},
{"1e", false},
{"+1", false},
};
for (const auto& c : cases)
{
const std::string& number = c.first;
const bool expected = c.second;
CAPTURE(number)
CAPTURE(expected)
// accept() takes the fast path (skips conversion when possible)
CHECK(json::accept(number) == expected);
// parse() always performs the full conversion; it must agree
json j;
CHECK_NOTHROW(json::parser(nlohmann::detail::input_adapter(number), nullptr, false).parse(true, j));
CHECK(!j.is_discarded() == expected);
// wrap in an array so get_token() is exercised beyond the
// very first (constructor-time) scan as well
std::string wrapped = "[";
wrapped += number;
wrapped += ",";
wrapped += number;
wrapped += "]";
CHECK(json::accept(wrapped) == expected);
}
}
}
}
@@ -1394,6 +1489,71 @@ TEST_CASE("parser class")
CHECK(accept_helper("\"\\uD80C\\uFFFF\"") == false);
}
#if !defined(JSON_NOEXCEPTION)
SECTION("issue #5412 - whitespace skipping bookkeeping (compact vs. pretty-printed)")
{
// lexer::skip_whitespace() reads its first character with get() (to
// honor a possibly pending unget() from the previous token) and every
// further whitespace character with get_ignoring_pending_unget() (a
// get() variant that skips the then-always-false next_unget check).
// This must not change the reported byte offset, line, or column of
// a syntax error, even when a long run of whitespace containing
// multiple newlines is skipped beforehand (as with pretty-printed
// input). The expected values below were captured from the
// unmodified do-while(get()) loop, so any regression that miscounts
// characters or newlines while skipping whitespace changes them.
const auto check_error = [](const std::string & input, std::size_t expected_byte,
const std::string & expected_what)
{
CAPTURE(input)
try
{
json _ = json::parse(input);
FAIL_CHECK("expected a parse_error, but parsing succeeded");
}
catch (const json::parse_error& e)
{
CHECK(e.byte == expected_byte);
CHECK(std::string(e.what()) == expected_what);
}
};
// a nested document, serialized both compactly and pretty-printed
// (dump(4)), each truncated right before the final closing '}' so
// that the parser hits EOF after skipping all of the (in the
// pretty-printed case, substantial) indentation whitespace
const json doc =
{
{"a", 1},
{"b", json::array({true, false, nullptr, "x"})},
{"c", json::object({{"d", 3.14}, {"e", json::array({1, 2, 3})}})}
};
const std::string compact = doc.dump();
const std::string pretty = doc.dump(4);
check_error(compact.substr(0, compact.size() - 1), 60,
"[json.exception.parse_error.101] parse error at line 1, column 60: syntax error while parsing object - unexpected end of input; expected '}'");
check_error(pretty.substr(0, pretty.size() - 1), 193,
"[json.exception.parse_error.101] parse error at line 17, column 1: syntax error while parsing object - unexpected end of input; expected '}'");
// an invalid token appearing after several indented, multi-line
// whitespace runs vs. the same document without any of that
// whitespace
check_error(R"({
"a": 1,
"b": [
true,
false
],
"c": @
})", 70,
"[json.exception.parse_error.101] parse error at line 7, column 10: syntax error while parsing value - invalid literal; last read: '\"c\": @'");
check_error(R"({"a":1,"b":[true,false],"c":@})", 29,
"[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing value - invalid literal; last read: '\"c\":@'");
}
#endif
SECTION("tests found by mutate++")
{
// test case to make sure no comma precedes the first key
@@ -22,6 +22,8 @@ using nlohmann::json;
#include <valarray>
#include "test_utils.hpp"
namespace
{
class SaxEventLogger
@@ -629,7 +631,8 @@ TEST_CASE("parser class")
SECTION("overflow")
{
// overflows during parsing yield an exception
CHECK_THROWS_WITH_AS(parser_helper("1.18973e+4932").empty(), "[json.exception.out_of_range.406] number overflow parsing '1.18973e+4932'", json::out_of_range&);
// empty() is nodiscard; the exception is thrown by parser_helper() itself, before empty() would run
CHECK_THROWS_WITH_AS(utils::ignore_return_value(parser_helper("1.18973e+4932").empty()), "[json.exception.out_of_range.406] number overflow parsing '1.18973e+4932'", json::out_of_range&);
}
SECTION("invalid numbers")
+31
View File
@@ -1389,6 +1389,37 @@ TEST_CASE("value conversion")
// CHECK(m5["one"] == "eins");
}
SECTION("reserve is called on containers that support it (#5406)")
{
// build a larger object so that a missing/incorrect reserve()
// call would be more likely to corrupt or drop elements
json j_large;
for (int i = 0; i < 100; ++i)
{
j_large[std::to_string(i)] = i;
}
SECTION("std::unordered_map (supports reserve)")
{
const auto m = j_large.get<std::unordered_map<std::string, int>>();
CHECK(m.size() == 100);
for (int i = 0; i < 100; ++i)
{
CHECK(m.at(std::to_string(i)) == i);
}
}
SECTION("std::map (no reserve, fallback path)")
{
const auto m = j_large.get<std::map<std::string, int>>();
CHECK(m.size() == 100);
for (int i = 0; i < 100; ++i)
{
CHECK(m.at(std::to_string(i)) == i);
}
}
}
SECTION("std::multimap")
{
j1.get<std::multimap<std::string, int>>();
+42
View File
@@ -0,0 +1,42 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
// This file contains the C++17-only part of unit-items.cpp (structured
// bindings support for json::items()). It is kept in a separate
// translation unit so the (much larger) unit-items.cpp does not need to
// be compiled a second time just for this one SECTION.
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
using nlohmann::json;
#ifdef JSON_HAS_CPP_17
#include <map>
#include <string>
TEST_CASE("items()")
{
SECTION("object")
{
SECTION("structured bindings")
{
json j = { {"A", 1}, {"B", 2} };
std::map<std::string, int> m;
for (auto const&[key, value] : j.items())
{
m.emplace(key, value);
}
CHECK(j.get<decltype(m)>() == m);
}
}
}
#endif
-16
View File
@@ -862,22 +862,6 @@ TEST_CASE("items()")
CHECK(counter == 3);
}
#ifdef JSON_HAS_CPP_17
SECTION("structured bindings")
{
json j = { {"A", 1}, {"B", 2} };
std::map<std::string, int> m;
for (auto const&[key, value] : j.items())
{
m.emplace(key, value);
}
CHECK(j.get<decltype(m)>() == m);
}
#endif
}
SECTION("const object")
+186
View File
@@ -1389,6 +1389,192 @@ TEST_CASE("JSON patch - add to a primitive parent (regression #4292)")
}
}
TEST_CASE("JSON patch - remove with primitive or null parent (regression #5396)")
{
// Regression test for https://github.com/nlohmann/json/issues/5396
//
// RFC 6902 (§4.2) requires the target location of a "remove" operation
// to exist. When the target's parent resolves to a primitive value or
// null, the operation must fail. Previously operation_remove silently
// did nothing in this case (neither the "is_object" nor the "is_array"
// branch matched, and there was no final "else"), so the patch appeared
// to succeed without changing the document. It now throws
// out_of_range.413.
SECTION("parent is a primitive (number)")
{
json const doc = {{"a", 1}};
json const patch = {{{"op", "remove"}, {"path", "/a/b"}}};
#if JSON_DIAGNOSTICS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] (/a) cannot remove value: the JSON Patch 'remove' target's parent is of type number, but must be an object or array", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] cannot remove value: the JSON Patch 'remove' target's parent is of type number, but must be an object or array", json::out_of_range&);
#endif
}
SECTION("parent is a primitive (string)")
{
json const doc = {{"foo", {{"bar", "a string"}}}};
json const patch = {{{"op", "remove"}, {"path", "/foo/bar/baz"}}};
#if JSON_DIAGNOSTICS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] (/foo/bar) cannot remove value: the JSON Patch 'remove' target's parent is of type string, but must be an object or array", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] cannot remove value: the JSON Patch 'remove' target's parent is of type string, but must be an object or array", json::out_of_range&);
#endif
}
SECTION("top-level document is null")
{
json const doc = nullptr;
json const patch = {{{"op", "remove"}, {"path", "/a"}}};
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.413] cannot remove value: the JSON Patch 'remove' target's parent is of type null, but must be an object or array", json::out_of_range&);
}
SECTION("legitimate removes still work")
{
// object member
json const doc1 = {{"a", 1}, {"b", 2}};
json const patch1 = {{{"op", "remove"}, {"path", "/a"}}};
CHECK(doc1.patch(patch1) == json({{"b", 2}}));
// array element
json const doc2 = R"([1, 2, 3])"_json;
json const patch2 = {{{"op", "remove"}, {"path", "/1"}}};
CHECK(doc2.patch(patch2) == R"([1, 3])"_json);
}
}
TEST_CASE("JSON patch - move where 'from' is a proper prefix of 'path' (regression #5397)")
{
// Regression test for https://github.com/nlohmann/json/issues/5397
//
// RFC 6902 (§4.4) forbids "from" from being a proper prefix of "path"
// for a "move" operation: "a location cannot be moved into one of its
// children." "move" is implemented as remove-then-add; for an object
// target this happened to throw anyway as a side effect of the "add"
// step re-resolving through the now-removed parent, but for an array
// target the removal shifted subsequent indices, so "path" silently
// re-resolved to a different element and the operation "succeeded"
// with a corrupted result. It now throws out_of_range.414 for both
// object and array targets.
SECTION("array target (from the issue)")
{
json const doc = R"([[1,2],[3]])"_json;
json const patch = {{{"op", "move"}, {"from", "/0"}, {"path", "/0/0"}}};
#if JSON_DIAGNOSTIC_POSITIONS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-11) cannot move value: 'from' path '/0' is a proper prefix of 'path' '/0/0'", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '/0' is a proper prefix of 'path' '/0/0'", json::out_of_range&);
#endif
}
SECTION("object target")
{
json const doc = R"({"a": {"b": 1}})"_json;
json const patch = {{{"op", "move"}, {"from", "/a"}, {"path", "/a/b"}}};
#if JSON_DIAGNOSTIC_POSITIONS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-15) cannot move value: 'from' path '/a' is a proper prefix of 'path' '/a/b'", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '/a' is a proper prefix of 'path' '/a/b'", json::out_of_range&);
#endif
}
SECTION("from == path is not a proper prefix and must not be rejected")
{
// "from" equal to "path" is a no-op move; it is not a *proper*
// prefix relationship, so this new check must not reject it.
json const doc = R"({"a": 1, "b": 2})"_json;
json const patch = {{{"op", "move"}, {"from", "/a"}, {"path", "/a"}}};
CHECK(doc.patch(patch) == doc);
}
SECTION("raw string prefix that is not a pointer-token prefix must be allowed")
{
// "/ab" is a string-prefix of "/abc/x" as raw text, but "ab" and
// "abc" are different reference tokens, so this is NOT a
// pointer-token prefix relationship and the move must succeed.
// This is the key case proving the check compares tokens, not
// raw pointer text (a naive std::string prefix/rfind check on
// the undecoded pointer would wrongly reject this).
json const doc = R"({"ab": 1, "abc": {"x": 2}})"_json;
json const patch = {{{"op", "move"}, {"from", "/ab"}, {"path", "/abc/x"}}};
json const result = R"({"abc": {"x": 1}})"_json;
CHECK(doc.patch(patch) == result);
}
SECTION("escaped reference tokens are compared unescaped")
{
// "from" is the single token "a/b" (escaped as "a~1b"); "path"
// addresses member "x" of that same value, so "from" is a
// proper (token-level) prefix of "path" and must be rejected.
json const doc = R"({"a/b": {"x": 1}})"_json;
json const patch = {{{"op", "move"}, {"from", "/a~1b"}, {"path", "/a~1b/x"}}};
#if JSON_DIAGNOSTIC_POSITIONS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-17) cannot move value: 'from' path '/a~1b' is a proper prefix of 'path' '/a~1b/x'", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '/a~1b' is a proper prefix of 'path' '/a~1b/x'", json::out_of_range&);
#endif
}
SECTION("ordinary valid moves still work")
{
// unrelated top-level members
json const doc1 = R"({"a": 1, "b": 2})"_json;
json const patch1 = {{{"op", "move"}, {"from", "/a"}, {"path", "/c"}}};
CHECK(doc1.patch(patch1) == R"({"b": 2, "c": 1})"_json);
// sibling paths that share a textual prefix but are unrelated
json const doc2 = R"({"a": {"x": 1}, "b": {"y": 2}})"_json;
json const patch2 = {{{"op", "move"}, {"from", "/a/x"}, {"path", "/b/z"}}};
CHECK(doc2.patch(patch2) == R"({"a": {}, "b": {"y": 2, "z": 1}})"_json);
// "path" is a proper prefix of "from" (the reverse relationship,
// which RFC 6902 does not forbid)
json const doc3 = R"({"a": {"b": 1}})"_json;
json const patch3 = {{{"op", "move"}, {"from", "/a/b"}, {"path", "/a"}}};
CHECK(doc3.patch(patch3) == R"({"a": 1})"_json);
}
SECTION("root 'from' is a proper prefix of every non-root 'path'")
{
// the whole document is a proper prefix of any location inside it
json const doc = R"({"a": 1})"_json;
json const patch = {{{"op", "move"}, {"from", ""}, {"path", "/a"}}};
#if JSON_DIAGNOSTIC_POSITIONS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-8) cannot move value: 'from' path '' is a proper prefix of 'path' '/a'", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '' is a proper prefix of 'path' '/a'", json::out_of_range&);
#endif
}
SECTION("root 'path' is never a proper prefix violation for a non-root 'from'")
{
// the reverse of the above: moving a non-root location to the root
// is the "path is a prefix of from" relationship, which RFC 6902
// permits (already covered generally above; this pins the root
// case specifically, since root is the one path with no reference
// tokens at all)
json const doc = R"({"a": {"b": 1}})"_json;
json const patch = {{{"op", "move"}, {"from", "/a"}, {"path", ""}}};
CHECK(doc.patch(patch) == R"({"b": 1})"_json);
}
SECTION("the array-append token '-' is an ordinary child token")
{
// "-" (append-to-array) addresses a location *inside* the array,
// so "from" pointing at the array is still a proper prefix of
// "path" ending in "-" and must be rejected like any other child.
json const doc = R"({"a": [1, 2]})"_json;
json const patch = {{{"op", "move"}, {"from", "/a"}, {"path", "/a/-"}}};
#if JSON_DIAGNOSTIC_POSITIONS
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] (bytes 0-13) cannot move value: 'from' path '/a' is a proper prefix of 'path' '/a/-'", json::out_of_range&);
#else
CHECK_THROWS_WITH_AS(doc.patch(patch), "[json.exception.out_of_range.414] cannot move value: 'from' path '/a' is a proper prefix of 'path' '/a/-'", json::out_of_range&);
#endif
}
}
TEST_CASE("JSON patch - diff emits array removals in descending index order")
{
SECTION("array shrunk to empty")
+42
View File
@@ -319,6 +319,44 @@ TEST_CASE("JSON pointers")
CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&);
// #5395: contains() must not throw for a reference token that is a
// syntactically valid array index but numerically exceeds ULLONG_MAX
// (causing strtoull() to set errno to ERANGE) -- it should just report
// that the pointer does not resolve to an element
CHECK(!j.contains(jp));
CHECK(!j_const.contains(jp));
}
{
// #5395: same as above, but using the exact reproduction from the issue
json::json_pointer const jp("/99999999999999999999");
std::string const throw_msg = "[json.exception.out_of_range.404] unresolved reference token '99999999999999999999'";
CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j.at(jp) = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const.at(jp) == 1, throw_msg.c_str(), json::out_of_range&);
CHECK(!j.contains(jp));
CHECK(!j_const.contains(jp));
}
{
// #5395: a reference token that is numerically representable in
// unsigned long long but exceeds size_type's max (e.g. ULLONG_MAX
// itself on typical 64-bit platforms, where size_type's max equals
// ULLONG_MAX) must not make contains() throw either
json::json_pointer const jp("/18446744073709551615");
std::string const throw_msg = "[json.exception.out_of_range.410] array index 18446744073709551615 exceeds size_type";
CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j.at(jp) = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const.at(jp) == 1, throw_msg.c_str(), json::out_of_range&);
CHECK(!j.contains(jp));
CHECK(!j_const.contains(jp));
}
// on some machines, the check below is not constant
@@ -334,6 +372,10 @@ TEST_CASE("JSON pointers")
CHECK_THROWS_WITH_AS(j[jp] = 1, throw_msg.c_str(), json::out_of_range&);
CHECK_THROWS_WITH_AS(j_const[jp] == 1, throw_msg.c_str(), json::out_of_range&);
// #5395: contains() must not throw for a reference token exceeding size_type's max
CHECK(!j.contains(jp));
CHECK(!j_const.contains(jp));
}
DOCTEST_MSVC_SUPPRESS_WARNING_POP
+34
View File
@@ -0,0 +1,34 @@
// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-License-Identifier: MIT
// This file makes sure that none of the internal JSON_HEDLEY_* macros (vendored
// from https://nemequ.github.io/hedley/, see
// include/nlohmann/thirdparty/hedley/hedley.hpp) leak into the including
// translation unit. include/nlohmann/detail/macro_unscope.hpp is supposed to
// #undef every JSON_HEDLEY_* macro (via hedley_undef.hpp) once json.hpp has
// been fully processed. See https://github.com/nlohmann/json/issues/5408,
// where JSON_HEDLEY_PRAGMA, JSON_HEDLEY_PREDICT_TRUE, JSON_HEDLEY_PREDICT_FALSE,
// and JSON_HEDLEY_CLANG_HAS_DECLSPEC_ATTRIBUTE escaped this cleanup because
// hedley_undef.hpp had no matching #undef for them.
//
// hedley_undef_checks.inc (included below) is generated at CMake configure/
// build time by cmake/scripts/gen_hedley_undef_check.cmake, which derives the
// full list of JSON_HEDLEY_* macro names directly from hedley.hpp. That way
// this test covers every macro Hedley actually defines -- not a hardcoded
// snapshot that would silently go stale the next time `make update_hedley`
// runs -- and can never drift from the vendored header.
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
TEST_CASE("JSON_HEDLEY macros do not leak after including json.hpp")
{
#include "hedley_undef_checks.inc"
CHECK(true); // keep an assertion when nothing leaked
}
+3 -5
View File
@@ -29,10 +29,7 @@ using nlohmann::json;
#include <limits>
#include <cstdio>
#include "make_test_data_available.hpp"
#ifdef JSON_HAS_CPP_17
#include <variant>
#endif
#include "test_utils.hpp"
#include "fifo_map.hpp"
@@ -1373,7 +1370,8 @@ TEST_CASE("regression tests 1")
std::array<uint8_t, 28> key1 = {{ 103, 92, 117, 48, 48, 48, 55, 92, 114, 215, 126, 214, 95, 92, 34, 174, 40, 71, 38, 174, 40, 71, 38, 223, 134, 247, 127, 0 }};
std::string const key1_str(reinterpret_cast<char*>(key1.data()));
json const j = key1_str;
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 10: 0x7E", json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 10: 0x7E", json::type_error&);
}
#if JSON_USE_IMPLICIT_CONVERSIONS
+11 -6
View File
@@ -31,6 +31,8 @@ using ordered_json = nlohmann::ordered_json;
#include <type_traits>
#include <utility>
#include "test_utils.hpp"
#ifdef JSON_HAS_CPP_17
#include <any>
#include <variant>
@@ -639,7 +641,8 @@ TEST_CASE("regression tests 2")
s += static_cast<char>(i);
}
dump_test["1"] = s;
dump_test.dump(-1, ' ', true, nlohmann::json::error_handler_t::replace);
// dump() is nodiscard; this only checks that dumping does not throw/crash
utils::ignore_return_value(dump_test.dump(-1, ' ', true, nlohmann::json::error_handler_t::replace));
}
}
@@ -731,12 +734,14 @@ TEST_CASE("regression tests 2")
{
const std::array<unsigned char, 23> data = {{0x81, 0xA4, 0x64, 0x61, 0x74, 0x61, 0xC4, 0x0F, 0x33, 0x30, 0x30, 0x32, 0x33, 0x34, 0x30, 0x31, 0x30, 0x37, 0x30, 0x35, 0x30, 0x31, 0x30}};
const json j = json::from_msgpack(data.data(), data.size());
// dump() is nodiscard; this only checks that dumping does not throw
CHECK_NOTHROW(
j.dump(4, // Indent
' ', // Indent char
false, // Ensure ascii
json::error_handler_t::strict // Error
));
utils::ignore_return_value(
j.dump(4, // Indent
' ', // Indent char
false, // Ensure ascii
json::error_handler_t::strict // Error
)));
}
SECTION("PR #2181 - regression bug with lvalue")
+11 -6
View File
@@ -15,6 +15,8 @@ using nlohmann::json;
#include <sstream>
#include <iomanip>
#include "test_utils.hpp"
TEST_CASE("serialization")
{
SECTION("operator<<")
@@ -84,8 +86,9 @@ TEST_CASE("serialization")
{
const json j = "ä\xA9ü";
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&);
CHECK_THROWS_WITH_AS(j.dump(1, ' ', false, json::error_handler_t::strict), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&);
CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump(1, ' ', false, json::error_handler_t::strict)), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&);
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"äü\"");
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"ä\xEF\xBF\xBDü\"");
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"\\u00e4\\ufffd\\u00fc\"");
@@ -95,8 +98,9 @@ TEST_CASE("serialization")
{
const json j = "123\xC2";
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] incomplete UTF-8 string; last byte: 0xC2", json::type_error&);
CHECK_THROWS_AS(j.dump(1, ' ', false, json::error_handler_t::strict), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] incomplete UTF-8 string; last byte: 0xC2", json::type_error&);
CHECK_THROWS_AS(utils::ignore_return_value(j.dump(1, ' ', false, json::error_handler_t::strict)), json::type_error&);
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123\"");
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\"");
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd\"");
@@ -106,8 +110,9 @@ TEST_CASE("serialization")
{
const json j = "123\xF1\xB0\x34\x35\x36";
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0x34", json::type_error&);
CHECK_THROWS_AS(j.dump(1, ' ', false, json::error_handler_t::strict), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0x34", json::type_error&);
CHECK_THROWS_AS(utils::ignore_return_value(j.dump(1, ' ', false, json::error_handler_t::strict)), json::type_error&);
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123456\"");
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\x34\x35\x36\"");
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd456\"");
+5 -2
View File
@@ -17,6 +17,7 @@ using nlohmann::json;
#include <sstream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
TEST_CASE("Unicode (1/5)" * doctest::skip())
{
@@ -240,7 +241,8 @@ void roundtrip(bool success_expected, const std::string& s)
if (success_expected)
{
// serialization succeeds
CHECK_NOTHROW(j.dump());
// dump() is nodiscard; this only checks that dumping does not throw
CHECK_NOTHROW(utils::ignore_return_value(j.dump()));
// exclude parse test for U+0000
if (s[0] != '\0')
@@ -259,7 +261,8 @@ void roundtrip(bool success_expected, const std::string& s)
else
{
// serialization fails
CHECK_THROWS_AS(j.dump(), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// parsing JSON text fails
CHECK_THROWS_AS(_ = json::parse(ps), json::parse_error&);
+3 -1
View File
@@ -19,6 +19,7 @@ using nlohmann::json;
#include <iostream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
// this test suite uses static variables with non-trivial destructors
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
@@ -97,7 +98,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
else
{
// strict mode must throw if success is not expected
CHECK_THROWS_AS(j.dump(), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// ignore and replace must create different dumps
CHECK(s_ignored != s_replaced);
+3 -1
View File
@@ -19,6 +19,7 @@ using nlohmann::json;
#include <iostream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
// this test suite uses static variables with non-trivial destructors
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
@@ -97,7 +98,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
else
{
// strict mode must throw if success is not expected
CHECK_THROWS_AS(j.dump(), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// ignore and replace must create different dumps
CHECK(s_ignored != s_replaced);
+3 -1
View File
@@ -19,6 +19,7 @@ using nlohmann::json;
#include <iostream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
// this test suite uses static variables with non-trivial destructors
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
@@ -97,7 +98,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
else
{
// strict mode must throw if success is not expected
CHECK_THROWS_AS(j.dump(), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// ignore and replace must create different dumps
CHECK(s_ignored != s_replaced);
+3 -1
View File
@@ -19,6 +19,7 @@ using nlohmann::json;
#include <iostream>
#include <iomanip>
#include "make_test_data_available.hpp"
#include "test_utils.hpp"
// this test suite uses static variables with non-trivial destructors
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
@@ -97,7 +98,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3
else
{
// strict mode must throw if success is not expected
CHECK_THROWS_AS(j.dump(), json::type_error&);
// dump() is nodiscard; the exception is thrown by dump() itself before it would return
CHECK_THROWS_AS(utils::ignore_return_value(j.dump()), json::type_error&);
// ignore and replace must create different dumps
CHECK(s_ignored != s_replaced);