diff --git a/.github/CONTRIBUTING.md b/.github/CONTRIBUTING.md index 9c3bb2c04..95d804ada 100644 --- a/.github/CONTRIBUTING.md +++ b/.github/CONTRIBUTING.md @@ -111,6 +111,8 @@ context which existing file needs to be extended, and only very few cases requir When fixing a bug, edit `unit-regression3.cpp` and add a section referencing the fixed issue. `unit-regression2.cpp` holds the older tests; the two files exist because a single one grew large enough for the MinGW linker to fail relocating it, so please keep adding to the smaller file rather than growing the larger one. +Regression tests that call `sax_parse` go into `unit-sax_parse.cpp` instead: every call instantiates the parser and +the binary reader that recover from errors, which grows a test file considerably. #### Exceptions diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index 4428add5b..fc503b897 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -483,6 +483,7 @@ class binary_reader /// @copydoc skip_unsupported_bson_element template < bool Recover = AllowRecovery, enable_if_t < !Recover, int > = 0 > + JSON_HEDLEY_ALWAYS_INLINE constexpr std::false_type skip_unsupported_bson_element(const char_int_type /*element_type*/) const noexcept { return {}; @@ -3905,6 +3906,7 @@ class binary_reader /// @copydoc recover_high_precision_number template < bool Recover = AllowRecovery, enable_if_t < !Recover, int > = 0 > + JSON_HEDLEY_ALWAYS_INLINE constexpr std::false_type recover_high_precision_number(const std::vector& /*number_vector*/, const std::size_t /*remaining*/ = 0) const noexcept { return {}; @@ -5014,6 +5016,7 @@ class binary_reader @return false, so that the caller stops reading */ template + JSON_HEDLEY_ALWAYS_INLINE bool report_error(const std::size_t position, const std::string& last_token, const Exception& ex) { close_requested = sax->parse_error(position, last_token, ex); @@ -5050,6 +5053,7 @@ class binary_reader /// @copydoc repair_requested template < bool Recover = AllowRecovery, enable_if_t < !Recover, int > = 0 > + JSON_HEDLEY_ALWAYS_INLINE constexpr std::false_type repair_requested() const noexcept { return {}; @@ -5074,6 +5078,7 @@ class binary_reader /// @copydoc value_failed template < bool Recover = AllowRecovery, enable_if_t < !Recover, int > = 0 > + JSON_HEDLEY_ALWAYS_INLINE constexpr std::false_type value_failed() const noexcept { return {}; @@ -5124,6 +5129,7 @@ class binary_reader /// @copydoc close_open_containers template < bool Recover = AllowRecovery, enable_if_t < !Recover, int > = 0 > + JSON_HEDLEY_ALWAYS_INLINE void close_open_containers() const noexcept {} /*! @@ -5159,6 +5165,7 @@ class binary_reader /// @copydoc resync template < bool Recover = AllowRecovery, enable_if_t < !Recover, int > = 0 > + JSON_HEDLEY_ALWAYS_INLINE constexpr std::false_type resync() const noexcept { return {}; diff --git a/include/nlohmann/detail/input/parser.hpp b/include/nlohmann/detail/input/parser.hpp index 3eb54ced2..f27e08522 100644 --- a/include/nlohmann/detail/input/parser.hpp +++ b/include/nlohmann/detail/input/parser.hpp @@ -430,7 +430,7 @@ class parser } // recover: keep what can be read of the token - recover_token(); + recover_token(allow_recovery); if (last_token != token_type::uninitialized) { // a string or a number @@ -439,7 +439,7 @@ class parser if (states.empty()) { // look for the value after the garbage - if (!skip_to_value()) + if (!skip_to_value(allow_recovery)) { return false; } @@ -473,12 +473,12 @@ class parser // there is no value return false; } - if (!recover_missing_value(sax, states)) + if (!recover_missing_value(sax, states, allow_recovery)) { return false; } // the state evaluation reads the token again - m_lexer.unget_token(); + unget_token(allow_recovery); skip_to_state_evaluation = true; continue; } @@ -499,7 +499,7 @@ class parser if (states.empty()) { // look for the value after the garbage - if (!skip_to_value()) + if (!skip_to_value(allow_recovery)) { return false; } @@ -511,12 +511,12 @@ class parser get_token(); continue; } - if (!recover_missing_value(sax, states)) + if (!recover_missing_value(sax, states, allow_recovery)) { return false; } // the state evaluation reads the token again - m_lexer.unget_token(); + unget_token(allow_recovery); skip_to_state_evaluation = true; continue; } @@ -578,7 +578,7 @@ class parser if (last_token == token_type::end_of_input) { // the input ends inside the array - return close_containers(sax, states); + return close_containers(sax, states, allow_recovery); } if (last_token == token_type::end_object) { @@ -664,7 +664,7 @@ class parser if (last_token == token_type::end_of_input) { // the input ends inside the object - return close_containers(sax, states); + return close_containers(sax, states, allow_recovery); } if (last_token == token_type::end_array) { @@ -700,6 +700,7 @@ class parser } /// the parser for parse() and accept() never recovers: stop parsing + JSON_HEDLEY_ALWAYS_INLINE static std::false_type continue_after(std::false_type /*step*/, bool& /*skip_to_state_evaluation*/) noexcept { return {}; @@ -755,8 +756,8 @@ class parser @param[in] value the value that is not finite @return whether to continue parsing */ - template - bool overflow_error(SAX* sax, const number_float_t value, AllowRecovery allow_recovery) + template + bool overflow_error(SAX* sax, const number_float_t value, std::true_type allow_recovery) { if (!report_error(sax, out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr), allow_recovery)) { @@ -765,6 +766,14 @@ class parser return sax->number_float(value, m_lexer.get_string()); } + /// @copydoc overflow_error + template + JSON_HEDLEY_ALWAYS_INLINE + std::false_type overflow_error(SAX* sax, const number_float_t /*value*/, std::false_type allow_recovery) + { + return report_error(sax, out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr), allow_recovery); + } + /*! @brief report a missing key, or a missing name separator (:) after the key; the parser for parse() and accept() never recovers @@ -774,6 +783,7 @@ class parser @return std::false_type, see report_error() */ template + JSON_HEDLEY_ALWAYS_INLINE std::false_type key_error(SAX* sax, std::false_type allow_recovery, const bool key_read) { return report_error(sax, parse_error::create(101, m_lexer.get_position(), key_read @@ -838,6 +848,7 @@ class parser for recovering is not generated */ template + JSON_HEDLEY_ALWAYS_INLINE std::false_type report_error(SAX* sax, const Exception& ex, std::false_type /*allow_recovery*/) { error_reported = true; @@ -888,9 +899,54 @@ class parser return last_token; } + // The functions below are called from sax_parse_internal() after an error + // was reported. The parser for parse() and accept() stopped there, so for + // it they are stubs: the code for recovering is then not even referenced, + // and unoptimized builds do not emit it either. + + /// @copydoc recover_token() + JSON_HEDLEY_ALWAYS_INLINE + token_type recover_token(std::true_type /*allow_recovery*/) + { + return recover_token(); + } + + /// @copydoc recover_token() + JSON_HEDLEY_ALWAYS_INLINE + static std::false_type recover_token(std::false_type /*allow_recovery*/) noexcept + { + return {}; + } + + /// return the token that was read last to the lexer (see lexer::unget_token()) + JSON_HEDLEY_ALWAYS_INLINE + void unget_token(std::true_type /*allow_recovery*/) + { + m_lexer.unget_token(); + } + + /// @copydoc unget_token(std::true_type) + JSON_HEDLEY_ALWAYS_INLINE + static void unget_token(std::false_type /*allow_recovery*/) noexcept {} + + /// @copydoc skip_to_value + JSON_HEDLEY_ALWAYS_INLINE + static std::false_type skip_to_value(std::false_type /*allow_recovery*/) noexcept + { + return {}; + } + + /// @copydoc recover_missing_value + template + JSON_HEDLEY_ALWAYS_INLINE + static std::false_type recover_missing_value(SAX* /*sax*/, const std::vector& /*states*/, std::false_type /*allow_recovery*/) noexcept + { + return {}; + } + /// pass the end events of all open containers template - bool close_containers(SAX* sax, std::vector& states) + bool close_containers(SAX* sax, std::vector& states, std::true_type /*allow_recovery*/) { while (!states.empty()) { @@ -904,12 +960,20 @@ class parser return true; } + /// @copydoc close_containers + template + JSON_HEDLEY_ALWAYS_INLINE + static std::false_type close_containers(SAX* /*sax*/, std::vector& /*states*/, std::false_type /*allow_recovery*/) noexcept + { + return {}; + } + /*! @brief read tokens until one begins a value, skipping everything before the top-level value @return whether a value begins with last_token */ - bool skip_to_value() + bool skip_to_value(std::true_type /*allow_recovery*/) { while (true) { @@ -1017,7 +1081,7 @@ class parser ends there just ends. */ template - bool recover_missing_value(SAX* sax, const std::vector& states) + bool recover_missing_value(SAX* sax, const std::vector& states, std::true_type /*allow_recovery*/) { JSON_ASSERT(!states.empty()); if (!states.back() || last_token == token_type::value_separator) @@ -1123,6 +1187,7 @@ class parser /// the parser for parse() and accept() never recovers (and does not come /// here, as report_error() returned false) template + JSON_HEDLEY_ALWAYS_INLINE std::false_type recover_member(SAX* /*sax*/, std::false_type /*allow_recovery*/) const noexcept { return {}; diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 026c97175..16e8d2684 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -14777,6 +14777,7 @@ class binary_reader /// @copydoc skip_unsupported_bson_element template < bool Recover = AllowRecovery, enable_if_t < !Recover, int > = 0 > + JSON_HEDLEY_ALWAYS_INLINE constexpr std::false_type skip_unsupported_bson_element(const char_int_type /*element_type*/) const noexcept { return {}; @@ -18199,6 +18200,7 @@ class binary_reader /// @copydoc recover_high_precision_number template < bool Recover = AllowRecovery, enable_if_t < !Recover, int > = 0 > + JSON_HEDLEY_ALWAYS_INLINE constexpr std::false_type recover_high_precision_number(const std::vector& /*number_vector*/, const std::size_t /*remaining*/ = 0) const noexcept { return {}; @@ -19308,6 +19310,7 @@ class binary_reader @return false, so that the caller stops reading */ template + JSON_HEDLEY_ALWAYS_INLINE bool report_error(const std::size_t position, const std::string& last_token, const Exception& ex) { close_requested = sax->parse_error(position, last_token, ex); @@ -19344,6 +19347,7 @@ class binary_reader /// @copydoc repair_requested template < bool Recover = AllowRecovery, enable_if_t < !Recover, int > = 0 > + JSON_HEDLEY_ALWAYS_INLINE constexpr std::false_type repair_requested() const noexcept { return {}; @@ -19368,6 +19372,7 @@ class binary_reader /// @copydoc value_failed template < bool Recover = AllowRecovery, enable_if_t < !Recover, int > = 0 > + JSON_HEDLEY_ALWAYS_INLINE constexpr std::false_type value_failed() const noexcept { return {}; @@ -19418,6 +19423,7 @@ class binary_reader /// @copydoc close_open_containers template < bool Recover = AllowRecovery, enable_if_t < !Recover, int > = 0 > + JSON_HEDLEY_ALWAYS_INLINE void close_open_containers() const noexcept {} /*! @@ -19453,6 +19459,7 @@ class binary_reader /// @copydoc resync template < bool Recover = AllowRecovery, enable_if_t < !Recover, int > = 0 > + JSON_HEDLEY_ALWAYS_INLINE constexpr std::false_type resync() const noexcept { return {}; @@ -20207,7 +20214,7 @@ class parser } // recover: keep what can be read of the token - recover_token(); + recover_token(allow_recovery); if (last_token != token_type::uninitialized) { // a string or a number @@ -20216,7 +20223,7 @@ class parser if (states.empty()) { // look for the value after the garbage - if (!skip_to_value()) + if (!skip_to_value(allow_recovery)) { return false; } @@ -20250,12 +20257,12 @@ class parser // there is no value return false; } - if (!recover_missing_value(sax, states)) + if (!recover_missing_value(sax, states, allow_recovery)) { return false; } // the state evaluation reads the token again - m_lexer.unget_token(); + unget_token(allow_recovery); skip_to_state_evaluation = true; continue; } @@ -20276,7 +20283,7 @@ class parser if (states.empty()) { // look for the value after the garbage - if (!skip_to_value()) + if (!skip_to_value(allow_recovery)) { return false; } @@ -20288,12 +20295,12 @@ class parser get_token(); continue; } - if (!recover_missing_value(sax, states)) + if (!recover_missing_value(sax, states, allow_recovery)) { return false; } // the state evaluation reads the token again - m_lexer.unget_token(); + unget_token(allow_recovery); skip_to_state_evaluation = true; continue; } @@ -20355,7 +20362,7 @@ class parser if (last_token == token_type::end_of_input) { // the input ends inside the array - return close_containers(sax, states); + return close_containers(sax, states, allow_recovery); } if (last_token == token_type::end_object) { @@ -20441,7 +20448,7 @@ class parser if (last_token == token_type::end_of_input) { // the input ends inside the object - return close_containers(sax, states); + return close_containers(sax, states, allow_recovery); } if (last_token == token_type::end_array) { @@ -20477,6 +20484,7 @@ class parser } /// the parser for parse() and accept() never recovers: stop parsing + JSON_HEDLEY_ALWAYS_INLINE static std::false_type continue_after(std::false_type /*step*/, bool& /*skip_to_state_evaluation*/) noexcept { return {}; @@ -20532,8 +20540,8 @@ class parser @param[in] value the value that is not finite @return whether to continue parsing */ - template - bool overflow_error(SAX* sax, const number_float_t value, AllowRecovery allow_recovery) + template + bool overflow_error(SAX* sax, const number_float_t value, std::true_type allow_recovery) { if (!report_error(sax, out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr), allow_recovery)) { @@ -20542,6 +20550,14 @@ class parser return sax->number_float(value, m_lexer.get_string()); } + /// @copydoc overflow_error + template + JSON_HEDLEY_ALWAYS_INLINE + std::false_type overflow_error(SAX* sax, const number_float_t /*value*/, std::false_type allow_recovery) + { + return report_error(sax, out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr), allow_recovery); + } + /*! @brief report a missing key, or a missing name separator (:) after the key; the parser for parse() and accept() never recovers @@ -20551,6 +20567,7 @@ class parser @return std::false_type, see report_error() */ template + JSON_HEDLEY_ALWAYS_INLINE std::false_type key_error(SAX* sax, std::false_type allow_recovery, const bool key_read) { return report_error(sax, parse_error::create(101, m_lexer.get_position(), key_read @@ -20615,6 +20632,7 @@ class parser for recovering is not generated */ template + JSON_HEDLEY_ALWAYS_INLINE std::false_type report_error(SAX* sax, const Exception& ex, std::false_type /*allow_recovery*/) { error_reported = true; @@ -20665,9 +20683,54 @@ class parser return last_token; } + // The functions below are called from sax_parse_internal() after an error + // was reported. The parser for parse() and accept() stopped there, so for + // it they are stubs: the code for recovering is then not even referenced, + // and unoptimized builds do not emit it either. + + /// @copydoc recover_token() + JSON_HEDLEY_ALWAYS_INLINE + token_type recover_token(std::true_type /*allow_recovery*/) + { + return recover_token(); + } + + /// @copydoc recover_token() + JSON_HEDLEY_ALWAYS_INLINE + static std::false_type recover_token(std::false_type /*allow_recovery*/) noexcept + { + return {}; + } + + /// return the token that was read last to the lexer (see lexer::unget_token()) + JSON_HEDLEY_ALWAYS_INLINE + void unget_token(std::true_type /*allow_recovery*/) + { + m_lexer.unget_token(); + } + + /// @copydoc unget_token(std::true_type) + JSON_HEDLEY_ALWAYS_INLINE + static void unget_token(std::false_type /*allow_recovery*/) noexcept {} + + /// @copydoc skip_to_value + JSON_HEDLEY_ALWAYS_INLINE + static std::false_type skip_to_value(std::false_type /*allow_recovery*/) noexcept + { + return {}; + } + + /// @copydoc recover_missing_value + template + JSON_HEDLEY_ALWAYS_INLINE + static std::false_type recover_missing_value(SAX* /*sax*/, const std::vector& /*states*/, std::false_type /*allow_recovery*/) noexcept + { + return {}; + } + /// pass the end events of all open containers template - bool close_containers(SAX* sax, std::vector& states) + bool close_containers(SAX* sax, std::vector& states, std::true_type /*allow_recovery*/) { while (!states.empty()) { @@ -20681,12 +20744,20 @@ class parser return true; } + /// @copydoc close_containers + template + JSON_HEDLEY_ALWAYS_INLINE + static std::false_type close_containers(SAX* /*sax*/, std::vector& /*states*/, std::false_type /*allow_recovery*/) noexcept + { + return {}; + } + /*! @brief read tokens until one begins a value, skipping everything before the top-level value @return whether a value begins with last_token */ - bool skip_to_value() + bool skip_to_value(std::true_type /*allow_recovery*/) { while (true) { @@ -20794,7 +20865,7 @@ class parser ends there just ends. */ template - bool recover_missing_value(SAX* sax, const std::vector& states) + bool recover_missing_value(SAX* sax, const std::vector& states, std::true_type /*allow_recovery*/) { JSON_ASSERT(!states.empty()); if (!states.back() || last_token == token_type::value_separator) @@ -20900,6 +20971,7 @@ class parser /// the parser for parse() and accept() never recovers (and does not come /// here, as report_error() returned false) template + JSON_HEDLEY_ALWAYS_INLINE std::false_type recover_member(SAX* /*sax*/, std::false_type /*allow_recovery*/) const noexcept { return {}; diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index 3ee9995e0..d084e6432 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -1029,605 +1029,6 @@ TEST_CASE("regression test - excessive binary container size honors allow_except CHECK(json::from_cbor(std::vector {0x9b, 0, 0, 0, 0, 0, 0, 0, 0x02}, true, false).is_discarded()); } -namespace -{ -/// builds a value from SAX events, asks the parser to recover from its first -/// 100 errors, and checks that the events are balanced (see #3989) -template -class BasicRecoveringParser -{ - public: - explicit BasicRecoveringParser(BasicJsonType& j) - : dom(j, false) - {} - - bool null() - { - value(); - return dom.null(); - } - - bool boolean(bool val) - { - value(); - return dom.boolean(val); - } - - bool number_integer(typename BasicJsonType::number_integer_t val) - { - value(); - return dom.number_integer(val); - } - - bool number_unsigned(typename BasicJsonType::number_unsigned_t val) - { - value(); - return dom.number_unsigned(val); - } - - bool number_float(typename BasicJsonType::number_float_t val, const std::string& s) - { - value(); - return dom.number_float(val, s); - } - - bool string(std::string& val) - { - value(); - return dom.string(val); - } - - bool binary(typename BasicJsonType::binary_t& val) - { - value(); - return dom.binary(val); - } - - bool start_object(std::size_t elements) - { - value(); - stack.push_back('o'); - return dom.start_object(elements); - } - - bool key(std::string& val) - { - if (stack.empty() || stack.back() != 'o') - { - well_formed = false; - return false; - } - stack.back() = 'v'; - return dom.key(val); - } - - bool end_object() - { - if (stack.empty() || stack.back() != 'o') - { - well_formed = false; - return false; - } - stack.pop_back(); - return dom.end_object(); - } - - bool start_array(std::size_t elements) - { - value(); - stack.push_back('a'); - return dom.start_array(elements); - } - - bool end_array() - { - if (stack.empty() || stack.back() != 'a') - { - well_formed = false; - return false; - } - stack.pop_back(); - return dom.end_array(); - } - - bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& ex) - { - messages.emplace_back(ex.what()); - // a limit, so that a reader that does not stop fails the test - // instead of making it hang - return ++errors < 100; - } - - /// whether the events were balanced and every key was followed by a value - bool balanced() const - { - return well_formed && stack.empty(); - } - - /// builds the value - nlohmann::detail::json_sax_dom_parser dom; - std::size_t errors = 0; - std::vector messages {}; // NOLINT(readability-redundant-member-init) - std::vector stack {}; // NOLINT(readability-redundant-member-init) - bool well_formed = true; - - private: - void value() - { - if (!stack.empty()) - { - if (stack.back() == 'v') - { - stack.back() = 'o'; - } - else if (stack.back() == 'o') - { - well_formed = false; - } - } - } - -}; - -using RecoveringParser = BasicRecoveringParser; - -struct BinaryParseResult -{ - json value; - std::size_t errors; - std::vector messages; - bool ok; - bool balanced; -}; - -BinaryParseResult parse_binary_recovering(const std::vector& input, const json::input_format_t format) -{ - json j; - RecoveringParser sax(j); - const bool ok = json::sax_parse(input, &sax, format); - return {j, sax.errors, sax.messages, ok, sax.balanced()}; -} - -#if !defined(JSON_NOEXCEPTION) -/// the message of the exception that reading @a input into a JSON value -/// throws, or an empty string if reading succeeds -std::string binary_error_message(const std::vector& input, const json::input_format_t format) -{ - try - { - json _; - switch (format) - { - case json::input_format_t::cbor: - _ = json::from_cbor(input); - break; - case json::input_format_t::msgpack: - _ = json::from_msgpack(input); - break; - case json::input_format_t::ubjson: - _ = json::from_ubjson(input); - break; - case json::input_format_t::bjdata: - _ = json::from_bjdata(input); - break; - case json::input_format_t::bson: - _ = json::from_bson(input); - break; - case json::input_format_t::bon8: - _ = json::from_bon8(input); - break; - case json::input_format_t::json: - default: - break; - } - } - catch (const json::exception& e) - { - return e.what(); - } - return ""; -} -#endif - -/// a BSON element: its type, its name, and its value -std::vector bson_element(const std::uint8_t type, const std::string& name, const std::vector& value) -{ - std::vector result = {type}; - result.insert(result.end(), name.begin(), name.end()); - result.push_back(0x00); - result.insert(result.end(), value.begin(), value.end()); - return result; -} - -/// a BSON document of the given elements; @a size_offset is added to the -/// size it declares -std::vector bson_document(const std::vector>& elements, const int size_offset = 0) -{ - std::vector body; - for (const auto& element : elements) - { - body.insert(body.end(), element.begin(), element.end()); - } - const auto size = static_cast(static_cast(body.size()) + 5 + size_offset); - std::vector result = {static_cast(size & 0xFFu), static_cast((size >> 8u) & 0xFFu), - static_cast((size >> 16u) & 0xFFu), static_cast((size >> 24u) & 0xFFu) - }; - result.insert(result.end(), body.begin(), body.end()); - result.push_back(0x00); - return result; -} - -/// a BSON int32 value -std::vector bson_int32(const std::int32_t value) -{ - const auto u = static_cast(value); - return {static_cast(u & 0xFFu), static_cast((u >> 8u) & 0xFFu), - static_cast((u >> 16u) & 0xFFu), static_cast((u >> 24u) & 0xFFu)}; -} - -/// a BSON string value, whose length is @a length_offset off -std::vector bson_string(const std::string& value, const std::int32_t length_offset = 0) -{ - auto result = bson_int32(static_cast(value.size() + 1) + length_offset); - result.insert(result.end(), value.begin(), value.end()); - result.push_back(0x00); - return result; -} - -/// @a count bytes of value 0xAB -std::vector bytes(const std::size_t count) -{ - return std::vector(count, 0xAB); -} - -template -std::vector concatenated(const std::vector& first, const Parts& ... rest) -{ - std::vector result = first; - for (const auto& part : std::initializer_list> {rest...}) - { - result.insert(result.end(), part.begin(), part.end()); - } - return result; -} - -/// U+FFFD REPLACEMENT CHARACTER -std::string replacement_character() -{ - return "\xEF\xBF\xBD"; -} -} // namespace - -TEST_CASE("regression test - #3989 SAX parse_error() returning true") -{ - SECTION("binary formats complete what was read before the input ends") - { - const json j = {{"a", {1, -2, {{"b", "c"}}, json::array()}}, {"d", {{"e", nullptr}, {"f", true}}}, {"g", 1.5}, {"h", json::binary({1, 2, 3})}}; - - const std::vector>> encodings = - { - {json::input_format_t::cbor, json::to_cbor(j)}, - {json::input_format_t::msgpack, json::to_msgpack(j)}, - {json::input_format_t::ubjson, json::to_ubjson(j)}, - {json::input_format_t::ubjson, json::to_ubjson(j, true, true)}, - {json::input_format_t::bjdata, json::to_bjdata(j)}, - {json::input_format_t::bjdata, json::to_bjdata(j, true, true)}, - {json::input_format_t::bson, json::to_bson(j)}, - {json::input_format_t::bon8, json::to_bon8(j)}, - }; - - for (const auto& encoding : encodings) - { - const auto format = encoding.first; - const auto& bytes = encoding.second; - CAPTURE(format) - - // every prefix is truncated input - for (std::size_t length = 0; length < bytes.size(); ++length) - { - CAPTURE(length) - const auto result = parse_binary_recovering(std::vector(bytes.begin(), bytes.begin() + static_cast(length)), format); - CHECK(!result.ok); - CHECK(result.errors == 1); - CHECK(result.balanced); - } - - // the complete input is read as usual (binary values do not - // round-trip through every format, so compare with a plain parse) - json expected; - nlohmann::detail::json_sax_dom_parser dom(expected); - CHECK(json::sax_parse(bytes, &dom, format)); - const auto complete = parse_binary_recovering(bytes, format); - CHECK(complete.ok); - CHECK(complete.errors == 0); - CHECK(complete.value == expected); - - // a byte after the value - auto trailing_bytes = bytes; - trailing_bytes.push_back(0x01); - const auto trailing = parse_binary_recovering(trailing_bytes, format); - CHECK(!trailing.ok); - CHECK(trailing.errors == 1); - CHECK(trailing.value == expected); - } - } - - SECTION("containers without an end") - { - // these made the readers loop, or read on, after the error - const auto cbor_array = parse_binary_recovering({0x9F}, json::input_format_t::cbor); - CHECK(cbor_array.errors == 1); - CHECK(cbor_array.value == json::array()); - - const auto cbor_map = parse_binary_recovering({0xBF, 0x61, 'a'}, json::input_format_t::cbor); - CHECK(cbor_map.errors == 1); - CHECK(cbor_map.value == json({{"a", nullptr}})); - - const auto msgpack_array = parse_binary_recovering({0xDD, 0xFF, 0xFF, 0xFF, 0xFF}, json::input_format_t::msgpack); - CHECK(msgpack_array.errors == 1); - CHECK(msgpack_array.value == json::array()); - - const auto msgpack_map = parse_binary_recovering({0x81, 0xA1, 'a', 0x92, 0x01}, json::input_format_t::msgpack); - CHECK(msgpack_map.errors == 1); - CHECK(msgpack_map.value == json({{"a", {1}}})); - } - - SECTION("BJData ndarray") - { - // a 2x3 int8 array with two of its six elements; the annotated array - // format opens an object and two arrays of its own - const auto result = parse_binary_recovering({'[', '$', 'i', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2}, json::input_format_t::bjdata); - CHECK(result.errors == 1); - CHECK(result.balanced); - CHECK(result.value == json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2}}})); - } - - SECTION("binary formats repair items whose end is known") - { - struct Repair - { - json::input_format_t format; - std::vector input; - json expected; - std::size_t errors; - }; - - const std::vector repairs = - { - // CBOR: tags are ignored (here tag 1 and the self-describe tag 55799) - {json::input_format_t::cbor, {0x82, 0xC1, 0x05, 0xD9, 0xD9, 0xF7, 0x06}, {5, 6}, 2}, - // CBOR: undefined and other simple values become null - {json::input_format_t::cbor, {0x84, 0xF7, 0xE0, 0xF8, 0x20, 0x01}, {nullptr, nullptr, nullptr, 1}, 3}, - // CBOR: members whose key is not a string are skipped, whatever their key and value - {json::input_format_t::cbor, {0xA4, 0x01, 0x02, 0x82, 0x01, 0x02, 0xA1, 0x61, 'x', 0x9F, 0xFF, 0xC1, 0x01, 0x5F, 0x41, 0x00, 0xFF, 0x61, 'a', 0x03}, {{"a", 3}}, 3}, - {json::input_format_t::cbor, {0xBF, 0xF5, 0xBF, 0x61, 'x', 0x7F, 0x61, 'y', 0xFF, 0xFF, 0x61, 'a', 0x03, 0xFF}, {{"a", 3}}, 1}, - // MessagePack: members whose key is not a string are skipped - {json::input_format_t::msgpack, {0x84, 0x01, 0x02, 0x81, 0xA1, 'x', 0x01, 0x92, 0x01, 0x02, 0xD4, 0x01, 0x02, 0xC0, 0xA1, 'a', 0x04}, {{"a", 4}}, 3}, - // UBJSON: a char that is not ASCII becomes U+FFFD - {json::input_format_t::ubjson, {'[', 'C', 0x80, 'C', 'A', ']'}, {replacement_character(), "A"}, 1}, - // UBJSON: the longest beginning of a high-precision number is kept - {json::input_format_t::ubjson, {'[', 'H', 'i', 5, '1', '2', 'a', 'b', 'c', 'H', 'i', 2, '1', '.', 'H', 'i', 3, 'a', 'b', 'c', 'H', 'i', 3, '4', '.', '5', ']'}, {12, 1, nullptr, 4.5}, 3}, - // BJData, too - {json::input_format_t::bjdata, {'[', 'C', 0xFF, 'H', 'i', 2, '-', '1', 'H', 'i', 2, '-', 'x', ']'}, {replacement_character(), -1, nullptr}, 2}, - // a NUL ends a high-precision number, as it ends JSON text - {json::input_format_t::ubjson, {'[', 'H', 'i', 4, '1', '2', 0, '9', 'H', 'i', 2, 0, '1', 'i', 3, ']'}, {12, nullptr, 3}, 2}, - // BON8: members whose key is not a string are skipped - {json::input_format_t::bon8, {0x89, 0x91, 0x92, 0xC9, 0x40, 0x82, 0x91, 0x92, 0x61, 0x93}, {{"a", 3}}, 2}, - {json::input_format_t::bon8, {0x8B, 0x91, 0x85, 0x91, 0xFE, 0xFA, 0x8B, 'x', 0x91, 0xFE, 0x61, 0x93, 0xFE}, {{"a", 3}}, 2}, - // BSON: elements of types the library does not read become null - { - json::input_format_t::bson, bson_document( - { - bson_element(0x07, "_id", bytes(12)), // ObjectId - bson_element(0x09, "date", bytes(8)), // UTC datetime - bson_element(0x13, "decimal", bytes(16)), // 128-bit decimal - bson_element(0x0B, "regex", {'a', '+', 0, 'i', 0}), // regular expression - bson_element(0x0D, "code", bson_string("f()")), // JavaScript code - bson_element(0x0E, "symbol", bson_string("s")), // symbol - bson_element(0x0C, "pointer", concatenated(bson_string("c"), bytes(12))), // DBPointer - bson_element(0x0F, "scope", concatenated(bson_int32(15), bson_string("g"), bson_document({}))), // code with scope - bson_element(0x06, "undefined", {}), // undefined - bson_element(0xFF, "min", {}), // min key - bson_element(0x7F, "max", {}), // max key - bson_element(0x10, "z", bson_int32(7)), - }), - {{"_id", nullptr}, {"date", nullptr}, {"decimal", nullptr}, {"regex", nullptr}, {"code", nullptr}, {"symbol", nullptr}, {"pointer", nullptr}, {"scope", nullptr}, {"undefined", nullptr}, {"min", nullptr}, {"max", nullptr}, {"z", 7}}, - 11 - }, - // BSON: an element of an unknown type becomes null, and the rest of its document is skipped - { - json::input_format_t::bson, bson_document( - { - bson_element(0x03, "inner", bson_document({bson_element(0x10, "a", bson_int32(1)), bson_element(0x42, "x", bytes(3)), bson_element(0x10, "b", bson_int32(2))})), - bson_element(0x04, "array", bson_document({bson_element(0x10, "0", bson_int32(1)), bson_element(0x42, "1", bytes(3))})), - bson_element(0x10, "after", bson_int32(3)), - }), - {{"inner", {{"a", 1}, {"x", nullptr}}}, {"array", {1, nullptr}}, {"after", 3}}, - 2 - }, - // BSON: so does a string or byte array whose length cannot be right - { - json::input_format_t::bson, bson_document( - { - bson_element(0x03, "inner", bson_document({bson_element(0x02, "s", bson_string("abc", -10)), bson_element(0x10, "b", bson_int32(2))})), - bson_element(0x03, "bin", bson_document({bson_element(0x05, "b", concatenated(bson_int32(-1), bytes(1))), bson_element(0x10, "b", bson_int32(2))})), - bson_element(0x10, "after", bson_int32(3)), - }), - {{"inner", {{"s", nullptr}}}, {"bin", {{"b", nullptr}}}, {"after", 3}}, - 2 - }, - // BSON: a string without its terminator, and a document whose size does not match, are kept - { - json::input_format_t::bson, bson_document( - { - bson_element(0x02, "s", {2, 0, 0, 0, 'a', 'X'}), - bson_element(0x03, "inner", bson_document({bson_element(0x10, "a", bson_int32(1))}, 1)), - }), - {{"s", "a"}, {"inner", {{"a", 1}}}}, - 2 - }, - }; - - for (const auto& repair : repairs) - { - CAPTURE(repair.format) - CAPTURE(repair.input) - const auto result = parse_binary_recovering(repair.input, repair.format); - CHECK(!result.ok); - CHECK(result.balanced); - CHECK(result.errors == repair.errors); - CHECK(result.value == repair.expected); - REQUIRE(!result.messages.empty()); -#if !defined(JSON_NOEXCEPTION) - // the first error is the one reported without recovering; under - // JSON_NOEXCEPTION, reading without recovering aborts instead of - // throwing, so there is no message to compare with - CHECK(result.messages.front() == binary_error_message(repair.input, repair.format)); -#endif - } - } - - SECTION("binary formats repair numbers that are out of range") - { - // CBOR: a double too large for a float number_float_t - float_json cbor; - BasicRecoveringParser sax(cbor); - const std::vector cbor_input = {0x82, 0xFB, 0x7E, 0x37, 0xE4, 0x3C, 0x88, 0x00, 0x75, 0x9C, 0x01}; // [1e300, 1] - CHECK(!float_json::sax_parse(cbor_input, &sax, float_json::input_format_t::cbor)); - CHECK(sax.errors == 1); - CHECK(sax.messages.front() == "[json.exception.out_of_range.406] syntax error while parsing CBOR value: number overflow"); - REQUIRE(cbor.size() == 2); - CHECK(std::isinf(cbor[0].get())); - CHECK(cbor[1] == 1); - - // UBJSON: a high-precision number too large for number_float_t - const auto ubjson = parse_binary_recovering({'H', 'i', 5, '1', 'e', '9', '9', '9'}, json::input_format_t::ubjson); - CHECK(ubjson.errors == 1); - CHECK(ubjson.value.is_number_float()); - CHECK(std::isinf(ubjson.value.get())); - } - - SECTION("binary formats stop where the end of an item is not known") - { - // a byte that begins no item - const auto cbor = parse_binary_recovering({0x82, 0x01, 0x1C, 0x02}, json::input_format_t::cbor); - CHECK(cbor.errors == 1); - CHECK(cbor.value == json({1})); - - // a key that is no item: the unused MessagePack byte, a CBOR break - // in a map of known size, and the end of a BON8 container - const auto msgpack = parse_binary_recovering({0x82, 0xA1, 'a', 0x01, 0xC1, 0x02}, json::input_format_t::msgpack); - CHECK(msgpack.errors == 1); - CHECK(msgpack.value == json({{"a", 1}})); - const auto cbor_break = parse_binary_recovering({0xA2, 0x61, 'a', 0x01, 0xFF, 0x02}, json::input_format_t::cbor); - CHECK(cbor_break.errors == 1); - CHECK(cbor_break.value == json({{"a", 1}})); - const auto bon8 = parse_binary_recovering({0x88, 0x61, 0x91, 0xFE}, json::input_format_t::bon8); - CHECK(bon8.errors == 1); - CHECK(bon8.value == json({{"a", 1}})); - - // an indefinite-length string inside an indefinite-length string - const auto nested = parse_binary_recovering({0x82, 0x01, 0x7F, 0x7F, 0x61, 'a', 0xFF, 0xFF}, json::input_format_t::cbor); - CHECK(nested.errors == 1); - CHECK(nested.value == json({1})); - - // a BJData ndarray whose element type has no name: the object that - // holds the ndarray was already begun - const auto ndarray = parse_binary_recovering({'[', '$', 0x01, '#', '[', '$', 'i', '#', 'i', 2, 2, 3}, json::input_format_t::bjdata); - CHECK(ndarray.errors == 1); - CHECK(ndarray.balanced); - CHECK(ndarray.value == json::object()); - - // a skipped member that the input ends in - const auto truncated = parse_binary_recovering({0xA2, 0x01, 0x82, 0x01}, json::input_format_t::cbor); - CHECK(truncated.errors == 2); - CHECK(truncated.balanced); - CHECK(truncated.value == json::object()); - - // a BSON element of an unknown type in a document whose size cannot be right - const auto bson = parse_binary_recovering(bson_document({bson_element(0x10, "a", bson_int32(1)), bson_element(0x42, "x", bytes(3))}, -10), json::input_format_t::bson); - CHECK(bson.errors == 1); - CHECK(bson.value == json({{"a", 1}, {"x", nullptr}})); - } - - SECTION("changed bytes in binary input") - { - const json j = {{"a", {1, -2, {{"b", "c"}}, json::array()}}, {"d", {{"e", nullptr}, {"f", true}}}, {"g", 1.5}, {"h", json::binary({1, 2, 3})}, {"i", "\xC3\xA4"}}; - - const std::vector>> encodings = - { - {json::input_format_t::cbor, json::to_cbor(j)}, - {json::input_format_t::msgpack, json::to_msgpack(j)}, - {json::input_format_t::ubjson, json::to_ubjson(j)}, - {json::input_format_t::ubjson, json::to_ubjson(j, true, true)}, - {json::input_format_t::bjdata, json::to_bjdata(j)}, - {json::input_format_t::bjdata, json::to_bjdata(j, true, true)}, - {json::input_format_t::bson, json::to_bson(j)}, - {json::input_format_t::bon8, json::to_bon8(j)}, - }; - const std::vector replacements = {0x00, 0x01, 0x7F, 0x80, 0xC1, 0xD9, 0xE0, 0xF7, 0xFE, 0xFF}; - - for (const auto& encoding : encodings) - { - const auto format = encoding.first; - const auto& original = encoding.second; - CAPTURE(format) - - std::vector> inputs; - for (std::size_t position = 0; position < original.size(); ++position) - { - for (const auto replacement : replacements) - { - auto changed = original; - changed[position] = replacement; - inputs.push_back(changed); - } - auto removed = original; - removed.erase(removed.begin() + static_cast(position)); - inputs.push_back(removed); - } - - for (const auto& input : inputs) - { - CAPTURE(input) - const auto result = parse_binary_recovering(input, format); - CHECK(result.balanced); - CHECK(result.errors <= input.size() + 1); -#if !defined(JSON_NOEXCEPTION) - // an error is reported exactly if reading into a JSON value - // fails, and the first one is the same (under JSON_NOEXCEPTION, - // that reading aborts instead of throwing) - const auto message = binary_error_message(input, format); - CHECK(result.ok == message.empty()); - if (!result.ok && result.errors < 100) - { - CHECK(result.messages.front() == message); - } -#endif - } - } - } - - SECTION("JSON text") - { - // the parser stopped, but reported success - json j; - RecoveringParser sax(j); - CHECK(!json::sax_parse("[1,2,3,]", &sax)); - CHECK(sax.errors == 1); - CHECK(j == json({1, 2, 3})); - } - - SECTION("the SAX parsers of the library stop") - { - json _; - CHECK(json::from_cbor(std::vector {0x9F}, true, false).is_discarded()); - CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector {0x9F}), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&); - CHECK(json::parse("[1,2,3,]", nullptr, false).is_discarded()); - CHECK(!json::accept("[1,2,3,]")); - } -} - #if (defined(__cpp_exceptions) || defined(__EXCEPTIONS) || defined(_CPPUNWIND)) && !defined(JSON_NOEXCEPTION) TEST_CASE("regression test #5135 - destructor never allocates, even under memory pressure") { diff --git a/tests/src/unit-regression3.cpp b/tests/src/unit-regression3.cpp index 69a95fb9b..21e047894 100644 --- a/tests/src/unit-regression3.cpp +++ b/tests/src/unit-regression3.cpp @@ -22,14 +22,6 @@ // scoped enum, so get() (needed below to get>() // from a plain JSON array, not just from an already-binary value) relies on // enum serialization being enabled -// capture whether JSON_DELETE_DEPRECATED_FUNCTIONS was enabled on the command -// line *before* including json.hpp, since the library #undefs it once the header -// has been fully processed (see include/nlohmann/detail/macro_unscope.hpp); the -// tests of deprecated functions are skipped if these functions are deleted -#if defined(JSON_DELETE_DEPRECATED_FUNCTIONS) && (JSON_DELETE_DEPRECATED_FUNCTIONS == 1) - #define JSON_TEST_DEPRECATED_FUNCTIONS_DELETED -#endif - #if defined(JSON_DISABLE_ENUM_SERIALIZATION) && (JSON_DISABLE_ENUM_SERIALIZATION == 1) #define SKIP_TESTS_FOR_ENUM_SERIALIZATION #endif @@ -870,50 +862,6 @@ TEST_CASE("issue #5338 - truncated CBOR tagged binary subtype is rejected") } } -TEST_CASE("issue #5676 - SAX parsing of CBOR tags") -{ - const json expected = json::binary({1, 2, 3}, 42); - const auto cbor = json::to_cbor(expected); - - nlohmann::detail::json_sax_acceptor acceptor; - CHECK_FALSE(json::sax_parse(cbor, &acceptor, json::input_format_t::cbor)); - CHECK_FALSE(json::sax_parse(cbor, &acceptor, json::input_format_t::cbor, - true, false, false, json::cbor_tag_handler_t::error)); - - CHECK(json::sax_parse(cbor, &acceptor, json::input_format_t::cbor, - true, false, false, json::cbor_tag_handler_t::ignore)); - - json parsed; - nlohmann::detail::json_sax_dom_parser sax(parsed); - CHECK(json::sax_parse(cbor, &sax, json::input_format_t::cbor, - true, false, false, json::cbor_tag_handler_t::store)); - CHECK(parsed == expected); - - json iterator_parsed; - nlohmann::detail::json_sax_dom_parser iterator_sax(iterator_parsed); - CHECK(json::sax_parse(cbor.begin(), cbor.end(), &iterator_sax, json::input_format_t::cbor, - true, false, false, json::cbor_tag_handler_t::store)); - CHECK(iterator_parsed == expected); - -#ifndef JSON_TEST_DEPRECATED_FUNCTIONS_DELETED - json span_parsed; - nlohmann::detail::json_sax_dom_parser span_sax(span_parsed); - CHECK(json::sax_parse(nlohmann::detail::span_input_adapter(cbor.data(), cbor.size()), &span_sax, - json::input_format_t::cbor, true, false, false, json::cbor_tag_handler_t::store)); - CHECK(span_parsed == expected); -#endif - - const std::string text = "null"; - CHECK(json::sax_parse(text, &acceptor, json::input_format_t::json, - true, false, false, json::cbor_tag_handler_t::store)); - CHECK(json::sax_parse(text.begin(), text.end(), &acceptor, json::input_format_t::json, - true, false, false, json::cbor_tag_handler_t::store)); -#ifndef JSON_TEST_DEPRECATED_FUNCTIONS_DELETED - CHECK(json::sax_parse(nlohmann::detail::span_input_adapter(text.data(), text.size()), &acceptor, - json::input_format_t::json, true, false, false, json::cbor_tag_handler_t::store)); -#endif -} - TEST_CASE("issue #5402 - update(merge_objects=true) overwrites a primitive with an object") { json t = {{"k", 1}}; diff --git a/tests/src/unit-sax_parse.cpp b/tests/src/unit-sax_parse.cpp new file mode 100644 index 000000000..51d549e04 --- /dev/null +++ b/tests/src/unit-sax_parse.cpp @@ -0,0 +1,689 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +///////////////////////////////////////////////////////////////////// +// Tests that call basic_json::sax_parse have a file of their own: every +// sax_parse call instantiates the parser and binary reader that recover from +// errors (see #3989), and in unit-regression2.cpp and unit-regression3.cpp +// this made the objects too large for the MinGW linker to relocate (see +// #5511). +///////////////////////////////////////////////////////////////////// + +#include "doctest_compatibility.h" + +// capture whether JSON_DELETE_DEPRECATED_FUNCTIONS was enabled on the command +// line *before* including json.hpp, since the library #undefs it once the header +// has been fully processed (see include/nlohmann/detail/macro_unscope.hpp); the +// tests of deprecated functions are skipped if these functions are deleted +#if defined(JSON_DELETE_DEPRECATED_FUNCTIONS) && (JSON_DELETE_DEPRECATED_FUNCTIONS == 1) + #define JSON_TEST_DEPRECATED_FUNCTIONS_DELETED +#endif + +#include +using json = nlohmann::json; + +#include +#include +#include +#include +#include +#include +#include + +// a narrow number_float_t, so that a double read from binary input can +// overflow it +using float_json = nlohmann::basic_json; + +DOCTEST_CLANG_SUPPRESS_WARNING_PUSH +DOCTEST_CLANG_SUPPRESS_WARNING("-Wexit-time-destructors") + +namespace +{ +/// builds a value from SAX events, asks the parser to recover from its first +/// 100 errors, and checks that the events are balanced (see #3989) +template +class BasicRecoveringParser +{ + public: + explicit BasicRecoveringParser(BasicJsonType& j) + : dom(j, false) + {} + + bool null() + { + value(); + return dom.null(); + } + + bool boolean(bool val) + { + value(); + return dom.boolean(val); + } + + bool number_integer(typename BasicJsonType::number_integer_t val) + { + value(); + return dom.number_integer(val); + } + + bool number_unsigned(typename BasicJsonType::number_unsigned_t val) + { + value(); + return dom.number_unsigned(val); + } + + bool number_float(typename BasicJsonType::number_float_t val, const std::string& s) + { + value(); + return dom.number_float(val, s); + } + + bool string(std::string& val) + { + value(); + return dom.string(val); + } + + bool binary(typename BasicJsonType::binary_t& val) + { + value(); + return dom.binary(val); + } + + bool start_object(std::size_t elements) + { + value(); + stack.push_back('o'); + return dom.start_object(elements); + } + + bool key(std::string& val) + { + if (stack.empty() || stack.back() != 'o') + { + well_formed = false; + return false; + } + stack.back() = 'v'; + return dom.key(val); + } + + bool end_object() + { + if (stack.empty() || stack.back() != 'o') + { + well_formed = false; + return false; + } + stack.pop_back(); + return dom.end_object(); + } + + bool start_array(std::size_t elements) + { + value(); + stack.push_back('a'); + return dom.start_array(elements); + } + + bool end_array() + { + if (stack.empty() || stack.back() != 'a') + { + well_formed = false; + return false; + } + stack.pop_back(); + return dom.end_array(); + } + + bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& ex) + { + messages.emplace_back(ex.what()); + // a limit, so that a reader that does not stop fails the test + // instead of making it hang + return ++errors < 100; + } + + /// whether the events were balanced and every key was followed by a value + bool balanced() const + { + return well_formed && stack.empty(); + } + + /// builds the value + nlohmann::detail::json_sax_dom_parser dom; + std::size_t errors = 0; + std::vector messages {}; // NOLINT(readability-redundant-member-init) + std::vector stack {}; // NOLINT(readability-redundant-member-init) + bool well_formed = true; + + private: + void value() + { + if (!stack.empty()) + { + if (stack.back() == 'v') + { + stack.back() = 'o'; + } + else if (stack.back() == 'o') + { + well_formed = false; + } + } + } + +}; + +using RecoveringParser = BasicRecoveringParser; + +struct BinaryParseResult +{ + json value; + std::size_t errors; + std::vector messages; + bool ok; + bool balanced; +}; + +BinaryParseResult parse_binary_recovering(const std::vector& input, const json::input_format_t format) +{ + json j; + RecoveringParser sax(j); + const bool ok = json::sax_parse(input, &sax, format); + return {j, sax.errors, sax.messages, ok, sax.balanced()}; +} + +#if !defined(JSON_NOEXCEPTION) +/// the message of the exception that reading @a input into a JSON value +/// throws, or an empty string if reading succeeds +std::string binary_error_message(const std::vector& input, const json::input_format_t format) +{ + try + { + json _; + switch (format) + { + case json::input_format_t::cbor: + _ = json::from_cbor(input); + break; + case json::input_format_t::msgpack: + _ = json::from_msgpack(input); + break; + case json::input_format_t::ubjson: + _ = json::from_ubjson(input); + break; + case json::input_format_t::bjdata: + _ = json::from_bjdata(input); + break; + case json::input_format_t::bson: + _ = json::from_bson(input); + break; + case json::input_format_t::bon8: + _ = json::from_bon8(input); + break; + case json::input_format_t::json: + default: + break; + } + } + catch (const json::exception& e) + { + return e.what(); + } + return ""; +} +#endif + +/// a BSON element: its type, its name, and its value +std::vector bson_element(const std::uint8_t type, const std::string& name, const std::vector& value) +{ + std::vector result = {type}; + result.insert(result.end(), name.begin(), name.end()); + result.push_back(0x00); + result.insert(result.end(), value.begin(), value.end()); + return result; +} + +/// a BSON document of the given elements; @a size_offset is added to the +/// size it declares +std::vector bson_document(const std::vector>& elements, const int size_offset = 0) +{ + std::vector body; + for (const auto& element : elements) + { + body.insert(body.end(), element.begin(), element.end()); + } + const auto size = static_cast(static_cast(body.size()) + 5 + size_offset); + std::vector result = {static_cast(size & 0xFFu), static_cast((size >> 8u) & 0xFFu), + static_cast((size >> 16u) & 0xFFu), static_cast((size >> 24u) & 0xFFu) + }; + result.insert(result.end(), body.begin(), body.end()); + result.push_back(0x00); + return result; +} + +/// a BSON int32 value +std::vector bson_int32(const std::int32_t value) +{ + const auto u = static_cast(value); + return {static_cast(u & 0xFFu), static_cast((u >> 8u) & 0xFFu), + static_cast((u >> 16u) & 0xFFu), static_cast((u >> 24u) & 0xFFu)}; +} + +/// a BSON string value, whose length is @a length_offset off +std::vector bson_string(const std::string& value, const std::int32_t length_offset = 0) +{ + auto result = bson_int32(static_cast(value.size() + 1) + length_offset); + result.insert(result.end(), value.begin(), value.end()); + result.push_back(0x00); + return result; +} + +/// @a count bytes of value 0xAB +std::vector bytes(const std::size_t count) +{ + return std::vector(count, 0xAB); +} + +template +std::vector concatenated(const std::vector& first, const Parts& ... rest) +{ + std::vector result = first; + for (const auto& part : std::initializer_list> {rest...}) + { + result.insert(result.end(), part.begin(), part.end()); + } + return result; +} + +/// U+FFFD REPLACEMENT CHARACTER +std::string replacement_character() +{ + return "\xEF\xBF\xBD"; +} +} // namespace + +TEST_CASE("regression test - #3989 SAX parse_error() returning true") +{ + SECTION("binary formats complete what was read before the input ends") + { + const json j = {{"a", {1, -2, {{"b", "c"}}, json::array()}}, {"d", {{"e", nullptr}, {"f", true}}}, {"g", 1.5}, {"h", json::binary({1, 2, 3})}}; + + const std::vector>> encodings = + { + {json::input_format_t::cbor, json::to_cbor(j)}, + {json::input_format_t::msgpack, json::to_msgpack(j)}, + {json::input_format_t::ubjson, json::to_ubjson(j)}, + {json::input_format_t::ubjson, json::to_ubjson(j, true, true)}, + {json::input_format_t::bjdata, json::to_bjdata(j)}, + {json::input_format_t::bjdata, json::to_bjdata(j, true, true)}, + {json::input_format_t::bson, json::to_bson(j)}, + {json::input_format_t::bon8, json::to_bon8(j)}, + }; + + for (const auto& encoding : encodings) + { + const auto format = encoding.first; + const auto& bytes = encoding.second; + CAPTURE(format) + + // every prefix is truncated input + for (std::size_t length = 0; length < bytes.size(); ++length) + { + CAPTURE(length) + const auto result = parse_binary_recovering(std::vector(bytes.begin(), bytes.begin() + static_cast(length)), format); + CHECK(!result.ok); + CHECK(result.errors == 1); + CHECK(result.balanced); + } + + // the complete input is read as usual (binary values do not + // round-trip through every format, so compare with a plain parse) + json expected; + nlohmann::detail::json_sax_dom_parser dom(expected); + CHECK(json::sax_parse(bytes, &dom, format)); + const auto complete = parse_binary_recovering(bytes, format); + CHECK(complete.ok); + CHECK(complete.errors == 0); + CHECK(complete.value == expected); + + // a byte after the value + auto trailing_bytes = bytes; + trailing_bytes.push_back(0x01); + const auto trailing = parse_binary_recovering(trailing_bytes, format); + CHECK(!trailing.ok); + CHECK(trailing.errors == 1); + CHECK(trailing.value == expected); + } + } + + SECTION("containers without an end") + { + // these made the readers loop, or read on, after the error + const auto cbor_array = parse_binary_recovering({0x9F}, json::input_format_t::cbor); + CHECK(cbor_array.errors == 1); + CHECK(cbor_array.value == json::array()); + + const auto cbor_map = parse_binary_recovering({0xBF, 0x61, 'a'}, json::input_format_t::cbor); + CHECK(cbor_map.errors == 1); + CHECK(cbor_map.value == json({{"a", nullptr}})); + + const auto msgpack_array = parse_binary_recovering({0xDD, 0xFF, 0xFF, 0xFF, 0xFF}, json::input_format_t::msgpack); + CHECK(msgpack_array.errors == 1); + CHECK(msgpack_array.value == json::array()); + + const auto msgpack_map = parse_binary_recovering({0x81, 0xA1, 'a', 0x92, 0x01}, json::input_format_t::msgpack); + CHECK(msgpack_map.errors == 1); + CHECK(msgpack_map.value == json({{"a", {1}}})); + } + + SECTION("BJData ndarray") + { + // a 2x3 int8 array with two of its six elements; the annotated array + // format opens an object and two arrays of its own + const auto result = parse_binary_recovering({'[', '$', 'i', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2}, json::input_format_t::bjdata); + CHECK(result.errors == 1); + CHECK(result.balanced); + CHECK(result.value == json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2}}})); + } + + SECTION("binary formats repair items whose end is known") + { + struct Repair + { + json::input_format_t format; + std::vector input; + json expected; + std::size_t errors; + }; + + const std::vector repairs = + { + // CBOR: tags are ignored (here tag 1 and the self-describe tag 55799) + {json::input_format_t::cbor, {0x82, 0xC1, 0x05, 0xD9, 0xD9, 0xF7, 0x06}, {5, 6}, 2}, + // CBOR: undefined and other simple values become null + {json::input_format_t::cbor, {0x84, 0xF7, 0xE0, 0xF8, 0x20, 0x01}, {nullptr, nullptr, nullptr, 1}, 3}, + // CBOR: members whose key is not a string are skipped, whatever their key and value + {json::input_format_t::cbor, {0xA4, 0x01, 0x02, 0x82, 0x01, 0x02, 0xA1, 0x61, 'x', 0x9F, 0xFF, 0xC1, 0x01, 0x5F, 0x41, 0x00, 0xFF, 0x61, 'a', 0x03}, {{"a", 3}}, 3}, + {json::input_format_t::cbor, {0xBF, 0xF5, 0xBF, 0x61, 'x', 0x7F, 0x61, 'y', 0xFF, 0xFF, 0x61, 'a', 0x03, 0xFF}, {{"a", 3}}, 1}, + // MessagePack: members whose key is not a string are skipped + {json::input_format_t::msgpack, {0x84, 0x01, 0x02, 0x81, 0xA1, 'x', 0x01, 0x92, 0x01, 0x02, 0xD4, 0x01, 0x02, 0xC0, 0xA1, 'a', 0x04}, {{"a", 4}}, 3}, + // UBJSON: a char that is not ASCII becomes U+FFFD + {json::input_format_t::ubjson, {'[', 'C', 0x80, 'C', 'A', ']'}, {replacement_character(), "A"}, 1}, + // UBJSON: the longest beginning of a high-precision number is kept + {json::input_format_t::ubjson, {'[', 'H', 'i', 5, '1', '2', 'a', 'b', 'c', 'H', 'i', 2, '1', '.', 'H', 'i', 3, 'a', 'b', 'c', 'H', 'i', 3, '4', '.', '5', ']'}, {12, 1, nullptr, 4.5}, 3}, + // BJData, too + {json::input_format_t::bjdata, {'[', 'C', 0xFF, 'H', 'i', 2, '-', '1', 'H', 'i', 2, '-', 'x', ']'}, {replacement_character(), -1, nullptr}, 2}, + // a NUL ends a high-precision number, as it ends JSON text + {json::input_format_t::ubjson, {'[', 'H', 'i', 4, '1', '2', 0, '9', 'H', 'i', 2, 0, '1', 'i', 3, ']'}, {12, nullptr, 3}, 2}, + // BON8: members whose key is not a string are skipped + {json::input_format_t::bon8, {0x89, 0x91, 0x92, 0xC9, 0x40, 0x82, 0x91, 0x92, 0x61, 0x93}, {{"a", 3}}, 2}, + {json::input_format_t::bon8, {0x8B, 0x91, 0x85, 0x91, 0xFE, 0xFA, 0x8B, 'x', 0x91, 0xFE, 0x61, 0x93, 0xFE}, {{"a", 3}}, 2}, + // BSON: elements of types the library does not read become null + { + json::input_format_t::bson, bson_document( + { + bson_element(0x07, "_id", bytes(12)), // ObjectId + bson_element(0x09, "date", bytes(8)), // UTC datetime + bson_element(0x13, "decimal", bytes(16)), // 128-bit decimal + bson_element(0x0B, "regex", {'a', '+', 0, 'i', 0}), // regular expression + bson_element(0x0D, "code", bson_string("f()")), // JavaScript code + bson_element(0x0E, "symbol", bson_string("s")), // symbol + bson_element(0x0C, "pointer", concatenated(bson_string("c"), bytes(12))), // DBPointer + bson_element(0x0F, "scope", concatenated(bson_int32(15), bson_string("g"), bson_document({}))), // code with scope + bson_element(0x06, "undefined", {}), // undefined + bson_element(0xFF, "min", {}), // min key + bson_element(0x7F, "max", {}), // max key + bson_element(0x10, "z", bson_int32(7)), + }), + {{"_id", nullptr}, {"date", nullptr}, {"decimal", nullptr}, {"regex", nullptr}, {"code", nullptr}, {"symbol", nullptr}, {"pointer", nullptr}, {"scope", nullptr}, {"undefined", nullptr}, {"min", nullptr}, {"max", nullptr}, {"z", 7}}, + 11 + }, + // BSON: an element of an unknown type becomes null, and the rest of its document is skipped + { + json::input_format_t::bson, bson_document( + { + bson_element(0x03, "inner", bson_document({bson_element(0x10, "a", bson_int32(1)), bson_element(0x42, "x", bytes(3)), bson_element(0x10, "b", bson_int32(2))})), + bson_element(0x04, "array", bson_document({bson_element(0x10, "0", bson_int32(1)), bson_element(0x42, "1", bytes(3))})), + bson_element(0x10, "after", bson_int32(3)), + }), + {{"inner", {{"a", 1}, {"x", nullptr}}}, {"array", {1, nullptr}}, {"after", 3}}, + 2 + }, + // BSON: so does a string or byte array whose length cannot be right + { + json::input_format_t::bson, bson_document( + { + bson_element(0x03, "inner", bson_document({bson_element(0x02, "s", bson_string("abc", -10)), bson_element(0x10, "b", bson_int32(2))})), + bson_element(0x03, "bin", bson_document({bson_element(0x05, "b", concatenated(bson_int32(-1), bytes(1))), bson_element(0x10, "b", bson_int32(2))})), + bson_element(0x10, "after", bson_int32(3)), + }), + {{"inner", {{"s", nullptr}}}, {"bin", {{"b", nullptr}}}, {"after", 3}}, + 2 + }, + // BSON: a string without its terminator, and a document whose size does not match, are kept + { + json::input_format_t::bson, bson_document( + { + bson_element(0x02, "s", {2, 0, 0, 0, 'a', 'X'}), + bson_element(0x03, "inner", bson_document({bson_element(0x10, "a", bson_int32(1))}, 1)), + }), + {{"s", "a"}, {"inner", {{"a", 1}}}}, + 2 + }, + }; + + for (const auto& repair : repairs) + { + CAPTURE(repair.format) + CAPTURE(repair.input) + const auto result = parse_binary_recovering(repair.input, repair.format); + CHECK(!result.ok); + CHECK(result.balanced); + CHECK(result.errors == repair.errors); + CHECK(result.value == repair.expected); + REQUIRE(!result.messages.empty()); +#if !defined(JSON_NOEXCEPTION) + // the first error is the one reported without recovering; under + // JSON_NOEXCEPTION, reading without recovering aborts instead of + // throwing, so there is no message to compare with + CHECK(result.messages.front() == binary_error_message(repair.input, repair.format)); +#endif + } + } + + SECTION("binary formats repair numbers that are out of range") + { + // CBOR: a double too large for a float number_float_t + float_json cbor; + BasicRecoveringParser sax(cbor); + const std::vector cbor_input = {0x82, 0xFB, 0x7E, 0x37, 0xE4, 0x3C, 0x88, 0x00, 0x75, 0x9C, 0x01}; // [1e300, 1] + CHECK(!float_json::sax_parse(cbor_input, &sax, float_json::input_format_t::cbor)); + CHECK(sax.errors == 1); + CHECK(sax.messages.front() == "[json.exception.out_of_range.406] syntax error while parsing CBOR value: number overflow"); + REQUIRE(cbor.size() == 2); + CHECK(std::isinf(cbor[0].get())); + CHECK(cbor[1] == 1); + + // UBJSON: a high-precision number too large for number_float_t + const auto ubjson = parse_binary_recovering({'H', 'i', 5, '1', 'e', '9', '9', '9'}, json::input_format_t::ubjson); + CHECK(ubjson.errors == 1); + CHECK(ubjson.value.is_number_float()); + CHECK(std::isinf(ubjson.value.get())); + } + + SECTION("binary formats stop where the end of an item is not known") + { + // a byte that begins no item + const auto cbor = parse_binary_recovering({0x82, 0x01, 0x1C, 0x02}, json::input_format_t::cbor); + CHECK(cbor.errors == 1); + CHECK(cbor.value == json({1})); + + // a key that is no item: the unused MessagePack byte, a CBOR break + // in a map of known size, and the end of a BON8 container + const auto msgpack = parse_binary_recovering({0x82, 0xA1, 'a', 0x01, 0xC1, 0x02}, json::input_format_t::msgpack); + CHECK(msgpack.errors == 1); + CHECK(msgpack.value == json({{"a", 1}})); + const auto cbor_break = parse_binary_recovering({0xA2, 0x61, 'a', 0x01, 0xFF, 0x02}, json::input_format_t::cbor); + CHECK(cbor_break.errors == 1); + CHECK(cbor_break.value == json({{"a", 1}})); + const auto bon8 = parse_binary_recovering({0x88, 0x61, 0x91, 0xFE}, json::input_format_t::bon8); + CHECK(bon8.errors == 1); + CHECK(bon8.value == json({{"a", 1}})); + + // an indefinite-length string inside an indefinite-length string + const auto nested = parse_binary_recovering({0x82, 0x01, 0x7F, 0x7F, 0x61, 'a', 0xFF, 0xFF}, json::input_format_t::cbor); + CHECK(nested.errors == 1); + CHECK(nested.value == json({1})); + + // a BJData ndarray whose element type has no name: the object that + // holds the ndarray was already begun + const auto ndarray = parse_binary_recovering({'[', '$', 0x01, '#', '[', '$', 'i', '#', 'i', 2, 2, 3}, json::input_format_t::bjdata); + CHECK(ndarray.errors == 1); + CHECK(ndarray.balanced); + CHECK(ndarray.value == json::object()); + + // a skipped member that the input ends in + const auto truncated = parse_binary_recovering({0xA2, 0x01, 0x82, 0x01}, json::input_format_t::cbor); + CHECK(truncated.errors == 2); + CHECK(truncated.balanced); + CHECK(truncated.value == json::object()); + + // a BSON element of an unknown type in a document whose size cannot be right + const auto bson = parse_binary_recovering(bson_document({bson_element(0x10, "a", bson_int32(1)), bson_element(0x42, "x", bytes(3))}, -10), json::input_format_t::bson); + CHECK(bson.errors == 1); + CHECK(bson.value == json({{"a", 1}, {"x", nullptr}})); + } + + SECTION("changed bytes in binary input") + { + const json j = {{"a", {1, -2, {{"b", "c"}}, json::array()}}, {"d", {{"e", nullptr}, {"f", true}}}, {"g", 1.5}, {"h", json::binary({1, 2, 3})}, {"i", "\xC3\xA4"}}; + + const std::vector>> encodings = + { + {json::input_format_t::cbor, json::to_cbor(j)}, + {json::input_format_t::msgpack, json::to_msgpack(j)}, + {json::input_format_t::ubjson, json::to_ubjson(j)}, + {json::input_format_t::ubjson, json::to_ubjson(j, true, true)}, + {json::input_format_t::bjdata, json::to_bjdata(j)}, + {json::input_format_t::bjdata, json::to_bjdata(j, true, true)}, + {json::input_format_t::bson, json::to_bson(j)}, + {json::input_format_t::bon8, json::to_bon8(j)}, + }; + const std::vector replacements = {0x00, 0x01, 0x7F, 0x80, 0xC1, 0xD9, 0xE0, 0xF7, 0xFE, 0xFF}; + + for (const auto& encoding : encodings) + { + const auto format = encoding.first; + const auto& original = encoding.second; + CAPTURE(format) + + std::vector> inputs; + for (std::size_t position = 0; position < original.size(); ++position) + { + for (const auto replacement : replacements) + { + auto changed = original; + changed[position] = replacement; + inputs.push_back(changed); + } + auto removed = original; + removed.erase(removed.begin() + static_cast(position)); + inputs.push_back(removed); + } + + for (const auto& input : inputs) + { + CAPTURE(input) + const auto result = parse_binary_recovering(input, format); + CHECK(result.balanced); + CHECK(result.errors <= input.size() + 1); +#if !defined(JSON_NOEXCEPTION) + // an error is reported exactly if reading into a JSON value + // fails, and the first one is the same (under JSON_NOEXCEPTION, + // that reading aborts instead of throwing) + const auto message = binary_error_message(input, format); + CHECK(result.ok == message.empty()); + if (!result.ok && result.errors < 100) + { + CHECK(result.messages.front() == message); + } +#endif + } + } + } + + SECTION("JSON text") + { + // the parser stopped, but reported success + json j; + RecoveringParser sax(j); + CHECK(!json::sax_parse("[1,2,3,]", &sax)); + CHECK(sax.errors == 1); + CHECK(j == json({1, 2, 3})); + } + + SECTION("the SAX parsers of the library stop") + { + json _; + CHECK(json::from_cbor(std::vector {0x9F}, true, false).is_discarded()); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector {0x9F}), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&); + CHECK(json::parse("[1,2,3,]", nullptr, false).is_discarded()); + CHECK(!json::accept("[1,2,3,]")); + } +} + + +TEST_CASE("issue #5676 - SAX parsing of CBOR tags") +{ + const json expected = json::binary({1, 2, 3}, 42); + const auto cbor = json::to_cbor(expected); + + nlohmann::detail::json_sax_acceptor acceptor; + CHECK_FALSE(json::sax_parse(cbor, &acceptor, json::input_format_t::cbor)); + CHECK_FALSE(json::sax_parse(cbor, &acceptor, json::input_format_t::cbor, + true, false, false, json::cbor_tag_handler_t::error)); + + CHECK(json::sax_parse(cbor, &acceptor, json::input_format_t::cbor, + true, false, false, json::cbor_tag_handler_t::ignore)); + + json parsed; + nlohmann::detail::json_sax_dom_parser sax(parsed); + CHECK(json::sax_parse(cbor, &sax, json::input_format_t::cbor, + true, false, false, json::cbor_tag_handler_t::store)); + CHECK(parsed == expected); + + json iterator_parsed; + nlohmann::detail::json_sax_dom_parser iterator_sax(iterator_parsed); + CHECK(json::sax_parse(cbor.begin(), cbor.end(), &iterator_sax, json::input_format_t::cbor, + true, false, false, json::cbor_tag_handler_t::store)); + CHECK(iterator_parsed == expected); + +#ifndef JSON_TEST_DEPRECATED_FUNCTIONS_DELETED + json span_parsed; + nlohmann::detail::json_sax_dom_parser span_sax(span_parsed); + CHECK(json::sax_parse(nlohmann::detail::span_input_adapter(cbor.data(), cbor.size()), &span_sax, + json::input_format_t::cbor, true, false, false, json::cbor_tag_handler_t::store)); + CHECK(span_parsed == expected); +#endif + + const std::string text = "null"; + CHECK(json::sax_parse(text, &acceptor, json::input_format_t::json, + true, false, false, json::cbor_tag_handler_t::store)); + CHECK(json::sax_parse(text.begin(), text.end(), &acceptor, json::input_format_t::json, + true, false, false, json::cbor_tag_handler_t::store)); +#ifndef JSON_TEST_DEPRECATED_FUNCTIONS_DELETED + CHECK(json::sax_parse(nlohmann::detail::span_input_adapter(text.data(), text.size()), &acceptor, + json::input_format_t::json, true, false, false, json::cbor_tag_handler_t::store)); +#endif +} + +DOCTEST_CLANG_SUPPRESS_WARNING_POP