From 677794f07670af102a1444a3b4a0bf717afce4c4 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Mon, 28 Sep 2026 12:17:11 +0200 Subject: [PATCH] Fix CI: recover numbers with the string operations every string_t has recover_number() used back(), pop_back(), front(), and find_first_of(const char*), which the minimal alt_string of unit-alt-string.cpp does not provide. Since the binary readers recover UBJSON/BJData high-precision numbers with it, from_ubjson() instantiated it, too, and the test no longer compiled. It now uses only size(), operator[], and resize(), and unit-alt-string.cpp recovers from errors in JSON text, UBJSON, and CBOR. Signed-off-by: Niels Lohmann --- include/nlohmann/detail/input/lexer.hpp | 26 +++++++++++++----- single_include/nlohmann/json.hpp | 26 +++++++++++++----- tests/src/unit-alt-string.cpp | 35 +++++++++++++++++++++++++ 3 files changed, 73 insertions(+), 14 deletions(-) diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index ed48359f2..60f315e20 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -2748,30 +2748,42 @@ scan_number_done: */ token_type recover_number() { - while (!token_buffer.empty() && (token_buffer.back() < '0' || token_buffer.back() > '9')) + // only size(), operator[], and resize() are used, which every string + // type the library supports provides + std::size_t length = token_buffer.size(); + while (length != 0 && (token_buffer[length - 1] < '0' || token_buffer[length - 1] > '9')) { - token_buffer.pop_back(); + --length; } + token_buffer.resize(length); - if (token_buffer.empty()) + if (length == 0) { skip_to_delimiter(); return token_type::uninitialized; } - if (decimal_point_position >= token_buffer.size()) + if (decimal_point_position >= length) { decimal_point_position = std::string::npos; } - const std::size_t exponent = token_buffer.find_first_of("eE"); - const std::size_t mantissa_end = (exponent == std::string::npos) ? token_buffer.size() : exponent; + std::size_t exponent = std::string::npos; + for (std::size_t i = 0; i < length; ++i) + { + if (token_buffer[i] == 'e' || token_buffer[i] == 'E') + { + exponent = i; + break; + } + } + const std::size_t mantissa_end = (exponent == std::string::npos) ? length : exponent; token_type number_type = token_type::value_unsigned; if (decimal_point_position != std::string::npos || exponent != std::string::npos) { number_type = token_type::value_float; } - else if (token_buffer.front() == '-') + else if (token_buffer[0] == '-') { number_type = token_type::value_integer; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4c87b1afd..6f3fdbf9b 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -11919,30 +11919,42 @@ scan_number_done: */ token_type recover_number() { - while (!token_buffer.empty() && (token_buffer.back() < '0' || token_buffer.back() > '9')) + // only size(), operator[], and resize() are used, which every string + // type the library supports provides + std::size_t length = token_buffer.size(); + while (length != 0 && (token_buffer[length - 1] < '0' || token_buffer[length - 1] > '9')) { - token_buffer.pop_back(); + --length; } + token_buffer.resize(length); - if (token_buffer.empty()) + if (length == 0) { skip_to_delimiter(); return token_type::uninitialized; } - if (decimal_point_position >= token_buffer.size()) + if (decimal_point_position >= length) { decimal_point_position = std::string::npos; } - const std::size_t exponent = token_buffer.find_first_of("eE"); - const std::size_t mantissa_end = (exponent == std::string::npos) ? token_buffer.size() : exponent; + std::size_t exponent = std::string::npos; + for (std::size_t i = 0; i < length; ++i) + { + if (token_buffer[i] == 'e' || token_buffer[i] == 'E') + { + exponent = i; + break; + } + } + const std::size_t mantissa_end = (exponent == std::string::npos) ? length : exponent; token_type number_type = token_type::value_unsigned; if (decimal_point_position != std::string::npos || exponent != std::string::npos) { number_type = token_type::value_float; } - else if (token_buffer.front() == '-') + else if (token_buffer[0] == '-') { number_type = token_type::value_integer; } diff --git a/tests/src/unit-alt-string.cpp b/tests/src/unit-alt-string.cpp index cddaadb7e..43e400702 100644 --- a/tests/src/unit-alt-string.cpp +++ b/tests/src/unit-alt-string.cpp @@ -374,4 +374,39 @@ TEST_CASE("alternative string type") const auto j2 = j.flatten(); CHECK(j2.dump() == R"({"/foo/0":"bar","/foo/1":"baz"})"); } + + SECTION("error recovery") + { + // a SAX parser that recovers from every error (see #3989) + struct recovering_parser : nlohmann::detail::json_sax_dom_parser + { + explicit recovering_parser(alt_json& j) + : nlohmann::detail::json_sax_dom_parser(j, false) + {} + + bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const nlohmann::detail::exception& /*unused*/) + { + ++errors; + return true; + } + + std::size_t errors = 0; + }; + + alt_json j; + recovering_parser sax(j); + CHECK(!alt_json::sax_parse(R"([1., "a\qb", tru, {"k" 2}])", &sax)); + CHECK(sax.errors == 4); + CHECK(j.dump() == R"([1,"aqb",null,{"k":2}])"); + + // a UBJSON high-precision number, a CBOR key that is not a string + alt_json u; + recovering_parser ubjson_sax(u); + CHECK(!alt_json::sax_parse(std::vector {'[', 'H', 'i', 2, '1', '.', ']'}, &ubjson_sax, alt_json::input_format_t::ubjson)); + CHECK(u.dump() == "[1]"); + alt_json c; + recovering_parser cbor_sax(c); + CHECK(!alt_json::sax_parse(std::vector {0xA2, 0x01, 0x02, 0x61, 'a', 0x03}, &cbor_sax, alt_json::input_format_t::cbor)); + CHECK(c.dump() == R"({"a":3})"); + } }