From aa7f0b02a01c258351d2d57af84b072b42bb7db4 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:32:20 +0200 Subject: [PATCH] Fuzz json_document with exact-size buffers and parse options The fuzzer only parsed a std::string with default options. Also parse an exact-size byte vector, which has no NUL after its last byte and takes the bounds-checked path, and derive ignore_comments and ignore_trailing_commas from the first input byte, comparing against json::parse and json::accept with the same options. Signed-off-by: Niels Lohmann --- tests/fuzzing.md | 5 +- tests/src/fuzzer-parse_json_view.cpp | 123 +++++++++++++++++++-------- 2 files changed, 93 insertions(+), 35 deletions(-) diff --git a/tests/fuzzing.md b/tests/fuzzing.md index de2d66716..ca305cf48 100644 --- a/tests/fuzzing.md +++ b/tests/fuzzing.md @@ -6,7 +6,10 @@ Each parser of the library (JSON, BJData, BON8, BSON, CBOR, MessagePack, and UBJ Additionally, `parse_json_view_fuzzer` (`tests/src/fuzzer-parse_json_view.cpp`) cross-checks `json_document`/`json_view` (the zero-copy, read-only view declared in `json_view.hpp`) against `basic_json` on the same JSON text: it asserts that `json_document::accept` agrees with `json::accept`, that an accepted input materializes to the same value `json::parse` -produces, and that a rejected input makes both parsers throw with an identical `what()`. It takes plain JSON text, so it +produces, and that a rejected input makes both parsers throw with an identical `what()`. It checks this for a +`std::string` input (borrowed, with a NUL after the last byte) and for an exact-size `std::vector` +(borrowed, with nothing after the last byte), and for the `ignore_comments` and `ignore_trailing_commas` options, which +are taken from the low bits of the first input byte (the byte stays part of the text). It takes plain JSON text, so it reuses the `corpus_json` corpus rather than a format of its own. ## What the fuzzers check diff --git a/tests/src/fuzzer-parse_json_view.cpp b/tests/src/fuzzer-parse_json_view.cpp index 59e32a556..9b9f5a02a 100644 --- a/tests/src/fuzzer-parse_json_view.cpp +++ b/tests/src/fuzzer-parse_json_view.cpp @@ -9,7 +9,9 @@ /* This file implements a parser test suitable for fuzz testing. It checks that json_document (the zero-copy, read-only view of a parsed JSON text declared in -json_view.hpp) agrees with basic_json on every input: +json_view.hpp) agrees with basic_json on every input, for the parse options +selected by the low bits of the first input byte (bit 0: ignore_comments, bit 1: +ignore_trailing_commas; the byte stays part of the text): - json_document::accept(data) must equal json::accept(data) - if the input is accepted, json_document::parse(data).root().materialize() @@ -18,12 +20,20 @@ json_view.hpp) agrees with basic_json on every input: enabled) must throw a json::parse_error or json::out_of_range whose what() is identical to the one json::parse(data) throws +This is checked for two kinds of input: a std::string, which the document +borrows and which ends in the NUL the parser uses as sentinel, and an +exact-size byte vector, which has no NUL after its last byte and takes the +parser's bounds-checked path (AddressSanitizer reports any read past the end). + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ #include +#include +#include #include +#include #include #include @@ -35,65 +45,110 @@ drivers. using json = nlohmann::json; using json_document = nlohmann::json_document; -// see http://llvm.org/docs/LibFuzzer.html -extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) +namespace { - // json_document::accept only has a single-argument overload; wrap the raw - // bytes in a (borrowed) std::string so the same bytes can be handed to it - const std::string input(reinterpret_cast(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) +// what json::parse does with a text: the value, or the message of the exception +struct reference_result +{ + bool accepted = false; + json value{}; + std::string what{}; // NOLINT(readability-redundant-member-init) +}; - const bool accepted_by_json = json::accept(data, data + size); - const bool accepted_by_view = json_document::accept(input); +reference_result parse_reference(const std::uint8_t* data, std::size_t size, bool comments, bool trailing_commas) +{ + reference_result r; + r.accepted = json::accept(data, data + size, comments, trailing_commas); + bool json_threw = false; + try + { + r.value = json::parse(data, data + size, nullptr, true, comments, trailing_commas); + } + catch (const json::parse_error& e) + { + r.what = e.what(); + json_threw = true; + } + catch (const json::out_of_range& e) + { + r.what = e.what(); + json_threw = true; + } + // json::accept and json::parse must agree + assert(json_threw == !r.accepted); + static_cast(json_threw); + return r; +} +// json_document must agree with the reference for this input (a container +// that json_document::parse borrows) +template +void check_input(const Input& input, const reference_result& expected, bool comments, bool trailing_commas) +{ // json_document::accept must agree with json::accept on every input - assert(accepted_by_json == accepted_by_view); + const bool accepted_by_view = json_document::accept(input, comments, trailing_commas); + assert(expected.accepted == accepted_by_view); + static_cast(accepted_by_view); - if (accepted_by_json) + if (expected.accepted) { // both parsers must agree on the resulting value - json const j1 = json::parse(data, data + size); - json_document const doc = json_document::parse(input); + json_document const doc = json_document::parse(input, true, comments, trailing_commas); + assert(!doc.is_discarded()); json const j2 = doc.root().materialize(); - assert(j1 == j2); + assert(expected.value == j2); + static_cast(j2); + + // (without exceptions, the same document) + json_document const quiet = json_document::parse(input, false, comments, trailing_commas); + assert(!quiet.is_discarded()); + assert(quiet.node_count() == doc.node_count()); } else { // both parsers must reject the input the same way when exceptions are used - std::string expected_what; - bool json_threw = false; - try - { - static_cast(json::parse(data, data + size)); - } - catch (const json::parse_error& e) - { - expected_what = e.what(); - json_threw = true; - } - catch (const json::out_of_range& e) - { - expected_what = e.what(); - json_threw = true; - } - assert(json_threw); - bool view_threw = false; try { - static_cast(json_document::parse(input)); + static_cast(json_document::parse(input, true, comments, trailing_commas)); } catch (const json::parse_error& e) { - assert(e.what() == expected_what); + assert(e.what() == expected.what); view_threw = true; } catch (const json::out_of_range& e) { - assert(e.what() == expected_what); + assert(e.what() == expected.what); view_threw = true; } assert(view_threw); + static_cast(view_threw); + + // and without exceptions, the document is discarded + json_document const quiet = json_document::parse(input, false, comments, trailing_commas); + assert(quiet.is_discarded()); + static_cast(quiet); } +} +} // namespace + +// see http://llvm.org/docs/LibFuzzer.html +extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) +{ + // the parse options are taken from the low bits of the first byte + const bool comments = size > 0 && (data[0] & 1U) != 0; + const bool trailing_commas = size > 0 && (data[0] & 2U) != 0; + + const reference_result expected = parse_reference(data, size, comments, trailing_commas); + + // a std::string: borrowed, with the NUL of std::string as sentinel + const std::string input(reinterpret_cast(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + check_input(input, expected, comments, trailing_commas); + + // an exact-size byte vector: borrowed, with nothing after its last byte + const std::vector exact(data, data + size); + check_input(exact, expected, comments, trailing_commas); // return 0 - non-zero return values are reserved for future use return 0;