Fuzz json_document with exact-size buffers and parse options

The fuzzer only parsed a std::string with default options. Also parse an
exact-size byte vector, which has no NUL after its last byte and takes the
bounds-checked path, and derive ignore_comments and ignore_trailing_commas
from the first input byte, comparing against json::parse and json::accept
with the same options.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann committed 2026-10-09 16:32:20 +02:00
1 parent 0c76b316fc
commit aa7f0b02a0
2 files changed
+93 -35

No files matched your search

+4 -1
View File
@@ -6,7 +6,10 @@ Each parser of the library (JSON, BJData, BON8, BSON, CBOR, MessagePack, and UBJ
Additionally, `parse_json_view_fuzzer` (`tests/src/fuzzer-parse_json_view.cpp`) cross-checks `json_document`/`json_view`
(the zero-copy, read-only view declared in `json_view.hpp`) against `basic_json` on the same JSON text: it asserts that
`json_document::accept` agrees with `json::accept`, that an accepted input materializes to the same value `json::parse`
produces, and that a rejected input makes both parsers throw with an identical `what()`. It takes plain JSON text, so it
produces, and that a rejected input makes both parsers throw with an identical `what()`. It checks this for a
`std::string` input (borrowed, with a NUL after the last byte) and for an exact-size `std::vector<std::uint8_t>`
(borrowed, with nothing after the last byte), and for the `ignore_comments` and `ignore_trailing_commas` options, which
are taken from the low bits of the first input byte (the byte stays part of the text). It takes plain JSON text, so it
reuses the `corpus_json` corpus rather than a format of its own.
## What the fuzzers check
+89 -34
View File
@@ -9,7 +9,9 @@
/*
This file implements a parser test suitable for fuzz testing. It checks that
json_document (the zero-copy, read-only view of a parsed JSON text declared in
json_view.hpp) agrees with basic_json on every input:
json_view.hpp) agrees with basic_json on every input, for the parse options
selected by the low bits of the first input byte (bit 0: ignore_comments, bit 1:
ignore_trailing_commas; the byte stays part of the text):
- json_document::accept(data) must equal json::accept(data)
- if the input is accepted, json_document::parse(data).root().materialize()
@@ -18,12 +20,20 @@ json_view.hpp) agrees with basic_json on every input:
enabled) must throw a json::parse_error or json::out_of_range whose what()
is identical to the one json::parse(data) throws
This is checked for two kinds of input: a std::string, which the document
borrows and which ends in the NUL the parser uses as sentinel, and an
exact-size byte vector, which has no NUL after its last byte and takes the
parser's bounds-checked path (AddressSanitizer reports any read past the end).
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
drivers.
*/
#include <cassert>
#include <cstddef>
#include <cstdint>
#include <string>
#include <vector>
#include <nlohmann/json.hpp>
#include <nlohmann/json_view.hpp>
@@ -35,65 +45,110 @@ drivers.
using json = nlohmann::json;
using json_document = nlohmann::json_document;
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
namespace
{
// json_document::accept only has a single-argument overload; wrap the raw
// bytes in a (borrowed) std::string so the same bytes can be handed to it
const std::string input(reinterpret_cast<const char*>(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
// what json::parse does with a text: the value, or the message of the exception
struct reference_result
{
bool accepted = false;
json value{};
std::string what{}; // NOLINT(readability-redundant-member-init)
};
const bool accepted_by_json = json::accept(data, data + size);
const bool accepted_by_view = json_document::accept(input);
reference_result parse_reference(const std::uint8_t* data, std::size_t size, bool comments, bool trailing_commas)
{
reference_result r;
r.accepted = json::accept(data, data + size, comments, trailing_commas);
bool json_threw = false;
try
{
r.value = json::parse(data, data + size, nullptr, true, comments, trailing_commas);
}
catch (const json::parse_error& e)
{
r.what = e.what();
json_threw = true;
}
catch (const json::out_of_range& e)
{
r.what = e.what();
json_threw = true;
}
// json::accept and json::parse must agree
assert(json_threw == !r.accepted);
static_cast<void>(json_threw);
return r;
}
// json_document must agree with the reference for this input (a container
// that json_document::parse borrows)
template<typename Input>
void check_input(const Input& input, const reference_result& expected, bool comments, bool trailing_commas)
{
// json_document::accept must agree with json::accept on every input
assert(accepted_by_json == accepted_by_view);
const bool accepted_by_view = json_document::accept(input, comments, trailing_commas);
assert(expected.accepted == accepted_by_view);
static_cast<void>(accepted_by_view);
if (accepted_by_json)
if (expected.accepted)
{
// both parsers must agree on the resulting value
json const j1 = json::parse(data, data + size);
json_document const doc = json_document::parse(input);
json_document const doc = json_document::parse(input, true, comments, trailing_commas);
assert(!doc.is_discarded());
json const j2 = doc.root().materialize();
assert(j1 == j2);
assert(expected.value == j2);
static_cast<void>(j2);
// (without exceptions, the same document)
json_document const quiet = json_document::parse(input, false, comments, trailing_commas);
assert(!quiet.is_discarded());
assert(quiet.node_count() == doc.node_count());
}
else
{
// both parsers must reject the input the same way when exceptions are used
std::string expected_what;
bool json_threw = false;
try
{
static_cast<void>(json::parse(data, data + size));
}
catch (const json::parse_error& e)
{
expected_what = e.what();
json_threw = true;
}
catch (const json::out_of_range& e)
{
expected_what = e.what();
json_threw = true;
}
assert(json_threw);
bool view_threw = false;
try
{
static_cast<void>(json_document::parse(input));
static_cast<void>(json_document::parse(input, true, comments, trailing_commas));
}
catch (const json::parse_error& e)
{
assert(e.what() == expected_what);
assert(e.what() == expected.what);
view_threw = true;
}
catch (const json::out_of_range& e)
{
assert(e.what() == expected_what);
assert(e.what() == expected.what);
view_threw = true;
}
assert(view_threw);
static_cast<void>(view_threw);
// and without exceptions, the document is discarded
json_document const quiet = json_document::parse(input, false, comments, trailing_commas);
assert(quiet.is_discarded());
static_cast<void>(quiet);
}
}
} // namespace
// see http://llvm.org/docs/LibFuzzer.html
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
{
// the parse options are taken from the low bits of the first byte
const bool comments = size > 0 && (data[0] & 1U) != 0;
const bool trailing_commas = size > 0 && (data[0] & 2U) != 0;
const reference_result expected = parse_reference(data, size, comments, trailing_commas);
// a std::string: borrowed, with the NUL of std::string as sentinel
const std::string input(reinterpret_cast<const char*>(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
check_input(input, expected, comments, trailing_commas);
// an exact-size byte vector: borrowed, with nothing after its last byte
const std::vector<std::uint8_t> exact(data, data + size);
check_input(exact, expected, comments, trailing_commas);
// return 0 - non-zero return values are reserved for future use
return 0;