mirror of
https://github.com/nlohmann/json.git
synced 2026-10-10 16:37:14 +00:00
Compare commits
36
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
aeba637593 | ||
|
|
466e56d92d | ||
|
|
b7b75aac8a | ||
|
|
516b9b516a | ||
|
|
1d2dea7769 | ||
|
|
f8ff0942b2 | ||
|
|
d1e6439187 | ||
|
|
dcf7e2222c | ||
|
|
d99932fb3f | ||
|
|
1d2cb49ff3 | ||
|
|
a2766df787 | ||
|
|
2ea5bed0e3 | ||
|
|
632f5369fe | ||
|
|
c2619bf460 | ||
|
|
351aeb7440 | ||
|
|
7b0af0072e | ||
|
|
650886d126 | ||
|
|
a759afc99f | ||
|
|
1143da4faa | ||
|
|
8de151f928 | ||
|
|
ab52c98f71 | ||
|
|
5e93415d91 | ||
|
|
7e8e8e219b | ||
|
|
5f1727cef2 | ||
|
|
c16dd7e4f5 | ||
|
|
792853d725 | ||
|
|
4bdf1b7e74 | ||
|
|
b1e9d98e41 | ||
|
|
f855d257df | ||
|
|
3e683e9c04 | ||
|
|
d1d84ed9af | ||
|
|
de8529f99b | ||
|
|
677794f076 | ||
|
|
437a95cfdb | ||
|
|
c8735246d0 | ||
|
|
9adb510a0d |
No files matched your search
@@ -111,6 +111,8 @@ context which existing file needs to be extended, and only very few cases requir
|
||||
When fixing a bug, edit `unit-regression3.cpp` and add a section referencing the fixed issue.
|
||||
`unit-regression2.cpp` holds the older tests; the two files exist because a single one grew large enough for the
|
||||
MinGW linker to fail relocating it, so please keep adding to the smaller file rather than growing the larger one.
|
||||
Tests that call `sax_parse` go into `unit-sax_parse.cpp` rather than into a large test file: every call instantiates
|
||||
the parser and the binary reader that recover from errors, which grows a test file considerably.
|
||||
|
||||
#### Exceptions
|
||||
|
||||
@@ -149,10 +151,8 @@ make build -C docs/mkdocs # strict build: fails on broken links, anchor
|
||||
make check_mermaid -C docs/mkdocs # checks the Mermaid diagrams (requires Node.js)
|
||||
```
|
||||
|
||||
The search index of the docset is generated from [`mkdocs.yml`](https://github.com/nlohmann/json/blob/develop/docs/mkdocs/mkdocs.yml)
|
||||
and each page's title (H1) and declaration by
|
||||
[`docs/docset/generate_docset.py`](https://github.com/nlohmann/json/blob/develop/docs/docset/generate_docset.py);
|
||||
`make build` reports API pages that cannot be classified.
|
||||
A new API page also needs an entry in [`docs/docset/docSet.sql`](https://github.com/nlohmann/json/blob/develop/docs/docset/docSet.sql),
|
||||
the search index of the docset; `make build` reports missing entries.
|
||||
|
||||
### Amalgamate the source code
|
||||
|
||||
|
||||
@@ -494,7 +494,7 @@ bool key(string_t& val);
|
||||
bool parse_error(std::size_t position, const std::string& last_token, const detail::exception& ex);
|
||||
```
|
||||
|
||||
The return value of each function determines whether parsing should proceed.
|
||||
The return value of each function determines whether parsing should proceed. For `parse_error`, returning `true` [recovers from the error](https://json.nlohmann.me/features/parsing/error_recovery/): the parser repairs the input and continues.
|
||||
|
||||
To implement your own SAX handler, proceed as follows:
|
||||
|
||||
@@ -502,7 +502,7 @@ To implement your own SAX handler, proceed as follows:
|
||||
2. Create an object of your SAX interface class, e.g. `my_sax`.
|
||||
3. Call `bool json::sax_parse(input, &my_sax)`; where the first parameter can be any input like a string or an input stream and the second parameter is a pointer to your SAX interface.
|
||||
|
||||
Note the `sax_parse` function only returns a `bool` indicating the result of the last executed SAX event. It does not return a `json` value - it is up to you to decide what to do with the SAX events. Furthermore, no exceptions are thrown in case of a parse error -- it is up to you what to do with the exception object passed to your `parse_error` implementation. Internally, the SAX interface is used for the DOM parser (class `json_sax_dom_parser`) as well as the acceptor (`json_sax_acceptor`), see file [`json_sax.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/json_sax.hpp).
|
||||
Note the `sax_parse` function only returns a `bool` indicating whether the input was parsed without errors and no SAX event returned `false`. It does not return a `json` value - it is up to you to decide what to do with the SAX events. Furthermore, no exceptions are thrown in case of a parse error -- it is up to you what to do with the exception object passed to your `parse_error` implementation. Internally, the SAX interface is used for the DOM parser (class `json_sax_dom_parser`) as well as the acceptor (`json_sax_acceptor`), see file [`json_sax.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/json_sax.hpp).
|
||||
|
||||
### STL-like access
|
||||
|
||||
|
||||
+55
-11
@@ -1,24 +1,48 @@
|
||||
SHELL=/usr/bin/env bash
|
||||
PYTHON=../mkdocs/venv/bin/python3
|
||||
SED ?= $(shell which gsed 2>/dev/null || which sed)
|
||||
|
||||
MKDOCS_PAGES=$(shell cd ../mkdocs/docs/ && find * -type f -name '*.md' | sort)
|
||||
|
||||
.PHONY: all
|
||||
all: JSON_for_Modern_C++.tgz
|
||||
|
||||
# generate the search index (the docset target does this itself, this is
|
||||
# only handy for inspecting the index)
|
||||
docSet.dsidx: generate_docset.py
|
||||
$(PYTHON) generate_docset.py index docSet.dsidx
|
||||
docSet.dsidx: docSet.sql
|
||||
# generate index
|
||||
sqlite3 docSet.dsidx <docSet.sql
|
||||
|
||||
# build the documentation and turn it into a self-contained docset
|
||||
.PHONY: JSON_for_Modern_C++.docset
|
||||
JSON_for_Modern_C++.docset:
|
||||
JSON_for_Modern_C++.docset: Info.plist docSet.dsidx
|
||||
rm -fr JSON_for_Modern_C++.docset JSON_for_Modern_C++.tgz
|
||||
mkdir -p JSON_for_Modern_C++.docset/Contents/Resources/Documents/
|
||||
cp icon*.png JSON_for_Modern_C++.docset
|
||||
cp Info.plist JSON_for_Modern_C++.docset/Contents
|
||||
# build and copy documentation
|
||||
$(MAKE) install_venv -C ../mkdocs
|
||||
$(MAKE) build -C ../mkdocs
|
||||
$(PYTHON) generate_docset.py docset ../mkdocs/site .
|
||||
cp -r ../mkdocs/site/* JSON_for_Modern_C++.docset/Contents/Resources/Documents
|
||||
# patch CSS to hide navigation items
|
||||
echo -e "\n\nheader, footer, nav.md-tabs, nav.md-tabs--active, div.md-sidebar--primary, a.md-content__button { display: none; }" >> "$$(ls JSON_for_Modern_C++.docset/Contents/Resources/Documents/assets/stylesheets/main.*.min.css)"
|
||||
# fix spacing
|
||||
echo -e "\n\ndiv.md-sidebar div.md-sidebar--secondary, div.md-main__inner { top: 0; margin-top: 0 }" >> "$$(ls JSON_for_Modern_C++.docset/Contents/Resources/Documents/assets/stylesheets/main.*.min.css)"
|
||||
# remove "JSON for Modern C++" from page titles (fallback)
|
||||
find JSON_for_Modern_C++.docset/Contents/Resources/Documents -type f -exec $(SED) -i 's| - JSON for Modern C++</title>|</title>|' {} +
|
||||
# replace page titles with name from index, if available
|
||||
for page in $(MKDOCS_PAGES); do \
|
||||
case "$$page" in \
|
||||
*/index.md) path=$${page/\/index.md/} ;; \
|
||||
*) path=$${page/.md/} ;; \
|
||||
esac; \
|
||||
title=$$(sqlite3 docSet.dsidx "SELECT name FROM searchIndex WHERE path='$$path/index.html'" | tr '\n' ',' | $(SED) -e 's/,/, /g' -e 's/, $$/\n/'); \
|
||||
if [ "x$$title" != "x" ]; then \
|
||||
$(SED) -i "s%<title>.*</title>%<title>$$title</title>%" "JSON_for_Modern_C++.docset/Contents/Resources/Documents/$$path/index.html"; \
|
||||
fi \
|
||||
done
|
||||
# clean up
|
||||
rm JSON_for_Modern_C++.docset/Contents/Resources/Documents/sitemap.*
|
||||
# copy index
|
||||
cp docSet.dsidx JSON_for_Modern_C++.docset/Contents/Resources/
|
||||
|
||||
.PHONY: JSON_for_Modern_C++.tgz
|
||||
JSON_for_Modern_C++.tgz: JSON_for_Modern_C++.docset
|
||||
$(PYTHON) generate_docset.py tgz .
|
||||
tar --exclude='.DS_Store' -cvzf JSON_for_Modern_C++.tgz JSON_for_Modern_C++.docset
|
||||
|
||||
# install docset for Zeal documentation browser (https://zealdocs.org/)
|
||||
.PHONY: install_docset_zeal
|
||||
@@ -28,6 +52,26 @@ install_docset_zeal: JSON_for_Modern_C++.docset
|
||||
mkdir -p $$docset_root; \
|
||||
cp -r JSON_for_Modern_C++.docset $$docset_root/
|
||||
|
||||
# both targets below compare the docset search index with the mkdocs page
|
||||
# set. They share the same normalization (docs/foo/index.md and
|
||||
# docs/foo.md both become foo/index.html, the URL mkdocs itself would
|
||||
# give the page; the top-level index.md is excluded, as it is not part
|
||||
# of the hand-curated docSet.sql) and use comm(1) on two sorted lists
|
||||
# instead of running a sqlite3 query, or an O(n*m) nested shell loop,
|
||||
# once per page.
|
||||
DOCSET_INDEX_PATHS=$(shell sqlite3 docSet.dsidx "SELECT DISTINCT path FROM searchIndex" | sort)
|
||||
DOCSET_PAGE_PATHS=$(shell echo '$(MKDOCS_PAGES)' | tr ' ' '\n' | grep -v '^index\.md$$' | $(SED) -E 's@/index\.md$$@/index.html@; s@\.md$$@/index.html@' | sort)
|
||||
|
||||
# list mkdocs pages missing from the docset index
|
||||
.PHONY: list_missing_pages
|
||||
list_missing_pages: docSet.dsidx
|
||||
@comm -23 <(echo '$(DOCSET_PAGE_PATHS)' | tr ' ' '\n') <(echo '$(DOCSET_INDEX_PATHS)' | tr ' ' '\n')
|
||||
|
||||
# list paths in the docset index without a corresponding mkdocs page
|
||||
.PHONY: list_removed_paths
|
||||
list_removed_paths: docSet.dsidx
|
||||
@comm -13 <(echo '$(DOCSET_PAGE_PATHS)' | tr ' ' '\n') <(echo '$(DOCSET_INDEX_PATHS)' | tr ' ' '\n')
|
||||
|
||||
.PHONY: clean
|
||||
clean:
|
||||
rm -f docSet.dsidx
|
||||
|
||||
@@ -4,8 +4,7 @@ The folder contains the required files to create a [docset](https://kapeli.com/d
|
||||
documentation browsers like [Dash](https://kapeli.com/dash), [Velocity](https://velocity.silverlakesoftware.com), or
|
||||
[Zeal](https://zealdocs.org).
|
||||
|
||||
The docset (pages and search index) is generated by `generate_docset.py` from the mkdocs site and `mkdocs.yml`. It
|
||||
can be created with
|
||||
The docset can be created with
|
||||
|
||||
```sh
|
||||
make JSON_for_Modern_C++.docset
|
||||
@@ -13,7 +12,6 @@ make JSON_for_Modern_C++.docset
|
||||
|
||||
The generated folder `JSON_for_Modern_C++.docset` can then be opened in the documentation browser. `make all` builds a
|
||||
`JSON_for_Modern_C++.tgz` archive instead, and `make install_docset_zeal` installs the docset for Zeal directly.
|
||||
`make docSet.dsidx` builds only the search index.
|
||||
|
||||
A recent version is also part of the [Dash user contributions](https://github.com/Kapeli/Dash-User-Contributions/tree/master/docsets/JSON_for_Modern_C%2B%2B).
|
||||
|
||||
|
||||
@@ -0,0 +1,289 @@
|
||||
DROP TABLE IF EXISTS searchIndex;
|
||||
CREATE TABLE searchIndex(id INTEGER PRIMARY KEY, name TEXT, type TEXT, path TEXT);
|
||||
CREATE UNIQUE INDEX anchor ON searchIndex (name, type, path);
|
||||
|
||||
-- API
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('adl_serializer', 'Class', 'api/adl_serializer/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('adl_serializer::from_json', 'Function', 'api/adl_serializer/from_json/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('adl_serializer::to_json', 'Function', 'api/adl_serializer/to_json/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('byte_container_with_subtype', 'Class', 'api/byte_container_with_subtype/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('byte_container_with_subtype::byte_container_with_subtype', 'Constructor', 'api/byte_container_with_subtype/byte_container_with_subtype/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('byte_container_with_subtype::clear_subtype', 'Method', 'api/byte_container_with_subtype/clear_subtype/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('byte_container_with_subtype::has_subtype', 'Method', 'api/byte_container_with_subtype/has_subtype/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('byte_container_with_subtype::operator!=', 'Operator', 'api/byte_container_with_subtype/operator_ne/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('byte_container_with_subtype::operator==', 'Operator', 'api/byte_container_with_subtype/operator_eq/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('byte_container_with_subtype::set_subtype', 'Method', 'api/byte_container_with_subtype/set_subtype/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('byte_container_with_subtype::subtype', 'Method', 'api/byte_container_with_subtype/subtype/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json', 'Class', 'api/basic_json/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('format_as', 'Function', 'api/basic_json/format_as/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::accept', 'Function', 'api/basic_json/accept/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::array', 'Function', 'api/basic_json/array/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::array_t', 'Type', 'api/basic_json/array_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::as_base_class', 'Method', 'api/basic_json/as_base_class/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::at', 'Method', 'api/basic_json/at/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::back', 'Method', 'api/basic_json/back/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::basic_json', 'Constructor', 'api/basic_json/basic_json/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::begin', 'Method', 'api/basic_json/begin/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::binary', 'Function', 'api/basic_json/binary/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::binary_t', 'Type', 'api/basic_json/binary_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::boolean_t', 'Type', 'api/basic_json/boolean_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::cbegin', 'Method', 'api/basic_json/cbegin/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::cbor_tag_handler_t', 'Enum', 'api/basic_json/cbor_tag_handler_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::cend', 'Method', 'api/basic_json/cend/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::clear', 'Method', 'api/basic_json/clear/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::contains', 'Method', 'api/basic_json/contains/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::count', 'Method', 'api/basic_json/count/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::crbegin', 'Method', 'api/basic_json/crbegin/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::crend', 'Method', 'api/basic_json/crend/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::default_object_comparator_t', 'Type', 'api/basic_json/default_object_comparator_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::diff', 'Function', 'api/basic_json/diff/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::dump', 'Method', 'api/basic_json/dump/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::emplace', 'Method', 'api/basic_json/emplace/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::emplace_back', 'Method', 'api/basic_json/emplace_back/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::empty', 'Method', 'api/basic_json/empty/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::end', 'Method', 'api/basic_json/end/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::end_pos', 'Method', 'api/basic_json/end_pos/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::erase', 'Method', 'api/basic_json/erase/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::error_handler_t', 'Enum', 'api/basic_json/error_handler_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::exception', 'Class', 'api/basic_json/exception/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::find', 'Method', 'api/basic_json/find/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::flatten', 'Method', 'api/basic_json/flatten/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::from_bjdata', 'Function', 'api/basic_json/from_bjdata/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::from_bson', 'Function', 'api/basic_json/from_bson/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::from_cbor', 'Function', 'api/basic_json/from_cbor/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::from_msgpack', 'Function', 'api/basic_json/from_msgpack/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::from_bon8', 'Function', 'api/basic_json/from_bon8/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::from_ubjson', 'Function', 'api/basic_json/from_ubjson/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::front', 'Method', 'api/basic_json/front/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::get', 'Method', 'api/basic_json/get/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::get_allocator', 'Function', 'api/basic_json/get_allocator/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::get_binary', 'Method', 'api/basic_json/get_binary/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::get_ptr', 'Method', 'api/basic_json/get_ptr/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::get_ref', 'Method', 'api/basic_json/get_ref/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::get_to', 'Method', 'api/basic_json/get_to/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::input_format_t', 'Enum', 'api/basic_json/input_format_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::insert', 'Method', 'api/basic_json/insert/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::invalid_iterator', 'Class', 'api/basic_json/invalid_iterator/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_array', 'Method', 'api/basic_json/is_array/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_binary', 'Method', 'api/basic_json/is_binary/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_boolean', 'Method', 'api/basic_json/is_boolean/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_discarded', 'Method', 'api/basic_json/is_discarded/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_null', 'Method', 'api/basic_json/is_null/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_number', 'Method', 'api/basic_json/is_number/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_number_float', 'Method', 'api/basic_json/is_number_float/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_number_integer', 'Method', 'api/basic_json/is_number_integer/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_number_unsigned', 'Method', 'api/basic_json/is_number_unsigned/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_object', 'Method', 'api/basic_json/is_object/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_primitive', 'Method', 'api/basic_json/is_primitive/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_string', 'Method', 'api/basic_json/is_string/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::is_structured', 'Method', 'api/basic_json/is_structured/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::items', 'Method', 'api/basic_json/items/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::json_base_class_t', 'Type', 'api/basic_json/json_base_class_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::json_serializer', 'Class', 'api/basic_json/json_serializer/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::max_size', 'Method', 'api/basic_json/max_size/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::merge_patch', 'Method', 'api/basic_json/merge_patch/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::meta', 'Function', 'api/basic_json/meta/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::number_float_t', 'Type', 'api/basic_json/number_float_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::number_integer_t', 'Type', 'api/basic_json/number_integer_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::number_unsigned_t', 'Type', 'api/basic_json/number_unsigned_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::object', 'Function', 'api/basic_json/object/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::object_comparator_t', 'Type', 'api/basic_json/object_comparator_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::object_t', 'Type', 'api/basic_json/object_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::operator ValueType', 'Operator', 'api/basic_json/operator_ValueType/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::operator value_t', 'Operator', 'api/basic_json/operator_value_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::operator[]', 'Operator', 'api/basic_json/operator[]/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::operator=', 'Operator', 'api/basic_json/operator=/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::operator+=', 'Operator', 'api/basic_json/operator+=/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::operator==', 'Operator', 'api/basic_json/operator_eq/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::operator!=', 'Operator', 'api/basic_json/operator_ne/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::operator<', 'Operator', 'api/basic_json/operator_lt/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::operator<=', 'Operator', 'api/basic_json/operator_le/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::operator>', 'Operator', 'api/basic_json/operator_gt/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::operator>=', 'Operator', 'api/basic_json/operator_ge/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::operator<=>', 'Operator', 'api/basic_json/operator_spaceship/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::out_of_range', 'Class', 'api/basic_json/out_of_range/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::other_error', 'Class', 'api/basic_json/other_error/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::parse', 'Function', 'api/basic_json/parse/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::parse_error', 'Class', 'api/basic_json/parse_error/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::parse_event_t', 'Enum', 'api/basic_json/parse_event_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::parser_callback_t', 'Type', 'api/basic_json/parser_callback_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::patch', 'Method', 'api/basic_json/patch/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::patch_inplace', 'Method', 'api/basic_json/patch_inplace/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::push_back', 'Method', 'api/basic_json/push_back/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::rbegin', 'Method', 'api/basic_json/rbegin/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::rend', 'Method', 'api/basic_json/rend/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::sax_parse', 'Function', 'api/basic_json/sax_parse/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::size', 'Method', 'api/basic_json/size/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::start_pos', 'Method', 'api/basic_json/start_pos/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::string_t', 'Type', 'api/basic_json/string_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::swap', 'Method', 'api/basic_json/swap/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::type', 'Method', 'api/basic_json/type/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::type_error', 'Class', 'api/basic_json/type_error/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::type_name', 'Method', 'api/basic_json/type_name/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::unflatten', 'Method', 'api/basic_json/unflatten/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::update', 'Method', 'api/basic_json/update/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_bjdata', 'Function', 'api/basic_json/to_bjdata/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_bson', 'Function', 'api/basic_json/to_bson/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_cbor', 'Function', 'api/basic_json/to_cbor/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_msgpack', 'Function', 'api/basic_json/to_msgpack/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_bon8', 'Function', 'api/basic_json/to_bon8/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_string', 'Method', 'api/basic_json/to_string/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_ubjson', 'Function', 'api/basic_json/to_ubjson/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::value', 'Method', 'api/basic_json/value/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::value_t', 'Enum', 'api/basic_json/value_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::with_t', 'Type', 'api/basic_json/with_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::~basic_json', 'Method', 'api/basic_json/~basic_json/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json', 'Class', 'api/json/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer', 'Class', 'api/json_pointer/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::back', 'Method', 'api/json_pointer/back/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::empty', 'Method', 'api/json_pointer/empty/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::front', 'Method', 'api/json_pointer/front/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::json_pointer', 'Constructor', 'api/json_pointer/json_pointer/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::operator==', 'Operator', 'api/json_pointer/operator_eq/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::operator!=', 'Operator', 'api/json_pointer/operator_ne/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::operator/', 'Operator', 'api/json_pointer/operator_slash/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::operator/=', 'Operator', 'api/json_pointer/operator_slasheq/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::operator string_t', 'Operator', 'api/json_pointer/operator_string_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::operator<=>', 'Operator', 'api/json_pointer/operator_spaceship/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::parent_pointer', 'Method', 'api/json_pointer/parent_pointer/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::pop_back', 'Method', 'api/json_pointer/pop_back/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::pop_front', 'Method', 'api/json_pointer/pop_front/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::push_back', 'Method', 'api/json_pointer/push_back/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::push_front', 'Method', 'api/json_pointer/push_front/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::string_t', 'Type', 'api/json_pointer/string_t/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::to_string', 'Method', 'api/json_pointer/to_string/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax', 'Class', 'api/json_sax/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::binary', 'Method', 'api/json_sax/binary/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::boolean', 'Method', 'api/json_sax/boolean/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::end_array', 'Method', 'api/json_sax/end_array/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::end_object', 'Method', 'api/json_sax/end_object/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::key', 'Method', 'api/json_sax/key/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::null', 'Method', 'api/json_sax/null/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::number_float', 'Method', 'api/json_sax/number_float/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::number_integer', 'Method', 'api/json_sax/number_integer/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::number_unsigned', 'Method', 'api/json_sax/number_unsigned/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::parse_error', 'Method', 'api/json_sax/parse_error/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::start_array', 'Method', 'api/json_sax/start_array/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::start_object', 'Method', 'api/json_sax/start_object/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('json_sax::string', 'Method', 'api/json_sax/string/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('operator""_json', 'Literal', 'api/operator_literal_json/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('operator""_json_pointer', 'Literal', 'api/operator_literal_json_pointer/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('operator<<', 'Operator', 'api/operator_ltlt/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('operator>>', 'Operator', 'api/operator_gtgt/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('ordered_json', 'Class', 'api/ordered_json/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('ordered_map', 'Class', 'api/ordered_map/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('std::formatter<basic_json>', 'Class', 'api/basic_json/std_formatter/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('std::hash<basic_json>', 'Class', 'api/basic_json/std_hash/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('std::swap<basic_json>', 'Function', 'api/basic_json/std_swap/index.html');
|
||||
|
||||
-- Features
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Arbitrary Type Conversions', 'Guide', 'features/arbitrary_types/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats', 'Guide', 'features/binary_formats/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats: BJData', 'Guide', 'features/binary_formats/bjdata/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats: BSON', 'Guide', 'features/binary_formats/bson/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats: CBOR', 'Guide', 'features/binary_formats/cbor/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats: MessagePack', 'Guide', 'features/binary_formats/messagepack/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats: BON8', 'Guide', 'features/binary_formats/bon8/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats: UBJSON', 'Guide', 'features/binary_formats/ubjson/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Binary Values', 'Guide', 'features/binary_values/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Comments', 'Guide', 'features/comments/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Element Access', 'Guide', 'features/element_access/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Element Access: Access with default value: value', 'Guide', 'features/element_access/default_value/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Element Access: Checked access: at', 'Guide', 'features/element_access/checked_access/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Element Access: Unchecked access: operator[]', 'Guide', 'features/element_access/unchecked_access/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Exceptions', 'Guide', 'home/exceptions/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Integration: Migration Guide', 'Guide', 'integration/migration_guide/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Integration: CMake', 'Guide', 'integration/cmake/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Integration: Header only', 'Guide', 'integration/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Integration: Package Managers', 'Guide', 'integration/package_managers/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Integration: Pkg-config', 'Guide', 'integration/pkg-config/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Iterators', 'Guide', 'features/iterators/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON Merge Patch', 'Guide', 'features/merge_patch/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON Patch and Diff', 'Guide', 'features/json_patch/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON Pointer', 'Guide', 'features/json_pointer/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('nlohmann Namespace', 'Guide', 'features/namespace/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Types', 'Guide', 'features/types/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Types: Number Handling', 'Guide', 'features/types/number_handling/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Object Order', 'Guide', 'features/object_order/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Performance', 'Guide', 'features/performance/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Parsing', 'Guide', 'features/parsing/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Parsing: JSON Lines', 'Guide', 'features/parsing/json_lines/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Parsing: Parser Callbacks', 'Guide', 'features/parsing/parser_callbacks/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Parsing: Parsing and Exceptions', 'Guide', 'features/parsing/parse_exceptions/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Parsing: SAX Interface', 'Guide', 'features/parsing/sax_interface/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Parsing: Untrusted Input', 'Guide', 'features/parsing/untrusted_input/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Runtime Assertions', 'Guide', 'features/assertions/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Specializing enum conversion', 'Guide', 'features/enum_conversion/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Supported Macros', 'Guide', 'features/macros/index.html');
|
||||
|
||||
-- Macros
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_ASSERT', 'Macro', 'api/macros/json_assert/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_BRACE_INIT_COPY_SEMANTICS', 'Macro', 'api/macros/json_brace_init_copy_semantics/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_CATCH_USER', 'Macro', 'api/macros/json_throw_user/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DELETE_DEPRECATED_FUNCTIONS', 'Macro', 'api/macros/json_delete_deprecated_functions/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DIAGNOSTICS', 'Macro', 'api/macros/json_diagnostics/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DIAGNOSTIC_POSITIONS', 'Macro', 'api/macros/json_diagnostic_positions/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DISABLE_ENUM_SERIALIZATION', 'Macro', 'api/macros/json_disable_enum_serialization/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DISABLE_TUPLE_REFERENCE_CONVERSION', 'Macro', 'api/macros/json_disable_tuple_reference_conversion/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_CPP_11', 'Macro', 'api/macros/json_has_cpp_11/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_CPP_14', 'Macro', 'api/macros/json_has_cpp_11/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_CPP_17', 'Macro', 'api/macros/json_has_cpp_11/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_CPP_20', 'Macro', 'api/macros/json_has_cpp_11/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_CPP_23', 'Macro', 'api/macros/json_has_cpp_11/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_CPP_26', 'Macro', 'api/macros/json_has_cpp_11/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_EXPERIMENTAL_FILESYSTEM', 'Macro', 'api/macros/json_has_filesystem/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_FILESYSTEM', 'Macro', 'api/macros/json_has_filesystem/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_RANGES', 'Macro', 'api/macros/json_has_ranges/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_STATIC_RTTI', 'Macro', 'api/macros/json_has_static_rtti/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_STD_FORMAT', 'Macro', 'api/macros/json_has_std_format/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_THREE_WAY_COMPARISON', 'Macro', 'api/macros/json_has_three_way_comparison/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_NOEXCEPTION', 'Macro', 'api/macros/json_noexception/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_NO_AUTOMATIC_UDLS', 'Macro', 'api/macros/json_no_automatic_udls/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_NO_IO', 'Macro', 'api/macros/json_no_io/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_NO_THREAD_LOCAL', 'Macro', 'api/macros/json_no_thread_local/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_PRECISE_STREAM_POSITION', 'Macro', 'api/macros/json_precise_stream_position/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_SKIP_LIBRARY_VERSION_CHECK', 'Macro', 'api/macros/json_skip_library_version_check/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_SKIP_UNSUPPORTED_COMPILER_CHECK', 'Macro', 'api/macros/json_skip_unsupported_compiler_check/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_STRICT_BINARY_UTF8', 'Macro', 'api/macros/json_strict_binary_utf8/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_STRICT_NUL_HANDLING', 'Macro', 'api/macros/json_strict_nul_handling/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_THROW_USER', 'Macro', 'api/macros/json_throw_user/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_TRY_USER', 'Macro', 'api/macros/json_throw_user/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_USE_GLOBAL_UDLS', 'Macro', 'api/macros/json_use_global_udls/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_USE_IMPLICIT_CONVERSIONS', 'Macro', 'api/macros/json_use_implicit_conversions/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON', 'Macro', 'api/macros/json_use_legacy_discarded_value_comparison/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_USE_OBJECTS_FOR_ENUM_KEYED_MAPS', 'Macro', 'api/macros/json_use_objects_for_enum_keyed_maps/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_USE_SIMDUTF', 'Macro', 'api/macros/json_use_simdutf/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('Macros', 'Macro', 'api/macros/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE', 'Macro', 'api/macros/nlohmann_define_derived_type/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE', 'Macro', 'api/macros/nlohmann_define_derived_type/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES', 'Macro', 'api/macros/nlohmann_define_type_with_names/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT', 'Macro', 'api/macros/nlohmann_define_derived_type/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT_WITH_NAMES', 'Macro', 'api/macros/nlohmann_define_type_with_names/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_NAMES', 'Macro', 'api/macros/nlohmann_define_type_with_names/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE', 'Macro', 'api/macros/nlohmann_define_derived_type/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE', 'Macro', 'api/macros/nlohmann_define_derived_type/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES', 'Macro', 'api/macros/nlohmann_define_type_with_names/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT', 'Macro', 'api/macros/nlohmann_define_derived_type/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT_WITH_NAMES', 'Macro', 'api/macros/nlohmann_define_type_with_names/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_NAMES', 'Macro', 'api/macros/nlohmann_define_type_with_names/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_TYPE_INTRUSIVE', 'Macro', 'api/macros/nlohmann_define_type_intrusive/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE', 'Macro', 'api/macros/nlohmann_define_type_intrusive/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES', 'Macro', 'api/macros/nlohmann_define_type_with_names/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT', 'Macro', 'api/macros/nlohmann_define_type_intrusive/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT_WITH_NAMES', 'Macro', 'api/macros/nlohmann_define_type_with_names/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_NAMES', 'Macro', 'api/macros/nlohmann_define_type_with_names/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE', 'Macro', 'api/macros/nlohmann_define_type_non_intrusive/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE', 'Macro', 'api/macros/nlohmann_define_type_non_intrusive/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES', 'Macro', 'api/macros/nlohmann_define_type_with_names/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT', 'Macro', 'api/macros/nlohmann_define_type_non_intrusive/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT_WITH_NAMES', 'Macro', 'api/macros/nlohmann_define_type_with_names/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_NAMES', 'Macro', 'api/macros/nlohmann_define_type_with_names/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_NAMESPACE', 'Macro', 'api/macros/nlohmann_json_namespace/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_NAMESPACE_BEGIN', 'Macro', 'api/macros/nlohmann_json_namespace_begin/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_NAMESPACE_END', 'Macro', 'api/macros/nlohmann_json_namespace_begin/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_NAMESPACE_NO_VERSION', 'Macro', 'api/macros/nlohmann_json_namespace_no_version/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_SERIALIZE_ENUM', 'Macro', 'api/macros/nlohmann_json_serialize_enum/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_SERIALIZE_ENUM_STRICT', 'Macro', 'api/macros/nlohmann_json_serialize_enum_strict/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_VERSION_MAJOR', 'Macro', 'api/macros/nlohmann_json_version_major/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_VERSION_MINOR', 'Macro', 'api/macros/nlohmann_json_version_major/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_VERSION_PATCH', 'Macro', 'api/macros/nlohmann_json_version_major/index.html');
|
||||
@@ -1,532 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
"""Generate the Dash docset search index from the mkdocs sources."""
|
||||
|
||||
import argparse
|
||||
import glob
|
||||
import hashlib
|
||||
import html
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import sqlite3
|
||||
import sys
|
||||
import tarfile
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
|
||||
import yaml
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
MKDOCS_YML = os.path.join(HERE, '..', 'mkdocs', 'mkdocs.yml')
|
||||
PAGES = os.path.join(HERE, '..', 'mkdocs', 'docs')
|
||||
|
||||
# api pages whose (name, type) cannot be derived by the heuristics
|
||||
OVERRIDES = {
|
||||
}
|
||||
|
||||
DOCSET = 'JSON_for_Modern_C++.docset'
|
||||
TITLE_SUFFIX = ' - JSON for Modern C++</title>'
|
||||
|
||||
# CSS rules appended to the stylesheet: hide navigation items and fix spacing
|
||||
# hide the navigation (the documentation browser has its own); Material's class selectors would win over element
|
||||
# selectors, hence the classes and !important
|
||||
CSS_PATCH = (
|
||||
'\n\n.md-header, .md-footer, .md-tabs, .md-sidebar--primary, .md-content__button { display: none !important; }'
|
||||
'\n\n.md-sidebar--secondary, .md-main__inner { top: 0 !important; margin-top: 0 !important; }'
|
||||
)
|
||||
|
||||
USER_AGENT = ('Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/124.0 Safari/537.36')
|
||||
CONTENT_TYPE_EXT = {'image/svg+xml': '.svg', 'image/png': '.png', 'image/jpeg': '.jpg',
|
||||
'image/gif': '.gif', 'image/webp': '.webp'}
|
||||
|
||||
# remote loads that are allowed to remain (URL -> reason)
|
||||
ALLOWED_REMOTE = {
|
||||
# Material only loads this polyfill if the browser has no ResizeObserver
|
||||
'https://unpkg.com/resize-observer-polyfill': 'fallback for browsers without ResizeObserver',
|
||||
}
|
||||
|
||||
problems = []
|
||||
|
||||
|
||||
def problem(page, reason) -> None:
|
||||
"""Record a problem; all problems are reported at the end."""
|
||||
problems.append(f'generate_docset.py: {page}: {reason}')
|
||||
|
||||
|
||||
class Loader(yaml.SafeLoader):
|
||||
"""YAML loader that tolerates the custom tags used in mkdocs.yml."""
|
||||
|
||||
|
||||
Loader.add_multi_constructor('', lambda loader, suffix, node: None)
|
||||
|
||||
|
||||
def walk_nav(items, groups=()):
|
||||
"""Yield (group titles, nav title or None, md path) for all nav leaves."""
|
||||
for item in items:
|
||||
if isinstance(item, str):
|
||||
yield list(groups), None, item
|
||||
elif isinstance(item, dict):
|
||||
for title, value in item.items():
|
||||
if isinstance(value, list):
|
||||
yield from walk_nav(value, groups + (str(title),))
|
||||
else:
|
||||
yield list(groups), str(title), value
|
||||
|
||||
|
||||
def page_path(md_path) -> str:
|
||||
"""Map a markdown path to the HTML path of the rendered page."""
|
||||
if md_path.endswith('/index.md'):
|
||||
return md_path[:-len('index.md')] + 'index.html'
|
||||
return md_path[:-len('.md')] + '/index.html'
|
||||
|
||||
|
||||
def read_page(md_path) -> str:
|
||||
"""Read a page (resolving a snippet include), return '' if it does not exist."""
|
||||
try:
|
||||
with open(os.path.join(PAGES, md_path), encoding='utf-8') as f:
|
||||
text = f.read()
|
||||
m = re.match(r'--8<-- "(.+)"\s*$', text)
|
||||
if m:
|
||||
with open(os.path.join(PAGES, m.group(1)), encoding='utf-8') as f:
|
||||
text = f.read()
|
||||
return text
|
||||
except OSError:
|
||||
return ''
|
||||
|
||||
|
||||
def clean(text) -> str:
|
||||
"""Strip tags and entities from a heading and normalize whitespace."""
|
||||
text = text.replace('\\>', '\x00') # escaped '>' is not the end of a tag
|
||||
text = html.unescape(re.sub(r'</?[a-zA-Z][^>]*>', '', text)).replace('\x00', '>')
|
||||
return re.sub(r'\s+', ' ', text).strip()
|
||||
|
||||
|
||||
def strip_fences(text) -> str:
|
||||
"""Remove fenced code blocks (their lines may start with '# ')."""
|
||||
return re.sub(r'^(```|~~~).*?^\1[^\n]*$', '', text, flags=re.M | re.S)
|
||||
|
||||
|
||||
def get_h1(text):
|
||||
"""Return the cleaned first H1 of a page or None."""
|
||||
body = strip_fences(text)
|
||||
m = re.search(r'^# (.+)$', body, flags=re.M)
|
||||
if m:
|
||||
return clean(m.group(1))
|
||||
m = re.search(r'<h1>(.*?)</h1>', body, flags=re.S)
|
||||
return clean(m.group(1)) if m else None
|
||||
|
||||
|
||||
def split_top_level(text, seps=','):
|
||||
"""Split at separators that are not nested in <>, () or []."""
|
||||
parts, depth, current = [], 0, ''
|
||||
for c in text:
|
||||
if c in '<([':
|
||||
depth += 1
|
||||
elif c in '>)]':
|
||||
depth -= 1
|
||||
if c in seps and depth <= 0:
|
||||
parts.append(current)
|
||||
current = ''
|
||||
else:
|
||||
current += c
|
||||
parts.append(current)
|
||||
return [p.strip() for p in parts if p.strip()]
|
||||
|
||||
|
||||
OPERATOR_SYMBOLS = ('<=>', '<<', '>>', '<=', '>=', '<', '>')
|
||||
|
||||
|
||||
def api_names(h1) -> list:
|
||||
"""Derive the entry names from the H1 of an api page."""
|
||||
# hide the angle brackets of operators from the nesting detection
|
||||
for i, sym in enumerate(OPERATOR_SYMBOLS):
|
||||
h1 = h1.replace('operator' + sym, f'operator\x01{i}\x01')
|
||||
names = []
|
||||
for part in split_top_level(h1):
|
||||
name = part.replace('\\', '').replace('nlohmann::', '')
|
||||
name = re.sub(r'\x01(\d)\x01', lambda m: OPERATOR_SYMBOLS[int(m.group(1))], name)
|
||||
# drop qualifiers like "operator<<(basic_json)", but keep "operator()"
|
||||
if not name.endswith('operator()'):
|
||||
name = re.sub(r'(?<=\w|[<>=!+\-*/\[\]])\([^()]*\)$', '', name)
|
||||
if name not in names:
|
||||
names.append(name)
|
||||
return names
|
||||
|
||||
|
||||
def first_cpp_block(text):
|
||||
"""Return the first ```cpp block following the H1."""
|
||||
m = re.search(r'^# .*?^```cpp\n(.*?)^```', text, flags=re.M | re.S)
|
||||
return m.group(1) if m else None
|
||||
|
||||
|
||||
def api_type(name, decl):
|
||||
"""Determine the Dash entry type from name and declaration."""
|
||||
parts = name.split('::')
|
||||
last = parts[-1]
|
||||
if name.startswith('operator""'):
|
||||
return 'Literal'
|
||||
if last.startswith('operator'):
|
||||
return 'Operator'
|
||||
if decl is None:
|
||||
return None
|
||||
if re.search(r'\benum\b', decl):
|
||||
return 'Enum'
|
||||
if re.search(r'^\s*(template\s*<.*>\s*)?(class|struct)\s+\w+\s*(final\b|[:{;<]|$)', decl, flags=re.M):
|
||||
return 'Class'
|
||||
if re.search(r'\busing\s+\w+\s*=', decl) or 'typedef' in decl:
|
||||
return 'Type'
|
||||
if len(parts) > 1 and last == parts[-2]:
|
||||
return 'Constructor'
|
||||
if last.startswith('~'):
|
||||
return 'Method'
|
||||
if re.search(r'\bstatic\b', decl) or len(parts) == 1 or parts[0] == 'std':
|
||||
return 'Function'
|
||||
return 'Method'
|
||||
|
||||
|
||||
def macro_names(h1) -> list:
|
||||
"""Split the H1 of a macro page into macro names."""
|
||||
return [n.strip() for n in re.split(r'[,/]', h1) if n.strip()]
|
||||
|
||||
|
||||
def api_entries(md_path):
|
||||
"""Return the (name, type) pairs of an api page."""
|
||||
if md_path in OVERRIDES:
|
||||
return OVERRIDES[md_path]
|
||||
if md_path == 'api/macros/index.md':
|
||||
return [('Macros', 'Macro')]
|
||||
text = read_page(md_path)
|
||||
h1 = get_h1(text)
|
||||
if not h1:
|
||||
problem(md_path, 'no H1 found')
|
||||
return []
|
||||
if md_path.startswith('api/macros/'):
|
||||
return [(n, 'Macro') for n in macro_names(h1)]
|
||||
names = api_names(h1)
|
||||
if not names:
|
||||
problem(md_path, 'no names found')
|
||||
return []
|
||||
decl = first_cpp_block(text)
|
||||
result = []
|
||||
for name in names:
|
||||
kind = api_type(name, decl)
|
||||
if kind is None:
|
||||
problem(md_path, f'no type determinable for {name}')
|
||||
else:
|
||||
result.append((name, kind))
|
||||
return result
|
||||
|
||||
|
||||
def guide_entries(nav):
|
||||
"""Yield (name, type, md path) for all non-api nav pages."""
|
||||
for groups, title, md_path in walk_nav(nav):
|
||||
if md_path.startswith('api/') or md_path == 'index.md':
|
||||
continue
|
||||
if not md_path.endswith('.md'):
|
||||
continue
|
||||
groups = groups[1:] # drop the top-level tab
|
||||
if title is None:
|
||||
title = get_h1(read_page(md_path))
|
||||
if not title:
|
||||
problem(md_path, 'no title found')
|
||||
continue
|
||||
# "Parsing: Parsing Untrusted Input" -> "Parsing: Untrusted Input"
|
||||
if groups and title.startswith(groups[-1] + ' ') and title[len(groups[-1]) + 1:][:1].isupper():
|
||||
title = title[len(groups[-1]) + 1:]
|
||||
if md_path.endswith('/index.md') and groups and title == groups[-1]:
|
||||
name = ': '.join(groups)
|
||||
else:
|
||||
name = ': '.join(groups + [title])
|
||||
yield name, 'Guide', md_path
|
||||
|
||||
|
||||
def build_entries() -> list:
|
||||
"""Return the sorted list of (name, type, path) index entries."""
|
||||
nav = load_mkdocs_yml()['nav']
|
||||
|
||||
entries = set()
|
||||
for name, kind, md_path in guide_entries(nav):
|
||||
entries.add((name, kind, page_path(md_path)))
|
||||
|
||||
api_root = os.path.join(PAGES, 'api')
|
||||
on_disk = set()
|
||||
for root, _, files in os.walk(api_root):
|
||||
for file in files:
|
||||
if file.endswith('.md'):
|
||||
rel = os.path.relpath(os.path.join(root, file), PAGES)
|
||||
on_disk.add(rel.replace(os.sep, '/'))
|
||||
for _, _, md_path in walk_nav(nav):
|
||||
if md_path.startswith('api/') and md_path not in on_disk:
|
||||
problem(md_path, 'listed in nav but missing on disk')
|
||||
for md_path in sorted(on_disk):
|
||||
for name, kind in api_entries(md_path):
|
||||
entries.add((name, kind, page_path(md_path)))
|
||||
return sorted(entries)
|
||||
|
||||
|
||||
def write_index(entries, out) -> None:
|
||||
"""Write the entries into a SQLite search index."""
|
||||
if os.path.exists(out):
|
||||
os.remove(out)
|
||||
con = sqlite3.connect(out)
|
||||
con.execute('CREATE TABLE searchIndex(id INTEGER PRIMARY KEY, name TEXT, type TEXT, path TEXT)')
|
||||
con.execute('CREATE UNIQUE INDEX anchor ON searchIndex (name, type, path)')
|
||||
con.executemany('INSERT INTO searchIndex(name, type, path) VALUES (?, ?, ?)', entries)
|
||||
con.commit()
|
||||
con.close()
|
||||
|
||||
|
||||
def html_files(root):
|
||||
"""Yield all HTML files below root."""
|
||||
for base, _, files in os.walk(root):
|
||||
for file in files:
|
||||
if file.endswith('.html'):
|
||||
yield os.path.join(base, file)
|
||||
|
||||
|
||||
def read(path) -> str:
|
||||
with open(path, encoding='utf-8') as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
def write(path, text) -> None:
|
||||
with open(path, 'w', encoding='utf-8') as f:
|
||||
f.write(text)
|
||||
|
||||
|
||||
def remove_source_widget(docs) -> None:
|
||||
"""Drop data-md-component=source so Material does not query api.github.com for the stars and version."""
|
||||
pattern = re.compile(r'\sdata-md-component=(?:"source"|source\b)')
|
||||
for path in html_files(docs):
|
||||
text = read(path)
|
||||
new = pattern.sub('', text)
|
||||
if new != text:
|
||||
write(path, new)
|
||||
|
||||
|
||||
def patch_titles(docs, entries) -> None:
|
||||
"""Strip the site name from all titles; use the index names where available."""
|
||||
names = {}
|
||||
for name, _, path in entries:
|
||||
names.setdefault(path, []).append(name)
|
||||
for file in html_files(docs):
|
||||
rel = os.path.relpath(file, docs).replace(os.sep, '/')
|
||||
text = read(file).replace(TITLE_SUFFIX, '</title>')
|
||||
if rel in names:
|
||||
title = html.escape(', '.join(names[rel]), quote=False)
|
||||
text = re.sub(r'<title>.*?</title>', lambda _: f'<title>{title}</title>', text, count=1, flags=re.S)
|
||||
write(file, text)
|
||||
|
||||
|
||||
IMG_RE = re.compile(r'<img\b[^>]*>', re.I)
|
||||
ATTR_RE = r'''(?:{0})\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))'''
|
||||
|
||||
|
||||
def attr(tag, name):
|
||||
"""Return the value of an attribute in a tag or None."""
|
||||
m = re.search(r'(?<![\w-])' + ATTR_RE.format(name), tag, flags=re.I)
|
||||
return next(g for g in m.groups() if g is not None) if m else None
|
||||
|
||||
|
||||
def is_remote(url) -> bool:
|
||||
return re.match(r'(https?:)?//', url.strip(), flags=re.I) is not None
|
||||
|
||||
|
||||
def download(url, docs) -> str:
|
||||
"""Download url into assets/external and return the path relative to docs."""
|
||||
u = urllib.parse.urlparse(url if not url.startswith('//') else 'https:' + url)
|
||||
if u.scheme.lower() not in ('http', 'https'):
|
||||
raise ValueError(f'not an http(s) URL: {url}')
|
||||
req = urllib.request.Request(u.geturl(), headers={'User-Agent': USER_AGENT})
|
||||
# (the scheme is checked above)
|
||||
with urllib.request.urlopen(req, timeout=20) as r: # nosec B310
|
||||
data = r.read()
|
||||
ctype = r.headers.get_content_type()
|
||||
path = urllib.parse.unquote(u.path).lstrip('/')
|
||||
ext = os.path.splitext(path)[1]
|
||||
if u.query or not ext or path.endswith('/'):
|
||||
digest = hashlib.sha1(url.encode(), usedforsecurity=False).hexdigest()[:12]
|
||||
path = os.path.join(os.path.dirname(path), digest + CONTENT_TYPE_EXT.get(ctype, ext or '.bin'))
|
||||
rel = os.path.normpath(os.path.join('assets', 'external', u.hostname, path))
|
||||
out = os.path.join(docs, rel)
|
||||
os.makedirs(os.path.dirname(out), exist_ok=True)
|
||||
with open(out, 'wb') as f:
|
||||
f.write(data)
|
||||
return rel.replace(os.sep, '/')
|
||||
|
||||
|
||||
def localize_images(docs) -> None:
|
||||
"""Download remote images and rewrite their src; drop them on failure."""
|
||||
cache = {}
|
||||
for file in html_files(docs):
|
||||
text = read(file)
|
||||
|
||||
def repl(m):
|
||||
tag = m.group(0)
|
||||
src = attr(tag, 'src')
|
||||
if src is None or not is_remote(src):
|
||||
return tag
|
||||
if src not in cache:
|
||||
try:
|
||||
cache[src] = download(src, docs)
|
||||
except Exception as e: # noqa: BLE001
|
||||
print(f'generate_docset.py: warning: cannot download {src}: {e}', file=sys.stderr)
|
||||
cache[src] = None
|
||||
if cache[src] is None:
|
||||
return attr(tag, 'alt') or ''
|
||||
local = os.path.relpath(os.path.join(docs, cache[src]), os.path.dirname(file))
|
||||
local = local.replace(os.sep, '/')
|
||||
return re.sub(ATTR_RE.format('src'), lambda _: f'src="{local}"', tag, count=1, flags=re.I)
|
||||
|
||||
new = IMG_RE.sub(repl, text)
|
||||
if new != text:
|
||||
write(file, new)
|
||||
|
||||
|
||||
def load_mkdocs_yml() -> dict:
|
||||
"""Load mkdocs.yml, ignoring tags like !ENV and !!python/name."""
|
||||
with open(MKDOCS_YML, encoding='utf-8') as f:
|
||||
# (Loader is a yaml.SafeLoader)
|
||||
return yaml.load(f, Loader=Loader) # nosec B506
|
||||
|
||||
|
||||
def localize_site_urls(docs, site_url) -> None:
|
||||
"""Load assets that the theme's JavaScript references by absolute site URL from the docset.
|
||||
|
||||
The privacy plugin rewrites the mermaid loader to "<site_url>assets/external/unpkg.com/mermaid@11/...", so the
|
||||
docset would fetch mermaid from the live site. __md_scope is the site root that Material defines in every page.
|
||||
"""
|
||||
pattern = re.compile(r'"' + re.escape(site_url) + r'(assets/[^"]*)"')
|
||||
for base, _, files in os.walk(docs):
|
||||
for file in (f for f in files if f.endswith('.js')):
|
||||
path = os.path.join(base, file)
|
||||
text = read(path)
|
||||
new = pattern.sub(r'new URL("\1",__md_scope).href', text)
|
||||
if new != text:
|
||||
write(path, new)
|
||||
|
||||
|
||||
def remote_loads(docs, site_url) -> list:
|
||||
"""Return 'file: url' strings for resources that would be loaded remotely."""
|
||||
found = []
|
||||
cdn = re.compile(r'https://(?:unpkg\.com|cdn\.jsdelivr\.net|cdnjs\.cloudflare\.com|'
|
||||
r'fonts\.googleapis\.com|fonts\.gstatic\.com)/[^\s"\'`)\\]*')
|
||||
css_url = re.compile(r'url\(\s*["\']?((?:https?:)?//[^)"\']+)', re.I)
|
||||
css_import = re.compile(r'@import\s+(?:url\(\s*)?["\']?((?:https?:)?//[^)"\'; ]+)', re.I)
|
||||
for base, _, files in os.walk(docs):
|
||||
for file in files:
|
||||
path = os.path.join(base, file)
|
||||
rel = os.path.relpath(path, docs)
|
||||
urls = []
|
||||
if file.endswith('.html'):
|
||||
text = read(path)
|
||||
for m in re.finditer(r'<[a-zA-Z][^>]*>', text):
|
||||
tag = m.group(0)
|
||||
for name in ('src', 'poster'):
|
||||
v = attr(tag, name)
|
||||
if v and is_remote(v):
|
||||
urls.append(v)
|
||||
v = attr(tag, 'srcset')
|
||||
if v:
|
||||
urls += [c.split()[0] for c in v.split(',') if c.strip() and is_remote(c.strip())]
|
||||
if re.match(r'<link\b', tag, flags=re.I):
|
||||
rel_attr = (attr(tag, 'rel') or '').lower()
|
||||
v = attr(tag, 'href')
|
||||
if v and is_remote(v) and re.search(r'stylesheet|icon|preload|modulepreload|manifest', rel_attr):
|
||||
urls.append(v)
|
||||
urls += css_url.findall(text) + css_import.findall(text)
|
||||
elif file.endswith('.css'):
|
||||
text = read(path)
|
||||
urls += css_url.findall(text) + css_import.findall(text)
|
||||
elif file.endswith('.js'):
|
||||
text = read(path)
|
||||
urls += cdn.findall(text)
|
||||
urls += re.findall(re.escape(site_url) + r'assets/[^\s"\'`)\\]*', text)
|
||||
found += [f'{rel}: {u}' for u in urls if u not in ALLOWED_REMOTE]
|
||||
return found
|
||||
|
||||
|
||||
def make_docset(site, out_dir) -> int:
|
||||
entries = build_entries()
|
||||
if problems:
|
||||
print('\n'.join(problems), file=sys.stderr)
|
||||
return 1
|
||||
docset = os.path.join(out_dir, DOCSET)
|
||||
docs = os.path.join(docset, 'Contents', 'Resources', 'Documents')
|
||||
if os.path.exists(docset):
|
||||
shutil.rmtree(docset)
|
||||
shutil.copytree(site, docs)
|
||||
for icon in ('icon.png', 'icon@2x.png'):
|
||||
shutil.copy(os.path.join(HERE, icon), docset)
|
||||
shutil.copy(os.path.join(HERE, 'Info.plist'), os.path.join(docset, 'Contents'))
|
||||
write_index(entries, os.path.join(docset, 'Contents', 'Resources', 'docSet.dsidx'))
|
||||
|
||||
# patch CSS to hide navigation items and fix spacing
|
||||
css = glob.glob(os.path.join(docs, 'assets', 'stylesheets', 'main.*.min.css'))
|
||||
if len(css) != 1:
|
||||
print(f'generate_docset.py: expected exactly one main.*.min.css, found {len(css)}', file=sys.stderr)
|
||||
return 1
|
||||
with open(css[0], 'a', encoding='utf-8') as f:
|
||||
f.write(CSS_PATCH)
|
||||
|
||||
patch_titles(docs, entries)
|
||||
remove_source_widget(docs)
|
||||
for sitemap in glob.glob(os.path.join(docs, 'sitemap.*')):
|
||||
os.remove(sitemap)
|
||||
|
||||
# make the docset self-contained
|
||||
site_url = load_mkdocs_yml()['site_url']
|
||||
localize_images(docs)
|
||||
localize_site_urls(docs, site_url)
|
||||
remote = remote_loads(docs, site_url)
|
||||
if remote:
|
||||
print('generate_docset.py: remote resources remain in the docset:', file=sys.stderr)
|
||||
print('\n'.join(' ' + r for r in remote), file=sys.stderr)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
def make_tgz(out_dir) -> int:
|
||||
docset = os.path.join(out_dir, DOCSET)
|
||||
if not os.path.isdir(docset):
|
||||
print(f'generate_docset.py: {docset} does not exist', file=sys.stderr)
|
||||
return 1
|
||||
with tarfile.open(os.path.join(out_dir, 'JSON_for_Modern_C++.tgz'), 'w:gz') as tar:
|
||||
tar.add(docset, arcname=DOCSET, filter=lambda i: None if os.path.basename(i.name) == '.DS_Store' else i)
|
||||
return 0
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
sub = parser.add_subparsers(dest='command', required=True)
|
||||
sub.add_parser('list', help='print the index entries as TSV')
|
||||
p = sub.add_parser('index', help='write the SQLite search index')
|
||||
p.add_argument('out')
|
||||
p = sub.add_parser('docset', help='build the docset from a built mkdocs site')
|
||||
p.add_argument('site_dir')
|
||||
p.add_argument('out_dir')
|
||||
p = sub.add_parser('tgz', help='pack the docset into a tarball')
|
||||
p.add_argument('out_dir')
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.command == 'docset':
|
||||
return make_docset(args.site_dir, args.out_dir)
|
||||
if args.command == 'tgz':
|
||||
return make_tgz(args.out_dir)
|
||||
|
||||
entries = build_entries()
|
||||
if problems:
|
||||
print('\n'.join(problems), file=sys.stderr)
|
||||
return 1
|
||||
if args.command == 'list':
|
||||
for entry in entries:
|
||||
print('\t'.join(entry))
|
||||
elif args.command == 'index':
|
||||
write_index(entries, args.out)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
@@ -17,7 +17,6 @@ build: style_check
|
||||
|
||||
style_check:
|
||||
@cd docs ; ../venv/bin/python3 ../scripts/check_structure.py
|
||||
@venv/bin/python3 ../docset/generate_docset.py list > /dev/null
|
||||
|
||||
# check that all Mermaid diagrams parse (needs Node.js)
|
||||
# This target is used in the CI (ci_test_documentation_mermaid).
|
||||
|
||||
@@ -96,7 +96,9 @@ The SAX event lister must follow the interface of [`json_sax`](../json_sax/index
|
||||
|
||||
## Return value
|
||||
|
||||
return value of the last processed SAX event
|
||||
`#!cpp true` if the input was parsed without errors and no SAX event returned `#!cpp false`; `#!cpp false` otherwise.
|
||||
In particular, the result is `#!cpp false` for input with errors, even if the SAX parser recovered from all of them
|
||||
(see [error recovery](../../features/parsing/error_recovery.md)).
|
||||
|
||||
## Exception safety
|
||||
|
||||
@@ -145,6 +147,7 @@ A UTF-8 byte order mark is silently ignored.
|
||||
- Added `ignore_trailing_commas` in version 3.13.0.
|
||||
- Added `tag_handler` in version 3.13.0.
|
||||
- Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0.
|
||||
- Recovering from parse errors (see [`parse_error`](../json_sax/parse_error.md)) added in version 3.13.0.
|
||||
- Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0.
|
||||
- `JSON_PRECISE_STREAM_POSITION` added in version 3.13.0 to optionally leave a `#!cpp std::istream` positioned right
|
||||
after the parsed value when `strict` is `#!cpp false`.
|
||||
|
||||
@@ -123,6 +123,4 @@ Linear in the size of the JSON value `j`.
|
||||
that is not valid UTF-8 unchanged, as before; `strict` (the default if
|
||||
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`.
|
||||
- Throws `type_error.321` for a discarded value since version 3.13.0; previously, a discarded value nested in an
|
||||
array or object was silently skipped, producing invalid BJData.
|
||||
- Writes unsigned integers wider than 64 bits as high-precision numbers since version 3.13.0; previously, they were
|
||||
silently truncated to 64 bits.
|
||||
array or object was silently skipped, producing invalid BJData.
|
||||
@@ -37,9 +37,8 @@ With (2), the bytes written before the exception remain in the output adapter.
|
||||
|
||||
## Exceptions
|
||||
|
||||
- Throws [out_of_range.407](../../home/exceptions.md#jsonexceptionout_of_range407) if `j` contains an integer outside
|
||||
the range of int64 (an unsigned integer above 9223372036854775807, or, with a number type wider than 64 bits, any
|
||||
integer beyond int64), which BON8 cannot represent
|
||||
- Throws [out_of_range.407](../../home/exceptions.md#jsonexceptionout_of_range407) if `j` contains an unsigned integer
|
||||
above 9223372036854775807, which BON8 cannot represent
|
||||
- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if `j` contains a string that is not
|
||||
valid UTF-8
|
||||
|
||||
|
||||
@@ -47,10 +47,6 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
|
||||
|
||||
- Throws [`type_error.317`](../../home/exceptions.md#jsonexceptiontype_error317) if the top-level type of the JSON value
|
||||
is not an object; example: `"to serialize to BSON, top-level type must be object, but is string"`
|
||||
- Throws [`out_of_range.407`](../../home/exceptions.md#jsonexceptionout_of_range407) if `j` contains a signed integer
|
||||
outside the range of int64 or an unsigned integer outside the range of uint64, which is only possible with a number
|
||||
type wider than 64 bits; example:
|
||||
`"integer number 9223372036854775808 cannot be represented by BSON as it does not fit int64"`
|
||||
- Throws [`out_of_range.409`](../../home/exceptions.md#jsonexceptionout_of_range409) if a key in the JSON object contains
|
||||
a null byte (code point U+0000); example: `"BSON key cannot contain code point U+0000 (at byte 2)"`
|
||||
- Throws [`out_of_range.412`](../../home/exceptions.md#jsonexceptionout_of_range412) if the length of a document, array,
|
||||
@@ -123,5 +119,3 @@ pass before anything is written.
|
||||
that is not valid UTF-8 unchanged, as before; `strict` (the default if
|
||||
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316` before anything
|
||||
is written.
|
||||
- Throws `out_of_range.407` for integers that do not fit 64 bits since version 3.13.0; previously, integers of a
|
||||
number type wider than 64 bits were silently truncated.
|
||||
@@ -46,9 +46,6 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
|
||||
|
||||
## Exceptions
|
||||
|
||||
- Throws [`out_of_range.407`](../../home/exceptions.md#jsonexceptionout_of_range407) if `j` contains an integer
|
||||
outside [-2^64, 2^64-1], which is only possible with a number type wider than 64 bits; example:
|
||||
`"integer number 18446744073709551616 cannot be represented by CBOR as it does not fit [-2^64, 2^64-1]"`
|
||||
- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is
|
||||
not valid UTF-8 and `error_handler` is `strict` (the default only if
|
||||
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled)
|
||||
@@ -93,5 +90,3 @@ Linear in the size of the JSON value `j`.
|
||||
[`JSON_STRICT_BINARY_UTF8`](../macros/json_strict_binary_utf8.md) is enabled) throws `type_error.316`.
|
||||
- Throws `type_error.321` for a discarded value since version 3.13.0; previously, a discarded value nested in an
|
||||
array or object was silently skipped, producing invalid CBOR.
|
||||
- Throws `out_of_range.407` for integers that do not fit 64 bits since version 3.13.0; previously, integers of a
|
||||
number type wider than 64 bits were silently truncated.
|
||||
@@ -46,9 +46,6 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
|
||||
|
||||
## Exceptions
|
||||
|
||||
- Throws [`out_of_range.407`](../../home/exceptions.md#jsonexceptionout_of_range407) if `j` contains an integer
|
||||
outside [-2^63, 2^64-1], which is only possible with a number type wider than 64 bits; example:
|
||||
`"integer number 18446744073709551616 cannot be represented by MessagePack as it does not fit [-2^63, 2^64-1]"`
|
||||
- Throws [`out_of_range.412`](../../home/exceptions.md#jsonexceptionout_of_range412) if the length of a string, binary
|
||||
value, array, or object exceeds 4294967295, the maximum MessagePack can store; example:
|
||||
`"MessagePack length 4294967296 exceeds maximum of 4294967295"`
|
||||
@@ -115,5 +112,3 @@ Linear in the size of the JSON value `j`.
|
||||
`number_unsigned_t`.
|
||||
- Throws `type_error.321` for a discarded value since version 3.13.0; previously, a discarded value nested in an
|
||||
array or object was silently skipped, producing invalid MessagePack.
|
||||
- Throws `out_of_range.407` for integers that do not fit 64 bits since version 3.13.0; previously, integers of a
|
||||
number type wider than 64 bits were silently truncated.
|
||||
@@ -7,7 +7,8 @@ struct json_sax;
|
||||
|
||||
This class describes the SAX interface used by [sax_parse](../basic_json/sax_parse.md). Each function is called in
|
||||
different situations while the input is parsed. The boolean return value informs the parser whether to continue
|
||||
processing the input.
|
||||
processing the input; for [`parse_error`](parse_error.md), it decides whether to
|
||||
[recover from the error](../../features/parsing/error_recovery.md).
|
||||
|
||||
For instance, parsing the JSON text `{"a": [1, true]}` triggers the following callbacks, in order:
|
||||
|
||||
|
||||
@@ -21,11 +21,18 @@ A parse error occurred.
|
||||
|
||||
## Return value
|
||||
|
||||
Whether parsing should proceed (**must return `#!cpp false`**).
|
||||
Whether to recover from the error:
|
||||
|
||||
- `#!cpp false` stops parsing.
|
||||
- `#!cpp true` recovers from the error: the error is repaired and parsing continues. If that is not possible, which
|
||||
happens in the binary formats when the end of the item with the error is unknown, the value read so far is completed
|
||||
and parsing stops. See [error recovery](../../features/parsing/error_recovery.md) for how errors are repaired.
|
||||
|
||||
Either way, [`sax_parse`](../basic_json/sax_parse.md) returns `#!cpp false`.
|
||||
|
||||
## Examples
|
||||
|
||||
??? example
|
||||
??? example "Example: (1) the SAX interface"
|
||||
|
||||
The example below shows how the SAX interface is used.
|
||||
|
||||
@@ -39,12 +46,29 @@ Whether parsing should proceed (**must return `#!cpp false`**).
|
||||
--8<-- "examples/sax_parse.output"
|
||||
```
|
||||
|
||||
??? example "Example: (2) recovering from errors"
|
||||
|
||||
The example below shows how a SAX parser recovers from errors.
|
||||
|
||||
```cpp
|
||||
--8<-- "examples/sax_parse__error_recovery.cpp"
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```
|
||||
--8<-- "examples/sax_parse__error_recovery.output"
|
||||
```
|
||||
|
||||
## See also
|
||||
|
||||
- [sax_parse](../basic_json/sax_parse.md) - SAX parser
|
||||
- [Parsing and Exceptions](../../features/parsing/parse_exceptions.md) - the article on handling parse errors without
|
||||
exceptions
|
||||
- [Error Recovery](../../features/parsing/error_recovery.md) - the article on recovering from parse errors
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.2.0.
|
||||
- Returning `#!cpp true` recovers from the error since version 3.13.0; before, parsing stopped, but the result of
|
||||
[`sax_parse`](../basic_json/sax_parse.md) could be wrong.
|
||||
@@ -0,0 +1,43 @@
|
||||
#include <iostream>
|
||||
#include <iomanip>
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// a SAX parser that creates a JSON value like json::parse does, but that
|
||||
// recovers from parse errors instead of stopping at the first one
|
||||
class recovering_parser : public nlohmann::detail::json_sax_dom_parser<json>
|
||||
{
|
||||
public:
|
||||
explicit recovering_parser(json& result)
|
||||
: nlohmann::detail::json_sax_dom_parser<json>(result, false)
|
||||
{}
|
||||
|
||||
bool parse_error(std::size_t position,
|
||||
const std::string& /*last_token*/,
|
||||
const json::exception& ex)
|
||||
{
|
||||
std::cout << "byte " << position << ": " << ex.what() << '\n';
|
||||
|
||||
// repair the input and continue
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
int main()
|
||||
{
|
||||
// JSON text with several mistakes that ends too early
|
||||
const std::string text = R"({
|
||||
"name": "Hello World",
|
||||
"tags": ["a" "b",],
|
||||
"valid": tru,
|
||||
"size": 1.,
|
||||
"nested": {"x": 1)";
|
||||
|
||||
json result;
|
||||
recovering_parser sax(result);
|
||||
const bool valid = json::sax_parse(text, &sax);
|
||||
|
||||
std::cout << "\nvalid JSON: " << std::boolalpha << valid << '\n'
|
||||
<< std::setw(4) << result << std::endl;
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
byte 49: [json.exception.parse_error.101] parse error at line 3, column 20: syntax error while parsing array - unexpected string literal; expected ']'
|
||||
byte 51: [json.exception.parse_error.101] parse error at line 3, column 22: syntax error while parsing value - unexpected ']'; expected '[', '{', or a literal
|
||||
byte 70: [json.exception.parse_error.101] parse error at line 4, column 17: syntax error while parsing value - invalid literal; last read: '"valid": tru,'
|
||||
byte 86: [json.exception.parse_error.101] parse error at line 5, column 15: syntax error while parsing value - invalid number; expected digit after '.'; last read: '1.,'
|
||||
byte 109: [json.exception.parse_error.101] parse error at line 6, column 22: syntax error while parsing object - unexpected end of input; expected '}'
|
||||
|
||||
valid JSON: false
|
||||
{
|
||||
"name": "Hello World",
|
||||
"nested": {
|
||||
"x": 1
|
||||
},
|
||||
"size": 1,
|
||||
"tags": [
|
||||
"a",
|
||||
"b"
|
||||
],
|
||||
"valid": null
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
# Error Recovery
|
||||
|
||||
By default, parsing stops at the first error. With the [SAX interface](sax_interface.md), you can instead ask the
|
||||
parser to *recover*: to repair the error and continue, so that you get as much as possible out of malformed input, for
|
||||
instance a file that was cut off, JSON edited by hand, or the output of a language model.
|
||||
|
||||
## Recovering from errors
|
||||
|
||||
The SAX parser's [`parse_error`](../../api/json_sax/parse_error.md) function is called for every error. Its return value
|
||||
decides what happens next:
|
||||
|
||||
- `#!cpp false` stops parsing. This is what the SAX parsers of the library do, so [`parse`](../../api/basic_json/parse.md)
|
||||
and [`accept`](../../api/basic_json/accept.md) never recover.
|
||||
- `#!cpp true` repairs the error and continues parsing.
|
||||
|
||||
When recovering, the SAX parser still receives well-formed events: every `start_object` or `start_array` is followed by
|
||||
the matching `end_object` or `end_array`, and every `key` is followed by exactly one value. A SAX parser that creates a
|
||||
JSON value, such as the one in the example below, therefore gets a complete value. Parsing always ends, and
|
||||
[`sax_parse`](../../api/basic_json/sax_parse.md) returns `#!cpp false` for input that is not valid JSON, even if every
|
||||
error was repaired. Each token is reported at most once, and the SAX parser can stop at any error by returning
|
||||
`#!cpp false`.
|
||||
|
||||
!!! example
|
||||
|
||||
The example below derives a SAX parser from the library's parser for `json` values (`json_sax_dom_parser`),
|
||||
and recovers from all errors.
|
||||
|
||||
```cpp
|
||||
--8<-- "examples/sax_parse__error_recovery.cpp"
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```
|
||||
--8<-- "examples/sax_parse__error_recovery.output"
|
||||
```
|
||||
|
||||
## How errors are repaired
|
||||
|
||||
Each error is repaired with the smallest local edit: a missing separator is inserted, a stray token is removed, what can
|
||||
be read of a broken string or number is kept, and a value that cannot be read at all becomes `#!json null`.
|
||||
|
||||
| Mistake | Repair | Example | Result |
|
||||
|---------------------------|--------------------------------------------------------------------------------|------------------------------------------|----------------------------|
|
||||
| missing `,` or `:` | inserted | `#!json [1 2]`, `#!json {"a" 1}` | `[1,2]`, `{"a":1}` |
|
||||
| missing value | `#!json null` for an object key or between commas in an array | `#!json {"a":}`, `#!json [1,,2]` | `{"a":null}`, `[1,null,2]` |
|
||||
| trailing comma | removed | `#!json [1,2,]` | `[1,2]` |
|
||||
| broken string | invalid escapes and bytes are replaced (see below); a line break ends the string | `#!json ["a\qb"]` | `["aqb"]` |
|
||||
| broken number | the longest valid beginning is kept | `#!json [1., 2e+]` | `[1,2]` |
|
||||
| unreadable value | `#!json null` | `#!json [1, NaN, tru]` | `[1,null,null]` |
|
||||
| number too large | passed as infinity, together with its text | `#!json [1e999]` | infinity (see below) |
|
||||
| stray `:` | removed | `#!json ["a":1]` | `["a",1]` |
|
||||
| member without a key | skipped up to the next `,` or `}` | `#!json {1:2, "b":3}` | `{"b":3}` |
|
||||
| wrong closing bracket | closes the innermost array or object | `#!json {"a":[1,2}, "b":3}` | `{"a":[1,2],"b":3}` |
|
||||
| input ends too early | all open arrays and objects are closed | `#!json {"a":[1,2` | `{"a":[1,2]}` |
|
||||
| text before the value | skipped | `#!json )]}'{"a":1}` | `{"a":1}` |
|
||||
|
||||
In a string, an unknown escape like `\q` stands for the escaped character (`q`), as in JavaScript. An invalid `\u`
|
||||
escape, a lone surrogate, and ill-formed UTF-8 are each replaced by U+FFFD (REPLACEMENT CHARACTER), and control
|
||||
characters are kept. A string without its closing quote ends at the next line break or at the end of the input.
|
||||
|
||||
The input after the top-level value is not repaired: as without recovery, it is reported as an error, and parsing stops.
|
||||
|
||||
## Binary formats
|
||||
|
||||
The binary formats ([BJData](../binary_formats/bjdata.md), [BON8](../binary_formats/bon8.md),
|
||||
[BSON](../binary_formats/bson.md), [CBOR](../binary_formats/cbor.md), [MessagePack](../binary_formats/messagepack.md),
|
||||
and [UBJSON](../binary_formats/ubjson.md)) have no delimiters to find the next value by. So what can be repaired depends
|
||||
on whether the end of the item with the error is known, a distinction that
|
||||
[RFC 8949, Section 5.3](https://www.rfc-editor.org/rfc/rfc8949.html#section-5.3) makes for CBOR, too.
|
||||
|
||||
If the item is complete, but cannot be passed on as it is, it is replaced, and parsing continues after it:
|
||||
|
||||
| Mistake | Formats | Repair |
|
||||
|---------------------------------------------------------------------|-------------------------|-------------------------------------------------------------------------|
|
||||
| tag, if `tag_handler` is `cbor_tag_handler_t::error` (the default) | CBOR | ignored |
|
||||
| simple value other than `false`, `true`, and `null`, like undefined | CBOR | `#!json null` |
|
||||
| number too large for a custom `number_float_t`, like `float` | all | infinity |
|
||||
| character (`C`) that is not ASCII | BJData, UBJSON | U+FFFD |
|
||||
| invalid high-precision number (`H`) | BJData, UBJSON | the longest valid beginning is kept, as for JSON text, or `#!json null` |
|
||||
| high-precision number too large | BJData, UBJSON | passed as infinity, together with its text |
|
||||
| object key that is not a string | BON8, CBOR, MessagePack | the member is skipped |
|
||||
| element of a type the library does not read, like ObjectId or date | BSON | `#!json null` |
|
||||
| string without its terminator | BSON | kept |
|
||||
| document whose size does not match its content | BSON | kept |
|
||||
|
||||
CBOR tags and simple values are repaired as [RFC 8949, Section 6.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-6.1)
|
||||
suggests for converting CBOR to JSON. By default, [`sax_parse`](../../api/basic_json/sax_parse.md) reports every tag as
|
||||
an error; when recovering, tags are then ignored like with
|
||||
[`cbor_tag_handler_t::ignore`](../../api/basic_json/cbor_tag_handler_t.md). Strings that are not valid UTF-8 are no
|
||||
error: like [`from_cbor`](../../api/basic_json/from_cbor.md) and the other functions by default, `sax_parse` passes
|
||||
them on as they are.
|
||||
|
||||
After any other error, the end of the item is unknown: the input ended, a byte is not a valid type marker, or a size
|
||||
cannot be right. Parsing then stops, and the value read so far is completed: a key that waits for its value gets
|
||||
`#!json null`, and all open arrays and objects are closed. This keeps everything before the error of an input that was
|
||||
cut off. The exception is BSON, which stores the size of every document: an element whose end is unknown gets
|
||||
`#!json null`, the rest of its document is skipped, and parsing continues after the document.
|
||||
|
||||
## Limitations
|
||||
|
||||
- A repair is a guess. For example, `#!json {"a" "b": 1}` could be meant as `#!json {"a": "b"}` or as
|
||||
`#!json {"a": null, "b": 1}`; it is repaired to the former. Treat recovered values as a best effort, and check the
|
||||
reported errors.
|
||||
- A closing bracket always closes the innermost array or object. If a bracket is missing rather than wrong, the
|
||||
repair differs from the intention: `#!json {"a": {"b": [1, 2}, "c": 3}` is repaired to
|
||||
`#!json {"a": {"b": [1, 2], "c": 3}}`, although `#!json {"a": {"b": [1, 2]}, "c": 3}` may have been meant.
|
||||
- Keys without quotes, and strings in single quotes, are not supported; such members are skipped.
|
||||
- In the binary formats, a member that is skipped because its key is not a string is lost, and so are the elements of a
|
||||
BSON document after one whose end is unknown.
|
||||
- A number that is too large for `number_float_t` is passed as positive or negative infinity. The SAX parser's
|
||||
`number_float` also gets the number's text, but a JSON value cannot store it, and
|
||||
[`dump`](../../api/basic_json/dump.md) serializes infinity as `#!json null`.
|
||||
- When parsing is not strict (see [`sax_parse`](../../api/basic_json/sax_parse.md)), a repair may read parts of the
|
||||
input after the value, for instance of the next value in a stream of concatenated values.
|
||||
|
||||
## See also
|
||||
|
||||
- [SAX interface](sax_interface.md) - implement a custom SAX handler
|
||||
- [`parse_error`](../../api/json_sax/parse_error.md) - the SAX event for parse errors
|
||||
- [`sax_parse`](../../api/basic_json/sax_parse.md) - generate SAX events
|
||||
- [parsing and exceptions](parse_exceptions.md) - control error handling
|
||||
@@ -75,7 +75,7 @@ You can influence a DOM parse without switching to the SAX interface by passing
|
||||
When the input is not valid JSON, the `parse` function throws an exception by default. If exceptions are undesired or
|
||||
unavailable, the parser can instead return a discarded value, or [`accept`](../../api/basic_json/accept.md) can be used
|
||||
to only check whether an input is valid JSON. See [parsing and exceptions](parse_exceptions.md) for the available
|
||||
options.
|
||||
options. To get as much as possible out of malformed input, a SAX parser can [recover from errors](error_recovery.md).
|
||||
|
||||
## See also
|
||||
|
||||
@@ -86,4 +86,5 @@ options.
|
||||
- [parser callbacks](parser_callbacks.md) - influence the parsing by a callback function
|
||||
- [SAX interface](sax_interface.md) - implement a custom SAX handler
|
||||
- [parsing and exceptions](parse_exceptions.md) - control error handling
|
||||
- [error recovery](error_recovery.md) - get as much as possible out of malformed input
|
||||
- [parsing untrusted input](untrusted_input.md) - what to consider when parsing input from untrusted sources
|
||||
@@ -64,7 +64,8 @@ bool parse_error(std::size_t position,
|
||||
const json::exception& ex);
|
||||
```
|
||||
|
||||
The return value indicates whether the parsing should continue, so the function should usually return `#!cpp false`.
|
||||
The return value decides whether to stop parsing (`#!cpp false`) or to repair the error and continue
|
||||
(`#!cpp true`); see [error recovery](error_recovery.md) for the latter.
|
||||
|
||||
??? example "Example: report parse errors without exceptions"
|
||||
|
||||
|
||||
@@ -60,7 +60,8 @@ bool key(string_t& val);
|
||||
bool parse_error(std::size_t position, const std::string& last_token, const json::exception& ex);
|
||||
```
|
||||
|
||||
The return value of each function determines whether parsing should proceed.
|
||||
The return value of each function determines whether parsing should proceed. For `parse_error`, returning
|
||||
`#!cpp true` [recovers from the error](error_recovery.md).
|
||||
|
||||
To implement your own SAX handler, proceed as follows:
|
||||
|
||||
@@ -68,7 +69,7 @@ To implement your own SAX handler, proceed as follows:
|
||||
2. Create an object of your SAX interface class, e.g. `my_sax`.
|
||||
3. Call `#!cpp bool json::sax_parse(input, &my_sax);` where the first parameter can be any input like a string or an input stream and the second parameter is a pointer to your SAX interface.
|
||||
|
||||
Note the `sax_parse` function only returns a `#!cpp bool` indicating the result of the last executed SAX event. It does not return `json` value - it is up to you to decide what to do with the SAX events. Furthermore, no exceptions are thrown in case of a parse error - it is up to you what to do with the exception object passed to your `parse_error` implementation. Internally, the SAX interface is used for the DOM parser (class `json_sax_dom_parser`) as well as the acceptor (`json_sax_acceptor`), see file `json_sax.hpp`.
|
||||
Note the `sax_parse` function only returns a `#!cpp bool` indicating whether the input was parsed without errors and no SAX event returned `#!cpp false`. It does not return `json` value - it is up to you to decide what to do with the SAX events. Furthermore, no exceptions are thrown in case of a parse error - it is up to you what to do with the exception object passed to your `parse_error` implementation. Internally, the SAX interface is used for the DOM parser (class `json_sax_dom_parser`) as well as the acceptor (`json_sax_acceptor`), see file `json_sax.hpp`.
|
||||
|
||||
## See also
|
||||
|
||||
|
||||
@@ -906,34 +906,14 @@ double-precision number when `number_float_t` is `#!cpp float`.
|
||||
|
||||
### json.exception.out_of_range.407
|
||||
|
||||
An integer number cannot be represented by the binary format it is serialized to:
|
||||
This exception previously indicated that the UBJSON and BSON binary formats did not support integer numbers greater than
|
||||
9223372036854775807 due to limitations in the implemented mapping. However, these limitations have since been resolved,
|
||||
and this exception no longer occurs.
|
||||
|
||||
- [BON8](../features/binary_formats/bon8.md) only stores integers that fit into int64.
|
||||
- [CBOR](../features/binary_formats/cbor.md), [MessagePack](../features/binary_formats/msgpack.md), and
|
||||
[BSON](../features/binary_formats/bson.md) store integers in at most 64 bits. With the default number types, every
|
||||
integer fits, but a [`number_integer_t`](../api/basic_json/number_integer_t.md) or
|
||||
[`number_unsigned_t`](../api/basic_json/number_unsigned_t.md) wider than 64 bits (e.g., `__int128`) can hold values
|
||||
outside the range of the format: [-2^64, 2^64-1] for CBOR, [-2^63, 2^64-1] for MessagePack, and the range of int64
|
||||
(signed integers) or uint64 (unsigned integers) for BSON.
|
||||
!!! success "Exception cannot occur any more"
|
||||
|
||||
[UBJSON](../features/binary_formats/ubjson.md) and [BJData](../features/binary_formats/bjdata.md) never throw this
|
||||
exception, because they serialize integers beyond 64 bits as high-precision numbers.
|
||||
|
||||
!!! failure "Example messages"
|
||||
|
||||
```
|
||||
integer number 9223372036854775808 cannot be represented by BON8 as it does not fit int64
|
||||
```
|
||||
```
|
||||
integer number 1267650600228229401496703205376 cannot be represented by CBOR as it does not fit [-2^64, 2^64-1]
|
||||
```
|
||||
|
||||
!!! note
|
||||
|
||||
Before version 3.13.0, CBOR, MessagePack, and BSON silently truncated integers wider than 64 bits, and BJData
|
||||
truncated unsigned integers wider than 64 bits. This exception was previously thrown by UBJSON and BSON for
|
||||
integers greater than 9223372036854775807; since version 3.9.0, such integers are serialized as high-precision
|
||||
UBJSON numbers, and since version 3.12.0 as uint64 BSON numbers.
|
||||
- Since version 3.9.0, integer numbers beyond int64 are serialized as high-precision UBJSON numbers.
|
||||
- Since version 3.12.0, integer numbers beyond int64 are serialized as uint64 BSON numbers.
|
||||
|
||||
### json.exception.out_of_range.408
|
||||
|
||||
|
||||
@@ -88,6 +88,7 @@ nav:
|
||||
- features/performance.md
|
||||
- Parsing:
|
||||
- features/parsing/index.md
|
||||
- features/parsing/error_recovery.md
|
||||
- features/parsing/json_lines.md
|
||||
- features/parsing/parse_exceptions.md
|
||||
- features/parsing/parser_callbacks.md
|
||||
|
||||
@@ -288,6 +288,36 @@ def check_header_links() -> None:
|
||||
f'link to "{match.group(0)}" does not point to a documentation page')
|
||||
|
||||
|
||||
def check_docset() -> None:
|
||||
"""Every API page and every macro has an entry in the docset index; no entry points to a missing page."""
|
||||
entry_re = re.compile(r"VALUES \('((?:[^']|'')*)', '(\w+)', '([^']*)'\);")
|
||||
names_by_path = {}
|
||||
with open("../../docset/docSet.sql", encoding="utf-8") as sql:
|
||||
for name, _, path in entry_re.findall(sql.read()):
|
||||
names_by_path.setdefault(path, set()).add(name.replace("''", "'"))
|
||||
|
||||
def to_path(page):
|
||||
if os.path.basename(page) == "index.md":
|
||||
return page[:-len("index.md")] + "index.html"
|
||||
return page[:-len(".md")] + "/index.html"
|
||||
|
||||
pages = sorted(glob.glob("**/*.md", recursive=True))
|
||||
for path in sorted(set(names_by_path) - {to_path(p) for p in pages}):
|
||||
report("docset/stale_entry", "../../docset/docSet.sql", f'entry "{path}" has no documentation page')
|
||||
for page in (p for p in pages if p.startswith("api/")):
|
||||
names = names_by_path.get(to_path(page))
|
||||
if not names:
|
||||
report("docset/missing_entry", page, "page has no entry in docs/docset/docSet.sql")
|
||||
elif page.startswith("api/macros/") and os.path.basename(page) != "index.md":
|
||||
with open(page, encoding="utf-8") as content:
|
||||
text = content.read()
|
||||
match = re.search(r"^# (.+)$", text, re.MULTILINE) or re.search(r"<h1>(.*?)</h1>", text, re.DOTALL)
|
||||
title = re.sub(r"<[^>]+>|\s+", " ", match.group(1))
|
||||
for macro in filter(None, (x.strip() for x in re.split(r"[,/]", title))):
|
||||
if macro not in names:
|
||||
report("docset/missing_macro", page, f'macro "{macro}" has no entry in docs/docset/docSet.sql')
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(120 * "-")
|
||||
check_structure()
|
||||
@@ -297,6 +327,7 @@ if __name__ == "__main__":
|
||||
check_heading_levels()
|
||||
check_image_alt_text()
|
||||
check_header_links()
|
||||
check_docset()
|
||||
print(120 * "-")
|
||||
|
||||
if warnings > 0:
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -132,7 +132,9 @@ struct json_sax
|
||||
@param[in] position the position in the input where the error occurs
|
||||
@param[in] last_token the last read token
|
||||
@param[in] ex an exception object describing the error
|
||||
@return whether parsing should proceed (must return false)
|
||||
@return whether to recover from the error: false stops parsing; true
|
||||
repairs the error and continues, or, if that is not possible,
|
||||
stops after completing the value read so far
|
||||
*/
|
||||
virtual bool parse_error(std::size_t position,
|
||||
const std::string& last_token,
|
||||
@@ -269,9 +271,12 @@ a pointer to the respective array or object for each recursion depth.
|
||||
After successful parsing, the value that is passed by reference to the
|
||||
constructor contains the parsed value.
|
||||
|
||||
@tparam BasicJsonType the JSON type
|
||||
@tparam BasicJsonType the JSON type
|
||||
@tparam InputAdapterType the input adapter of the lexer that can be passed to
|
||||
the constructor to record diagnostic positions; it
|
||||
does not matter if no lexer is passed
|
||||
*/
|
||||
template<typename BasicJsonType, typename InputAdapterType>
|
||||
template<typename BasicJsonType, typename InputAdapterType = string_input_adapter_type>
|
||||
class json_sax_dom_parser
|
||||
{
|
||||
public:
|
||||
@@ -518,7 +523,7 @@ class json_sax_dom_parser
|
||||
lexer_t* m_lexer_ref = nullptr;
|
||||
};
|
||||
|
||||
template<typename BasicJsonType, typename InputAdapterType>
|
||||
template<typename BasicJsonType, typename InputAdapterType = string_input_adapter_type>
|
||||
class json_sax_dom_callback_parser
|
||||
{
|
||||
public:
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstdint> // uint8_t, uint32_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <initializer_list> // initializer_list
|
||||
#include <string> // char_traits, string
|
||||
@@ -439,8 +439,16 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
if (0xD800 <= codepoint1 && codepoint1 <= 0xDBFF)
|
||||
{
|
||||
// expect next \uxxxx entry
|
||||
if (JSON_HEDLEY_LIKELY(get() == '\\' && get() == 'u'))
|
||||
if (JSON_HEDLEY_LIKELY(get() == '\\'))
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(get() != 'u'))
|
||||
{
|
||||
// current is the character escaped by the backslash
|
||||
error_message = "invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF";
|
||||
string_error_resume = resume_kind::escaped_character;
|
||||
return token_type::parse_error;
|
||||
}
|
||||
|
||||
const int codepoint2 = get_codepoint();
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(codepoint2 == -1))
|
||||
@@ -465,7 +473,11 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
}
|
||||
else
|
||||
{
|
||||
// the second escape was read completely and is a
|
||||
// code point of its own
|
||||
error_message = "invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF";
|
||||
string_error_resume = resume_kind::after_escape;
|
||||
string_error_codepoint = codepoint2;
|
||||
return token_type::parse_error;
|
||||
}
|
||||
}
|
||||
@@ -479,7 +491,9 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(0xDC00 <= codepoint1 && codepoint1 <= 0xDFFF))
|
||||
{
|
||||
// the escape was read completely
|
||||
error_message = "invalid string: surrogate U+DC00..U+DFFF must follow U+D800..U+DBFF";
|
||||
string_error_resume = resume_kind::after_escape;
|
||||
return token_type::parse_error;
|
||||
}
|
||||
}
|
||||
@@ -2133,6 +2147,574 @@ scan_number_done:
|
||||
}
|
||||
}
|
||||
|
||||
public:
|
||||
/////////////////////
|
||||
// error recovery
|
||||
/////////////////////
|
||||
|
||||
/*!
|
||||
@brief make the best of the token that scan() rejected
|
||||
|
||||
Called by the parser after scan() returned token_type::parse_error and the
|
||||
SAX parser asked to recover from the error (see #3989). Keeps what can be
|
||||
read of the token and skips the rest:
|
||||
|
||||
- A string keeps its characters. An unknown escape stands for the escaped
|
||||
character itself (as in JavaScript), an invalid Unicode escape and ill-formed
|
||||
UTF-8 become U+FFFD, and a control character is kept. A line break or the
|
||||
end of the input ends a string that lacks its closing quote.
|
||||
- A number keeps its longest valid prefix, e.g. `1` for `1.` or `1e+`.
|
||||
- A block comment that is not closed runs to the end of the input.
|
||||
- Anything else is skipped.
|
||||
|
||||
The rest of an invalid token is skipped up to the next delimiter
|
||||
(whitespace, a structural character, or a quote). A delimiter that the
|
||||
invalid token consumed is returned to the input, so that the next scan()
|
||||
reads it.
|
||||
|
||||
@return token_type::value_string or a number token type if a string or a
|
||||
number could be read, token_type::end_of_input for a block comment
|
||||
that is not closed, token_type::uninitialized otherwise
|
||||
*/
|
||||
token_type recover_token()
|
||||
{
|
||||
const resume_kind resume = string_error_resume;
|
||||
const int codepoint = string_error_codepoint;
|
||||
string_error_resume = resume_kind::character;
|
||||
string_error_codepoint = -1;
|
||||
|
||||
if (error_message_starts_with("invalid string"))
|
||||
{
|
||||
return recover_string(resume, codepoint);
|
||||
}
|
||||
|
||||
if (error_message_starts_with("invalid number"))
|
||||
{
|
||||
return recover_number();
|
||||
}
|
||||
|
||||
if (error_message_starts_with("invalid comment; missing"))
|
||||
{
|
||||
// the comment runs to the end of the input
|
||||
return token_type::end_of_input;
|
||||
}
|
||||
|
||||
skip_to_delimiter();
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief return the token that scan() read last to the input, so that the
|
||||
next scan() reads it again
|
||||
|
||||
Called by the parser when recovering from an error. The token must be a
|
||||
single character (',', ':', '[', ']', '{', or '}') or the end of the
|
||||
input, and scan() must have read it last.
|
||||
*/
|
||||
void unget_token()
|
||||
{
|
||||
JSON_ASSERT(!next_unget);
|
||||
unget();
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief let the token string for the next error begin at the current character
|
||||
|
||||
The token string of an error reaches back to the beginning of the last
|
||||
string or number. After an error, the parser calls this function so that
|
||||
the next error does not report (and, with many errors, copy) everything
|
||||
read since then.
|
||||
*/
|
||||
void restart_token_string()
|
||||
{
|
||||
restart_token_string_impl(std::integral_constant<bool, lazy_token_string> {});
|
||||
}
|
||||
|
||||
private:
|
||||
/// how recover_string() continues after the error scan_string() reported
|
||||
enum class resume_kind : std::uint8_t
|
||||
{
|
||||
/// current is the next character of the string (or the end of input)
|
||||
character,
|
||||
/// current is the character escaped by the preceding backslash
|
||||
escaped_character,
|
||||
/// current is the last character of a complete escape
|
||||
after_escape
|
||||
};
|
||||
|
||||
/// whether error_message begins with @a prefix
|
||||
bool error_message_starts_with(const char* prefix) const noexcept
|
||||
{
|
||||
const char* message = error_message;
|
||||
while (*prefix != '\0')
|
||||
{
|
||||
if (*message++ != *prefix++)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// whether current ends an invalid token (see recover_token())
|
||||
bool current_is_delimiter() const noexcept
|
||||
{
|
||||
switch (current)
|
||||
{
|
||||
case ' ':
|
||||
case '\t':
|
||||
case '\n':
|
||||
case '\r':
|
||||
case '[':
|
||||
case ']':
|
||||
case '{':
|
||||
case '}':
|
||||
case ',':
|
||||
case ':':
|
||||
case '\"':
|
||||
#if !JSON_STRICT_NUL_HANDLING
|
||||
case '\0':
|
||||
#endif
|
||||
case char_traits<char_type>::eof():
|
||||
return true;
|
||||
|
||||
case '/':
|
||||
return ignore_comments;
|
||||
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/// skip the rest of an invalid token and return its delimiter to the input
|
||||
void skip_to_delimiter()
|
||||
{
|
||||
while (!current_is_delimiter())
|
||||
{
|
||||
get();
|
||||
}
|
||||
|
||||
if (current != char_traits<char_type>::eof())
|
||||
{
|
||||
unget();
|
||||
}
|
||||
}
|
||||
|
||||
/// append U+FFFD REPLACEMENT CHARACTER to token_buffer
|
||||
void add_replacement_character()
|
||||
{
|
||||
add(0xEF);
|
||||
add(0xBF);
|
||||
add(0xBD);
|
||||
}
|
||||
|
||||
/// append the UTF-8 encoding of @a codepoint (not a surrogate) to token_buffer
|
||||
void add_codepoint(const int codepoint)
|
||||
{
|
||||
JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF);
|
||||
const auto cp = static_cast<unsigned int>(codepoint);
|
||||
if (cp < 0x80)
|
||||
{
|
||||
add(static_cast<char_int_type>(cp));
|
||||
}
|
||||
else if (cp <= 0x7FF)
|
||||
{
|
||||
add(static_cast<char_int_type>(0xC0u | (cp >> 6u)));
|
||||
add(static_cast<char_int_type>(0x80u | (cp & 0x3Fu)));
|
||||
}
|
||||
else if (cp <= 0xFFFF)
|
||||
{
|
||||
add(static_cast<char_int_type>(0xE0u | (cp >> 12u)));
|
||||
add(static_cast<char_int_type>(0x80u | ((cp >> 6u) & 0x3Fu)));
|
||||
add(static_cast<char_int_type>(0x80u | (cp & 0x3Fu)));
|
||||
}
|
||||
else
|
||||
{
|
||||
add(static_cast<char_int_type>(0xF0u | (cp >> 18u)));
|
||||
add(static_cast<char_int_type>(0x80u | ((cp >> 12u) & 0x3Fu)));
|
||||
add(static_cast<char_int_type>(0x80u | ((cp >> 6u) & 0x3Fu)));
|
||||
add(static_cast<char_int_type>(0x80u | (cp & 0x3Fu)));
|
||||
}
|
||||
}
|
||||
|
||||
/// append a code point read from a Unicode escape; a surrogate becomes U+FFFD
|
||||
void add_escaped_codepoint(const int codepoint)
|
||||
{
|
||||
if (0xD800 <= codepoint && codepoint <= 0xDFFF)
|
||||
{
|
||||
add_replacement_character();
|
||||
}
|
||||
else
|
||||
{
|
||||
add_codepoint(codepoint);
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief remove an incomplete UTF-8 sequence from the end of token_buffer
|
||||
|
||||
next_byte_in_range() adds the bytes of a sequence as it checks them, so
|
||||
when it rejects a byte, the beginning of the sequence is already in
|
||||
token_buffer, which otherwise holds only complete sequences.
|
||||
|
||||
@return whether an incomplete sequence was removed
|
||||
*/
|
||||
bool remove_incomplete_utf8_sequence()
|
||||
{
|
||||
std::size_t lead = token_buffer.size();
|
||||
std::size_t continuation_bytes = 0;
|
||||
while (lead > 0 && continuation_bytes < 3
|
||||
&& (static_cast<unsigned char>(token_buffer[lead - 1]) & 0xC0u) == 0x80u)
|
||||
{
|
||||
--lead;
|
||||
++continuation_bytes;
|
||||
}
|
||||
if (lead == 0)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
const auto lead_byte = static_cast<unsigned char>(token_buffer[lead - 1]);
|
||||
std::size_t expected = 0;
|
||||
if (lead_byte >= 0xF0)
|
||||
{
|
||||
expected = 3;
|
||||
}
|
||||
else if (lead_byte >= 0xE0)
|
||||
{
|
||||
expected = 2;
|
||||
}
|
||||
else if (lead_byte >= 0xC0)
|
||||
{
|
||||
expected = 1;
|
||||
}
|
||||
if (continuation_bytes >= expected)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
token_buffer.resize(lead - 1);
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief read the UTF-8 sequence that begins with current, which is not ASCII
|
||||
@return whether the next character must be read; false if current still
|
||||
needs to be handled, because it does not belong to the sequence
|
||||
*/
|
||||
bool recover_utf8_sequence()
|
||||
{
|
||||
// the number of continuation bytes and the range of the first one;
|
||||
// see the ranges in scan_string()
|
||||
std::size_t count = 0;
|
||||
char_int_type low = 0x80;
|
||||
char_int_type high = 0xBF;
|
||||
if (current >= 0xC2 && current <= 0xDF)
|
||||
{
|
||||
count = 1;
|
||||
}
|
||||
else if (current >= 0xE0 && current <= 0xEF)
|
||||
{
|
||||
count = 2;
|
||||
low = (current == 0xE0) ? 0xA0 : 0x80;
|
||||
high = (current == 0xED) ? 0x9F : 0xBF;
|
||||
}
|
||||
else if (current >= 0xF0 && current <= 0xF4)
|
||||
{
|
||||
count = 3;
|
||||
low = (current == 0xF0) ? 0x90 : 0x80;
|
||||
high = (current == 0xF4) ? 0x8F : 0xBF;
|
||||
}
|
||||
else
|
||||
{
|
||||
// an ill-formed byte
|
||||
add_replacement_character();
|
||||
return true;
|
||||
}
|
||||
|
||||
const std::size_t start = token_buffer.size();
|
||||
add(current);
|
||||
for (std::size_t i = 0; i < count; ++i)
|
||||
{
|
||||
get();
|
||||
if (current < low || current > high)
|
||||
{
|
||||
token_buffer.resize(start);
|
||||
add_replacement_character();
|
||||
return false;
|
||||
}
|
||||
add(current);
|
||||
low = 0x80;
|
||||
high = 0xBF;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief read the low surrogate that must follow the high surrogate @a high
|
||||
@return whether the next character must be read; false if current still
|
||||
needs to be handled
|
||||
*/
|
||||
bool recover_low_surrogate(int high)
|
||||
{
|
||||
while (true)
|
||||
{
|
||||
if (get() != '\\')
|
||||
{
|
||||
add_replacement_character();
|
||||
return false;
|
||||
}
|
||||
if (get() != 'u')
|
||||
{
|
||||
add_replacement_character();
|
||||
// not 'u', so this does not come back here
|
||||
return recover_escape();
|
||||
}
|
||||
|
||||
const int low = get_codepoint();
|
||||
if (low == -1)
|
||||
{
|
||||
add_replacement_character();
|
||||
return false;
|
||||
}
|
||||
if (0xDC00 <= low && low <= 0xDFFF)
|
||||
{
|
||||
add_codepoint(static_cast<int>((static_cast<unsigned int>(high) << 10u)
|
||||
+ static_cast<unsigned int>(low) - 0x35FDC00u));
|
||||
return true;
|
||||
}
|
||||
|
||||
// high has no low surrogate
|
||||
add_replacement_character();
|
||||
if (low < 0xD800 || low > 0xDBFF)
|
||||
{
|
||||
add_codepoint(low);
|
||||
return true;
|
||||
}
|
||||
// another high surrogate
|
||||
high = low;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief read the escape whose backslash was read; current is the escaped character
|
||||
@return whether the next character must be read; false if current still
|
||||
needs to be handled
|
||||
*/
|
||||
bool recover_escape()
|
||||
{
|
||||
switch (current)
|
||||
{
|
||||
case '\"':
|
||||
add('\"');
|
||||
return true;
|
||||
case '\\':
|
||||
add('\\');
|
||||
return true;
|
||||
case '/':
|
||||
add('/');
|
||||
return true;
|
||||
case 'b':
|
||||
add('\b');
|
||||
return true;
|
||||
case 'f':
|
||||
add('\f');
|
||||
return true;
|
||||
case 'n':
|
||||
add('\n');
|
||||
return true;
|
||||
case 'r':
|
||||
add('\r');
|
||||
return true;
|
||||
case 't':
|
||||
add('\t');
|
||||
return true;
|
||||
|
||||
case 'u':
|
||||
{
|
||||
const int codepoint = get_codepoint();
|
||||
if (codepoint == -1)
|
||||
{
|
||||
add_replacement_character();
|
||||
return false;
|
||||
}
|
||||
if (0xD800 <= codepoint && codepoint <= 0xDBFF)
|
||||
{
|
||||
return recover_low_surrogate(codepoint);
|
||||
}
|
||||
add_escaped_codepoint(codepoint);
|
||||
return true;
|
||||
}
|
||||
|
||||
// an unknown escape stands for the escaped character
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief read the rest of a string after scan_string() rejected it
|
||||
|
||||
token_buffer holds what scan_string() read before the error. See
|
||||
recover_token() for how errors are repaired.
|
||||
|
||||
@param[in] resume how to continue, see resume_kind
|
||||
@param[in] codepoint for a high surrogate followed by an escape of another
|
||||
code point: that code point; -1 otherwise
|
||||
*/
|
||||
token_type recover_string(const resume_kind resume, const int codepoint)
|
||||
{
|
||||
// whether the next character must be read before it can be handled
|
||||
bool fetch = false;
|
||||
|
||||
if (error_message_starts_with("invalid string: surrogate")
|
||||
|| error_message_starts_with("invalid string: '\\u'")
|
||||
|| (error_message_starts_with("invalid string: ill-formed UTF-8")
|
||||
&& remove_incomplete_utf8_sequence()))
|
||||
{
|
||||
add_replacement_character();
|
||||
}
|
||||
|
||||
switch (resume)
|
||||
{
|
||||
case resume_kind::escaped_character:
|
||||
fetch = recover_escape();
|
||||
break;
|
||||
case resume_kind::after_escape:
|
||||
if (0xD800 <= codepoint && codepoint <= 0xDBFF)
|
||||
{
|
||||
fetch = recover_low_surrogate(codepoint);
|
||||
}
|
||||
else
|
||||
{
|
||||
if (codepoint != -1)
|
||||
{
|
||||
add_escaped_codepoint(codepoint);
|
||||
}
|
||||
fetch = true;
|
||||
}
|
||||
break;
|
||||
case resume_kind::character:
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
while (true)
|
||||
{
|
||||
if (fetch)
|
||||
{
|
||||
get();
|
||||
}
|
||||
fetch = true;
|
||||
|
||||
switch (current)
|
||||
{
|
||||
case '\"':
|
||||
// a line break or the end of the input ends a string that
|
||||
// lacks its closing quote
|
||||
case '\n':
|
||||
case '\r':
|
||||
case char_traits<char_type>::eof():
|
||||
return token_type::value_string;
|
||||
|
||||
#if !JSON_STRICT_NUL_HANDLING
|
||||
case '\0':
|
||||
// the end of the input, see scan()
|
||||
unget();
|
||||
return token_type::value_string;
|
||||
#endif
|
||||
|
||||
case '\\':
|
||||
get();
|
||||
fetch = recover_escape();
|
||||
break;
|
||||
|
||||
default:
|
||||
if (current < 0x80)
|
||||
{
|
||||
// including control characters
|
||||
add(current);
|
||||
}
|
||||
else
|
||||
{
|
||||
fetch = recover_utf8_sequence();
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief keep the longest valid prefix of a number that scan_number() rejected
|
||||
|
||||
token_buffer holds the characters scan_number() accepted before the error,
|
||||
so the prefix ends at its last digit.
|
||||
*/
|
||||
token_type recover_number()
|
||||
{
|
||||
// only size(), operator[], and resize() are used, which every string
|
||||
// type the library supports provides
|
||||
std::size_t length = token_buffer.size();
|
||||
while (length != 0 && (token_buffer[length - 1] < '0' || token_buffer[length - 1] > '9'))
|
||||
{
|
||||
--length;
|
||||
}
|
||||
token_buffer.resize(length);
|
||||
|
||||
if (length == 0)
|
||||
{
|
||||
skip_to_delimiter();
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
|
||||
if (decimal_point_position >= length)
|
||||
{
|
||||
decimal_point_position = std::string::npos;
|
||||
}
|
||||
|
||||
std::size_t exponent = std::string::npos;
|
||||
for (std::size_t i = 0; i < length; ++i)
|
||||
{
|
||||
if (token_buffer[i] == 'e' || token_buffer[i] == 'E')
|
||||
{
|
||||
exponent = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
const std::size_t mantissa_end = (exponent == std::string::npos) ? length : exponent;
|
||||
token_type number_type = token_type::value_unsigned;
|
||||
if (decimal_point_position != std::string::npos || exponent != std::string::npos)
|
||||
{
|
||||
number_type = token_type::value_float;
|
||||
}
|
||||
else if (token_buffer[0] == '-')
|
||||
{
|
||||
number_type = token_type::value_integer;
|
||||
}
|
||||
|
||||
const token_type result = convert_number(number_type, mantissa_end);
|
||||
skip_to_delimiter();
|
||||
return result;
|
||||
}
|
||||
|
||||
/// seekable adapter: the token string begins at current, which was consumed
|
||||
void restart_token_string_impl(std::true_type /*lazy*/) noexcept
|
||||
{
|
||||
const std::size_t consumed = ia.get_consumed_count();
|
||||
token_string_start = (consumed > 0 && current != char_traits<char_type>::eof()) ? consumed - 1 : consumed;
|
||||
}
|
||||
|
||||
/// streaming adapter: the token string begins at current; a character
|
||||
/// that was put back is copied again when it is read again
|
||||
void restart_token_string_impl(std::false_type /*lazy*/)
|
||||
{
|
||||
token_string.clear();
|
||||
if (!next_unget && current != char_traits<char_type>::eof())
|
||||
{
|
||||
token_string.push_back(char_traits<char_type>::to_char_type(current));
|
||||
}
|
||||
}
|
||||
|
||||
/// input adapter
|
||||
InputAdapterType ia;
|
||||
|
||||
@@ -2172,6 +2754,13 @@ scan_number_done:
|
||||
/// a description of occurred lexer errors
|
||||
const char* error_message = "";
|
||||
|
||||
/// how recover_token() continues a string that scan_string() rejected;
|
||||
/// set only on the error paths that need more than error_message
|
||||
resume_kind string_error_resume = resume_kind::character;
|
||||
/// the code point of the second escape when a high surrogate is followed
|
||||
/// by an escape that is not a low surrogate; -1 otherwise
|
||||
int string_error_codepoint = -1;
|
||||
|
||||
// number values
|
||||
number_integer_t value_integer = 0;
|
||||
number_unsigned_t value_unsigned = 0;
|
||||
|
||||
@@ -139,26 +139,59 @@ class parser
|
||||
bool accept(const bool strict = true)
|
||||
{
|
||||
json_sax_acceptor<BasicJsonType> sax_acceptor;
|
||||
return sax_parse(&sax_acceptor, strict);
|
||||
return sax_parse_impl<false>(&sax_acceptor, strict);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief public SAX interface
|
||||
|
||||
If the SAX parser's parse_error() returns true, the parser recovers from
|
||||
the error: it repairs the input and continues (see #3989).
|
||||
|
||||
@param[in] sax the SAX parser
|
||||
@param[in] strict whether to expect the last token to be EOF
|
||||
@return whether the input was parsed without errors and no SAX event
|
||||
returned false
|
||||
*/
|
||||
template<typename SAX>
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
bool sax_parse(SAX* sax, const bool strict = true)
|
||||
{
|
||||
return sax_parse_impl<true>(sax, strict);
|
||||
}
|
||||
|
||||
private:
|
||||
/// what sax_parse_internal() does after an object key was expected
|
||||
enum class next_step : std::uint8_t
|
||||
{
|
||||
/// stop parsing
|
||||
stop,
|
||||
/// parse a value that begins with last_token
|
||||
parse_value,
|
||||
/// evaluate the state of the innermost container, which reads
|
||||
/// last_token again
|
||||
evaluate_state
|
||||
};
|
||||
|
||||
template<bool AllowRecovery, typename SAX>
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
bool sax_parse_impl(SAX* sax, const bool strict)
|
||||
{
|
||||
(void)detail::is_sax_static_asserts<SAX, BasicJsonType> {};
|
||||
const bool result = sax_parse_internal(sax);
|
||||
const bool result = sax_parse_internal<AllowRecovery>(sax);
|
||||
|
||||
if (result)
|
||||
{
|
||||
if (strict)
|
||||
{
|
||||
// strict mode: next byte must be EOF
|
||||
if (get_token() != token_type::end_of_input)
|
||||
// strict mode: next byte must be EOF; after recovering from an
|
||||
// error, the end of the input may already have been read
|
||||
if (last_token != token_type::end_of_input && get_token() != token_type::end_of_input)
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr));
|
||||
// the value is complete, so there is nothing to recover
|
||||
static_cast<void>(report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr),
|
||||
std::integral_constant<bool, AllowRecovery> {}));
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -169,10 +202,9 @@ class parser
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
return result && !error_reported;
|
||||
}
|
||||
|
||||
private:
|
||||
/*!
|
||||
@brief run a DOM SAX parser to completion and position the lexer
|
||||
|
||||
@@ -190,7 +222,7 @@ class parser
|
||||
template<typename DomSax>
|
||||
bool parse_dom(DomSax& sdp, const bool strict)
|
||||
{
|
||||
sax_parse_internal(&sdp);
|
||||
sax_parse_internal<false>(&sdp);
|
||||
|
||||
if (strict)
|
||||
{
|
||||
@@ -213,10 +245,20 @@ class parser
|
||||
return !sdp.is_errored();
|
||||
}
|
||||
|
||||
template<typename SAX>
|
||||
/*!
|
||||
@brief parse a JSON value and pass it to a SAX parser
|
||||
|
||||
@tparam AllowRecovery whether to recover from an error if the SAX parser's
|
||||
parse_error() returns true; false for the SAX parsers
|
||||
of parse() and accept(), which never do, so that no
|
||||
code for recovering is generated for them
|
||||
*/
|
||||
template<bool AllowRecovery, typename SAX>
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
bool sax_parse_internal(SAX* sax)
|
||||
{
|
||||
const std::integral_constant<bool, AllowRecovery> allow_recovery{};
|
||||
|
||||
// stack to remember the hierarchy of structured values we are parsing
|
||||
// true = array; false = object
|
||||
std::vector<bool> states;
|
||||
@@ -247,12 +289,18 @@ class parser
|
||||
break;
|
||||
}
|
||||
|
||||
// parse key
|
||||
// remember we are now inside an object
|
||||
states.push_back(false);
|
||||
|
||||
// parse key (the steps of parse_key(), which are
|
||||
// repeated here and below for speed)
|
||||
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr));
|
||||
if (!continue_after(key_error(sax, allow_recovery, false), skip_to_state_evaluation))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
|
||||
{
|
||||
@@ -262,14 +310,13 @@ class parser
|
||||
// parse separator (:)
|
||||
if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator)))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr));
|
||||
if (!continue_after(key_error(sax, allow_recovery, true), skip_to_state_evaluation))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// remember we are now inside an object
|
||||
states.push_back(false);
|
||||
|
||||
// parse values
|
||||
get_token();
|
||||
continue;
|
||||
@@ -305,9 +352,11 @@ class parser
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(!std::isfinite(res)))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr));
|
||||
if (!overflow_error(sax, res, allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->number_float(res, m_lexer.get_string())))
|
||||
@@ -375,23 +424,63 @@ class parser
|
||||
case token_type::parse_error:
|
||||
{
|
||||
// using "uninitialized" to avoid an "expected" message
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::uninitialized, "value"), nullptr));
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::uninitialized, "value"), nullptr), allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// recover: keep what can be read of the token
|
||||
recover_token(allow_recovery);
|
||||
if (last_token != token_type::uninitialized)
|
||||
{
|
||||
// a string or a number
|
||||
continue;
|
||||
}
|
||||
if (states.empty())
|
||||
{
|
||||
// look for the value after the garbage
|
||||
if (!skip_to_value(allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
// nothing could be read
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->null()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case token_type::end_of_input:
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(m_lexer.get_position().chars_read_total == 1))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(),
|
||||
"attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr));
|
||||
// there is nothing to recover
|
||||
static_cast<void>(report_error(sax, parse_error::create(101, m_lexer.get_position(),
|
||||
"attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr), allow_recovery));
|
||||
return false;
|
||||
}
|
||||
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr));
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr), allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// recover: the input ends where a value is missing
|
||||
if (states.empty())
|
||||
{
|
||||
// there is no value
|
||||
return false;
|
||||
}
|
||||
if (!recover_missing_value(sax, states, allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
// the state evaluation reads the token again
|
||||
unget_token(allow_recovery);
|
||||
skip_to_state_evaluation = true;
|
||||
continue;
|
||||
}
|
||||
case token_type::uninitialized:
|
||||
case token_type::end_array:
|
||||
@@ -401,9 +490,35 @@ class parser
|
||||
case token_type::literal_or_value:
|
||||
default: // the last token was unexpected
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr));
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr), allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// recover
|
||||
if (states.empty())
|
||||
{
|
||||
// look for the value after the garbage
|
||||
if (!skip_to_value(allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (last_token == token_type::name_separator)
|
||||
{
|
||||
// a stray ':'; the value may follow
|
||||
get_token();
|
||||
continue;
|
||||
}
|
||||
if (!recover_missing_value(sax, states, allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
// the state evaluation reads the token again
|
||||
unget_token(allow_recovery);
|
||||
skip_to_state_evaluation = true;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -454,9 +569,30 @@ class parser
|
||||
continue;
|
||||
}
|
||||
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_array, "array"), nullptr));
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_array, "array"), nullptr), allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// recover
|
||||
if (last_token == token_type::end_of_input)
|
||||
{
|
||||
// the input ends inside the array
|
||||
return close_containers(sax, states, allow_recovery);
|
||||
}
|
||||
if (last_token == token_type::end_object)
|
||||
{
|
||||
// a wrong closing bracket closes the innermost container
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->end_array()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
states.pop_back();
|
||||
skip_to_state_evaluation = true;
|
||||
}
|
||||
// otherwise, a missing ',' (or a stray ':', which value
|
||||
// parsing drops): the next value begins here
|
||||
continue;
|
||||
}
|
||||
|
||||
// states.back() is false -> object
|
||||
@@ -473,11 +609,12 @@ class parser
|
||||
// parse key
|
||||
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr));
|
||||
if (!continue_after(key_error(sax, allow_recovery, false), skip_to_state_evaluation))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
|
||||
{
|
||||
return false;
|
||||
@@ -486,9 +623,11 @@ class parser
|
||||
// parse separator (:)
|
||||
if (JSON_HEDLEY_UNLIKELY(!get_token_expecting(token_type::name_separator)))
|
||||
{
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr));
|
||||
if (!continue_after(key_error(sax, allow_recovery, true), skip_to_state_evaluation))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// parse values
|
||||
@@ -516,12 +655,546 @@ class parser
|
||||
continue;
|
||||
}
|
||||
|
||||
return sax->parse_error(m_lexer.get_position(),
|
||||
m_lexer.get_token_string(),
|
||||
parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_object, "object"), nullptr));
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_object, "object"), nullptr), allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// recover
|
||||
if (last_token == token_type::end_of_input)
|
||||
{
|
||||
// the input ends inside the object
|
||||
return close_containers(sax, states, allow_recovery);
|
||||
}
|
||||
if (last_token == token_type::end_array)
|
||||
{
|
||||
// a wrong closing bracket closes the innermost container
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->end_object()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
states.pop_back();
|
||||
skip_to_state_evaluation = true;
|
||||
continue;
|
||||
}
|
||||
if (!continue_after(recover_member(sax, allow_recovery), skip_to_state_evaluation))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief continue sax_parse_internal() after a recovery
|
||||
@return whether to continue parsing
|
||||
*/
|
||||
bool continue_after(const next_step step, bool& skip_to_state_evaluation)
|
||||
{
|
||||
if (step == next_step::evaluate_state)
|
||||
{
|
||||
// the state evaluation reads the token again
|
||||
m_lexer.unget_token();
|
||||
skip_to_state_evaluation = true;
|
||||
}
|
||||
return step != next_step::stop;
|
||||
}
|
||||
|
||||
/// the parser for parse() and accept() never recovers: stop parsing
|
||||
JSON_INTERNAL_ALWAYS_INLINE
|
||||
static std::false_type continue_after(std::false_type /*step*/, bool& /*skip_to_state_evaluation*/) noexcept
|
||||
{
|
||||
return {};
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief parse an object key and the name separator (:) after it
|
||||
|
||||
last_token is the token where the key is expected. sax_parse_internal()
|
||||
repeats these steps rather than calling this function, which is used
|
||||
when recovering from an error.
|
||||
|
||||
@return next_step::parse_value if the value follows, with last_token its
|
||||
first token; next_step::evaluate_state if the object's state is
|
||||
to be evaluated after recovering from an error; next_step::stop
|
||||
to stop parsing
|
||||
*/
|
||||
template<typename SAX>
|
||||
next_step parse_key(SAX* sax)
|
||||
{
|
||||
const std::true_type allow_recovery{};
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string))
|
||||
{
|
||||
return key_error(sax, allow_recovery, false);
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string())))
|
||||
{
|
||||
return next_step::stop;
|
||||
}
|
||||
|
||||
// parse separator (:)
|
||||
if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator))
|
||||
{
|
||||
return key_error(sax, allow_recovery, true);
|
||||
}
|
||||
|
||||
// the value begins with the next token
|
||||
get_token();
|
||||
return next_step::parse_value;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief report a number that is too large for number_float_t, and recover
|
||||
from the error by passing the value on; the SAX parser gets the
|
||||
number's text as well
|
||||
|
||||
This is a separate function, as reading other numbers is measurably
|
||||
slower if the error is handled where they are read.
|
||||
|
||||
@param[in] sax the SAX parser
|
||||
@param[in] value the value that is not finite
|
||||
@return whether to continue parsing
|
||||
*/
|
||||
template<typename SAX>
|
||||
bool overflow_error(SAX* sax, const number_float_t value, std::true_type allow_recovery)
|
||||
{
|
||||
if (!report_error(sax, out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr), allow_recovery))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
return sax->number_float(value, m_lexer.get_string());
|
||||
}
|
||||
|
||||
/// @copydoc overflow_error
|
||||
template<typename SAX>
|
||||
JSON_INTERNAL_ALWAYS_INLINE
|
||||
std::false_type overflow_error(SAX* sax, const number_float_t /*value*/, std::false_type allow_recovery)
|
||||
{
|
||||
return report_error(sax, out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr), allow_recovery);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief report a missing key, or a missing name separator (:) after the
|
||||
key; the parser for parse() and accept() never recovers
|
||||
|
||||
@param[in] key_read whether the key was read, so that the name separator
|
||||
is missing
|
||||
@return std::false_type, see report_error()
|
||||
*/
|
||||
template<typename SAX>
|
||||
JSON_INTERNAL_ALWAYS_INLINE
|
||||
std::false_type key_error(SAX* sax, std::false_type allow_recovery, const bool key_read)
|
||||
{
|
||||
return report_error(sax, parse_error::create(101, m_lexer.get_position(), key_read
|
||||
? exception_message(token_type::name_separator, "object separator")
|
||||
: exception_message(token_type::value_string, "object key"), nullptr), allow_recovery);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief report a missing key, or a missing name separator (:) after the
|
||||
key, and recover from it
|
||||
|
||||
@param[in] key_read whether the key was read, so that the name separator
|
||||
is missing
|
||||
*/
|
||||
template<typename SAX>
|
||||
next_step key_error(SAX* sax, std::true_type allow_recovery, const bool key_read)
|
||||
{
|
||||
if (!key_read)
|
||||
{
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr), allow_recovery))
|
||||
{
|
||||
return next_step::stop;
|
||||
}
|
||||
return recover_key(sax);
|
||||
}
|
||||
|
||||
if (!report_error(sax, parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr), allow_recovery))
|
||||
{
|
||||
return next_step::stop;
|
||||
}
|
||||
return recover_name_separator(sax);
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// error recovery
|
||||
/////////////////////
|
||||
|
||||
/*
|
||||
The functions below repair an error after the SAX parser's parse_error()
|
||||
returned true (see #3989). Each mistake is repaired by the smallest local
|
||||
edit: a missing ',' or ':' is inserted, a stray token is removed, what can
|
||||
be read of an invalid string or number is kept (see
|
||||
lexer::recover_token()), a missing value becomes null, a wrong closing
|
||||
bracket closes the innermost container, and the end of the input closes
|
||||
all of them. The events stay balanced, and every key() is followed by
|
||||
exactly one value.
|
||||
|
||||
A repair hands a token to the state evaluation, by returning it to the
|
||||
lexer (lexer::unget_token()) so that the state evaluation reads it again,
|
||||
only if it is ',', ']', '}', or the end of the input. The state evaluation
|
||||
hands a token to value or key parsing only if it is none of them, so a
|
||||
token is never handed back and forth. Every other step reads a token or
|
||||
closes a container, so parsing always ends.
|
||||
*/
|
||||
|
||||
/*!
|
||||
@brief report an error to the SAX parser; the parser for parse() and
|
||||
accept() never recovers
|
||||
|
||||
@return std::false_type rather than false: its value is known where the
|
||||
function is called even if the call is not inlined, so the code
|
||||
for recovering is not generated
|
||||
*/
|
||||
template<typename SAX, typename Exception>
|
||||
JSON_INTERNAL_ALWAYS_INLINE
|
||||
std::false_type report_error(SAX* sax, const Exception& ex, std::false_type /*allow_recovery*/)
|
||||
{
|
||||
static_cast<void>(sax); // MSVC 2015 does not count calling a static parse_error() as using it
|
||||
error_reported = true;
|
||||
static_cast<void>(sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), ex));
|
||||
return {};
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief report an error to the SAX parser
|
||||
@return whether to recover from the error
|
||||
*/
|
||||
template<typename SAX, typename Exception>
|
||||
bool report_error(SAX* sax, const Exception& ex, std::true_type /*allow_recovery*/)
|
||||
{
|
||||
static_cast<void>(sax); // MSVC 2015 does not count calling a static parse_error() as using it
|
||||
const std::size_t position = m_lexer.get_position().chars_read_total;
|
||||
if (error_reported && position == last_error_position && last_token == last_error_token)
|
||||
{
|
||||
// a repair handed on the token of the error it repaired; the
|
||||
// token was reported already, and the SAX parser asked to recover
|
||||
return true;
|
||||
}
|
||||
|
||||
error_reported = true;
|
||||
last_error_position = position;
|
||||
last_error_token = last_token;
|
||||
|
||||
if (!sax->parse_error(m_lexer.get_position(), m_lexer.get_token_string(), ex))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// the token string of the next error begins here
|
||||
m_lexer.restart_token_string();
|
||||
return true;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief keep what can be read of the token that the lexer rejected
|
||||
|
||||
The error was reported for the rejected token, so it is not reported again
|
||||
for the token it is repaired to (see lexer::recover_token()).
|
||||
*/
|
||||
token_type recover_token()
|
||||
{
|
||||
last_token = m_lexer.recover_token();
|
||||
last_error_position = m_lexer.get_position().chars_read_total;
|
||||
last_error_token = last_token;
|
||||
return last_token;
|
||||
}
|
||||
|
||||
// The functions below are called from sax_parse_internal() after an error
|
||||
// was reported. The parser for parse() and accept() stopped there, so for
|
||||
// it they are stubs: the code for recovering is then not even referenced,
|
||||
// and unoptimized builds do not emit it either.
|
||||
|
||||
/// @copydoc recover_token()
|
||||
JSON_INTERNAL_ALWAYS_INLINE
|
||||
token_type recover_token(std::true_type /*allow_recovery*/)
|
||||
{
|
||||
return recover_token();
|
||||
}
|
||||
|
||||
/// @copydoc recover_token()
|
||||
JSON_INTERNAL_ALWAYS_INLINE
|
||||
static std::false_type recover_token(std::false_type /*allow_recovery*/) noexcept
|
||||
{
|
||||
return {};
|
||||
}
|
||||
|
||||
/// return the token that was read last to the lexer (see lexer::unget_token())
|
||||
JSON_INTERNAL_ALWAYS_INLINE
|
||||
void unget_token(std::true_type /*allow_recovery*/)
|
||||
{
|
||||
m_lexer.unget_token();
|
||||
}
|
||||
|
||||
/// @copydoc unget_token(std::true_type)
|
||||
JSON_INTERNAL_ALWAYS_INLINE
|
||||
static void unget_token(std::false_type /*allow_recovery*/) noexcept {}
|
||||
|
||||
/// @copydoc skip_to_value
|
||||
JSON_INTERNAL_ALWAYS_INLINE
|
||||
static std::false_type skip_to_value(std::false_type /*allow_recovery*/) noexcept
|
||||
{
|
||||
return {};
|
||||
}
|
||||
|
||||
/// @copydoc recover_missing_value
|
||||
template<typename SAX>
|
||||
JSON_INTERNAL_ALWAYS_INLINE
|
||||
static std::false_type recover_missing_value(SAX* /*sax*/, const std::vector<bool>& /*states*/, std::false_type /*allow_recovery*/) noexcept
|
||||
{
|
||||
return {};
|
||||
}
|
||||
|
||||
/// pass the end events of all open containers
|
||||
template<typename SAX>
|
||||
bool close_containers(SAX* sax, std::vector<bool>& states, std::true_type /*allow_recovery*/)
|
||||
{
|
||||
while (!states.empty())
|
||||
{
|
||||
const bool is_array = states.back();
|
||||
states.pop_back();
|
||||
if (JSON_HEDLEY_UNLIKELY(is_array ? !sax->end_array() : !sax->end_object()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// @copydoc close_containers
|
||||
template<typename SAX>
|
||||
JSON_INTERNAL_ALWAYS_INLINE
|
||||
static std::false_type close_containers(SAX* /*sax*/, std::vector<bool>& /*states*/, std::false_type /*allow_recovery*/) noexcept
|
||||
{
|
||||
return {};
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief read tokens until one begins a value, skipping everything before
|
||||
the top-level value
|
||||
@return whether a value begins with last_token
|
||||
*/
|
||||
bool skip_to_value(std::true_type /*allow_recovery*/)
|
||||
{
|
||||
while (true)
|
||||
{
|
||||
switch (get_token())
|
||||
{
|
||||
case token_type::begin_array:
|
||||
case token_type::begin_object:
|
||||
case token_type::literal_false:
|
||||
case token_type::literal_null:
|
||||
case token_type::literal_true:
|
||||
case token_type::value_float:
|
||||
case token_type::value_integer:
|
||||
case token_type::value_string:
|
||||
case token_type::value_unsigned:
|
||||
return true;
|
||||
|
||||
case token_type::end_of_input:
|
||||
return false;
|
||||
|
||||
case token_type::parse_error:
|
||||
recover_token();
|
||||
if (last_token != token_type::uninitialized)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
break;
|
||||
|
||||
case token_type::uninitialized:
|
||||
case token_type::end_array:
|
||||
case token_type::end_object:
|
||||
case token_type::name_separator:
|
||||
case token_type::value_separator:
|
||||
case token_type::literal_or_value:
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief skip the rest of an object member that cannot be read
|
||||
|
||||
Reads tokens, beginning with last_token, until a ',', '}', or ']' that is
|
||||
not inside a container that begins in the skipped tokens, or the end of
|
||||
the input.
|
||||
*/
|
||||
void skip_member()
|
||||
{
|
||||
std::size_t depth = 0;
|
||||
while (true)
|
||||
{
|
||||
switch (last_token)
|
||||
{
|
||||
case token_type::begin_array:
|
||||
case token_type::begin_object:
|
||||
++depth;
|
||||
break;
|
||||
|
||||
case token_type::end_array:
|
||||
case token_type::end_object:
|
||||
if (depth == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
--depth;
|
||||
break;
|
||||
|
||||
case token_type::value_separator:
|
||||
if (depth == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
break;
|
||||
|
||||
case token_type::end_of_input:
|
||||
return;
|
||||
|
||||
case token_type::parse_error:
|
||||
recover_token();
|
||||
break;
|
||||
|
||||
case token_type::uninitialized:
|
||||
case token_type::literal_true:
|
||||
case token_type::literal_false:
|
||||
case token_type::literal_null:
|
||||
case token_type::value_string:
|
||||
case token_type::value_unsigned:
|
||||
case token_type::value_integer:
|
||||
case token_type::value_float:
|
||||
case token_type::name_separator:
|
||||
case token_type::literal_or_value:
|
||||
default:
|
||||
break;
|
||||
}
|
||||
get_token();
|
||||
}
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief pass a value where it is missing
|
||||
|
||||
last_token is ',', ']', '}', or the end of the input, where a value was
|
||||
expected. In an object, the key gets null; in an array, a ',' where a
|
||||
value is missing stands for null (as in JavaScript), while an array that
|
||||
ends there just ends.
|
||||
*/
|
||||
template<typename SAX>
|
||||
bool recover_missing_value(SAX* sax, const std::vector<bool>& states, std::true_type /*allow_recovery*/)
|
||||
{
|
||||
JSON_ASSERT(!states.empty());
|
||||
if (!states.back() || last_token == token_type::value_separator)
|
||||
{
|
||||
return sax->null();
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// recover from a missing key; last_token is where it was expected
|
||||
template<typename SAX>
|
||||
next_step recover_key(SAX* sax)
|
||||
{
|
||||
switch (last_token)
|
||||
{
|
||||
case token_type::value_separator:
|
||||
case token_type::end_object:
|
||||
case token_type::end_array:
|
||||
case token_type::end_of_input:
|
||||
// no member: the object's state handles the token
|
||||
return next_step::evaluate_state;
|
||||
|
||||
case token_type::parse_error:
|
||||
recover_token();
|
||||
if (last_token == token_type::value_string)
|
||||
{
|
||||
// a key that could be repaired
|
||||
return parse_key(sax);
|
||||
}
|
||||
skip_member();
|
||||
return next_step::evaluate_state;
|
||||
|
||||
case token_type::uninitialized:
|
||||
case token_type::literal_true:
|
||||
case token_type::literal_false:
|
||||
case token_type::literal_null:
|
||||
case token_type::value_string:
|
||||
case token_type::value_unsigned:
|
||||
case token_type::value_integer:
|
||||
case token_type::value_float:
|
||||
case token_type::begin_array:
|
||||
case token_type::begin_object:
|
||||
case token_type::name_separator:
|
||||
case token_type::literal_or_value:
|
||||
default:
|
||||
// a member without a key
|
||||
skip_member();
|
||||
return next_step::evaluate_state;
|
||||
}
|
||||
}
|
||||
|
||||
/// recover from a missing name separator (:) after the key; last_token
|
||||
/// is where it was expected
|
||||
template<typename SAX>
|
||||
next_step recover_name_separator(SAX* sax)
|
||||
{
|
||||
switch (last_token)
|
||||
{
|
||||
case token_type::value_separator:
|
||||
case token_type::end_object:
|
||||
case token_type::end_array:
|
||||
case token_type::end_of_input:
|
||||
// the value is missing as well
|
||||
return sax->null() ? next_step::evaluate_state : next_step::stop;
|
||||
|
||||
case token_type::uninitialized:
|
||||
case token_type::literal_true:
|
||||
case token_type::literal_false:
|
||||
case token_type::literal_null:
|
||||
case token_type::value_string:
|
||||
case token_type::value_unsigned:
|
||||
case token_type::value_integer:
|
||||
case token_type::value_float:
|
||||
case token_type::begin_array:
|
||||
case token_type::begin_object:
|
||||
case token_type::name_separator:
|
||||
case token_type::parse_error:
|
||||
case token_type::literal_or_value:
|
||||
default:
|
||||
// a missing ':'; the value begins here
|
||||
return next_step::parse_value;
|
||||
}
|
||||
}
|
||||
|
||||
/// recover from a token after an object member that is neither ',' nor
|
||||
/// '}' (nor ']' or the end of the input, which the caller handles)
|
||||
template<typename SAX>
|
||||
next_step recover_member(SAX* sax, std::true_type /*allow_recovery*/)
|
||||
{
|
||||
if (last_token == token_type::parse_error)
|
||||
{
|
||||
recover_token();
|
||||
}
|
||||
if (last_token == token_type::value_string)
|
||||
{
|
||||
// a missing ','; the next key begins here
|
||||
return parse_key(sax);
|
||||
}
|
||||
skip_member();
|
||||
return next_step::evaluate_state;
|
||||
}
|
||||
|
||||
/// the parser for parse() and accept() never recovers (and does not come
|
||||
/// here, as report_error() returned false)
|
||||
template<typename SAX>
|
||||
JSON_INTERNAL_ALWAYS_INLINE
|
||||
std::false_type recover_member(SAX* /*sax*/, std::false_type /*allow_recovery*/) const noexcept
|
||||
{
|
||||
return {};
|
||||
}
|
||||
|
||||
/// get next token from lexer
|
||||
token_type get_token()
|
||||
{
|
||||
@@ -575,6 +1248,12 @@ class parser
|
||||
const bool allow_exceptions = true;
|
||||
/// whether trailing commas in objects and arrays should be ignored (true) or signaled as errors (false)
|
||||
const bool ignore_trailing_commas = false;
|
||||
/// whether an error was reported to the SAX parser
|
||||
bool error_reported = false;
|
||||
/// the position of the last reported error
|
||||
std::size_t last_error_position = 0;
|
||||
/// the token of the last reported error
|
||||
token_type last_error_token = token_type::uninitialized;
|
||||
};
|
||||
|
||||
} // namespace detail
|
||||
|
||||
@@ -199,6 +199,17 @@
|
||||
#define JSON_NO_UNIQUE_ADDRESS
|
||||
#endif
|
||||
|
||||
// Inlines small functions even in unoptimized builds, so that they are not
|
||||
// emitted. The parsers for parse(), accept(), and from_*() use it for the
|
||||
// functions that stand in for the code recovering from errors (see #3989).
|
||||
// MSVC is left to decide, as it warns (C4714) where it does not inline a
|
||||
// __forceinline function.
|
||||
#if defined(_MSC_VER) && !defined(__clang__)
|
||||
#define JSON_INTERNAL_ALWAYS_INLINE
|
||||
#else
|
||||
#define JSON_INTERNAL_ALWAYS_INLINE JSON_HEDLEY_ALWAYS_INLINE
|
||||
#endif
|
||||
|
||||
// Clang targeting MinGW does not survive the thread_local storage the copy
|
||||
// constructor uses to bound its descent: every test that copies a value
|
||||
// segfaults with clang 11.0.1 and clang 18.1.8, while the same tests pass with
|
||||
|
||||
@@ -19,6 +19,7 @@
|
||||
#undef NLOHMANN_CAN_CALL_STD_FUNC_IMPL
|
||||
#undef JSON_INLINE_VARIABLE
|
||||
#undef JSON_NO_UNIQUE_ADDRESS
|
||||
#undef JSON_INTERNAL_ALWAYS_INLINE
|
||||
#undef JSON_DISABLE_ENUM_SERIALIZATION
|
||||
#undef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION
|
||||
|
||||
|
||||
@@ -207,10 +207,6 @@ class binary_writer
|
||||
{
|
||||
if (j.m_data.m_value.number_integer >= 0)
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::uint64_t>(j.m_data.m_value.number_integer)))
|
||||
{
|
||||
throw_integer_out_of_range(j, "CBOR", "[-2^64, 2^64-1]");
|
||||
}
|
||||
// CBOR does not differentiate between positive signed
|
||||
// integers and unsigned integers
|
||||
write_cbor_head(0x00, static_cast<std::uint64_t>(j.m_data.m_value.number_integer));
|
||||
@@ -218,10 +214,6 @@ class binary_writer
|
||||
else
|
||||
{
|
||||
// a negative integer n is encoded as -1 - n
|
||||
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::uint64_t>(-1 - j.m_data.m_value.number_integer)))
|
||||
{
|
||||
throw_integer_out_of_range(j, "CBOR", "[-2^64, 2^64-1]");
|
||||
}
|
||||
write_cbor_head(0x20, static_cast<std::uint64_t>(-1 - j.m_data.m_value.number_integer));
|
||||
}
|
||||
break;
|
||||
@@ -229,11 +221,7 @@ class binary_writer
|
||||
|
||||
case value_t::number_unsigned:
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::uint64_t>(j.m_data.m_value.number_unsigned)))
|
||||
{
|
||||
throw_integer_out_of_range(j, "CBOR", "uint64");
|
||||
}
|
||||
write_cbor_head(0x00, static_cast<std::uint64_t>(j.m_data.m_value.number_unsigned));
|
||||
write_cbor_head(0x00, j.m_data.m_value.number_unsigned);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -441,20 +429,12 @@ class binary_writer
|
||||
{
|
||||
if (j.m_data.m_value.number_integer >= 0)
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::uint64_t>(j.m_data.m_value.number_integer)))
|
||||
{
|
||||
throw_integer_out_of_range(j, "MessagePack", "[-2^63, 2^64-1]");
|
||||
}
|
||||
// MessagePack does not differentiate between positive
|
||||
// signed integers and unsigned integers.
|
||||
write_msgpack_unsigned(static_cast<std::uint64_t>(j.m_data.m_value.number_integer));
|
||||
}
|
||||
else
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(!integer_fits_min<std::int64_t>(j.m_data.m_value.number_integer)))
|
||||
{
|
||||
throw_integer_out_of_range(j, "MessagePack", "[-2^63, 2^64-1]");
|
||||
}
|
||||
if (j.m_data.m_value.number_integer >= -32)
|
||||
{
|
||||
// negative fixnum
|
||||
@@ -493,10 +473,6 @@ class binary_writer
|
||||
|
||||
case value_t::number_unsigned:
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::uint64_t>(j.m_data.m_value.number_unsigned)))
|
||||
{
|
||||
throw_integer_out_of_range(j, "MessagePack", "uint64");
|
||||
}
|
||||
write_msgpack_unsigned(static_cast<std::uint64_t>(j.m_data.m_value.number_unsigned));
|
||||
break;
|
||||
}
|
||||
@@ -858,84 +834,6 @@ class binary_writer
|
||||
JSON_THROW(type_error::create(321, concat("cannot serialize discarded value to ", format_name), &j));
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief whether integer @a n does not exceed the maximum of @a TargetType
|
||||
|
||||
The binary formats store integers in at most 64 bits, but number_integer_t
|
||||
and number_unsigned_t may be wider (e.g., __int128). The range of the
|
||||
integer is taken from std::numeric_limits, so a number type whose range
|
||||
fits into @a TargetType is never checked: the function is then constant
|
||||
true, and the default int64_t/uint64_t types pay nothing.
|
||||
*/
|
||||
template<typename TargetType, typename NumberType>
|
||||
static constexpr bool integer_fits_max(const NumberType n) noexcept
|
||||
{
|
||||
return integer_fits_max<TargetType>(n, std::integral_constant < bool,
|
||||
(std::numeric_limits<NumberType>::digits > std::numeric_limits<TargetType>::digits) > ());
|
||||
}
|
||||
|
||||
template<typename TargetType, typename NumberType>
|
||||
static constexpr bool integer_fits_max(const NumberType /*unused*/, std::false_type /*may_exceed*/) noexcept
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
template<typename TargetType, typename NumberType>
|
||||
static constexpr bool integer_fits_max(const NumberType n, std::true_type /*may_exceed*/) noexcept
|
||||
{
|
||||
// NumberType has more digits than TargetType, so it can hold its maximum
|
||||
return n <= static_cast<NumberType>((std::numeric_limits<TargetType>::max)());
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief whether integer @a n is not below the minimum of @a TargetType;
|
||||
constant true if NumberType cannot hold such a value
|
||||
*/
|
||||
template<typename TargetType, typename NumberType>
|
||||
static constexpr bool integer_fits_min(const NumberType n) noexcept
|
||||
{
|
||||
return integer_fits_min<TargetType>(n, std::integral_constant < bool, std::numeric_limits<NumberType>::is_signed &&
|
||||
(!std::numeric_limits<TargetType>::is_signed || std::numeric_limits<NumberType>::digits > std::numeric_limits<TargetType>::digits) > ());
|
||||
}
|
||||
|
||||
template<typename TargetType, typename NumberType>
|
||||
static constexpr bool integer_fits_min(const NumberType /*unused*/, std::false_type /*may_fall_below*/) noexcept
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
template<typename TargetType, typename NumberType>
|
||||
static constexpr bool integer_fits_min(const NumberType n, std::true_type /*may_fall_below*/) noexcept
|
||||
{
|
||||
// NumberType is signed and either TargetType is unsigned (minimum 0)
|
||||
// or NumberType has more digits, so it can hold the minimum
|
||||
return n >= static_cast<NumberType>((std::numeric_limits<TargetType>::min)());
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief whether integer @a n lies in the range of @a TargetType
|
||||
*/
|
||||
template<typename TargetType, typename NumberType>
|
||||
static constexpr bool integer_fits(const NumberType n) noexcept
|
||||
{
|
||||
return integer_fits_min<TargetType>(n) && integer_fits_max<TargetType>(n);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief throws because the integer @a j is too large for @a format_name
|
||||
@throw out_of_range.407 always
|
||||
*/
|
||||
JSON_HEDLEY_NO_RETURN static void throw_integer_out_of_range(const BasicJsonType& j, const char* format_name, const char* range)
|
||||
{
|
||||
// dump() rather than std::to_string(), which has no overload for
|
||||
// integer types wider than 64 bits
|
||||
const auto number = j.dump();
|
||||
static_cast<void>(number); // unused when JSON_NOEXCEPTION is defined
|
||||
static_cast<void>(format_name);
|
||||
static_cast<void>(range);
|
||||
JSON_THROW(out_of_range::create(407, concat("integer number ", std::string(number.begin(), number.end()), " cannot be represented by ", format_name, " as it does not fit ", range), &j));
|
||||
}
|
||||
|
||||
void write_msgpack_array_prefix(const std::size_t N, const BasicJsonType& j)
|
||||
{
|
||||
const auto n = to_msgpack_length(N, j);
|
||||
@@ -1692,8 +1590,6 @@ class binary_writer
|
||||
is neither an object nor an array
|
||||
@throw out_of_range.415 if @a j is binary with a subtype that does not fit
|
||||
into a byte, before anything is written
|
||||
@throw out_of_range.407 if @a j is an integer that does not fit the 64 bits
|
||||
of a BSON integer, before anything is written
|
||||
@throw type_error.316 if @a j is a string that is not valid UTF-8, before
|
||||
anything is written
|
||||
@throw type_error.321 if @a j is discarded
|
||||
@@ -1712,18 +1608,10 @@ class binary_writer
|
||||
return 8ul;
|
||||
|
||||
case value_t::number_integer:
|
||||
if (JSON_HEDLEY_UNLIKELY(!integer_fits<std::int64_t>(j.m_data.m_value.number_integer)))
|
||||
{
|
||||
throw_integer_out_of_range(j, "BSON", "int64");
|
||||
}
|
||||
return calc_bson_integer_size(static_cast<std::int64_t>(j.m_data.m_value.number_integer));
|
||||
return calc_bson_integer_size(j.m_data.m_value.number_integer);
|
||||
|
||||
case value_t::number_unsigned:
|
||||
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::uint64_t>(j.m_data.m_value.number_unsigned)))
|
||||
{
|
||||
throw_integer_out_of_range(j, "BSON", "uint64");
|
||||
}
|
||||
return calc_bson_unsigned_size(static_cast<std::uint64_t>(j.m_data.m_value.number_unsigned));
|
||||
return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned);
|
||||
|
||||
case value_t::string:
|
||||
return calc_bson_string_size(*j.m_data.m_value.string, j);
|
||||
@@ -1761,12 +1649,11 @@ class binary_writer
|
||||
case value_t::number_float:
|
||||
return write_bson_double(name, j.m_data.m_value.number_float);
|
||||
|
||||
// calc_bson_value_size() checked that integers fit 64 bits
|
||||
case value_t::number_integer:
|
||||
return write_bson_integer(name, static_cast<std::int64_t>(j.m_data.m_value.number_integer));
|
||||
return write_bson_integer(name, j.m_data.m_value.number_integer);
|
||||
|
||||
case value_t::number_unsigned:
|
||||
return write_bson_unsigned(name, static_cast<std::uint64_t>(j.m_data.m_value.number_unsigned));
|
||||
return write_bson_unsigned(name, j.m_data.m_value.number_unsigned);
|
||||
|
||||
case value_t::string:
|
||||
return write_bson_string(name, *j.m_data.m_value.string);
|
||||
@@ -2221,7 +2108,7 @@ class binary_writer
|
||||
{
|
||||
return 'L';
|
||||
}
|
||||
if (use_bjdata && std::is_unsigned<NumberType>::value && integer_fits_max<std::uint64_t>(n))
|
||||
if (use_bjdata && std::is_unsigned<NumberType>::value)
|
||||
{
|
||||
return 'M';
|
||||
}
|
||||
@@ -2346,16 +2233,14 @@ class binary_writer
|
||||
@brief checks whether a JSON number fits into @a TargetType
|
||||
@param[in] el a JSON number of either the signed or unsigned integer kind
|
||||
@return whether @a el's value can be represented by @a TargetType without
|
||||
wrapping, regardless of which of the two kinds it is stored as;
|
||||
false for a value that does not even fit the 64-bit type it is
|
||||
read as (possible for number types wider than 64 bits)
|
||||
wrapping, regardless of which of the two kinds it is stored as
|
||||
*/
|
||||
template<typename TargetType>
|
||||
static bool bjdata_ndarray_value_in_range(const BasicJsonType& el)
|
||||
{
|
||||
return el.is_number_unsigned()
|
||||
? integer_fits_max<std::uint64_t>(el.m_data.m_value.number_unsigned) && value_in_range_of<TargetType>(el.template get<std::uint64_t>())
|
||||
: integer_fits<std::int64_t>(el.m_data.m_value.number_integer) && value_in_range_of<TargetType>(el.template get<std::int64_t>());
|
||||
? value_in_range_of<TargetType>(el.template get<std::uint64_t>())
|
||||
: value_in_range_of<TargetType>(el.template get<std::int64_t>());
|
||||
}
|
||||
|
||||
/*!
|
||||
@@ -2587,11 +2472,10 @@ class binary_writer
|
||||
for (const auto& el : dims)
|
||||
{
|
||||
// a dimension is read as an unsigned value below, so anything that
|
||||
// is not a non-negative integer in the range of std::uint64_t is
|
||||
// rejected: a non-integer entry would pun unrelated bytes as the
|
||||
// dimension, and a negative or wider one would wrap into a
|
||||
// nonsensical length
|
||||
if (!el.is_number_integer() || !bjdata_ndarray_value_in_range<std::uint64_t>(el))
|
||||
// is not a non-negative integer is rejected: a non-integer entry
|
||||
// would pun unrelated bytes as the dimension, and a negative one
|
||||
// would wrap into a nonsensical length
|
||||
if (!el.is_number_integer() || (!el.is_number_unsigned() && el.template get<std::int64_t>() < 0))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
@@ -2705,9 +2589,9 @@ class binary_writer
|
||||
|
||||
case value_t::number_unsigned:
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(!integer_fits_max<std::int64_t>(j.m_data.m_value.number_unsigned)))
|
||||
if (j.m_data.m_value.number_unsigned > static_cast<typename BasicJsonType::number_unsigned_t>((std::numeric_limits<std::int64_t>::max)()))
|
||||
{
|
||||
throw_integer_out_of_range(j, "BON8", "int64");
|
||||
JSON_THROW(out_of_range::create(407, concat("integer number ", std::to_string(j.m_data.m_value.number_unsigned), " cannot be represented by BON8 as it does not fit int64"), &j));
|
||||
}
|
||||
write_bon8_integer(static_cast<std::int64_t>(j.m_data.m_value.number_unsigned));
|
||||
string_open = false;
|
||||
@@ -2716,10 +2600,6 @@ class binary_writer
|
||||
|
||||
case value_t::number_integer:
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(!integer_fits<std::int64_t>(j.m_data.m_value.number_integer)))
|
||||
{
|
||||
throw_integer_out_of_range(j, "BON8", "int64");
|
||||
}
|
||||
write_bon8_integer(static_cast<std::int64_t>(j.m_data.m_value.number_integer));
|
||||
string_open = false;
|
||||
break;
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint8_t, uint32_t
|
||||
#include <string> // string, to_string
|
||||
#include <utility> // move
|
||||
|
||||
#include <nlohmann/detail/abi_macros.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
@@ -14,6 +14,7 @@
|
||||
#include <cstdint> // uint8_t
|
||||
#include <limits> // numeric_limits
|
||||
#include <string> // string
|
||||
#include <type_traits> // is_signed
|
||||
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#if JSON_HAS_THREE_WAY_COMPARISON
|
||||
@@ -146,16 +147,13 @@ FloatType compare_integer_with_float(const IntegerType i, const FloatType f) noe
|
||||
|
||||
// values of IntegerType lie in [-bound, bound) when signed and in
|
||||
// [0, bound) when unsigned; digits excludes the sign bit, so bound is a
|
||||
// power of two that the float represents exactly; the signedness comes
|
||||
// from numeric_limits as well, because std::is_signed is false for class
|
||||
// types such as 128-bit or multiprecision integers
|
||||
using limits = std::numeric_limits<IntegerType>;
|
||||
const FloatType bound = std::ldexp(static_cast<FloatType>(1), limits::digits);
|
||||
// power of two that the float represents exactly
|
||||
const FloatType bound = std::ldexp(static_cast<FloatType>(1), std::numeric_limits<IntegerType>::digits);
|
||||
if (f >= bound)
|
||||
{
|
||||
return ordered(-1);
|
||||
}
|
||||
if (limits::is_signed ? (f < -bound) : (f < static_cast<FloatType>(0)))
|
||||
if (std::is_signed<IntegerType>::value ? (f < -bound) : (f < static_cast<FloatType>(0)))
|
||||
{
|
||||
return ordered(1);
|
||||
}
|
||||
|
||||
@@ -149,7 +149,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
friend class ::nlohmann::detail::iter_impl;
|
||||
template<typename BasicJsonType, typename CharType, typename OutputSinkType>
|
||||
friend class ::nlohmann::detail::binary_writer;
|
||||
template<typename BasicJsonType, typename InputType, typename SAX>
|
||||
template<typename BasicJsonType, typename InputType, typename SAX, bool AllowRecovery>
|
||||
friend class ::nlohmann::detail::binary_reader;
|
||||
template<typename BasicJsonType, typename InputAdapterType>
|
||||
friend class ::nlohmann::detail::json_sax_dom_parser;
|
||||
@@ -5689,7 +5689,7 @@ public:
|
||||
auto ia = detail::input_adapter(std::forward<InputType>(i));
|
||||
return format == input_format_t::json
|
||||
? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict)
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX>(std::move(ia), format).sax_parse(sax, strict, tag_handler);
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX, true>(std::move(ia), format).sax_parse(sax, strict, tag_handler);
|
||||
}
|
||||
|
||||
/// @brief generate SAX events (iterator pair, or iterator+sentinel pair for C++20 ranges support)
|
||||
@@ -5707,7 +5707,7 @@ public:
|
||||
auto ia = detail::input_adapter(std::move(first), std::move(last));
|
||||
return format == input_format_t::json
|
||||
? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict)
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX>(std::move(ia), format).sax_parse(sax, strict, tag_handler);
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX, true>(std::move(ia), format).sax_parse(sax, strict, tag_handler);
|
||||
}
|
||||
|
||||
/// @brief generate SAX events
|
||||
@@ -5746,7 +5746,7 @@ public:
|
||||
// NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg)
|
||||
? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict)
|
||||
// NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg)
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX>(std::move(ia), format).sax_parse(sax, strict, tag_handler);
|
||||
: detail::binary_reader<basic_json, decltype(ia), SAX, true>(std::move(ia), format).sax_parse(sax, strict, tag_handler);
|
||||
}
|
||||
#endif
|
||||
#if defined(__clang__)
|
||||
|
||||
+2475
-303
File diff suppressed because it is too large.
Load diff
@@ -47,6 +47,7 @@ inline namespace json_literals
|
||||
namespace detail
|
||||
{
|
||||
using NLOHMANN_JSON_NAMESPACE::detail::json_sax_dom_callback_parser;
|
||||
using NLOHMANN_JSON_NAMESPACE::detail::json_sax_dom_parser;
|
||||
using NLOHMANN_JSON_NAMESPACE::detail::unknown_size;
|
||||
} // namespace detail
|
||||
|
||||
|
||||
@@ -47,6 +47,10 @@ dumps is stable under exactly the same values that break operator==.
|
||||
The unit tests run the same checks on a fixed corpus (see the "BJData round-trip
|
||||
invariants" test case), so keep both in sync.
|
||||
|
||||
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that reading ends, and that it
|
||||
reports an error exactly when from_bjdata() fails (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -59,6 +63,8 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// compares dumps rather than values, because NaN != NaN; keep writes strings
|
||||
@@ -78,6 +84,9 @@ static bool is_value_stable(const json& lhs, const json& rhs)
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// recover from all errors, reading from memory and from a stream
|
||||
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bjdata).errors == 0;
|
||||
|
||||
std::vector<uint8_t> const vec1(data, data + size);
|
||||
|
||||
// step 0: parse input without exceptions; a parse error must then be
|
||||
@@ -110,6 +119,9 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
// without exceptions, the same input must give the same value
|
||||
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
|
||||
|
||||
// the recovering parser must not have reported an error either
|
||||
assert(recovered_without_errors);
|
||||
|
||||
try
|
||||
{
|
||||
// step 2.1: round trip without adding size annotations to container types
|
||||
@@ -143,6 +155,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// parse errors are ok, because input may be random bytes
|
||||
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
|
||||
assert(parsed || !recovered_without_errors);
|
||||
}
|
||||
catch (const json::type_error&)
|
||||
{
|
||||
@@ -153,6 +166,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// out of range errors may happen if provided sizes are excessive
|
||||
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
|
||||
assert(parsed || !recovered_without_errors);
|
||||
}
|
||||
|
||||
// return 0 - non-zero return values are reserved for future use
|
||||
|
||||
@@ -21,6 +21,10 @@ It also checks that reading the data from a stream, which reads strings byte by
|
||||
byte, gives the same value or error as reading it from contiguous memory, which
|
||||
copies strings in bulk.
|
||||
|
||||
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that reading ends, and that it
|
||||
reports an error exactly when from_bon8() fails (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -34,6 +38,8 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// compares dumps rather than values, because NaN != NaN; keep writes strings
|
||||
@@ -64,6 +70,9 @@ std::string read_bon8(InputType&& input)
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// recover from all errors, reading from memory and from a stream
|
||||
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bon8).errors == 0;
|
||||
|
||||
// contiguous and stream input must be read alike
|
||||
{
|
||||
std::istringstream stream(std::string(reinterpret_cast<const char*>(data), size));
|
||||
@@ -102,6 +111,9 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
// without exceptions, the same input must give the same value
|
||||
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
|
||||
|
||||
// the recovering parser must not have reported an error either
|
||||
assert(recovered_without_errors);
|
||||
|
||||
try
|
||||
{
|
||||
// step 2: round trip
|
||||
@@ -123,6 +135,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// parse errors are ok, because input may be random bytes
|
||||
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
|
||||
assert(parsed || !recovered_without_errors);
|
||||
}
|
||||
catch (const json::type_error&)
|
||||
{
|
||||
@@ -133,6 +146,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// out of range errors may happen if provided sizes are excessive
|
||||
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
|
||||
assert(parsed || !recovered_without_errors);
|
||||
}
|
||||
|
||||
// return 0 - non-zero return values are reserved for future use
|
||||
|
||||
@@ -17,6 +17,10 @@ array data, it performs the following steps:
|
||||
- j2 = from_bson(vec)
|
||||
- assert(to_bson(j2) == vec)
|
||||
|
||||
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that reading ends, and that it
|
||||
reports an error exactly when from_bson() fails (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -29,6 +33,8 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// compares dumps rather than values, because NaN != NaN; keep writes strings
|
||||
@@ -41,6 +47,9 @@ static bool same_value(const json& lhs, const json& rhs)
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// recover from all errors, reading from memory and from a stream
|
||||
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::bson).errors == 0;
|
||||
|
||||
std::vector<uint8_t> const vec1(data, data + size);
|
||||
|
||||
// step 0: parse input without exceptions; a parse error must then be
|
||||
@@ -73,6 +82,9 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
// without exceptions, the same input must give the same value
|
||||
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
|
||||
|
||||
// the recovering parser must not have reported an error either
|
||||
assert(recovered_without_errors);
|
||||
|
||||
try
|
||||
{
|
||||
// step 2: round trip
|
||||
@@ -94,6 +106,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// parse errors are ok, because input may be random bytes
|
||||
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
|
||||
assert(parsed || !recovered_without_errors);
|
||||
}
|
||||
catch (const json::type_error&)
|
||||
{
|
||||
@@ -104,6 +117,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// out of range errors can occur during parsing, too
|
||||
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
|
||||
assert(parsed || !recovered_without_errors);
|
||||
}
|
||||
|
||||
// return 0 - non-zero return values are reserved for future use
|
||||
|
||||
@@ -17,6 +17,10 @@ array data, it performs the following steps:
|
||||
- j2 = from_cbor(vec)
|
||||
- assert(to_cbor(j2) == vec)
|
||||
|
||||
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that reading ends, and that it
|
||||
reports an error exactly when from_cbor() fails (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -29,6 +33,8 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// compares dumps rather than values, because NaN != NaN; keep writes strings
|
||||
@@ -41,6 +47,9 @@ static bool same_value(const json& lhs, const json& rhs)
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// recover from all errors, reading from memory and from a stream
|
||||
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::cbor).errors == 0;
|
||||
|
||||
std::vector<uint8_t> const vec1(data, data + size);
|
||||
|
||||
// step 0: parse input without exceptions; a parse error must then be
|
||||
@@ -73,6 +82,9 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
// without exceptions, the same input must give the same value
|
||||
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
|
||||
|
||||
// the recovering parser must not have reported an error either
|
||||
assert(recovered_without_errors);
|
||||
|
||||
try
|
||||
{
|
||||
// step 2: round trip
|
||||
@@ -94,6 +106,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// parse errors are ok, because input may be random bytes
|
||||
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
|
||||
assert(parsed || !recovered_without_errors);
|
||||
}
|
||||
catch (const json::type_error&)
|
||||
{
|
||||
@@ -104,6 +117,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// out of range errors can occur during parsing, too
|
||||
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
|
||||
assert(parsed || !recovered_without_errors);
|
||||
}
|
||||
|
||||
// return 0 - non-zero return values are reserved for future use
|
||||
|
||||
@@ -18,6 +18,10 @@ array data, it performs the following steps:
|
||||
- s2 = serialize(j2)
|
||||
- assert(s1 == s2)
|
||||
|
||||
Furthermore, it parses data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that parsing ends, and that valid
|
||||
input is parsed without errors (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -30,6 +34,8 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// compares dumps rather than values, because NaN != NaN; keep writes strings
|
||||
@@ -42,6 +48,13 @@ static bool same_value(const json& lhs, const json& rhs)
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// recover from all errors, reading from memory and from a stream
|
||||
{
|
||||
const auto checker = check_recovering_parse(data, size, json::input_format_t::json);
|
||||
assert(checker.events <= (4 * size) + 4);
|
||||
assert((checker.errors == 0) == json::accept(data, data + size));
|
||||
}
|
||||
|
||||
// step 0: parse input without exceptions; a parse error must then be
|
||||
// reported as a discarded value, never thrown
|
||||
json j_noexcept;
|
||||
|
||||
@@ -17,6 +17,10 @@ array data, it performs the following steps:
|
||||
- j2 = from_msgpack(vec)
|
||||
- assert(to_msgpack(j2) == vec)
|
||||
|
||||
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that reading ends, and that it
|
||||
reports an error exactly when from_msgpack() fails (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -29,6 +33,8 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// compares dumps rather than values, because NaN != NaN; keep writes strings
|
||||
@@ -41,6 +47,9 @@ static bool same_value(const json& lhs, const json& rhs)
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// recover from all errors, reading from memory and from a stream
|
||||
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::msgpack).errors == 0;
|
||||
|
||||
std::vector<uint8_t> const vec1(data, data + size);
|
||||
|
||||
// step 0: parse input without exceptions; a parse error must then be
|
||||
@@ -73,6 +82,9 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
// without exceptions, the same input must give the same value
|
||||
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
|
||||
|
||||
// the recovering parser must not have reported an error either
|
||||
assert(recovered_without_errors);
|
||||
|
||||
try
|
||||
{
|
||||
// step 2: round trip
|
||||
@@ -94,6 +106,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// parse errors are ok, because input may be random bytes
|
||||
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
|
||||
assert(parsed || !recovered_without_errors);
|
||||
}
|
||||
catch (const json::type_error&)
|
||||
{
|
||||
@@ -104,6 +117,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// out of range errors may happen if provided sizes are excessive
|
||||
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
|
||||
assert(parsed || !recovered_without_errors);
|
||||
}
|
||||
|
||||
// return 0 - non-zero return values are reserved for future use
|
||||
|
||||
@@ -26,6 +26,10 @@ array data, it performs the following steps:
|
||||
The unit tests run the same checks on a fixed corpus (see the "UBJSON round-trip
|
||||
invariants" test case), so keep both in sync.
|
||||
|
||||
Furthermore, it reads data with a SAX parser that recovers from every error
|
||||
and checks that the events are balanced, that reading ends, and that it
|
||||
reports an error exactly when from_ubjson() fails (see #3989).
|
||||
|
||||
The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer
|
||||
drivers.
|
||||
*/
|
||||
@@ -38,6 +42,8 @@ drivers.
|
||||
#error "the fuzzer drivers must be built without NDEBUG"
|
||||
#endif
|
||||
|
||||
#include "fuzzer-recovering_checker.hpp"
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// compares dumps rather than values, because NaN != NaN; keep writes strings
|
||||
@@ -50,6 +56,9 @@ static bool same_value(const json& lhs, const json& rhs)
|
||||
// see http://llvm.org/docs/LibFuzzer.html
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// recover from all errors, reading from memory and from a stream
|
||||
const bool recovered_without_errors = check_recovering_parse(data, size, json::input_format_t::ubjson).errors == 0;
|
||||
|
||||
std::vector<uint8_t> const vec1(data, data + size);
|
||||
|
||||
// step 0: parse input without exceptions; a parse error must then be
|
||||
@@ -82,6 +91,9 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
// without exceptions, the same input must give the same value
|
||||
assert(!noexcept_threw && !j_noexcept.is_discarded() && same_value(j_noexcept, j1));
|
||||
|
||||
// the recovering parser must not have reported an error either
|
||||
assert(recovered_without_errors);
|
||||
|
||||
try
|
||||
{
|
||||
// step 2.1: round trip without adding size annotations to container types
|
||||
@@ -113,6 +125,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// parse errors are ok, because input may be random bytes
|
||||
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
|
||||
assert(parsed || !recovered_without_errors);
|
||||
}
|
||||
catch (const json::type_error&)
|
||||
{
|
||||
@@ -123,6 +136,7 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size)
|
||||
{
|
||||
// out of range errors may happen if provided sizes are excessive
|
||||
assert(parsed || noexcept_threw || j_noexcept.is_discarded());
|
||||
assert(parsed || !recovered_without_errors);
|
||||
}
|
||||
|
||||
// return 0 - non-zero return values are reserved for future use
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cassert>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
namespace
|
||||
{
|
||||
// a SAX parser that recovers from every error and checks that the events are
|
||||
// balanced and that every key is followed by exactly one value
|
||||
class recovering_checker : public nlohmann::json_sax<nlohmann::json>
|
||||
{
|
||||
public:
|
||||
bool null() override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool boolean(bool /*val*/) override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool number_integer(number_integer_t /*val*/) override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool number_unsigned(number_unsigned_t /*val*/) override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool number_float(number_float_t /*val*/, const string_t& /*s*/) override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool string(string_t& /*val*/) override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool binary(binary_t& /*val*/) override
|
||||
{
|
||||
return value();
|
||||
}
|
||||
|
||||
bool start_object(std::size_t /*elements*/) override
|
||||
{
|
||||
value();
|
||||
stack.push_back('o');
|
||||
return true;
|
||||
}
|
||||
|
||||
bool key(string_t& /*val*/) override
|
||||
{
|
||||
++events;
|
||||
assert(!stack.empty() && stack.back() == 'o');
|
||||
stack.back() = 'v';
|
||||
return true;
|
||||
}
|
||||
|
||||
bool end_object() override
|
||||
{
|
||||
++events;
|
||||
assert(!stack.empty() && stack.back() == 'o');
|
||||
stack.pop_back();
|
||||
return true;
|
||||
}
|
||||
|
||||
bool start_array(std::size_t /*elements*/) override
|
||||
{
|
||||
value();
|
||||
stack.push_back('a');
|
||||
return true;
|
||||
}
|
||||
|
||||
bool end_array() override
|
||||
{
|
||||
++events;
|
||||
assert(!stack.empty() && stack.back() == 'a');
|
||||
stack.pop_back();
|
||||
return true;
|
||||
}
|
||||
|
||||
bool parse_error(std::size_t /*position*/, const std::string& /*last_token*/, const nlohmann::detail::exception& /*ex*/) override
|
||||
{
|
||||
++errors;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool complete() const
|
||||
{
|
||||
return stack.empty();
|
||||
}
|
||||
|
||||
std::size_t events = 0;
|
||||
std::size_t errors = 0;
|
||||
|
||||
private:
|
||||
bool value()
|
||||
{
|
||||
++events;
|
||||
if (!stack.empty())
|
||||
{
|
||||
// an array element, or the value of a key
|
||||
assert(stack.back() != 'o');
|
||||
if (stack.back() == 'v')
|
||||
{
|
||||
stack.back() = 'o';
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// 'a' for an array, 'o' for an object that expects a key, 'v' for an
|
||||
// object that expects the value of a key
|
||||
std::vector<char> stack {}; // NOLINT(readability-redundant-member-init)
|
||||
};
|
||||
|
||||
/// parses @a data with a recovering_checker from memory and from a stream,
|
||||
/// checks that both see the same, that the events are balanced, and that the
|
||||
/// number of errors is bounded, and returns the checker (see #3989)
|
||||
inline recovering_checker check_recovering_parse(const std::uint8_t* data, const std::size_t size, const nlohmann::json::input_format_t format)
|
||||
{
|
||||
recovering_checker checker;
|
||||
const bool ok = nlohmann::json::sax_parse(data, data + size, &checker, format);
|
||||
assert(checker.complete());
|
||||
assert(checker.errors <= size + 1);
|
||||
assert(ok == (checker.errors == 0));
|
||||
|
||||
std::istringstream stream(std::string(reinterpret_cast<const char*>(data), size));
|
||||
recovering_checker stream_checker;
|
||||
assert(nlohmann::json::sax_parse(stream, &stream_checker, format) == ok);
|
||||
assert(stream_checker.complete());
|
||||
assert(stream_checker.events == checker.events);
|
||||
assert(stream_checker.errors == checker.errors);
|
||||
|
||||
return checker;
|
||||
}
|
||||
} // namespace
|
||||
@@ -415,6 +415,46 @@ TEST_CASE("alternative string type")
|
||||
CHECK(j2.dump() == R"({"/foo/0":"bar","/foo/1":"baz"})");
|
||||
}
|
||||
|
||||
SECTION("error recovery")
|
||||
{
|
||||
// a SAX parser that recovers from every error (see #3989)
|
||||
struct recovering_parser : nlohmann::detail::json_sax_dom_parser<alt_json>
|
||||
{
|
||||
explicit recovering_parser(alt_json& j)
|
||||
: nlohmann::detail::json_sax_dom_parser<alt_json>(j, false)
|
||||
{}
|
||||
|
||||
// sax_parse() calls the SAX parser's own parse_error(), so hiding
|
||||
// the one of the base class is what recovering takes
|
||||
// NOLINTNEXTLINE(bugprone-derived-method-shadowing-base-method)
|
||||
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const nlohmann::detail::exception& /*unused*/)
|
||||
{
|
||||
++errors;
|
||||
return true;
|
||||
}
|
||||
|
||||
std::size_t errors = 0;
|
||||
};
|
||||
|
||||
alt_json j;
|
||||
recovering_parser sax(j);
|
||||
// not inside CHECK(): MSVC reads the escape in a stringized raw string
|
||||
const std::string input = R"([1., "a\qb", tru, {"k" 2}])";
|
||||
CHECK(!alt_json::sax_parse(input, &sax));
|
||||
CHECK(sax.errors == 4);
|
||||
CHECK(j.dump() == R"([1,"aqb",null,{"k":2}])");
|
||||
|
||||
// a UBJSON high-precision number, a CBOR key that is not a string
|
||||
alt_json u;
|
||||
recovering_parser ubjson_sax(u);
|
||||
CHECK(!alt_json::sax_parse(std::vector<std::uint8_t> {'[', 'H', 'i', 2, '1', '.', ']'}, &ubjson_sax, alt_json::input_format_t::ubjson));
|
||||
CHECK(u.dump() == "[1]");
|
||||
alt_json c;
|
||||
recovering_parser cbor_sax(c);
|
||||
CHECK(!alt_json::sax_parse(std::vector<std::uint8_t> {0xA2, 0x01, 0x02, 0x61, 'a', 0x03}, &cbor_sax, alt_json::input_format_t::cbor));
|
||||
CHECK(c.dump() == R"({"a":3})");
|
||||
}
|
||||
|
||||
SECTION("conversion between basic_json specializations (#2649)")
|
||||
{
|
||||
// explicit conversions are always possible
|
||||
|
||||
@@ -143,11 +143,13 @@ class SaxEventLogger
|
||||
{
|
||||
errored = true;
|
||||
events.push_back("parse_error(" + std::to_string(position) + ")");
|
||||
return false;
|
||||
return recover;
|
||||
}
|
||||
|
||||
std::vector<std::string> events {}; // NOLINT(readability-redundant-member-init)
|
||||
bool errored = false;
|
||||
/// whether parse_error() asks the parser to recover from the error (see #3989)
|
||||
bool recover = false;
|
||||
};
|
||||
|
||||
class SaxCountdown : public nlohmann::json::json_sax_t
|
||||
@@ -3026,3 +3028,583 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX")
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
namespace
|
||||
{
|
||||
/// builds a value like json::parse(), but asks the parser to recover from
|
||||
/// errors (see #3989), and checks that the events it receives are balanced
|
||||
class RecoveringDomParser
|
||||
{
|
||||
public:
|
||||
explicit RecoveringDomParser(json& j, std::size_t max_errors_ = static_cast<std::size_t>(-1))
|
||||
: dom(j, false)
|
||||
, max_errors(max_errors_)
|
||||
{}
|
||||
|
||||
bool null()
|
||||
{
|
||||
value();
|
||||
return dom.null();
|
||||
}
|
||||
|
||||
bool boolean(bool val)
|
||||
{
|
||||
value();
|
||||
return dom.boolean(val);
|
||||
}
|
||||
|
||||
bool number_integer(json::number_integer_t val)
|
||||
{
|
||||
value();
|
||||
return dom.number_integer(val);
|
||||
}
|
||||
|
||||
bool number_unsigned(json::number_unsigned_t val)
|
||||
{
|
||||
value();
|
||||
return dom.number_unsigned(val);
|
||||
}
|
||||
|
||||
bool number_float(json::number_float_t val, const std::string& s)
|
||||
{
|
||||
value();
|
||||
return dom.number_float(val, s);
|
||||
}
|
||||
|
||||
bool string(std::string& val)
|
||||
{
|
||||
value();
|
||||
return dom.string(val);
|
||||
}
|
||||
|
||||
bool binary(json::binary_t& val)
|
||||
{
|
||||
value();
|
||||
return dom.binary(val);
|
||||
}
|
||||
|
||||
bool start_object(std::size_t elements)
|
||||
{
|
||||
value();
|
||||
stack.push_back('o');
|
||||
return dom.start_object(elements);
|
||||
}
|
||||
|
||||
bool key(std::string& val)
|
||||
{
|
||||
++events;
|
||||
if (stack.empty() || stack.back() != 'o')
|
||||
{
|
||||
well_formed = false;
|
||||
return false;
|
||||
}
|
||||
stack.back() = 'v';
|
||||
return dom.key(val);
|
||||
}
|
||||
|
||||
bool end_object()
|
||||
{
|
||||
++events;
|
||||
if (stack.empty() || stack.back() != 'o')
|
||||
{
|
||||
well_formed = false;
|
||||
return false;
|
||||
}
|
||||
stack.pop_back();
|
||||
return dom.end_object();
|
||||
}
|
||||
|
||||
bool start_array(std::size_t elements)
|
||||
{
|
||||
value();
|
||||
stack.push_back('a');
|
||||
return dom.start_array(elements);
|
||||
}
|
||||
|
||||
bool end_array()
|
||||
{
|
||||
++events;
|
||||
if (stack.empty() || stack.back() != 'a')
|
||||
{
|
||||
well_formed = false;
|
||||
return false;
|
||||
}
|
||||
stack.pop_back();
|
||||
return dom.end_array();
|
||||
}
|
||||
|
||||
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& ex)
|
||||
{
|
||||
errors.emplace_back(ex.what());
|
||||
return errors.size() < max_errors;
|
||||
}
|
||||
|
||||
/// whether the events were balanced and every key was followed by a value
|
||||
bool balanced() const
|
||||
{
|
||||
return well_formed && stack.empty();
|
||||
}
|
||||
|
||||
/// builds the value
|
||||
nlohmann::detail::json_sax_dom_parser<json> dom;
|
||||
std::vector<std::string> errors {}; // NOLINT(readability-redundant-member-init)
|
||||
std::size_t events = 0;
|
||||
/// the open containers: 'a' for an array, 'o' for an object that expects
|
||||
/// a key, 'v' for an object that expects the value of a key
|
||||
std::vector<char> stack {}; // NOLINT(readability-redundant-member-init)
|
||||
bool well_formed = true;
|
||||
std::size_t max_errors;
|
||||
|
||||
private:
|
||||
/// a value is passed: it is an array element, or the value of a key
|
||||
void value()
|
||||
{
|
||||
++events;
|
||||
if (!stack.empty())
|
||||
{
|
||||
if (stack.back() == 'v')
|
||||
{
|
||||
stack.back() = 'o';
|
||||
}
|
||||
else if (stack.back() == 'o')
|
||||
{
|
||||
// a value without a key
|
||||
well_formed = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
struct RecoveryResult
|
||||
{
|
||||
json value;
|
||||
std::vector<std::string> errors;
|
||||
std::size_t events;
|
||||
bool ok;
|
||||
bool balanced;
|
||||
};
|
||||
|
||||
template<typename InputType>
|
||||
RecoveryResult parse_recovering(InputType&& input, const bool strict = true,
|
||||
const bool ignore_comments = false, const bool ignore_trailing_commas = false)
|
||||
{
|
||||
json j;
|
||||
RecoveringDomParser sax(j);
|
||||
const bool ok = json::sax_parse(std::forward<InputType>(input), &sax, json::input_format_t::json,
|
||||
strict, ignore_comments, ignore_trailing_commas);
|
||||
return {j, sax.errors, sax.events, ok, sax.balanced()};
|
||||
}
|
||||
|
||||
/// stops after a number of events, but recovers from errors
|
||||
class RecoveringCountdown : public SaxCountdown
|
||||
{
|
||||
public:
|
||||
using SaxCountdown::SaxCountdown;
|
||||
|
||||
bool parse_error(std::size_t /*position*/, const std::string& /*last_token*/, const json::exception& /*ex*/) override
|
||||
{
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
/// a repaired input: the value it is repaired to, and the number of errors
|
||||
struct Repair
|
||||
{
|
||||
const char* input;
|
||||
const char* expected;
|
||||
std::size_t errors;
|
||||
};
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("parser error recovery (#3989)")
|
||||
{
|
||||
SECTION("repairs")
|
||||
{
|
||||
const std::vector<Repair> repairs =
|
||||
{
|
||||
// a missing separator is inserted
|
||||
{"[1 2]", "[1,2]", 1},
|
||||
{R"({"a":1 "b":2})", R"({"a":1,"b":2})", 1},
|
||||
{R"({"a" 1})", R"({"a":1})", 1},
|
||||
{"[1 tru 2]", "[1,null,2]", 2},
|
||||
{R"({"a" "b": 1})", R"({"a":"b"})", 2},
|
||||
|
||||
// a missing value is null in an object; in an array, a ',' stands
|
||||
// for null, while an array that ends there just ends
|
||||
{R"({"a":})", R"({"a":null})", 1},
|
||||
{R"({"a"})", R"({"a":null})", 1},
|
||||
{R"({"a","b":1})", R"({"a":null,"b":1})", 1},
|
||||
{"[1,,2]", "[1,null,2]", 1},
|
||||
{"[,1]", "[null,1]", 1},
|
||||
{"[1,]", "[1]", 1},
|
||||
{"[1,2,3,]", "[1,2,3]", 1},
|
||||
{R"({"a":1,})", R"({"a":1})", 1},
|
||||
|
||||
// a broken string keeps what can be read
|
||||
{R"(["a\qb"])", R"(["aqb"])", 1},
|
||||
{R"({"na\me":1})", R"({"name":1})", 1},
|
||||
{"[\"\xFF\"]", R"(["\uFFFD"])", 1},
|
||||
{"[\"a\xC3(\"]", R"(["a\uFFFD("])", 1},
|
||||
{"[\"\xE2\x82\"]", R"(["\uFFFD"])", 1},
|
||||
{"[\"\xC3\\\\\", 1]", R"(["\uFFFD\\",1])", 1},
|
||||
{R"(["\u12"])", R"(["\uFFFD"])", 1},
|
||||
{R"(["\u12G4"])", R"(["\uFFFDG4"])", 1},
|
||||
{R"(["\uDC00x"])", R"(["\uFFFDx"])", 1},
|
||||
{R"(["\uD800x"])", R"(["\uFFFDx"])", 1},
|
||||
{R"(["\uD800\u0041"])", R"(["\uFFFDA"])", 1},
|
||||
{R"(["\uD800\uD800\uDC00"])", R"(["\uFFFD\uD800\uDC00"])", 1},
|
||||
{R"(["\uD800\uD800\uD800x"])", R"(["\uFFFD\uFFFD\uFFFDx"])", 1},
|
||||
{
|
||||
R"(["\uD800\"x", 1])", R"(["\uFFFD\"x",1])", 1
|
||||
},
|
||||
{R"(["\uD800\q"])", R"(["\uFFFDq"])", 1},
|
||||
{"[\"a\tb\"]", R"(["a\tb"])", 1},
|
||||
{R"(["a\qb\u0041\x"])", R"(["aqbAx"])", 1},
|
||||
|
||||
// a broken number keeps its longest valid prefix
|
||||
{"[1.]", "[1]", 1},
|
||||
{"[-2.]", "[-2]", 1},
|
||||
{"[1.5e]", "[1.5]", 1},
|
||||
{"[1e+]", "[1]", 1},
|
||||
{"[1.x2, 3]", "[1,3]", 1},
|
||||
|
||||
// what cannot be read at all is null
|
||||
{"[1,NaN,3]", "[1,null,3]", 1},
|
||||
{"[tru]", "[null]", 1},
|
||||
{"[-]", "[null]", 1},
|
||||
{R"({"a":Infinity})", R"({"a":null})", 1},
|
||||
|
||||
// a stray token is dropped
|
||||
{"[:1]", "[1]", 1},
|
||||
{R"(["a":1])", R"(["a",1])", 1},
|
||||
{R"({"a"::1})", R"({"a":1})", 1},
|
||||
|
||||
// a member that cannot be read is skipped
|
||||
{R"({1:2,"b":3})", R"({"b":3})", 1},
|
||||
{R"({"a":1 2})", R"({"a":1})", 1},
|
||||
{R"({,"a":1})", R"({"a":1})", 1},
|
||||
{R"({"a":1,,"b":2})", R"({"a":1,"b":2})", 1},
|
||||
{"{a:1}", "{}", 1},
|
||||
{R"({"a":1 [1,{"b":2}], "c":3})", R"({"a":1,"c":3})", 1},
|
||||
{R"([{1}, "a"])", R"([{},"a"])", 1},
|
||||
|
||||
// a wrong closing bracket closes the innermost container
|
||||
{R"({"a":[1,2}, "b":3})", R"({"a":[1,2],"b":3})", 1},
|
||||
{R"([{"a":1], 2])", R"([{"a":1},2])", 1},
|
||||
{"{]", "{}", 1},
|
||||
{"[}", "[]", 1},
|
||||
|
||||
// the end of the input closes all containers
|
||||
{R"({"a":[1,2)", R"({"a":[1,2]})", 1},
|
||||
{"[", "[]", 1},
|
||||
{"{", "{}", 1},
|
||||
{R"({"a")", R"({"a":null})", 1},
|
||||
{R"({"a":)", R"({"a":null})", 1},
|
||||
{"[1,", "[1]", 1},
|
||||
{"[[[1", "[[[1]]]", 1},
|
||||
{
|
||||
R"(["abc)", R"(["abc"])", 2
|
||||
},
|
||||
{"[1,tr", "[1,null]", 2},
|
||||
{"\"abc", "\"abc\"", 1},
|
||||
{"[\"ab\ncd\"]", R"(["ab",null,"]"])", 4},
|
||||
|
||||
// what comes before the top-level value is skipped
|
||||
{")]}'\n{\"a\":1}", R"({"a":1})", 1},
|
||||
{R"(data: {"a":1})", R"({"a":1})", 1},
|
||||
{"\xEF\xBB[1]", "[1]", 1},
|
||||
|
||||
// what comes after it is an error that ends parsing
|
||||
{R"({"a":1}})", R"({"a":1})", 1},
|
||||
{"[1}]", "[1]", 2},
|
||||
{"[1] [2]", "[1]", 1},
|
||||
};
|
||||
|
||||
for (const auto& repair : repairs)
|
||||
{
|
||||
CAPTURE(repair.input)
|
||||
const auto result = parse_recovering(std::string(repair.input));
|
||||
CHECK(!result.ok);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.value == json::parse(repair.expected));
|
||||
CHECK(result.errors.size() == repair.errors);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("number overflow")
|
||||
{
|
||||
const auto result = parse_recovering(std::string("[1e999,-1e999]"));
|
||||
CHECK(!result.ok);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.errors.size() == 2);
|
||||
CHECK(result.errors[0] == "[json.exception.out_of_range.406] number overflow parsing '1e999'");
|
||||
REQUIRE(result.value.size() == 2);
|
||||
CHECK(result.value[0].is_number_float());
|
||||
CHECK(result.value[0].get<double>() == std::numeric_limits<double>::infinity());
|
||||
CHECK(result.value[1].get<double>() == -std::numeric_limits<double>::infinity());
|
||||
|
||||
// the SAX parser gets the number's text
|
||||
SaxEventLogger logger;
|
||||
logger.recover = true;
|
||||
CHECK(!json::sax_parse("1e999", &logger));
|
||||
CHECK(logger.events == std::vector<std::string>({"parse_error(5)", "number_float(1e999)"}));
|
||||
}
|
||||
|
||||
SECTION("nothing to recover")
|
||||
{
|
||||
for (const std::string s :
|
||||
{
|
||||
"", " ", "]", "tru", "NaN", ",:", "/* comment"
|
||||
})
|
||||
{
|
||||
CAPTURE(s)
|
||||
const auto result = parse_recovering(s, true, true);
|
||||
CHECK(!result.ok);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.events == 0);
|
||||
CHECK(result.value == nullptr);
|
||||
CHECK(result.errors.size() == 1);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("error messages")
|
||||
{
|
||||
// the first error is reported as without recovery
|
||||
for (const std::string s :
|
||||
{
|
||||
"[1 2]", R"({"a":1 "b":2})", R"({"a" 1})", R"({"a":})", "[1,]", "[1.]",
|
||||
R"(["a\qb"])", "[1e999]", "{1:2}", R"({"a":[1,2}})", "[1,", "[1] [2]", "{a:1}"
|
||||
})
|
||||
{
|
||||
CAPTURE(s)
|
||||
const auto result = parse_recovering(s);
|
||||
REQUIRE(!result.errors.empty());
|
||||
json _;
|
||||
CHECK_THROWS_WITH_STD_STR(_ = json::parse(s), result.errors.front());
|
||||
}
|
||||
|
||||
// the token of an error begins where the previous error was
|
||||
const auto result = parse_recovering(std::string("[tru, fals, nul]"));
|
||||
CHECK(result.errors == std::vector<std::string>(
|
||||
{
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: '[tru,'",
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 11: syntax error while parsing value - invalid literal; last read: ', fals,'",
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 16: syntax error while parsing value - invalid literal; last read: ', nul]'"
|
||||
}));
|
||||
CHECK(result.value == json::parse("[null,null,null]"));
|
||||
}
|
||||
|
||||
SECTION("events")
|
||||
{
|
||||
// see #4522
|
||||
SaxEventLogger logger;
|
||||
logger.recover = true;
|
||||
CHECK(!json::sax_parse(R"([{1}, "a"])", &logger));
|
||||
CHECK(logger.events == std::vector<std::string>(
|
||||
{
|
||||
"start_array()", "start_object()", "parse_error(3)", "end_object()", "string(a)", "end_array()"
|
||||
}));
|
||||
}
|
||||
|
||||
SECTION("options")
|
||||
{
|
||||
SECTION("strict")
|
||||
{
|
||||
const auto result = parse_recovering(std::string("[1 2] [3]"), false);
|
||||
CHECK(!result.ok);
|
||||
CHECK(result.value == json::parse("[1,2]"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
}
|
||||
|
||||
SECTION("ignore_trailing_commas")
|
||||
{
|
||||
for (const std::string s :
|
||||
{
|
||||
"[1,]", R"({"a":1,})", "[[1,],]"
|
||||
})
|
||||
{
|
||||
CAPTURE(s)
|
||||
const auto result = parse_recovering(s, true, false, true);
|
||||
CHECK(result.ok);
|
||||
CHECK(result.errors.empty());
|
||||
}
|
||||
|
||||
auto result = parse_recovering(std::string("[1,,]"), true, false, true);
|
||||
CHECK(result.value == json::parse("[1,null]"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
|
||||
result = parse_recovering(std::string(R"({"a":1,,})"), true, false, true);
|
||||
CHECK(result.value == json::parse(R"({"a":1})"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
}
|
||||
|
||||
SECTION("ignore_comments")
|
||||
{
|
||||
auto result = parse_recovering(std::string("[1 /* one */ 2]"), true, true);
|
||||
CHECK(result.value == json::parse("[1,2]"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
|
||||
// a comment that is not closed runs to the end of the input, which
|
||||
// is not reported again
|
||||
result = parse_recovering(std::string("[1, 2 /* unterminated"), true, true);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.value == json::parse("[1,2]"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
|
||||
// a '/' that does not begin a comment is garbage
|
||||
result = parse_recovering(std::string("[1, /x, 2]"), true, true);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.value == json::parse("[1,null,2]"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("null bytes")
|
||||
{
|
||||
// a null byte ends the input, unless JSON_STRICT_NUL_HANDLING is set
|
||||
const auto result = parse_recovering(std::string("[1,\0x", 5));
|
||||
CHECK(result.balanced);
|
||||
CHECK(!result.ok);
|
||||
#ifdef JSON_TEST_STRICT_NUL_HANDLING_ENABLED
|
||||
CHECK(result.value == json::parse("[1,null]"));
|
||||
#else
|
||||
CHECK(result.value == json::parse("[1]"));
|
||||
CHECK(result.errors.size() == 1);
|
||||
#endif
|
||||
|
||||
const auto in_string = parse_recovering(std::string("[\"a\0b\"]", 7));
|
||||
CHECK(in_string.balanced);
|
||||
#ifdef JSON_TEST_STRICT_NUL_HANDLING_ENABLED
|
||||
CHECK(in_string.value == json::array({std::string("a\0b", 3)}));
|
||||
#else
|
||||
CHECK(in_string.value == json::parse(R"(["a"])"));
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("the SAX parser stops recovering")
|
||||
{
|
||||
json j;
|
||||
RecoveringDomParser sax(j, 2);
|
||||
CHECK(!json::sax_parse("[1 2 3 4 5]", &sax));
|
||||
CHECK(sax.errors.size() == 2);
|
||||
|
||||
// an error at a delimiter that an invalid token consumed is reported
|
||||
// to the SAX parser, too
|
||||
json j2;
|
||||
RecoveringDomParser sax2(j2, 2);
|
||||
CHECK(!json::sax_parse("[tru}, 1]", &sax2));
|
||||
CHECK(sax2.errors.size() == 2);
|
||||
}
|
||||
|
||||
SECTION("an event stops parsing during a repair")
|
||||
{
|
||||
// start_object() and key() are passed, then null() for the missing
|
||||
// value returns false
|
||||
RecoveringCountdown countdown(2);
|
||||
CHECK(!json::sax_parse(R"({"a":})", &countdown));
|
||||
|
||||
// the end of the input: end_array() for the second array returns false
|
||||
RecoveringCountdown countdown2(4);
|
||||
CHECK(!json::sax_parse("[[1", &countdown2));
|
||||
}
|
||||
|
||||
SECTION("input adapters")
|
||||
{
|
||||
// the lexer reads contiguous and streaming input differently, and it
|
||||
// puts back a character that ended an invalid token
|
||||
for (const std::string s :
|
||||
{
|
||||
"[1 2]", "[tru}, 1]", R"({"a" "b\q", "c":[1.x, 2}})", "[\"\xFF\xC3(\", -, 1e+]", "{a:1,\"b\":2", ")]}' [1]"
|
||||
})
|
||||
{
|
||||
CAPTURE(s)
|
||||
const auto reference = parse_recovering(s);
|
||||
CHECK(reference.balanced);
|
||||
|
||||
const auto from_c_string = parse_recovering(s.c_str());
|
||||
CHECK(from_c_string.value == reference.value);
|
||||
CHECK(from_c_string.errors == reference.errors);
|
||||
|
||||
const std::list<char> l(s.begin(), s.end());
|
||||
json j;
|
||||
RecoveringDomParser sax(j);
|
||||
CHECK(!json::sax_parse(l.begin(), l.end(), &sax));
|
||||
CHECK(j == reference.value);
|
||||
CHECK(sax.errors == reference.errors);
|
||||
|
||||
std::istringstream ss(s);
|
||||
const auto from_stream = parse_recovering(ss);
|
||||
CHECK(from_stream.value == reference.value);
|
||||
CHECK(from_stream.errors == reference.errors);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("long runs of errors")
|
||||
{
|
||||
// no error may copy all the input read before it
|
||||
const auto closing = parse_recovering("[" + std::string(100000, '}'));
|
||||
CHECK(closing.balanced);
|
||||
CHECK(closing.value == json::array());
|
||||
|
||||
const auto garbage = parse_recovering("[" + std::string(100000, 'x') + "]");
|
||||
CHECK(garbage.balanced);
|
||||
CHECK(garbage.errors.size() == 1);
|
||||
|
||||
const auto commas = parse_recovering("{" + std::string(100000, ',') + "}");
|
||||
CHECK(commas.balanced);
|
||||
CHECK(commas.value == json::object());
|
||||
}
|
||||
|
||||
SECTION("mutations of valid input")
|
||||
{
|
||||
// whatever the input, the events are balanced, every error is reported
|
||||
// at most once, and valid input is parsed as usual
|
||||
const std::vector<std::string> documents =
|
||||
{
|
||||
R"({"name": "value", "list": [1, -2.5, true, null, {"x": [[]]}], "e": "\u00e9"})",
|
||||
R"([{"a": [1, 2, {"b": "c"}]}, [], {}, "\ud83d\ude00", 1e10])",
|
||||
"{\"\xC3\xA9\": \"\xF0\x9F\x98\x80\"}",
|
||||
R"( {"k" : [ "v" , 0 ] } )",
|
||||
};
|
||||
// each character that can be inserted, including a null byte
|
||||
const std::string insertions("[]{},:\"x\\\0\xFF", 11);
|
||||
|
||||
std::vector<std::string> inputs;
|
||||
for (const auto& doc : documents)
|
||||
{
|
||||
for (std::size_t i = 0; i <= doc.size(); ++i)
|
||||
{
|
||||
inputs.push_back(doc.substr(0, i));
|
||||
if (i < doc.size())
|
||||
{
|
||||
inputs.push_back(doc.substr(0, i) + doc.substr(i + 1));
|
||||
}
|
||||
for (const char c : insertions)
|
||||
{
|
||||
inputs.push_back(doc.substr(0, i) + c + doc.substr(i));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (const auto& s : inputs)
|
||||
{
|
||||
CAPTURE(s)
|
||||
const auto result = parse_recovering(s);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.errors.size() <= s.size() + 1);
|
||||
CHECK(result.events <= (4 * s.size()) + 4);
|
||||
if (json::accept(s))
|
||||
{
|
||||
CHECK(result.ok);
|
||||
CHECK(result.errors.empty());
|
||||
CHECK(result.value == json::parse(s));
|
||||
}
|
||||
else
|
||||
{
|
||||
CHECK(!result.ok);
|
||||
CHECK(!result.errors.empty());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,350 +0,0 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
#include <cstdint>
|
||||
#include <limits>
|
||||
#include <map>
|
||||
#include <string>
|
||||
#include <type_traits>
|
||||
#include <vector>
|
||||
|
||||
#if JSON_HAS_THREE_WAY_COMPARISON
|
||||
#include <compare>
|
||||
#endif
|
||||
|
||||
namespace custom_number_types
|
||||
{
|
||||
|
||||
// a signed integer of class type: std::is_signed is only true for arithmetic
|
||||
// types, so the library has to take the signedness from std::numeric_limits,
|
||||
// as it does for 128-bit and multiprecision integer classes such as
|
||||
// absl::int128 or boost::multiprecision::cpp_int
|
||||
class class_int
|
||||
{
|
||||
public:
|
||||
// trivial, like the 128-bit integer classes: the value is stored in a union
|
||||
class_int() = default;
|
||||
|
||||
// implicit from built-in integers and explicit from floats, like absl::int128
|
||||
template<typename T, typename std::enable_if<std::is_integral<T>::value, int>::type = 0>
|
||||
class_int(T v) : value(static_cast<std::int64_t>(v)) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions)
|
||||
|
||||
template<typename T, typename std::enable_if<std::is_floating_point<T>::value, int>::type = 0>
|
||||
explicit class_int(T v) : value(static_cast<std::int64_t>(v)) {}
|
||||
|
||||
template<typename T, typename std::enable_if<std::is_arithmetic<T>::value, int>::type = 0>
|
||||
explicit operator T() const
|
||||
{
|
||||
return static_cast<T>(value);
|
||||
}
|
||||
|
||||
friend bool operator==(class_int lhs, class_int rhs)
|
||||
{
|
||||
return lhs.value == rhs.value;
|
||||
}
|
||||
friend bool operator!=(class_int lhs, class_int rhs)
|
||||
{
|
||||
return lhs.value != rhs.value;
|
||||
}
|
||||
friend bool operator<(class_int lhs, class_int rhs)
|
||||
{
|
||||
return lhs.value < rhs.value;
|
||||
}
|
||||
#if JSON_HAS_THREE_WAY_COMPARISON
|
||||
friend std::strong_ordering operator<=>(class_int lhs, class_int rhs) // *NOPAD*
|
||||
{
|
||||
return lhs.value <=> rhs.value; // *NOPAD*
|
||||
}
|
||||
#endif
|
||||
|
||||
private:
|
||||
std::int64_t value;
|
||||
};
|
||||
|
||||
} // namespace custom_number_types
|
||||
|
||||
using custom_number_types::class_int;
|
||||
|
||||
namespace std
|
||||
{
|
||||
// only the members the library uses
|
||||
template<>
|
||||
class numeric_limits<class_int>
|
||||
{
|
||||
public:
|
||||
static constexpr bool is_signed = true;
|
||||
static constexpr int digits = std::numeric_limits<std::int64_t>::digits;
|
||||
};
|
||||
} // namespace std
|
||||
|
||||
namespace
|
||||
{
|
||||
|
||||
using class_int_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, class_int, std::uint64_t, double>;
|
||||
|
||||
class_int_json make_class_int(std::int64_t v)
|
||||
{
|
||||
class_int_json j(class_int_json::value_t::number_integer);
|
||||
j.get_ref<class_int&>() = class_int(v);
|
||||
return j;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
// the serializer cannot print an integer of class type, so doctest must not
|
||||
// try when an assertion fails
|
||||
namespace doctest
|
||||
{
|
||||
template<>
|
||||
struct StringMaker<class_int_json>
|
||||
{
|
||||
static String convert(const class_int_json& j)
|
||||
{
|
||||
return j.type_name();
|
||||
}
|
||||
};
|
||||
} // namespace doctest
|
||||
|
||||
// __int128 as number type needs the std::numeric_limits (for the range checks)
|
||||
// and std::is_integral (for the serializer) specializations, which libc++
|
||||
// always provides and libstdc++ only outside strict ISO modes; the MSVC
|
||||
// standard library has none
|
||||
#if defined(__SIZEOF_INT128__) && (defined(_LIBCPP_VERSION) || defined(__GLIBCXX_TYPE_INT_N_0))
|
||||
#define JSON_TEST_INT128_NUMBER_TYPES 1
|
||||
#else
|
||||
#define JSON_TEST_INT128_NUMBER_TYPES 0
|
||||
#endif
|
||||
|
||||
#if JSON_TEST_INT128_NUMBER_TYPES
|
||||
namespace
|
||||
{
|
||||
|
||||
// __extension__ keeps -Wpedantic from flagging the non-standard type
|
||||
__extension__ typedef __int128 int128; // NOLINT(modernize-use-using)
|
||||
__extension__ typedef unsigned __int128 uint128; // NOLINT(modernize-use-using)
|
||||
|
||||
static_assert(std::numeric_limits<int128>::digits == 127 && std::numeric_limits<uint128>::digits == 128 &&
|
||||
std::is_integral<int128>::value && std::is_integral<uint128>::value,
|
||||
"__int128 is not fully supported by the standard library");
|
||||
|
||||
using wide_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, int128, uint128, double>;
|
||||
|
||||
wide_json make_int(int128 v)
|
||||
{
|
||||
wide_json j(wide_json::value_t::number_integer);
|
||||
j.get_ref<int128&>() = v;
|
||||
return j;
|
||||
}
|
||||
|
||||
wide_json make_uint(uint128 v)
|
||||
{
|
||||
wide_json j(wide_json::value_t::number_unsigned);
|
||||
j.get_ref<uint128&>() = v;
|
||||
return j;
|
||||
}
|
||||
|
||||
wide_json wrap(const wide_json& value)
|
||||
{
|
||||
wide_json o(wide_json::value_t::object);
|
||||
o["v"] = value;
|
||||
return o;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("custom number types: integers beyond 64 bits")
|
||||
{
|
||||
using out_of_range = wide_json::out_of_range;
|
||||
|
||||
const int128 two_64 = static_cast<int128>(1) << 64;
|
||||
const int128 two_100 = static_cast<int128>(1) << 100;
|
||||
const int128 max_int64 = (std::numeric_limits<std::int64_t>::max)();
|
||||
const int128 min_int64 = (std::numeric_limits<std::int64_t>::min)();
|
||||
const uint128 max_uint64 = (std::numeric_limits<std::uint64_t>::max)();
|
||||
|
||||
// before the range checks, these values were silently truncated, e.g.,
|
||||
// 2^100 was written as 0
|
||||
const wide_json int_big = make_int(two_100);
|
||||
const wide_json int_big_negative = make_int(-two_100);
|
||||
const wide_json uint_big = make_uint(static_cast<uint128>(two_100));
|
||||
std::vector<std::uint8_t> _;
|
||||
|
||||
SECTION("CBOR")
|
||||
{
|
||||
CHECK_THROWS_WITH_AS(_ = wide_json::to_cbor(int_big), "[json.exception.out_of_range.407] integer number 1267650600228229401496703205376 cannot be represented by CBOR as it does not fit [-2^64, 2^64-1]", out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(_ = wide_json::to_cbor(int_big_negative), "[json.exception.out_of_range.407] integer number -1267650600228229401496703205376 cannot be represented by CBOR as it does not fit [-2^64, 2^64-1]", out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(_ = wide_json::to_cbor(uint_big), "[json.exception.out_of_range.407] integer number 1267650600228229401496703205376 cannot be represented by CBOR as it does not fit uint64", out_of_range&);
|
||||
CHECK_THROWS_AS(_ = wide_json::to_cbor(make_int(two_64)), out_of_range&);
|
||||
CHECK_THROWS_AS(_ = wide_json::to_cbor(make_int(-two_64 - 1)), out_of_range&);
|
||||
CHECK_THROWS_AS(_ = wide_json::to_cbor(make_uint(static_cast<uint128>(max_uint64) + 1)), out_of_range&);
|
||||
|
||||
// CBOR's integers cover [-2^64, 2^64-1]
|
||||
CHECK(wide_json::to_cbor(make_int(two_64 - 1)) == std::vector<std::uint8_t>({0x1B, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF}));
|
||||
CHECK(wide_json::to_cbor(make_int(-two_64)) == std::vector<std::uint8_t>({0x3B, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF}));
|
||||
CHECK(wide_json::to_cbor(make_uint(max_uint64)) == std::vector<std::uint8_t>({0x1B, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF}));
|
||||
}
|
||||
|
||||
SECTION("MessagePack")
|
||||
{
|
||||
CHECK_THROWS_WITH_AS(_ = wide_json::to_msgpack(int_big), "[json.exception.out_of_range.407] integer number 1267650600228229401496703205376 cannot be represented by MessagePack as it does not fit [-2^63, 2^64-1]", out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(_ = wide_json::to_msgpack(int_big_negative), "[json.exception.out_of_range.407] integer number -1267650600228229401496703205376 cannot be represented by MessagePack as it does not fit [-2^63, 2^64-1]", out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(_ = wide_json::to_msgpack(uint_big), "[json.exception.out_of_range.407] integer number 1267650600228229401496703205376 cannot be represented by MessagePack as it does not fit uint64", out_of_range&);
|
||||
CHECK_THROWS_AS(_ = wide_json::to_msgpack(make_int(two_64)), out_of_range&);
|
||||
CHECK_THROWS_AS(_ = wide_json::to_msgpack(make_int(min_int64 - 1)), out_of_range&);
|
||||
|
||||
// MessagePack's integers cover [-2^63, 2^64-1]
|
||||
CHECK(wide_json::to_msgpack(make_int(two_64 - 1)) == std::vector<std::uint8_t>({0xCF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF}));
|
||||
CHECK(wide_json::to_msgpack(make_int(min_int64)) == std::vector<std::uint8_t>({0xD3, 0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}));
|
||||
CHECK(wide_json::to_msgpack(make_uint(max_uint64)) == std::vector<std::uint8_t>({0xCF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF}));
|
||||
}
|
||||
|
||||
SECTION("BSON")
|
||||
{
|
||||
CHECK_THROWS_WITH_AS(_ = wide_json::to_bson(wrap(int_big)), "[json.exception.out_of_range.407] integer number 1267650600228229401496703205376 cannot be represented by BSON as it does not fit int64", out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(_ = wide_json::to_bson(wrap(int_big_negative)), "[json.exception.out_of_range.407] integer number -1267650600228229401496703205376 cannot be represented by BSON as it does not fit int64", out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(_ = wide_json::to_bson(wrap(uint_big)), "[json.exception.out_of_range.407] integer number 1267650600228229401496703205376 cannot be represented by BSON as it does not fit uint64", out_of_range&);
|
||||
CHECK_THROWS_AS(_ = wide_json::to_bson(wrap(make_int(max_int64 + 1))), out_of_range&);
|
||||
CHECK_THROWS_AS(_ = wide_json::to_bson(wrap(make_int(min_int64 - 1))), out_of_range&);
|
||||
|
||||
// nested values are checked as well, before anything is written
|
||||
wide_json nested(wide_json::value_t::object);
|
||||
nested["a"] = wide_json::array({wrap(int_big)});
|
||||
std::vector<std::uint8_t> out;
|
||||
CHECK_THROWS_AS(wide_json::to_bson(nested, out), out_of_range&);
|
||||
CHECK(out.empty());
|
||||
|
||||
CHECK(wide_json::from_bson(wide_json::to_bson(wrap(make_int(max_int64)))) == wrap(make_int(max_int64)));
|
||||
CHECK(wide_json::from_bson(wide_json::to_bson(wrap(make_int(min_int64)))) == wrap(make_int(min_int64)));
|
||||
CHECK_NOTHROW(wide_json::to_bson(wrap(make_uint(max_uint64))));
|
||||
}
|
||||
|
||||
SECTION("BON8")
|
||||
{
|
||||
CHECK_THROWS_WITH_AS(_ = wide_json::to_bon8(int_big), "[json.exception.out_of_range.407] integer number 1267650600228229401496703205376 cannot be represented by BON8 as it does not fit int64", out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(_ = wide_json::to_bon8(int_big_negative), "[json.exception.out_of_range.407] integer number -1267650600228229401496703205376 cannot be represented by BON8 as it does not fit int64", out_of_range&);
|
||||
CHECK_THROWS_WITH_AS(_ = wide_json::to_bon8(uint_big), "[json.exception.out_of_range.407] integer number 1267650600228229401496703205376 cannot be represented by BON8 as it does not fit int64", out_of_range&);
|
||||
CHECK_THROWS_AS(_ = wide_json::to_bon8(make_int(max_int64 + 1)), out_of_range&);
|
||||
CHECK_THROWS_AS(_ = wide_json::to_bon8(make_int(min_int64 - 1)), out_of_range&);
|
||||
|
||||
CHECK(wide_json::from_bon8(wide_json::to_bon8(make_int(max_int64))) == make_int(max_int64));
|
||||
CHECK(wide_json::from_bon8(wide_json::to_bon8(make_int(min_int64))) == make_int(min_int64));
|
||||
}
|
||||
|
||||
SECTION("UBJSON and BJData")
|
||||
{
|
||||
// integers beyond 64 bits are written exactly as high-precision numbers
|
||||
const std::string digits = "1267650600228229401496703205376";
|
||||
std::vector<std::uint8_t> expected = {'H', 'i', static_cast<std::uint8_t>(digits.size())};
|
||||
expected.insert(expected.end(), digits.begin(), digits.end());
|
||||
CHECK(wide_json::to_ubjson(int_big) == expected);
|
||||
CHECK(wide_json::to_ubjson(uint_big) == expected);
|
||||
CHECK(wide_json::to_bjdata(int_big) == expected);
|
||||
// BJData's uint64 marker 'M' was used for any unsigned value beyond
|
||||
// int64, which truncated the ones beyond 64 bits
|
||||
CHECK(wide_json::to_bjdata(uint_big) == expected);
|
||||
|
||||
// the uint64 marker is still used where the value fits
|
||||
CHECK(wide_json::to_bjdata(make_uint(max_uint64)) == std::vector<std::uint8_t>({'M', 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF}));
|
||||
|
||||
// the elements of an optimized container are written the same way;
|
||||
// BJData does not allow 'H' as the type of an optimized container
|
||||
std::vector<std::uint8_t> expected_ubjson_array = {'[', '$', 'H', '#', 'i', 2};
|
||||
std::vector<std::uint8_t> expected_bjdata_array = {'[', '#', 'i', 2};
|
||||
for (int i = 0; i < 2; ++i)
|
||||
{
|
||||
expected_ubjson_array.insert(expected_ubjson_array.end(), expected.begin() + 1, expected.end());
|
||||
expected_bjdata_array.insert(expected_bjdata_array.end(), expected.begin(), expected.end());
|
||||
}
|
||||
CHECK(wide_json::to_ubjson(wide_json::array({uint_big, uint_big}), true, true) == expected_ubjson_array);
|
||||
CHECK(wide_json::to_bjdata(wide_json::array({uint_big, uint_big}), true, true) == expected_bjdata_array);
|
||||
}
|
||||
|
||||
SECTION("BJData ND-array")
|
||||
{
|
||||
// an ND-array element beyond 64 bits does not fit any dtype, so the
|
||||
// annotated object is written as a plain object instead of truncating
|
||||
// the element into range
|
||||
wide_json element(wide_json::value_t::object);
|
||||
element["_ArrayType_"] = "uint8";
|
||||
element["_ArraySize_"] = wide_json::array({make_uint(2), make_uint(1)});
|
||||
element["_ArrayData_"] = wide_json::array({make_uint(1), make_uint(static_cast<uint128>(two_64) + 1)});
|
||||
const auto element_bytes = wide_json::to_bjdata(element, true, true);
|
||||
REQUIRE(!element_bytes.empty());
|
||||
CHECK(element_bytes[0] == '{');
|
||||
|
||||
wide_json signed_element = element;
|
||||
signed_element["_ArrayType_"] = "int8";
|
||||
signed_element["_ArrayData_"] = wide_json::array({make_int(1), make_int(-two_64 + 1)});
|
||||
const auto signed_element_bytes = wide_json::to_bjdata(signed_element, true, true);
|
||||
REQUIRE(!signed_element_bytes.empty());
|
||||
CHECK(signed_element_bytes[0] == '{');
|
||||
|
||||
// the same for a dimension beyond 64 bits, which wrapped into a
|
||||
// dimension matching the size of _ArrayData_ (here: 1)
|
||||
wide_json dimension = element;
|
||||
dimension["_ArraySize_"] = wide_json::array({make_uint(2), make_uint(static_cast<uint128>(two_64) + 1)});
|
||||
dimension["_ArrayData_"] = wide_json::array({make_uint(1), make_uint(2)});
|
||||
const auto dimension_bytes = wide_json::to_bjdata(dimension, true, true);
|
||||
REQUIRE(!dimension_bytes.empty());
|
||||
CHECK(dimension_bytes[0] == '{');
|
||||
|
||||
// in range, the ND-array is still written
|
||||
wide_json fits = element;
|
||||
fits["_ArrayData_"] = wide_json::array({make_uint(1), make_uint(2)});
|
||||
const auto fits_bytes = wide_json::to_bjdata(fits, true, true);
|
||||
REQUIRE(!fits_bytes.empty());
|
||||
CHECK(fits_bytes[0] == '[');
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
TEST_CASE("custom number types")
|
||||
{
|
||||
SECTION("signed integer of class type compared with a float")
|
||||
{
|
||||
// std::is_signed is false for a class type, which made the comparison
|
||||
// treat any float below zero as less than every integer
|
||||
const class_int_json minus_five = make_class_int(-5);
|
||||
const class_int_json minus_two = make_class_int(-2);
|
||||
const class_int_json five = make_class_int(5);
|
||||
const class_int_json minus_two_and_a_half = -2.5;
|
||||
const class_int_json two_and_a_half = 2.5;
|
||||
const class_int_json huge_negative = -1e30;
|
||||
const class_int_json huge_positive = 1e30;
|
||||
|
||||
CHECK(minus_five < minus_two_and_a_half);
|
||||
CHECK_FALSE(minus_two_and_a_half < minus_five);
|
||||
CHECK(minus_two_and_a_half > minus_five);
|
||||
CHECK(minus_five <= minus_two_and_a_half);
|
||||
CHECK(minus_two_and_a_half >= minus_five);
|
||||
CHECK(minus_five != minus_two_and_a_half);
|
||||
|
||||
CHECK(minus_two_and_a_half < minus_two);
|
||||
CHECK_FALSE(minus_two < minus_two_and_a_half);
|
||||
|
||||
CHECK(two_and_a_half < five);
|
||||
CHECK(minus_five < two_and_a_half);
|
||||
|
||||
// floats beyond the integer's range
|
||||
CHECK(huge_negative < minus_five);
|
||||
CHECK_FALSE(minus_five < huge_negative);
|
||||
CHECK(five < huge_positive);
|
||||
CHECK_FALSE(huge_positive < five);
|
||||
|
||||
// equality
|
||||
const class_int_json minus_two_float = -2.0;
|
||||
CHECK(minus_two == minus_two_float);
|
||||
CHECK(minus_two_float == minus_two);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -31,8 +31,6 @@ using nlohmann::json;
|
||||
#include "make_test_data_available.hpp"
|
||||
#include "round_trip_corpus.hpp"
|
||||
#include "test_utils.hpp"
|
||||
#include "sax_countdown.hpp"
|
||||
using utils::SaxCountdown;
|
||||
|
||||
|
||||
TEST_CASE("MessagePack")
|
||||
@@ -1645,30 +1643,6 @@ TEST_CASE("MessagePack")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("SAX aborts")
|
||||
{
|
||||
SECTION("start_array(len)")
|
||||
{
|
||||
std::vector<uint8_t> const v = {0x93, 0x01, 0x02, 0x03};
|
||||
SaxCountdown scp(0);
|
||||
CHECK(!json::sax_parse(v, &scp, json::input_format_t::msgpack));
|
||||
}
|
||||
|
||||
SECTION("start_object(len)")
|
||||
{
|
||||
std::vector<uint8_t> const v = {0x81, 0xa3, 0x66, 0x6F, 0x6F, 0xc2};
|
||||
SaxCountdown scp(0);
|
||||
CHECK(!json::sax_parse(v, &scp, json::input_format_t::msgpack));
|
||||
}
|
||||
|
||||
SECTION("key()")
|
||||
{
|
||||
std::vector<uint8_t> const v = {0x81, 0xa3, 0x66, 0x6F, 0x6F, 0xc2};
|
||||
SaxCountdown scp(1);
|
||||
CHECK(!json::sax_parse(v, &scp, json::input_format_t::msgpack));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("issue #5405 - array reserve for definite-length MessagePack arrays")
|
||||
@@ -1739,21 +1713,6 @@ TEST_CASE("issue #5405 - array reserve for definite-length MessagePack arrays")
|
||||
CHECK(json::from_msgpack(packed) == j);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
|
||||
{
|
||||
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
|
||||
// a custom SAX consumer that does not touch a DOM array sees identical events
|
||||
json j = json::array();
|
||||
for (int i = 0; i < 100; ++i)
|
||||
{
|
||||
j.push_back(i);
|
||||
}
|
||||
const auto packed = json::to_msgpack(j);
|
||||
|
||||
SaxCountdown scp(1000000); // large enough to never trigger an abort
|
||||
CHECK(json::sax_parse(packed, &scp, json::input_format_t::msgpack));
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("regression test - MessagePack ext type rejects a subtype that doesn't fit a single byte")
|
||||
@@ -1792,15 +1751,6 @@ TEST_CASE("MessagePack nesting does not consume the call stack")
|
||||
CHECK(json::from_msgpack(input, true, false).is_discarded());
|
||||
}
|
||||
|
||||
SECTION("a well-formed deep value is read through the SAX interface")
|
||||
{
|
||||
std::vector<uint8_t> input(300000, 0x91);
|
||||
input.push_back(0x01); // innermost value
|
||||
|
||||
SaxCountdown accept_all(600001);
|
||||
CHECK(json::sax_parse(input, &accept_all, json::input_format_t::msgpack));
|
||||
}
|
||||
|
||||
SECTION("a well-formed deep value is read into a value")
|
||||
{
|
||||
const std::size_t depth = 10000;
|
||||
@@ -1848,31 +1798,6 @@ TEST_CASE("MessagePack input that cannot be read is discarded by every overload"
|
||||
#endif
|
||||
}
|
||||
|
||||
TEST_CASE("MessagePack SAX parsing stops at every event")
|
||||
{
|
||||
// Containers are opened and closed by the loop that reads them; a SAX
|
||||
// handler that rejects any event - including the end of a nested
|
||||
// container - must stop the parse right there.
|
||||
const auto count_events = [](const std::vector<std::uint8_t>& input)
|
||||
{
|
||||
int events = 0;
|
||||
while (true)
|
||||
{
|
||||
SaxCountdown scp(events);
|
||||
if (json::sax_parse(input, &scp, json::input_format_t::msgpack))
|
||||
{
|
||||
return events;
|
||||
}
|
||||
++events;
|
||||
REQUIRE(events < 1000);
|
||||
}
|
||||
};
|
||||
|
||||
// 20 events: every container kind closes inside another one
|
||||
const json j = json::parse(R"({"a": [1, {"b": []}], "c": {"d": [[2]]}})");
|
||||
CHECK(count_events(json::to_msgpack(j)) == 20);
|
||||
}
|
||||
|
||||
TEST_CASE("single MessagePack roundtrip")
|
||||
{
|
||||
SECTION("sample.json")
|
||||
|
||||
@@ -22,14 +22,6 @@
|
||||
// scoped enum, so get<std::byte>() (needed below to get<std::vector<std::byte>>()
|
||||
// from a plain JSON array, not just from an already-binary value) relies on
|
||||
// enum serialization being enabled
|
||||
// capture whether JSON_DELETE_DEPRECATED_FUNCTIONS was enabled on the command
|
||||
// line *before* including json.hpp, since the library #undefs it once the header
|
||||
// has been fully processed (see include/nlohmann/detail/macro_unscope.hpp); the
|
||||
// tests of deprecated functions are skipped if these functions are deleted
|
||||
#if defined(JSON_DELETE_DEPRECATED_FUNCTIONS) && (JSON_DELETE_DEPRECATED_FUNCTIONS == 1)
|
||||
#define JSON_TEST_DEPRECATED_FUNCTIONS_DELETED
|
||||
#endif
|
||||
|
||||
#if defined(JSON_DISABLE_ENUM_SERIALIZATION) && (JSON_DISABLE_ENUM_SERIALIZATION == 1)
|
||||
#define SKIP_TESTS_FOR_ENUM_SERIALIZATION
|
||||
#endif
|
||||
@@ -859,50 +851,6 @@ TEST_CASE("issue #5338 - truncated CBOR tagged binary subtype is rejected")
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("issue #5676 - SAX parsing of CBOR tags")
|
||||
{
|
||||
const json expected = json::binary({1, 2, 3}, 42);
|
||||
const auto cbor = json::to_cbor(expected);
|
||||
|
||||
nlohmann::detail::json_sax_acceptor<json> acceptor;
|
||||
CHECK_FALSE(json::sax_parse(cbor, &acceptor, json::input_format_t::cbor));
|
||||
CHECK_FALSE(json::sax_parse(cbor, &acceptor, json::input_format_t::cbor,
|
||||
true, false, false, json::cbor_tag_handler_t::error));
|
||||
|
||||
CHECK(json::sax_parse(cbor, &acceptor, json::input_format_t::cbor,
|
||||
true, false, false, json::cbor_tag_handler_t::ignore));
|
||||
|
||||
json parsed;
|
||||
nlohmann::detail::json_sax_dom_parser<json, nlohmann::detail::string_input_adapter_type> sax(parsed);
|
||||
CHECK(json::sax_parse(cbor, &sax, json::input_format_t::cbor,
|
||||
true, false, false, json::cbor_tag_handler_t::store));
|
||||
CHECK(parsed == expected);
|
||||
|
||||
json iterator_parsed;
|
||||
nlohmann::detail::json_sax_dom_parser<json, nlohmann::detail::string_input_adapter_type> iterator_sax(iterator_parsed);
|
||||
CHECK(json::sax_parse(cbor.begin(), cbor.end(), &iterator_sax, json::input_format_t::cbor,
|
||||
true, false, false, json::cbor_tag_handler_t::store));
|
||||
CHECK(iterator_parsed == expected);
|
||||
|
||||
#ifndef JSON_TEST_DEPRECATED_FUNCTIONS_DELETED
|
||||
json span_parsed;
|
||||
nlohmann::detail::json_sax_dom_parser<json, nlohmann::detail::string_input_adapter_type> span_sax(span_parsed);
|
||||
CHECK(json::sax_parse(nlohmann::detail::span_input_adapter(cbor.data(), cbor.size()), &span_sax,
|
||||
json::input_format_t::cbor, true, false, false, json::cbor_tag_handler_t::store));
|
||||
CHECK(span_parsed == expected);
|
||||
#endif
|
||||
|
||||
const std::string text = "null";
|
||||
CHECK(json::sax_parse(text, &acceptor, json::input_format_t::json,
|
||||
true, false, false, json::cbor_tag_handler_t::store));
|
||||
CHECK(json::sax_parse(text.begin(), text.end(), &acceptor, json::input_format_t::json,
|
||||
true, false, false, json::cbor_tag_handler_t::store));
|
||||
#ifndef JSON_TEST_DEPRECATED_FUNCTIONS_DELETED
|
||||
CHECK(json::sax_parse(nlohmann::detail::span_input_adapter(text.data(), text.size()), &acceptor,
|
||||
json::input_format_t::json, true, false, false, json::cbor_tag_handler_t::store));
|
||||
#endif
|
||||
}
|
||||
|
||||
TEST_CASE("issue #5402 - update(merge_objects=true) overwrites a primitive with an object")
|
||||
{
|
||||
json t = {{"k", 1}};
|
||||
|
||||
@@ -0,0 +1,768 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
/////////////////////////////////////////////////////////////////////
|
||||
// Tests that call basic_json::sax_parse have a file of their own: every
|
||||
// sax_parse call instantiates the parser and binary reader that recover from
|
||||
// errors (see #3989), and in unit-regression2.cpp, unit-regression3.cpp, and
|
||||
// unit-msgpack.cpp this made the objects too large for the MinGW linker to
|
||||
// relocate (see #5511).
|
||||
/////////////////////////////////////////////////////////////////////
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
|
||||
// capture whether JSON_DELETE_DEPRECATED_FUNCTIONS was enabled on the command
|
||||
// line *before* including json.hpp, since the library #undefs it once the header
|
||||
// has been fully processed (see include/nlohmann/detail/macro_unscope.hpp); the
|
||||
// tests of deprecated functions are skipped if these functions are deleted
|
||||
#if defined(JSON_DELETE_DEPRECATED_FUNCTIONS) && (JSON_DELETE_DEPRECATED_FUNCTIONS == 1)
|
||||
#define JSON_TEST_DEPRECATED_FUNCTIONS_DELETED
|
||||
#endif
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
using json = nlohmann::json;
|
||||
|
||||
#include <cmath>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <initializer_list>
|
||||
#include <map>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "sax_countdown.hpp"
|
||||
using utils::SaxCountdown;
|
||||
|
||||
// a narrow number_float_t, so that a double read from binary input can
|
||||
// overflow it
|
||||
using float_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, float>;
|
||||
|
||||
DOCTEST_CLANG_SUPPRESS_WARNING_PUSH
|
||||
DOCTEST_CLANG_SUPPRESS_WARNING("-Wexit-time-destructors")
|
||||
|
||||
namespace
|
||||
{
|
||||
/// builds a value from SAX events, asks the parser to recover from its first
|
||||
/// 100 errors, and checks that the events are balanced (see #3989)
|
||||
template<typename BasicJsonType>
|
||||
class BasicRecoveringParser
|
||||
{
|
||||
public:
|
||||
explicit BasicRecoveringParser(BasicJsonType& j)
|
||||
: dom(j, false)
|
||||
{}
|
||||
|
||||
bool null()
|
||||
{
|
||||
value();
|
||||
return dom.null();
|
||||
}
|
||||
|
||||
bool boolean(bool val)
|
||||
{
|
||||
value();
|
||||
return dom.boolean(val);
|
||||
}
|
||||
|
||||
bool number_integer(typename BasicJsonType::number_integer_t val)
|
||||
{
|
||||
value();
|
||||
return dom.number_integer(val);
|
||||
}
|
||||
|
||||
bool number_unsigned(typename BasicJsonType::number_unsigned_t val)
|
||||
{
|
||||
value();
|
||||
return dom.number_unsigned(val);
|
||||
}
|
||||
|
||||
bool number_float(typename BasicJsonType::number_float_t val, const std::string& s)
|
||||
{
|
||||
value();
|
||||
return dom.number_float(val, s);
|
||||
}
|
||||
|
||||
bool string(std::string& val)
|
||||
{
|
||||
value();
|
||||
return dom.string(val);
|
||||
}
|
||||
|
||||
bool binary(typename BasicJsonType::binary_t& val)
|
||||
{
|
||||
value();
|
||||
return dom.binary(val);
|
||||
}
|
||||
|
||||
bool start_object(std::size_t elements)
|
||||
{
|
||||
value();
|
||||
stack.push_back('o');
|
||||
return dom.start_object(elements);
|
||||
}
|
||||
|
||||
bool key(std::string& val)
|
||||
{
|
||||
if (stack.empty() || stack.back() != 'o')
|
||||
{
|
||||
well_formed = false;
|
||||
return false;
|
||||
}
|
||||
stack.back() = 'v';
|
||||
return dom.key(val);
|
||||
}
|
||||
|
||||
bool end_object()
|
||||
{
|
||||
if (stack.empty() || stack.back() != 'o')
|
||||
{
|
||||
well_formed = false;
|
||||
return false;
|
||||
}
|
||||
stack.pop_back();
|
||||
return dom.end_object();
|
||||
}
|
||||
|
||||
bool start_array(std::size_t elements)
|
||||
{
|
||||
value();
|
||||
stack.push_back('a');
|
||||
return dom.start_array(elements);
|
||||
}
|
||||
|
||||
bool end_array()
|
||||
{
|
||||
if (stack.empty() || stack.back() != 'a')
|
||||
{
|
||||
well_formed = false;
|
||||
return false;
|
||||
}
|
||||
stack.pop_back();
|
||||
return dom.end_array();
|
||||
}
|
||||
|
||||
bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& ex)
|
||||
{
|
||||
messages.emplace_back(ex.what());
|
||||
// a limit, so that a reader that does not stop fails the test
|
||||
// instead of making it hang
|
||||
return ++errors < 100;
|
||||
}
|
||||
|
||||
/// whether the events were balanced and every key was followed by a value
|
||||
bool balanced() const
|
||||
{
|
||||
return well_formed && stack.empty();
|
||||
}
|
||||
|
||||
/// builds the value
|
||||
nlohmann::detail::json_sax_dom_parser<BasicJsonType> dom;
|
||||
std::size_t errors = 0;
|
||||
std::vector<std::string> messages {}; // NOLINT(readability-redundant-member-init)
|
||||
std::vector<char> stack {}; // NOLINT(readability-redundant-member-init)
|
||||
bool well_formed = true;
|
||||
|
||||
private:
|
||||
void value()
|
||||
{
|
||||
if (!stack.empty())
|
||||
{
|
||||
if (stack.back() == 'v')
|
||||
{
|
||||
stack.back() = 'o';
|
||||
}
|
||||
else if (stack.back() == 'o')
|
||||
{
|
||||
well_formed = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
using RecoveringParser = BasicRecoveringParser<json>;
|
||||
|
||||
struct BinaryParseResult
|
||||
{
|
||||
json value;
|
||||
std::size_t errors;
|
||||
std::vector<std::string> messages;
|
||||
bool ok;
|
||||
bool balanced;
|
||||
};
|
||||
|
||||
BinaryParseResult parse_binary_recovering(const std::vector<std::uint8_t>& input, const json::input_format_t format)
|
||||
{
|
||||
json j;
|
||||
RecoveringParser sax(j);
|
||||
const bool ok = json::sax_parse(input, &sax, format);
|
||||
return {j, sax.errors, sax.messages, ok, sax.balanced()};
|
||||
}
|
||||
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
/// the message of the exception that reading @a input into a JSON value
|
||||
/// throws, or an empty string if reading succeeds
|
||||
std::string binary_error_message(const std::vector<std::uint8_t>& input, const json::input_format_t format)
|
||||
{
|
||||
try
|
||||
{
|
||||
json _;
|
||||
switch (format)
|
||||
{
|
||||
case json::input_format_t::cbor:
|
||||
_ = json::from_cbor(input);
|
||||
break;
|
||||
case json::input_format_t::msgpack:
|
||||
_ = json::from_msgpack(input);
|
||||
break;
|
||||
case json::input_format_t::ubjson:
|
||||
_ = json::from_ubjson(input);
|
||||
break;
|
||||
case json::input_format_t::bjdata:
|
||||
_ = json::from_bjdata(input);
|
||||
break;
|
||||
case json::input_format_t::bson:
|
||||
_ = json::from_bson(input);
|
||||
break;
|
||||
case json::input_format_t::bon8:
|
||||
_ = json::from_bon8(input);
|
||||
break;
|
||||
case json::input_format_t::json:
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
catch (const json::exception& e)
|
||||
{
|
||||
return e.what();
|
||||
}
|
||||
return "";
|
||||
}
|
||||
#endif
|
||||
|
||||
/// a BSON element: its type, its name, and its value
|
||||
std::vector<std::uint8_t> bson_element(const std::uint8_t type, const std::string& name, const std::vector<std::uint8_t>& value)
|
||||
{
|
||||
std::vector<std::uint8_t> result = {type};
|
||||
result.insert(result.end(), name.begin(), name.end());
|
||||
result.push_back(0x00);
|
||||
result.insert(result.end(), value.begin(), value.end());
|
||||
return result;
|
||||
}
|
||||
|
||||
/// a BSON document of the given elements; @a size_offset is added to the
|
||||
/// size it declares
|
||||
std::vector<std::uint8_t> bson_document(const std::vector<std::vector<std::uint8_t>>& elements, const int size_offset = 0)
|
||||
{
|
||||
std::vector<std::uint8_t> body;
|
||||
for (const auto& element : elements)
|
||||
{
|
||||
body.insert(body.end(), element.begin(), element.end());
|
||||
}
|
||||
const auto size = static_cast<std::uint32_t>(static_cast<int>(body.size()) + 5 + size_offset);
|
||||
std::vector<std::uint8_t> result = {static_cast<std::uint8_t>(size & 0xFFu), static_cast<std::uint8_t>((size >> 8u) & 0xFFu),
|
||||
static_cast<std::uint8_t>((size >> 16u) & 0xFFu), static_cast<std::uint8_t>((size >> 24u) & 0xFFu)
|
||||
};
|
||||
result.insert(result.end(), body.begin(), body.end());
|
||||
result.push_back(0x00);
|
||||
return result;
|
||||
}
|
||||
|
||||
/// a BSON int32 value
|
||||
std::vector<std::uint8_t> bson_int32(const std::int32_t value)
|
||||
{
|
||||
const auto u = static_cast<std::uint32_t>(value);
|
||||
return {static_cast<std::uint8_t>(u & 0xFFu), static_cast<std::uint8_t>((u >> 8u) & 0xFFu),
|
||||
static_cast<std::uint8_t>((u >> 16u) & 0xFFu), static_cast<std::uint8_t>((u >> 24u) & 0xFFu)};
|
||||
}
|
||||
|
||||
/// a BSON string value, whose length is @a length_offset off
|
||||
std::vector<std::uint8_t> bson_string(const std::string& value, const std::int32_t length_offset = 0)
|
||||
{
|
||||
auto result = bson_int32(static_cast<std::int32_t>(value.size() + 1) + length_offset);
|
||||
result.insert(result.end(), value.begin(), value.end());
|
||||
result.push_back(0x00);
|
||||
return result;
|
||||
}
|
||||
|
||||
/// @a count bytes of value 0xAB
|
||||
std::vector<std::uint8_t> bytes(const std::size_t count)
|
||||
{
|
||||
return std::vector<std::uint8_t>(count, 0xAB);
|
||||
}
|
||||
|
||||
/// the bytes of @a parts, one after the other
|
||||
std::vector<std::uint8_t> concatenated(std::initializer_list<std::vector<std::uint8_t>> parts)
|
||||
{
|
||||
std::vector<std::uint8_t> result;
|
||||
for (const auto& part : parts)
|
||||
{
|
||||
result.insert(result.end(), part.begin(), part.end());
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/// U+FFFD REPLACEMENT CHARACTER
|
||||
std::string replacement_character()
|
||||
{
|
||||
return "\xEF\xBF\xBD";
|
||||
}
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("regression test - #3989 SAX parse_error() returning true")
|
||||
{
|
||||
SECTION("binary formats complete what was read before the input ends")
|
||||
{
|
||||
const json j = {{"a", {1, -2, {{"b", "c"}}, json::array()}}, {"d", {{"e", nullptr}, {"f", true}}}, {"g", 1.5}, {"h", json::binary({1, 2, 3})}};
|
||||
|
||||
const std::vector<std::pair<json::input_format_t, std::vector<std::uint8_t>>> encodings =
|
||||
{
|
||||
{json::input_format_t::cbor, json::to_cbor(j)},
|
||||
{json::input_format_t::msgpack, json::to_msgpack(j)},
|
||||
{json::input_format_t::ubjson, json::to_ubjson(j)},
|
||||
{json::input_format_t::ubjson, json::to_ubjson(j, true, true)},
|
||||
{json::input_format_t::bjdata, json::to_bjdata(j)},
|
||||
{json::input_format_t::bjdata, json::to_bjdata(j, true, true)},
|
||||
{json::input_format_t::bson, json::to_bson(j)},
|
||||
{json::input_format_t::bon8, json::to_bon8(j)},
|
||||
};
|
||||
|
||||
for (const auto& encoding : encodings)
|
||||
{
|
||||
const auto format = encoding.first;
|
||||
const auto& bytes = encoding.second;
|
||||
CAPTURE(format)
|
||||
|
||||
// every prefix is truncated input
|
||||
for (std::size_t length = 0; length < bytes.size(); ++length)
|
||||
{
|
||||
CAPTURE(length)
|
||||
const auto result = parse_binary_recovering(std::vector<std::uint8_t>(bytes.begin(), bytes.begin() + static_cast<std::ptrdiff_t>(length)), format);
|
||||
CHECK(!result.ok);
|
||||
CHECK(result.errors == 1);
|
||||
CHECK(result.balanced);
|
||||
}
|
||||
|
||||
// the complete input is read as usual (binary values do not
|
||||
// round-trip through every format, so compare with a plain parse)
|
||||
json expected;
|
||||
nlohmann::detail::json_sax_dom_parser<json> dom(expected);
|
||||
CHECK(json::sax_parse(bytes, &dom, format));
|
||||
const auto complete = parse_binary_recovering(bytes, format);
|
||||
CHECK(complete.ok);
|
||||
CHECK(complete.errors == 0);
|
||||
CHECK(complete.value == expected);
|
||||
|
||||
// a byte after the value
|
||||
auto trailing_bytes = bytes;
|
||||
trailing_bytes.push_back(0x01);
|
||||
const auto trailing = parse_binary_recovering(trailing_bytes, format);
|
||||
CHECK(!trailing.ok);
|
||||
CHECK(trailing.errors == 1);
|
||||
CHECK(trailing.value == expected);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("containers without an end")
|
||||
{
|
||||
// these made the readers loop, or read on, after the error
|
||||
const auto cbor_array = parse_binary_recovering({0x9F}, json::input_format_t::cbor);
|
||||
CHECK(cbor_array.errors == 1);
|
||||
CHECK(cbor_array.value == json::array());
|
||||
|
||||
const auto cbor_map = parse_binary_recovering({0xBF, 0x61, 'a'}, json::input_format_t::cbor);
|
||||
CHECK(cbor_map.errors == 1);
|
||||
CHECK(cbor_map.value == json({{"a", nullptr}}));
|
||||
|
||||
const auto msgpack_array = parse_binary_recovering({0xDD, 0xFF, 0xFF, 0xFF, 0xFF}, json::input_format_t::msgpack);
|
||||
CHECK(msgpack_array.errors == 1);
|
||||
CHECK(msgpack_array.value == json::array());
|
||||
|
||||
const auto msgpack_map = parse_binary_recovering({0x81, 0xA1, 'a', 0x92, 0x01}, json::input_format_t::msgpack);
|
||||
CHECK(msgpack_map.errors == 1);
|
||||
CHECK(msgpack_map.value == json({{"a", {1}}}));
|
||||
}
|
||||
|
||||
SECTION("BJData ndarray")
|
||||
{
|
||||
// a 2x3 int8 array with two of its six elements; the annotated array
|
||||
// format opens an object and two arrays of its own
|
||||
const auto result = parse_binary_recovering({'[', '$', 'i', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2}, json::input_format_t::bjdata);
|
||||
CHECK(result.errors == 1);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.value == json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2}}}));
|
||||
}
|
||||
|
||||
SECTION("binary formats repair items whose end is known")
|
||||
{
|
||||
struct Repair
|
||||
{
|
||||
json::input_format_t format;
|
||||
std::vector<std::uint8_t> input;
|
||||
json expected;
|
||||
std::size_t errors;
|
||||
};
|
||||
|
||||
const std::vector<Repair> repairs =
|
||||
{
|
||||
// CBOR: tags are ignored (here tag 1 and the self-describe tag 55799)
|
||||
{json::input_format_t::cbor, {0x82, 0xC1, 0x05, 0xD9, 0xD9, 0xF7, 0x06}, {5, 6}, 2},
|
||||
// CBOR: undefined and other simple values become null
|
||||
{json::input_format_t::cbor, {0x84, 0xF7, 0xE0, 0xF8, 0x20, 0x01}, {nullptr, nullptr, nullptr, 1}, 3},
|
||||
// CBOR: members whose key is not a string are skipped, whatever their key and value
|
||||
{json::input_format_t::cbor, {0xA4, 0x01, 0x02, 0x82, 0x01, 0x02, 0xA1, 0x61, 'x', 0x9F, 0xFF, 0xC1, 0x01, 0x5F, 0x41, 0x00, 0xFF, 0x61, 'a', 0x03}, {{"a", 3}}, 3},
|
||||
{json::input_format_t::cbor, {0xBF, 0xF5, 0xBF, 0x61, 'x', 0x7F, 0x61, 'y', 0xFF, 0xFF, 0x61, 'a', 0x03, 0xFF}, {{"a", 3}}, 1},
|
||||
// MessagePack: members whose key is not a string are skipped
|
||||
{json::input_format_t::msgpack, {0x84, 0x01, 0x02, 0x81, 0xA1, 'x', 0x01, 0x92, 0x01, 0x02, 0xD4, 0x01, 0x02, 0xC0, 0xA1, 'a', 0x04}, {{"a", 4}}, 3},
|
||||
// UBJSON: a char that is not ASCII becomes U+FFFD
|
||||
{json::input_format_t::ubjson, {'[', 'C', 0x80, 'C', 'A', ']'}, {replacement_character(), "A"}, 1},
|
||||
// UBJSON: the longest beginning of a high-precision number is kept
|
||||
{json::input_format_t::ubjson, {'[', 'H', 'i', 5, '1', '2', 'a', 'b', 'c', 'H', 'i', 2, '1', '.', 'H', 'i', 3, 'a', 'b', 'c', 'H', 'i', 3, '4', '.', '5', ']'}, {12, 1, nullptr, 4.5}, 3},
|
||||
// BJData, too
|
||||
{json::input_format_t::bjdata, {'[', 'C', 0xFF, 'H', 'i', 2, '-', '1', 'H', 'i', 2, '-', 'x', ']'}, {replacement_character(), -1, nullptr}, 2},
|
||||
// a NUL ends a high-precision number, as it ends JSON text
|
||||
{json::input_format_t::ubjson, {'[', 'H', 'i', 4, '1', '2', 0, '9', 'H', 'i', 2, 0, '1', 'i', 3, ']'}, {12, nullptr, 3}, 2},
|
||||
// BON8: members whose key is not a string are skipped
|
||||
{json::input_format_t::bon8, {0x89, 0x91, 0x92, 0xC9, 0x40, 0x82, 0x91, 0x92, 0x61, 0x93}, {{"a", 3}}, 2},
|
||||
{json::input_format_t::bon8, {0x8B, 0x91, 0x85, 0x91, 0xFE, 0xFA, 0x8B, 'x', 0x91, 0xFE, 0x61, 0x93, 0xFE}, {{"a", 3}}, 2},
|
||||
// BSON: elements of types the library does not read become null
|
||||
{
|
||||
json::input_format_t::bson, bson_document(
|
||||
{
|
||||
bson_element(0x07, "_id", bytes(12)), // ObjectId
|
||||
bson_element(0x09, "date", bytes(8)), // UTC datetime
|
||||
bson_element(0x13, "decimal", bytes(16)), // 128-bit decimal
|
||||
bson_element(0x0B, "regex", {'a', '+', 0, 'i', 0}), // regular expression
|
||||
bson_element(0x0D, "code", bson_string("f()")), // JavaScript code
|
||||
bson_element(0x0E, "symbol", bson_string("s")), // symbol
|
||||
bson_element(0x0C, "pointer", concatenated({bson_string("c"), bytes(12)})), // DBPointer
|
||||
bson_element(0x0F, "scope", concatenated({bson_int32(15), bson_string("g"), bson_document({})})), // code with scope
|
||||
bson_element(0x06, "undefined", {}), // undefined
|
||||
bson_element(0xFF, "min", {}), // min key
|
||||
bson_element(0x7F, "max", {}), // max key
|
||||
bson_element(0x10, "z", bson_int32(7)),
|
||||
}),
|
||||
{{"_id", nullptr}, {"date", nullptr}, {"decimal", nullptr}, {"regex", nullptr}, {"code", nullptr}, {"symbol", nullptr}, {"pointer", nullptr}, {"scope", nullptr}, {"undefined", nullptr}, {"min", nullptr}, {"max", nullptr}, {"z", 7}},
|
||||
11
|
||||
},
|
||||
// BSON: an element of an unknown type becomes null, and the rest of its document is skipped
|
||||
{
|
||||
json::input_format_t::bson, bson_document(
|
||||
{
|
||||
bson_element(0x03, "inner", bson_document({bson_element(0x10, "a", bson_int32(1)), bson_element(0x42, "x", bytes(3)), bson_element(0x10, "b", bson_int32(2))})),
|
||||
bson_element(0x04, "array", bson_document({bson_element(0x10, "0", bson_int32(1)), bson_element(0x42, "1", bytes(3))})),
|
||||
bson_element(0x10, "after", bson_int32(3)),
|
||||
}),
|
||||
{{"inner", {{"a", 1}, {"x", nullptr}}}, {"array", {1, nullptr}}, {"after", 3}},
|
||||
2
|
||||
},
|
||||
// BSON: so does a string or byte array whose length cannot be right
|
||||
{
|
||||
json::input_format_t::bson, bson_document(
|
||||
{
|
||||
bson_element(0x03, "inner", bson_document({bson_element(0x02, "s", bson_string("abc", -10)), bson_element(0x10, "b", bson_int32(2))})),
|
||||
bson_element(0x03, "bin", bson_document({bson_element(0x05, "b", concatenated({bson_int32(-1), bytes(1)})), bson_element(0x10, "b", bson_int32(2))})),
|
||||
bson_element(0x10, "after", bson_int32(3)),
|
||||
}),
|
||||
{{"inner", {{"s", nullptr}}}, {"bin", {{"b", nullptr}}}, {"after", 3}},
|
||||
2
|
||||
},
|
||||
// BSON: a string without its terminator, and a document whose size does not match, are kept
|
||||
{
|
||||
json::input_format_t::bson, bson_document(
|
||||
{
|
||||
bson_element(0x02, "s", {2, 0, 0, 0, 'a', 'X'}),
|
||||
bson_element(0x03, "inner", bson_document({bson_element(0x10, "a", bson_int32(1))}, 1)),
|
||||
}),
|
||||
{{"s", "a"}, {"inner", {{"a", 1}}}},
|
||||
2
|
||||
},
|
||||
};
|
||||
|
||||
for (const auto& repair : repairs)
|
||||
{
|
||||
CAPTURE(repair.format)
|
||||
CAPTURE(repair.input)
|
||||
const auto result = parse_binary_recovering(repair.input, repair.format);
|
||||
CHECK(!result.ok);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.errors == repair.errors);
|
||||
CHECK(result.value == repair.expected);
|
||||
REQUIRE(!result.messages.empty());
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
// the first error is the one reported without recovering; under
|
||||
// JSON_NOEXCEPTION, reading without recovering aborts instead of
|
||||
// throwing, so there is no message to compare with
|
||||
CHECK(result.messages.front() == binary_error_message(repair.input, repair.format));
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("binary formats repair numbers that are out of range")
|
||||
{
|
||||
// CBOR: a double too large for a float number_float_t
|
||||
float_json cbor;
|
||||
BasicRecoveringParser<float_json> sax(cbor);
|
||||
const std::vector<std::uint8_t> cbor_input = {0x82, 0xFB, 0x7E, 0x37, 0xE4, 0x3C, 0x88, 0x00, 0x75, 0x9C, 0x01}; // [1e300, 1]
|
||||
CHECK(!float_json::sax_parse(cbor_input, &sax, float_json::input_format_t::cbor));
|
||||
CHECK(sax.errors == 1);
|
||||
CHECK(sax.messages.front() == "[json.exception.out_of_range.406] syntax error while parsing CBOR value: number overflow");
|
||||
REQUIRE(cbor.size() == 2);
|
||||
CHECK(std::isinf(cbor[0].get<float>()));
|
||||
CHECK(cbor[1] == 1);
|
||||
|
||||
// UBJSON: a high-precision number too large for number_float_t
|
||||
const auto ubjson = parse_binary_recovering({'H', 'i', 5, '1', 'e', '9', '9', '9'}, json::input_format_t::ubjson);
|
||||
CHECK(ubjson.errors == 1);
|
||||
CHECK(ubjson.value.is_number_float());
|
||||
CHECK(std::isinf(ubjson.value.get<double>()));
|
||||
}
|
||||
|
||||
SECTION("binary formats stop where the end of an item is not known")
|
||||
{
|
||||
// a byte that begins no item
|
||||
const auto cbor = parse_binary_recovering({0x82, 0x01, 0x1C, 0x02}, json::input_format_t::cbor);
|
||||
CHECK(cbor.errors == 1);
|
||||
CHECK(cbor.value == json({1}));
|
||||
|
||||
// a key that is no item: the unused MessagePack byte, a CBOR break
|
||||
// in a map of known size, and the end of a BON8 container
|
||||
const auto msgpack = parse_binary_recovering({0x82, 0xA1, 'a', 0x01, 0xC1, 0x02}, json::input_format_t::msgpack);
|
||||
CHECK(msgpack.errors == 1);
|
||||
CHECK(msgpack.value == json({{"a", 1}}));
|
||||
const auto cbor_break = parse_binary_recovering({0xA2, 0x61, 'a', 0x01, 0xFF, 0x02}, json::input_format_t::cbor);
|
||||
CHECK(cbor_break.errors == 1);
|
||||
CHECK(cbor_break.value == json({{"a", 1}}));
|
||||
const auto bon8 = parse_binary_recovering({0x88, 0x61, 0x91, 0xFE}, json::input_format_t::bon8);
|
||||
CHECK(bon8.errors == 1);
|
||||
CHECK(bon8.value == json({{"a", 1}}));
|
||||
|
||||
// an indefinite-length string inside an indefinite-length string
|
||||
const auto nested = parse_binary_recovering({0x82, 0x01, 0x7F, 0x7F, 0x61, 'a', 0xFF, 0xFF}, json::input_format_t::cbor);
|
||||
CHECK(nested.errors == 1);
|
||||
CHECK(nested.value == json({1}));
|
||||
|
||||
// a BJData ndarray whose element type has no name: the object that
|
||||
// holds the ndarray was already begun
|
||||
const auto ndarray = parse_binary_recovering({'[', '$', 0x01, '#', '[', '$', 'i', '#', 'i', 2, 2, 3}, json::input_format_t::bjdata);
|
||||
CHECK(ndarray.errors == 1);
|
||||
CHECK(ndarray.balanced);
|
||||
CHECK(ndarray.value == json::object());
|
||||
|
||||
// a skipped member that the input ends in
|
||||
const auto truncated = parse_binary_recovering({0xA2, 0x01, 0x82, 0x01}, json::input_format_t::cbor);
|
||||
CHECK(truncated.errors == 2);
|
||||
CHECK(truncated.balanced);
|
||||
CHECK(truncated.value == json::object());
|
||||
|
||||
// a BSON element of an unknown type in a document whose size cannot be right
|
||||
const auto bson = parse_binary_recovering(bson_document({bson_element(0x10, "a", bson_int32(1)), bson_element(0x42, "x", bytes(3))}, -10), json::input_format_t::bson);
|
||||
CHECK(bson.errors == 1);
|
||||
CHECK(bson.value == json({{"a", 1}, {"x", nullptr}}));
|
||||
}
|
||||
|
||||
SECTION("changed bytes in binary input")
|
||||
{
|
||||
const json j = {{"a", {1, -2, {{"b", "c"}}, json::array()}}, {"d", {{"e", nullptr}, {"f", true}}}, {"g", 1.5}, {"h", json::binary({1, 2, 3})}, {"i", "\xC3\xA4"}};
|
||||
|
||||
const std::vector<std::pair<json::input_format_t, std::vector<std::uint8_t>>> encodings =
|
||||
{
|
||||
{json::input_format_t::cbor, json::to_cbor(j)},
|
||||
{json::input_format_t::msgpack, json::to_msgpack(j)},
|
||||
{json::input_format_t::ubjson, json::to_ubjson(j)},
|
||||
{json::input_format_t::ubjson, json::to_ubjson(j, true, true)},
|
||||
{json::input_format_t::bjdata, json::to_bjdata(j)},
|
||||
{json::input_format_t::bjdata, json::to_bjdata(j, true, true)},
|
||||
{json::input_format_t::bson, json::to_bson(j)},
|
||||
{json::input_format_t::bon8, json::to_bon8(j)},
|
||||
};
|
||||
const std::vector<std::uint8_t> replacements = {0x00, 0x01, 0x7F, 0x80, 0xC1, 0xD9, 0xE0, 0xF7, 0xFE, 0xFF};
|
||||
|
||||
for (const auto& encoding : encodings)
|
||||
{
|
||||
const auto format = encoding.first;
|
||||
const auto& original = encoding.second;
|
||||
CAPTURE(format)
|
||||
|
||||
std::vector<std::vector<std::uint8_t>> inputs;
|
||||
for (std::size_t position = 0; position < original.size(); ++position)
|
||||
{
|
||||
for (const auto replacement : replacements)
|
||||
{
|
||||
auto changed = original;
|
||||
changed[position] = replacement;
|
||||
inputs.push_back(changed);
|
||||
}
|
||||
auto removed = original;
|
||||
removed.erase(removed.begin() + static_cast<std::ptrdiff_t>(position));
|
||||
inputs.push_back(removed);
|
||||
}
|
||||
|
||||
for (const auto& input : inputs)
|
||||
{
|
||||
CAPTURE(input)
|
||||
const auto result = parse_binary_recovering(input, format);
|
||||
CHECK(result.balanced);
|
||||
CHECK(result.errors <= input.size() + 1);
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
// an error is reported exactly if reading into a JSON value
|
||||
// fails, and the first one is the same (under JSON_NOEXCEPTION,
|
||||
// that reading aborts instead of throwing)
|
||||
const auto message = binary_error_message(input, format);
|
||||
CHECK(result.ok == message.empty());
|
||||
if (!result.ok && result.errors < 100)
|
||||
{
|
||||
CHECK(result.messages.front() == message);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("JSON text")
|
||||
{
|
||||
// the parser stopped, but reported success
|
||||
json j;
|
||||
RecoveringParser sax(j);
|
||||
CHECK(!json::sax_parse("[1,2,3,]", &sax));
|
||||
CHECK(sax.errors == 1);
|
||||
CHECK(j == json({1, 2, 3}));
|
||||
}
|
||||
|
||||
SECTION("the SAX parsers of the library stop")
|
||||
{
|
||||
json _;
|
||||
CHECK(json::from_cbor(std::vector<std::uint8_t> {0x9F}, true, false).is_discarded());
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<std::uint8_t> {0x9F}), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&);
|
||||
CHECK(json::parse("[1,2,3,]", nullptr, false).is_discarded());
|
||||
CHECK(!json::accept("[1,2,3,]"));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
TEST_CASE("issue #5676 - SAX parsing of CBOR tags")
|
||||
{
|
||||
const json expected = json::binary({1, 2, 3}, 42);
|
||||
const auto cbor = json::to_cbor(expected);
|
||||
|
||||
nlohmann::detail::json_sax_acceptor<json> acceptor;
|
||||
CHECK_FALSE(json::sax_parse(cbor, &acceptor, json::input_format_t::cbor));
|
||||
CHECK_FALSE(json::sax_parse(cbor, &acceptor, json::input_format_t::cbor,
|
||||
true, false, false, json::cbor_tag_handler_t::error));
|
||||
|
||||
CHECK(json::sax_parse(cbor, &acceptor, json::input_format_t::cbor,
|
||||
true, false, false, json::cbor_tag_handler_t::ignore));
|
||||
|
||||
json parsed;
|
||||
nlohmann::detail::json_sax_dom_parser<json, nlohmann::detail::string_input_adapter_type> sax(parsed);
|
||||
CHECK(json::sax_parse(cbor, &sax, json::input_format_t::cbor,
|
||||
true, false, false, json::cbor_tag_handler_t::store));
|
||||
CHECK(parsed == expected);
|
||||
|
||||
json iterator_parsed;
|
||||
nlohmann::detail::json_sax_dom_parser<json, nlohmann::detail::string_input_adapter_type> iterator_sax(iterator_parsed);
|
||||
CHECK(json::sax_parse(cbor.begin(), cbor.end(), &iterator_sax, json::input_format_t::cbor,
|
||||
true, false, false, json::cbor_tag_handler_t::store));
|
||||
CHECK(iterator_parsed == expected);
|
||||
|
||||
#ifndef JSON_TEST_DEPRECATED_FUNCTIONS_DELETED
|
||||
json span_parsed;
|
||||
nlohmann::detail::json_sax_dom_parser<json, nlohmann::detail::string_input_adapter_type> span_sax(span_parsed);
|
||||
CHECK(json::sax_parse(nlohmann::detail::span_input_adapter(cbor.data(), cbor.size()), &span_sax,
|
||||
json::input_format_t::cbor, true, false, false, json::cbor_tag_handler_t::store));
|
||||
CHECK(span_parsed == expected);
|
||||
#endif
|
||||
|
||||
const std::string text = "null";
|
||||
CHECK(json::sax_parse(text, &acceptor, json::input_format_t::json,
|
||||
true, false, false, json::cbor_tag_handler_t::store));
|
||||
CHECK(json::sax_parse(text.begin(), text.end(), &acceptor, json::input_format_t::json,
|
||||
true, false, false, json::cbor_tag_handler_t::store));
|
||||
#ifndef JSON_TEST_DEPRECATED_FUNCTIONS_DELETED
|
||||
CHECK(json::sax_parse(nlohmann::detail::span_input_adapter(text.data(), text.size()), &acceptor,
|
||||
json::input_format_t::json, true, false, false, json::cbor_tag_handler_t::store));
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
TEST_CASE("MessagePack SAX aborts")
|
||||
{
|
||||
SECTION("start_array(len)")
|
||||
{
|
||||
std::vector<uint8_t> const v = {0x93, 0x01, 0x02, 0x03};
|
||||
SaxCountdown scp(0);
|
||||
CHECK(!json::sax_parse(v, &scp, json::input_format_t::msgpack));
|
||||
}
|
||||
|
||||
SECTION("start_object(len)")
|
||||
{
|
||||
std::vector<uint8_t> const v = {0x81, 0xa3, 0x66, 0x6F, 0x6F, 0xc2};
|
||||
SaxCountdown scp(0);
|
||||
CHECK(!json::sax_parse(v, &scp, json::input_format_t::msgpack));
|
||||
}
|
||||
|
||||
SECTION("key()")
|
||||
{
|
||||
std::vector<uint8_t> const v = {0x81, 0xa3, 0x66, 0x6F, 0x6F, 0xc2};
|
||||
SaxCountdown scp(1);
|
||||
CHECK(!json::sax_parse(v, &scp, json::input_format_t::msgpack));
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("issue #5405 - a user-defined SAX consumer is unaffected by the internal DOM reserve optimization")
|
||||
{
|
||||
// the reserve() call is local to json_sax_dom_parser / json_sax_dom_callback_parser;
|
||||
// a custom SAX consumer that does not touch a DOM array sees identical events
|
||||
json j = json::array();
|
||||
for (int i = 0; i < 100; ++i)
|
||||
{
|
||||
j.push_back(i);
|
||||
}
|
||||
const auto packed = json::to_msgpack(j);
|
||||
|
||||
SaxCountdown scp(1000000); // large enough to never trigger an abort
|
||||
CHECK(json::sax_parse(packed, &scp, json::input_format_t::msgpack));
|
||||
}
|
||||
|
||||
TEST_CASE("MessagePack nesting does not consume the call stack - SAX interface")
|
||||
{
|
||||
// see the test case of the same name in unit-msgpack.cpp (#5104)
|
||||
std::vector<uint8_t> input(300000, 0x91);
|
||||
input.push_back(0x01); // innermost value
|
||||
|
||||
SaxCountdown accept_all(600001);
|
||||
CHECK(json::sax_parse(input, &accept_all, json::input_format_t::msgpack));
|
||||
}
|
||||
|
||||
TEST_CASE("MessagePack SAX parsing stops at every event")
|
||||
{
|
||||
// Containers are opened and closed by the loop that reads them; a SAX
|
||||
// handler that rejects any event - including the end of a nested
|
||||
// container - must stop the parse right there.
|
||||
const auto count_events = [](const std::vector<std::uint8_t>& input)
|
||||
{
|
||||
int events = 0;
|
||||
while (true)
|
||||
{
|
||||
SaxCountdown scp(events);
|
||||
if (json::sax_parse(input, &scp, json::input_format_t::msgpack))
|
||||
{
|
||||
return events;
|
||||
}
|
||||
++events;
|
||||
REQUIRE(events < 1000);
|
||||
}
|
||||
};
|
||||
|
||||
// 20 events: every container kind closes inside another one
|
||||
const json j = json::parse(R"({"a": [1, {"b": []}], "c": {"d": [[2]]}})");
|
||||
CHECK(count_events(json::to_msgpack(j)) == 20);
|
||||
}
|
||||
|
||||
DOCTEST_CLANG_SUPPRESS_WARNING_POP
|
||||
Reference in new issue
Block a user