mirror of
https://github.com/nlohmann/json.git
synced 2026-09-29 19:20:30 +00:00
Compare commits
8
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a491cc7663 | ||
|
|
6536c1b869 | ||
|
|
038f448dec | ||
|
|
bf6b5b4719 | ||
|
|
1c7041908d | ||
|
|
20fd4c6a8b | ||
|
|
f57de1aa0a | ||
|
|
401af52511 |
@@ -52,6 +52,7 @@ labels:
|
||||
- "single_include/nlohmann/json_view\\.hpp"
|
||||
- "tests/src/unit-json_view.*"
|
||||
- "tests/src/fuzzer-parse_json_view\\.cpp"
|
||||
- "tests/benchmarks/json_view/.*"
|
||||
- "tools/amalgamate/config_json_view\\.json"
|
||||
- "docs/mkdocs/docs/features/json_view\\.md"
|
||||
- "docs/mkdocs/docs/api/basic_json_(document|view)/.*"
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
name: "json_view benchmarks"
|
||||
|
||||
# On demand only: runs the comparison of json_view with yyjson, simdjson, and
|
||||
# Boost.JSON (tests/benchmarks/json_view/compare.py) on GitHub-hosted runners,
|
||||
# for numbers from x86-64 and AArch64 Linux. It runs when started by hand, or
|
||||
# when a pull request gets the label "benchmark" (on both architectures, with
|
||||
# GCC and the default settings). Shared runners are noisy: the results show
|
||||
# where json_view stands, but published numbers need a quiet machine (see
|
||||
# tests/benchmarks/json_view/README.md).
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [labeled]
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
runner:
|
||||
description: "Runner image"
|
||||
type: choice
|
||||
options:
|
||||
- ubuntu-24.04
|
||||
- ubuntu-24.04-arm
|
||||
default: ubuntu-24.04
|
||||
compiler:
|
||||
description: "Compiler"
|
||||
type: choice
|
||||
options:
|
||||
- g++
|
||||
- clang++
|
||||
default: g++
|
||||
native:
|
||||
description: "Compile for the runner's CPU (-march=native)"
|
||||
type: boolean
|
||||
default: false
|
||||
rounds:
|
||||
description: "Rounds of bench_view"
|
||||
type: number
|
||||
default: 30
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
compare:
|
||||
if: github.event_name == 'workflow_dispatch' || github.event.label.name == 'benchmark'
|
||||
strategy:
|
||||
matrix:
|
||||
runner: ${{ fromJSON(github.event_name == 'workflow_dispatch' && format('["{0}"]', inputs.runner) || '["ubuntu-24.04", "ubuntu-24.04-arm"]') }}
|
||||
runs-on: ${{ matrix.runner }}
|
||||
steps:
|
||||
- name: Harden Runner
|
||||
uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1
|
||||
with:
|
||||
egress-policy: audit
|
||||
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Download test data
|
||||
run: |
|
||||
cmake -S . -B build -DJSON_BuildTests=On
|
||||
cmake --build build --target download_test_data
|
||||
|
||||
- name: Run the comparison
|
||||
env:
|
||||
CXX: ${{ inputs.compiler || 'g++' }}
|
||||
CC: ${{ inputs.compiler == 'clang++' && 'clang' || 'gcc' }}
|
||||
ROUNDS: ${{ inputs.rounds || 30 }}
|
||||
NATIVE: ${{ inputs.native && '--native' || '' }}
|
||||
run: python3 tests/benchmarks/json_view/compare.py --data build/test_files --download --rounds "$ROUNDS" $NATIVE
|
||||
|
||||
- name: Summary
|
||||
run: cat tests/benchmarks/json_view/results/*.md >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: json_view-benchmarks-${{ matrix.runner }}-${{ inputs.compiler || 'g++' }}
|
||||
path: tests/benchmarks/json_view/results/
|
||||
@@ -67,6 +67,7 @@ cc_library(
|
||||
"include/nlohmann/detail/string_utils.hpp",
|
||||
"include/nlohmann/detail/value_t.hpp",
|
||||
"include/nlohmann/detail/view/builder.hpp",
|
||||
"include/nlohmann/detail/view/compare.hpp",
|
||||
"include/nlohmann/detail/view/document_data.hpp",
|
||||
"include/nlohmann/detail/view/errors.hpp",
|
||||
"include/nlohmann/detail/view/input.hpp",
|
||||
|
||||
@@ -178,6 +178,8 @@ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::number_token
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator bool', 'Method', 'api/basic_json_view/operator_bool/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator<<', 'Operator', 'api/basic_json_view/operator_ltlt/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator[]', 'Operator', 'api/basic_json_view/operator[]/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator==', 'Operator', 'api/basic_json_view/operator_eq/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator!=', 'Operator', 'api/basic_json_view/operator_ne/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::size', 'Method', 'api/basic_json_view/size/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::source_offset', 'Method', 'api/basic_json_view/source_offset/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::type', 'Method', 'api/basic_json_view/type/index.html');
|
||||
|
||||
@@ -166,6 +166,8 @@ Linear.
|
||||
|
||||
- [operator!=](operator_ne.md) compare for inequality
|
||||
- [operator<=>](operator_spaceship.md) comparison: 3-way (C++20)
|
||||
- [basic_json_view::operator==](../basic_json_view/operator_eq.md) - the same comparison on a zero-copy view, without
|
||||
building a `basic_json` value for it
|
||||
|
||||
## Version history
|
||||
|
||||
|
||||
@@ -89,6 +89,12 @@ Linear.
|
||||
--8<-- "examples/operator__notequal__nullptr_t.output"
|
||||
```
|
||||
|
||||
## See also
|
||||
|
||||
- [operator==](operator_eq.md) compare for equality
|
||||
- [basic_json_view::operator!=](../basic_json_view/operator_ne.md) - the same comparison on a zero-copy view, without
|
||||
building a `basic_json` value for it
|
||||
|
||||
## Version history
|
||||
|
||||
1. Added in version 1.0.0. Added C++20 member functions in version 3.11.0. Changed in version 3.13.0 to remove
|
||||
|
||||
@@ -20,11 +20,12 @@ Moving the document itself does not invalidate its views: the index is heap-allo
|
||||
`basic_json_document` object.
|
||||
|
||||
`basic_json_view` provides the read-only part of the `BasicJsonType` interface: the type-inspection functions, element
|
||||
access, lookup, iteration, and conversion -- [`get<T>()`](get.md), [`get_string()`](get_string.md),
|
||||
access, lookup, iteration, conversion, and comparison -- [`get<T>()`](get.md), [`get_string()`](get_string.md),
|
||||
[`number_token()`](number_token.md), and [`materialize()`](materialize.md) to build the `BasicJsonType` value of a
|
||||
subtree on demand. [`operator[]`](operator%5B%5D.md), [`at`](at.md), [`contains`](contains.md), and
|
||||
[`value`](value.md) also accept a [`json_pointer`](../json_pointer/index.md). It does not (yet) provide
|
||||
comparison.
|
||||
[`value`](value.md) also accept a [`json_pointer`](../json_pointer/index.md). [`operator==`](operator_eq.md) and
|
||||
[`operator!=`](operator_ne.md) compare two views, or a view and a `BasicJsonType` value, without ever building a
|
||||
`BasicJsonType` value for a view; no ordering comparison (`#!cpp operator<`) is provided.
|
||||
|
||||
## Template parameters
|
||||
|
||||
@@ -107,6 +108,11 @@ comparison.
|
||||
- [**number_token**](number_token.md) - get a number's token text without a copy
|
||||
- [**materialize**](materialize.md) - build the `BasicJsonType` value of this subtree
|
||||
|
||||
### Comparison
|
||||
|
||||
- [**operator==**](operator_eq.md) - comparison: equal
|
||||
- [**operator!=**](operator_ne.md) - comparison: not equal
|
||||
|
||||
### Serialization
|
||||
|
||||
- [**dump**](dump.md) - serialize to a JSON-formatted string
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
# <small>nlohmann::basic_json_view::</small>operator==
|
||||
|
||||
```cpp
|
||||
// (1)
|
||||
bool operator==(const basic_json_view& lhs, const basic_json_view& rhs);
|
||||
|
||||
// (2)
|
||||
bool operator==(const basic_json_view& lhs, const BasicJsonType& rhs);
|
||||
bool operator==(const BasicJsonType& lhs, const basic_json_view& rhs);
|
||||
```
|
||||
|
||||
1. Compares two views for equality: whether the values [`BasicJsonType::parse()`](../basic_json/parse.md) would
|
||||
produce for `lhs` and `rhs` are equal, according to `BasicJsonType`'s [`operator==`](../basic_json/operator_eq.md).
|
||||
2. Compares a view and a `BasicJsonType` value for equality, in either order: whether the value `parse()` would
|
||||
produce for the view and the other operand are equal, according to `BasicJsonType`'s
|
||||
[`operator==`](../basic_json/operator_eq.md).
|
||||
|
||||
Neither overload builds a `BasicJsonType` value for a view to do the comparison (see [Notes](#notes) below). Numbers
|
||||
compare by value across their types (`#!cpp 1 == 1.0`), and an object compares by its members, with duplicate keys
|
||||
resolved exactly as `parse()` resolves them -- the last value, at the position of the first occurrence of the key.
|
||||
|
||||
## Parameters
|
||||
|
||||
`lhs` (in)
|
||||
: first value to consider
|
||||
|
||||
`rhs` (in)
|
||||
: second value to consider
|
||||
|
||||
## Return value
|
||||
|
||||
whether the values `lhs` and `rhs` are equal
|
||||
|
||||
## Exception safety
|
||||
|
||||
Strong exception safety: if an exception is thrown, there are no changes to either operand, or to the document(s) a
|
||||
view refers to.
|
||||
|
||||
## Exceptions
|
||||
|
||||
May throw `#!cpp std::bad_alloc`. Unlike the other comparison and most other `basic_json_view` functions,
|
||||
`operator==` is not `#!cpp noexcept`: resolving an object's members needs a temporary array to sort them by key (see
|
||||
[Complexity](#complexity) below), and that allocation can fail.
|
||||
|
||||
## Complexity
|
||||
|
||||
Linear in the size of the compared values: every number, string, array element, and object member is visited at most
|
||||
once, and the walk is iterative, so the nesting depth it can compare is limited by available memory only, not by the
|
||||
call stack (as for [`materialize()`](materialize.md)). Resolving an object's members takes an additional O(n log n)
|
||||
in the number of members at that level, since they are sorted by key to detect and resolve duplicates before being
|
||||
compared. Two arrays of different [`size()`](size.md) are rejected without visiting either one's elements.
|
||||
|
||||
## Notes
|
||||
|
||||
Only a single number, boolean, or `#!cpp null` value is ever materialized into a `BasicJsonType`, to reuse its
|
||||
`operator==` -- for numbers, so that values written differently in the source text but equal in value (e.g. an
|
||||
integer and a floating-point literal) still compare equal, following the same rules `BasicJsonType` does for special
|
||||
values such as `#!cpp NaN`. Constructing one of these scalars never allocates. Strings are compared directly, without
|
||||
allocating, either from the source text on both sides or, for overload 2, against `BasicJsonType`'s own string.
|
||||
Arrays and objects are never materialized at all; only their elements or members are visited, one pair at a time.
|
||||
|
||||
!!! info "How objects are compared"
|
||||
|
||||
For a [`json_view`](../json_view.md) (`BasicJsonType::object_t` is `#!cpp std::map`), members are compared by
|
||||
key, regardless of the order they appear in the source text. For an
|
||||
[`ordered_json_view`](../ordered_json_view.md) (`object_t` is `ordered_map`), they are compared in the order
|
||||
they occur, so the very same two objects with their members reordered can compare equal as `json_view`s but not
|
||||
as `ordered_json_view`s. This is exactly how [`json`](../json.md) and [`ordered_json`](../ordered_json.md)
|
||||
compare, see ["Comparing different `basic_json` specializations"](../basic_json/operator_eq.md#notes).
|
||||
|
||||
!!! info "Discarded views"
|
||||
|
||||
A [discarded](is_discarded.md) view compares the same way a discarded `BasicJsonType` value does, which is
|
||||
governed by
|
||||
[`JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../macros/json_use_legacy_discarded_value_comparison.md): by
|
||||
default, a discarded view is never equal to anything, not even another discarded view.
|
||||
|
||||
No ordering comparison (`#!cpp operator<`) is provided for `basic_json_view`; [`materialize()`](materialize.md) is
|
||||
the way to get a `BasicJsonType` value that supports it.
|
||||
|
||||
## Examples
|
||||
|
||||
??? example
|
||||
|
||||
The example below checks whether a newly received configuration differs from the previous one, and whether a
|
||||
received document matches what a test expects -- directly on views, without ever materializing a `BasicJsonType`
|
||||
value for either side.
|
||||
|
||||
```cpp
|
||||
--8<-- "examples/basic_json_view__operator_eq.cpp"
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```json
|
||||
--8<-- "examples/basic_json_view__operator_eq.output"
|
||||
```
|
||||
|
||||
## See also
|
||||
|
||||
- [operator!=](operator_ne.md) - compare for inequality
|
||||
- [materialize](materialize.md) - build a `BasicJsonType` value, e.g. to keep comparing after the document is gone
|
||||
- [`BasicJsonType::operator==`](../basic_json/operator_eq.md) - the corresponding function of `basic_json`
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.13.0.
|
||||
@@ -0,0 +1,82 @@
|
||||
# <small>nlohmann::basic_json_view::</small>operator!=
|
||||
|
||||
```cpp
|
||||
// (1)
|
||||
bool operator!=(const basic_json_view& lhs, const basic_json_view& rhs);
|
||||
|
||||
// (2)
|
||||
bool operator!=(const basic_json_view& lhs, const BasicJsonType& rhs);
|
||||
bool operator!=(const BasicJsonType& lhs, const basic_json_view& rhs);
|
||||
```
|
||||
|
||||
1. Compares two views for inequality. Returns `#!cpp !(lhs == rhs)`, see [operator==](operator_eq.md).
|
||||
2. Compares a view and a `BasicJsonType` value for inequality, in either order. Returns `#!cpp !(lhs == rhs)` (or,
|
||||
for the reversed order, `#!cpp !(rhs == lhs)`), see [operator==](operator_eq.md).
|
||||
|
||||
Since `operator!=` is defined as the negation of [`operator==`](operator_eq.md), it follows the same rules for
|
||||
special cases: for instance, since a [discarded](is_discarded.md) view is never equal to anything by default (see
|
||||
[operator=='s Notes](operator_eq.md#notes)), it is never *unequal* to anything either -- `#!cpp discarded != discarded`
|
||||
is also `#!cpp false`, exactly as for a discarded `BasicJsonType` value.
|
||||
|
||||
## Parameters
|
||||
|
||||
`lhs` (in)
|
||||
: first value to consider
|
||||
|
||||
`rhs` (in)
|
||||
: second value to consider
|
||||
|
||||
## Return value
|
||||
|
||||
whether the values `lhs` and `rhs` are not equal
|
||||
|
||||
## Exception safety
|
||||
|
||||
Strong exception safety: if an exception is thrown, there are no changes to either operand, or to the document(s) a
|
||||
view refers to.
|
||||
|
||||
## Exceptions
|
||||
|
||||
May throw `#!cpp std::bad_alloc`, propagated from [`operator==`](operator_eq.md#exceptions). Unlike most other
|
||||
`basic_json_view` functions, `operator!=` is not `#!cpp noexcept`.
|
||||
|
||||
## Complexity
|
||||
|
||||
Linear, as [`operator==`](operator_eq.md#complexity).
|
||||
|
||||
## Notes
|
||||
|
||||
See the [Notes](operator_eq.md#notes) of `operator==` -- in particular for how an object's members are compared
|
||||
(order matters for [`ordered_json_view`](../ordered_json_view.md) but not for [`json_view`](../json_view.md)) and
|
||||
for how discarded views compare.
|
||||
|
||||
No ordering comparison (`#!cpp operator<`) is provided for `basic_json_view`; [`materialize()`](materialize.md) is
|
||||
the way to get a `BasicJsonType` value that supports it.
|
||||
|
||||
## Examples
|
||||
|
||||
??? example
|
||||
|
||||
The example below asserts, as a test would, that a received document differs from an unwanted value, and shows
|
||||
that -- as for [`json`](../json.md)/[`ordered_json`](../ordered_json.md) -- reordering an object's members is
|
||||
detected as a difference for an `ordered_json_view` but not for a `json_view`.
|
||||
|
||||
```cpp
|
||||
--8<-- "examples/basic_json_view__operator_ne.cpp"
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```json
|
||||
--8<-- "examples/basic_json_view__operator_ne.output"
|
||||
```
|
||||
|
||||
## See also
|
||||
|
||||
- [operator==](operator_eq.md) - compare for equality
|
||||
- [materialize](materialize.md) - build a `BasicJsonType` value, e.g. to keep comparing after the document is gone
|
||||
- [`BasicJsonType::operator!=`](../basic_json/operator_ne.md) - the corresponding function of `basic_json`
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.13.0.
|
||||
@@ -0,0 +1,30 @@
|
||||
#include <iostream>
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
using json_document = nlohmann::json_document;
|
||||
using json = nlohmann::json;
|
||||
|
||||
int main()
|
||||
{
|
||||
// two snapshots of a polled configuration endpoint -- compare them
|
||||
// directly as views, without ever building a nlohmann::json value for
|
||||
// either one
|
||||
const json_document previous = json_document::parse(
|
||||
R"({"name": "cache", "port": 6379, "timeout": 30})");
|
||||
const json_document current = json_document::parse(
|
||||
R"({"port": 6379.0, "timeout": 30, "name": "cache"})");
|
||||
|
||||
// same members, reordered, and 6379 written as a float -- operator==
|
||||
// treats them the same way BasicJsonType::operator== would
|
||||
std::cout << std::boolalpha << (previous.root() == current.root()) << '\n';
|
||||
|
||||
// an actually changed value is detected the same way
|
||||
const json_document changed = json_document::parse(
|
||||
R"({"name": "cache", "port": 6380, "timeout": 30})");
|
||||
std::cout << (previous.root() == changed.root()) << '\n';
|
||||
|
||||
// comparing a view directly against an expected json value -- handy in a
|
||||
// test, without materializing the received document at all
|
||||
const json expected = {{"name", "cache"}, {"port", 6379}, {"timeout", 30}};
|
||||
std::cout << (previous.root() == expected) << '\n';
|
||||
}
|
||||
@@ -0,0 +1,3 @@
|
||||
true
|
||||
false
|
||||
true
|
||||
@@ -0,0 +1,28 @@
|
||||
#include <iostream>
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
using json_document = nlohmann::json_document;
|
||||
using ordered_json_document = nlohmann::ordered_json_document;
|
||||
using json = nlohmann::json;
|
||||
|
||||
int main()
|
||||
{
|
||||
// assert, as a test would, that a received document differs from an
|
||||
// unwanted shape -- without ever materializing it into a json value just
|
||||
// to compare
|
||||
const json_document received = json_document::parse(
|
||||
R"({"status": "ok", "code": 200})");
|
||||
const json unwanted = {{"status", "error"}, {"code", 500}};
|
||||
std::cout << std::boolalpha << (received.root() != unwanted) << '\n';
|
||||
|
||||
// json (std::map) compares object members regardless of order ...
|
||||
const json_document a = json_document::parse(R"({"a": 1, "b": 2})");
|
||||
const json_document b = json_document::parse(R"({"b": 2, "a": 1})");
|
||||
std::cout << (a.root() != b.root()) << '\n';
|
||||
|
||||
// ... but ordered_json (ordered_map) compares them in the order they
|
||||
// appear, so the very same reordering is detected as a difference
|
||||
const ordered_json_document oa = ordered_json_document::parse(R"({"a": 1, "b": 2})");
|
||||
const ordered_json_document ob = ordered_json_document::parse(R"({"b": 2, "a": 1})");
|
||||
std::cout << (oa.root() != ob.root()) << '\n';
|
||||
}
|
||||
@@ -0,0 +1,3 @@
|
||||
true
|
||||
false
|
||||
true
|
||||
@@ -139,7 +139,11 @@ whenever any of the other conditions above was not met.
|
||||
element access and lookup functions never carry the JSON Pointer path `JSON_DIAGNOSTICS` would otherwise add: the
|
||||
view has no `basic_json` value to point at, so the exception is created without one, regardless of how
|
||||
`BasicJsonType` was built.
|
||||
- **Comparison is not (yet) provided** by `basic_json_view`. For now,
|
||||
- **Ordering comparisons are not provided** by `basic_json_view` -- there is no `#!cpp operator<`.
|
||||
[`operator==`](../api/basic_json_view/operator_eq.md) and [`operator!=`](../api/basic_json_view/operator_ne.md) are
|
||||
provided, though: two views, or a view and a `BasicJsonType` value, compare equal exactly when
|
||||
[`materialize()`](../api/basic_json_view/materialize.md) or [`parse()`](../api/basic_json/parse.md) would produce
|
||||
equal values for them, without ever building a tree to do it. For ordering, too,
|
||||
[`materialize()`](../api/basic_json_view/materialize.md) is the way to get a value you can compare.
|
||||
|
||||
## Getting values out without copying
|
||||
|
||||
@@ -281,6 +281,8 @@ nav:
|
||||
- 'operator bool': api/basic_json_view/operator_bool.md
|
||||
- 'operator<<': api/basic_json_view/operator_ltlt.md
|
||||
- 'operator[]': api/basic_json_view/operator[].md
|
||||
- 'operator==': api/basic_json_view/operator_eq.md
|
||||
- 'operator!=': api/basic_json_view/operator_ne.md
|
||||
- 'size': api/basic_json_view/size.md
|
||||
- 'source_offset': api/basic_json_view/source_offset.md
|
||||
- 'type': api/basic_json_view/type.md
|
||||
|
||||
@@ -0,0 +1,317 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <algorithm> // sort, stable_sort
|
||||
#include <cstddef> // size_t
|
||||
#include <string> // string
|
||||
#include <utility> // move, pair
|
||||
#include <vector> // vector
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
namespace view
|
||||
{
|
||||
|
||||
// Equality of views, and of views with basic_json values, with the semantics
|
||||
// of basic_json's operator== applied to the values parse() would produce:
|
||||
// numbers compare by value across their types, an object is compared by its
|
||||
// members with duplicate keys resolved as parse() resolves them (the last
|
||||
// value, at the position of the first occurrence), and in document order if
|
||||
// the object type keeps an order (ordered_json), by key otherwise.
|
||||
|
||||
/// one side of a comparison: a view
|
||||
template<typename BasicJsonType, typename View>
|
||||
class view_side
|
||||
{
|
||||
public:
|
||||
using string_view_t = typename View::string_view_t;
|
||||
|
||||
explicit view_side(const View& v) noexcept
|
||||
: m_view(v)
|
||||
{}
|
||||
|
||||
value_t type() const noexcept
|
||||
{
|
||||
return m_view.type();
|
||||
}
|
||||
|
||||
std::size_t size() const noexcept
|
||||
{
|
||||
return m_view.size();
|
||||
}
|
||||
|
||||
string_view_t string() const
|
||||
{
|
||||
return m_view.get_string();
|
||||
}
|
||||
|
||||
/// a number, boolean, or null as a basic_json value (no allocation)
|
||||
BasicJsonType scalar() const
|
||||
{
|
||||
switch (m_view.type())
|
||||
{
|
||||
case value_t::number_integer:
|
||||
return BasicJsonType(m_view.template get<typename BasicJsonType::number_integer_t>());
|
||||
case value_t::number_unsigned:
|
||||
return BasicJsonType(m_view.template get<typename BasicJsonType::number_unsigned_t>());
|
||||
case value_t::number_float:
|
||||
return BasicJsonType(m_view.template get<typename BasicJsonType::number_float_t>());
|
||||
case value_t::boolean:
|
||||
return BasicJsonType(m_view.template get<bool>());
|
||||
case value_t::null:
|
||||
case value_t::object:
|
||||
case value_t::array:
|
||||
case value_t::string:
|
||||
case value_t::binary:
|
||||
case value_t::discarded:
|
||||
default:
|
||||
return BasicJsonType(nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void elements(std::vector<view_side>& out) const
|
||||
{
|
||||
out.reserve(m_view.size());
|
||||
for (const View e : m_view)
|
||||
{
|
||||
out.emplace_back(e);
|
||||
}
|
||||
}
|
||||
|
||||
/// the members as parse() keeps them: one per key, the last value at the
|
||||
/// position of the first occurrence; in that order, or sorted by key
|
||||
void members(std::vector<std::pair<string_view_t, view_side>>& out, bool ordered) const
|
||||
{
|
||||
struct member
|
||||
{
|
||||
string_view_t key;
|
||||
View value;
|
||||
std::size_t position;
|
||||
};
|
||||
std::vector<member> all;
|
||||
all.reserve(m_view.size());
|
||||
std::size_t position = 0;
|
||||
for (auto it = m_view.begin(); it != m_view.end(); ++it)
|
||||
{
|
||||
all.push_back(member{it.key(), it.value(), position++});
|
||||
}
|
||||
std::stable_sort(all.begin(), all.end(), [](const member & a, const member & b)
|
||||
{
|
||||
return a.key < b.key;
|
||||
});
|
||||
std::vector<member> unique;
|
||||
unique.reserve(all.size());
|
||||
for (std::size_t i = 0; i < all.size();)
|
||||
{
|
||||
std::size_t last = i;
|
||||
while (last + 1 < all.size() && all[last + 1].key == all[i].key)
|
||||
{
|
||||
++last;
|
||||
}
|
||||
unique.push_back(member{all[i].key, all[last].value, all[i].position});
|
||||
i = last + 1;
|
||||
}
|
||||
if (ordered)
|
||||
{
|
||||
std::sort(unique.begin(), unique.end(), [](const member & a, const member & b)
|
||||
{
|
||||
return a.position < b.position;
|
||||
});
|
||||
}
|
||||
out.reserve(unique.size());
|
||||
for (const member& m : unique)
|
||||
{
|
||||
out.emplace_back(m.key, view_side(m.value));
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
View m_view;
|
||||
};
|
||||
|
||||
/// the other side of a comparison: a basic_json value
|
||||
template<typename BasicJsonType, typename StringView>
|
||||
class json_side
|
||||
{
|
||||
public:
|
||||
using string_view_t = StringView;
|
||||
|
||||
explicit json_side(const BasicJsonType& j) noexcept
|
||||
: m_json(&j)
|
||||
{}
|
||||
|
||||
value_t type() const noexcept
|
||||
{
|
||||
return m_json->type();
|
||||
}
|
||||
|
||||
std::size_t size() const noexcept
|
||||
{
|
||||
return m_json->size();
|
||||
}
|
||||
|
||||
string_view_t string() const
|
||||
{
|
||||
const auto& s = m_json->template get_ref<const typename BasicJsonType::string_t&>();
|
||||
return string_view_t(s.data(), s.size());
|
||||
}
|
||||
|
||||
BasicJsonType scalar() const
|
||||
{
|
||||
return *m_json;
|
||||
}
|
||||
|
||||
void elements(std::vector<json_side>& out) const
|
||||
{
|
||||
out.reserve(m_json->size());
|
||||
for (const auto& e : *m_json)
|
||||
{
|
||||
out.emplace_back(e);
|
||||
}
|
||||
}
|
||||
|
||||
void members(std::vector<std::pair<string_view_t, json_side>>& out, bool ordered) const
|
||||
{
|
||||
out.reserve(m_json->size());
|
||||
for (auto it = m_json->cbegin(); it != m_json->cend(); ++it)
|
||||
{
|
||||
out.emplace_back(string_view_t(it.key().data(), it.key().size()), json_side(it.value()));
|
||||
}
|
||||
if (!ordered)
|
||||
{
|
||||
std::sort(out.begin(), out.end(), [](const std::pair<string_view_t, json_side>& a, const std::pair<string_view_t, json_side>& b)
|
||||
{
|
||||
return a.first < b.first;
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
const BasicJsonType* m_json;
|
||||
};
|
||||
|
||||
/// whether two sides are equal; iterative, so that the nesting depth is
|
||||
/// limited by memory only
|
||||
template<typename BasicJsonType, typename A, typename B>
|
||||
bool equal(const A& a0, const B& b0)
|
||||
{
|
||||
using string_view_t = typename A::string_view_t;
|
||||
const bool ordered = is_ordered_map<typename BasicJsonType::object_t>::value;
|
||||
|
||||
struct frame
|
||||
{
|
||||
std::vector<A> elements_a{};
|
||||
std::vector<B> elements_b{};
|
||||
std::vector<std::pair<string_view_t, A>> members_a{};
|
||||
std::vector<std::pair<string_view_t, B>> members_b{};
|
||||
bool object = false;
|
||||
std::size_t next = 0;
|
||||
};
|
||||
std::vector<frame> stack;
|
||||
A a = a0;
|
||||
B b = b0;
|
||||
for (;;)
|
||||
{
|
||||
const value_t ta = a.type();
|
||||
const value_t tb = b.type();
|
||||
const bool numbers = (ta == value_t::number_integer || ta == value_t::number_unsigned || ta == value_t::number_float)
|
||||
&& (tb == value_t::number_integer || tb == value_t::number_unsigned || tb == value_t::number_float);
|
||||
if (ta == value_t::discarded || tb == value_t::discarded)
|
||||
{
|
||||
// basic_json decides (JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON)
|
||||
if (ta != tb || !(BasicJsonType(value_t::discarded) == BasicJsonType(value_t::discarded)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if (!numbers && ta != tb)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if (ta == value_t::string)
|
||||
{
|
||||
if (!(a.string() == b.string()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if (ta == value_t::array || ta == value_t::object)
|
||||
{
|
||||
if (a.size() != b.size() && ta == value_t::array)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
frame f;
|
||||
f.object = ta == value_t::object;
|
||||
if (f.object)
|
||||
{
|
||||
a.members(f.members_a, ordered);
|
||||
b.members(f.members_b, ordered);
|
||||
if (f.members_a.size() != f.members_b.size())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
a.elements(f.elements_a);
|
||||
b.elements(f.elements_b);
|
||||
}
|
||||
stack.push_back(std::move(f));
|
||||
}
|
||||
else if (!(a.scalar() == b.scalar())) // numbers (also of different types), null, boolean
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// the next pair of values
|
||||
for (;;)
|
||||
{
|
||||
if (stack.empty())
|
||||
{
|
||||
return true;
|
||||
}
|
||||
frame& f = stack.back();
|
||||
const std::size_t count = f.object ? f.members_a.size() : f.elements_a.size();
|
||||
if (f.next == count)
|
||||
{
|
||||
stack.pop_back();
|
||||
continue;
|
||||
}
|
||||
if (f.object)
|
||||
{
|
||||
if (!(f.members_a[f.next].first == f.members_b[f.next].first))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
a = f.members_a[f.next].second;
|
||||
b = f.members_b[f.next].second;
|
||||
}
|
||||
else
|
||||
{
|
||||
a = f.elements_a[f.next];
|
||||
b = f.elements_b[f.next];
|
||||
}
|
||||
++f.next;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -81,7 +81,7 @@ BasicJsonType materialize(const document_data& d, const node* n)
|
||||
++n;
|
||||
break;
|
||||
case value_t::number_float:
|
||||
sax.number_float(float_value<typename BasicJsonType::number_float_t>(d.str(*n), *n), no_token);
|
||||
sax.number_float(float_value<typename BasicJsonType::number_float_t>(d, *n), no_token);
|
||||
++n;
|
||||
break;
|
||||
case value_t::boolean:
|
||||
|
||||
@@ -8,12 +8,19 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <cstring> // memcpy
|
||||
#include <string> // string
|
||||
#include <type_traits> // integral_constant, is_same
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/detail/view/document_data.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
#include <nlohmann/detail/view/node.hpp>
|
||||
#include <nlohmann/detail/view/scan.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -68,6 +75,96 @@ NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n)
|
||||
return v;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the double of a float token with at most 19 digits, from its layout
|
||||
|
||||
The digit layout recorded while parsing says where the integer digits, the
|
||||
fraction digits, and the exponent are, so the digits are read eight at a
|
||||
time without scanning. The result is correctly rounded (Clinger's fast path
|
||||
where both operands are exact, else the Eisel-Lemire algorithm, which needs
|
||||
no fallback for up to 19 digits), so it is the value parse() produces.
|
||||
|
||||
@param[in] p first character of the token
|
||||
@param[in] e end of the token
|
||||
@param[in] limit end of the readable memory (the source text)
|
||||
*/
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept
|
||||
{
|
||||
const bool negative = *p == '-';
|
||||
p += negative ? 1 : 0;
|
||||
std::uint64_t w = parse_upto19(p, int_digits, limit);
|
||||
p += int_digits;
|
||||
std::int64_t q = 0;
|
||||
if (frac_digits != 0)
|
||||
{
|
||||
w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit);
|
||||
p += 1 + frac_digits;
|
||||
q = -static_cast<std::int64_t>(frac_digits);
|
||||
}
|
||||
if (p != e)
|
||||
{
|
||||
// [eE][+-]digits; huge exponents saturate (the parser rejected overflow)
|
||||
++p;
|
||||
const bool exp_negative = *p == '-';
|
||||
p += (*p == '-' || *p == '+') ? 1 : 0;
|
||||
std::int64_t exp_value = 0;
|
||||
for (; p != e; ++p)
|
||||
{
|
||||
if (exp_value < 0x10000000)
|
||||
{
|
||||
exp_value = (exp_value * 10) + (*p - '0');
|
||||
}
|
||||
}
|
||||
q += exp_negative ? -exp_value : exp_value;
|
||||
}
|
||||
|
||||
double result = 0;
|
||||
if (w != 0)
|
||||
{
|
||||
#if !defined(FLT_EVAL_METHOD) || FLT_EVAL_METHOD == 0
|
||||
static const std::array<double, 23> pow10 = {{1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11, 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22}};
|
||||
if (q >= -22 && q <= 22 && w <= (std::uint64_t{1} << 53))
|
||||
{
|
||||
// Clinger's fast path: both operands exact, one rounding
|
||||
result = static_cast<double>(w);
|
||||
result = q < 0 ? result / pow10[static_cast<std::size_t>(-q)] : result * pow10[static_cast<std::size_t>(q)];
|
||||
return negative ? -result : result;
|
||||
}
|
||||
#endif
|
||||
const std::uint64_t bits = eisel_lemire(q, w);
|
||||
std::memcpy(&result, &bits, sizeof(result));
|
||||
}
|
||||
return negative ? -result : result;
|
||||
}
|
||||
|
||||
/// the value of the float token of a node, as parse() converts it; doubles
|
||||
/// with at most 19 digits are converted from the digit layout
|
||||
template<typename FloatType>
|
||||
FloatType float_value(const document_data& d, const node& n)
|
||||
{
|
||||
return float_value<FloatType>(d, n, std::is_same<FloatType, double> {});
|
||||
}
|
||||
|
||||
template<typename FloatType>
|
||||
FloatType float_value(const document_data& d, const node& n, std::true_type /*double*/)
|
||||
{
|
||||
const unsigned int_digits = n.extra & 0xFFu;
|
||||
const unsigned frac_digits = n.extra >> 8u;
|
||||
if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many")
|
||||
{
|
||||
// (a float token not written by an edit is in the text)
|
||||
const auto* const first = reinterpret_cast<const unsigned char*>(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
return layout_double(first, first + n.len, int_digits, frac_digits, reinterpret_cast<const unsigned char*>(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
}
|
||||
return float_value<FloatType>(d.str(n), n);
|
||||
}
|
||||
|
||||
template<typename FloatType>
|
||||
FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/)
|
||||
{
|
||||
return float_value<FloatType>(d.str(n), n);
|
||||
}
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
@@ -256,7 +256,7 @@ class view_serializer
|
||||
}
|
||||
else
|
||||
{
|
||||
write_float(float_value<number_float_t>(m_doc.str(n), n));
|
||||
write_float(float_value<number_float_t>(m_doc, n));
|
||||
}
|
||||
break;
|
||||
case value_t::object: // LCOV_EXCL_LINE (containers are written by dump())
|
||||
|
||||
@@ -49,7 +49,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE T arithmetic_value(const document_data& d, const nod
|
||||
case value_t::number_integer:
|
||||
return static_cast<T>(static_cast<typename BasicJsonType::number_integer_t>(static_cast<std::int64_t>(integer_bits(n))));
|
||||
case value_t::number_float:
|
||||
return static_cast<T>(float_value<typename BasicJsonType::number_float_t>(d.str(n), n));
|
||||
return static_cast<T>(float_value<typename BasicJsonType::number_float_t>(d, n));
|
||||
case value_t::boolean:
|
||||
return static_cast<T>((n.flags & node_flags::is_true) != 0);
|
||||
case value_t::null:
|
||||
|
||||
@@ -47,6 +47,7 @@
|
||||
#endif
|
||||
|
||||
#include <nlohmann/detail/view/builder.hpp>
|
||||
#include <nlohmann/detail/view/compare.hpp>
|
||||
#include <nlohmann/detail/view/document_data.hpp>
|
||||
#include <nlohmann/detail/view/errors.hpp>
|
||||
#include <nlohmann/detail/view/input.hpp>
|
||||
@@ -603,6 +604,44 @@ class basic_json_view
|
||||
}
|
||||
#endif
|
||||
|
||||
////////////////
|
||||
// comparison //
|
||||
////////////////
|
||||
|
||||
/// whether the values parse() would produce for two views are equal, as
|
||||
/// by BasicJsonType's operator== (numbers by value, objects by their
|
||||
/// members with duplicate keys resolved as parse() resolves them)
|
||||
friend bool operator==(const basic_json_view& a, const basic_json_view& b)
|
||||
{
|
||||
return detail::view::equal<BasicJsonType>(side(a), side(b));
|
||||
}
|
||||
|
||||
friend bool operator!=(const basic_json_view& a, const basic_json_view& b)
|
||||
{
|
||||
return !(a == b);
|
||||
}
|
||||
|
||||
/// whether the value parse() would produce for a view equals a value
|
||||
friend bool operator==(const basic_json_view& a, const BasicJsonType& j)
|
||||
{
|
||||
return detail::view::equal<BasicJsonType>(side(a), json_side_t(j));
|
||||
}
|
||||
|
||||
friend bool operator==(const BasicJsonType& j, const basic_json_view& a)
|
||||
{
|
||||
return a == j;
|
||||
}
|
||||
|
||||
friend bool operator!=(const basic_json_view& a, const BasicJsonType& j)
|
||||
{
|
||||
return !(a == j);
|
||||
}
|
||||
|
||||
friend bool operator!=(const BasicJsonType& j, const basic_json_view& a)
|
||||
{
|
||||
return !(a == j);
|
||||
}
|
||||
|
||||
/////////////////
|
||||
// materialize //
|
||||
/////////////////
|
||||
@@ -635,6 +674,13 @@ class basic_json_view
|
||||
: m_doc(d), m_node(n)
|
||||
{}
|
||||
|
||||
using json_side_t = detail::view::json_side<BasicJsonType, string_view_t>;
|
||||
|
||||
static detail::view::view_side<BasicJsonType, basic_json_view> side(const basic_json_view& v) noexcept
|
||||
{
|
||||
return detail::view::view_side<BasicJsonType, basic_json_view>(v);
|
||||
}
|
||||
|
||||
/// the number of source bytes of this value (estimated for values with
|
||||
/// decoded strings)
|
||||
std::size_t source_extent() const noexcept
|
||||
|
||||
@@ -1567,6 +1567,326 @@ inline bool build(document_data& d, const char* src, std::size_t size, bool comm
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
// #include <nlohmann/detail/view/compare.hpp>
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
|
||||
|
||||
#include <algorithm> // sort, stable_sort
|
||||
#include <cstddef> // size_t
|
||||
#include <string> // string
|
||||
#include <utility> // move, pair
|
||||
#include <vector> // vector
|
||||
|
||||
// #include <nlohmann/json.hpp>
|
||||
// #include <nlohmann/detail/view/macro_scope.hpp>
|
||||
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
namespace view
|
||||
{
|
||||
|
||||
// Equality of views, and of views with basic_json values, with the semantics
|
||||
// of basic_json's operator== applied to the values parse() would produce:
|
||||
// numbers compare by value across their types, an object is compared by its
|
||||
// members with duplicate keys resolved as parse() resolves them (the last
|
||||
// value, at the position of the first occurrence), and in document order if
|
||||
// the object type keeps an order (ordered_json), by key otherwise.
|
||||
|
||||
/// one side of a comparison: a view
|
||||
template<typename BasicJsonType, typename View>
|
||||
class view_side
|
||||
{
|
||||
public:
|
||||
using string_view_t = typename View::string_view_t;
|
||||
|
||||
explicit view_side(const View& v) noexcept
|
||||
: m_view(v)
|
||||
{}
|
||||
|
||||
value_t type() const noexcept
|
||||
{
|
||||
return m_view.type();
|
||||
}
|
||||
|
||||
std::size_t size() const noexcept
|
||||
{
|
||||
return m_view.size();
|
||||
}
|
||||
|
||||
string_view_t string() const
|
||||
{
|
||||
return m_view.get_string();
|
||||
}
|
||||
|
||||
/// a number, boolean, or null as a basic_json value (no allocation)
|
||||
BasicJsonType scalar() const
|
||||
{
|
||||
switch (m_view.type())
|
||||
{
|
||||
case value_t::number_integer:
|
||||
return BasicJsonType(m_view.template get<typename BasicJsonType::number_integer_t>());
|
||||
case value_t::number_unsigned:
|
||||
return BasicJsonType(m_view.template get<typename BasicJsonType::number_unsigned_t>());
|
||||
case value_t::number_float:
|
||||
return BasicJsonType(m_view.template get<typename BasicJsonType::number_float_t>());
|
||||
case value_t::boolean:
|
||||
return BasicJsonType(m_view.template get<bool>());
|
||||
case value_t::null:
|
||||
case value_t::object:
|
||||
case value_t::array:
|
||||
case value_t::string:
|
||||
case value_t::binary:
|
||||
case value_t::discarded:
|
||||
default:
|
||||
return BasicJsonType(nullptr);
|
||||
}
|
||||
}
|
||||
|
||||
void elements(std::vector<view_side>& out) const
|
||||
{
|
||||
out.reserve(m_view.size());
|
||||
for (const View e : m_view)
|
||||
{
|
||||
out.emplace_back(e);
|
||||
}
|
||||
}
|
||||
|
||||
/// the members as parse() keeps them: one per key, the last value at the
|
||||
/// position of the first occurrence; in that order, or sorted by key
|
||||
void members(std::vector<std::pair<string_view_t, view_side>>& out, bool ordered) const
|
||||
{
|
||||
struct member
|
||||
{
|
||||
string_view_t key;
|
||||
View value;
|
||||
std::size_t position;
|
||||
};
|
||||
std::vector<member> all;
|
||||
all.reserve(m_view.size());
|
||||
std::size_t position = 0;
|
||||
for (auto it = m_view.begin(); it != m_view.end(); ++it)
|
||||
{
|
||||
all.push_back(member{it.key(), it.value(), position++});
|
||||
}
|
||||
std::stable_sort(all.begin(), all.end(), [](const member & a, const member & b)
|
||||
{
|
||||
return a.key < b.key;
|
||||
});
|
||||
std::vector<member> unique;
|
||||
unique.reserve(all.size());
|
||||
for (std::size_t i = 0; i < all.size();)
|
||||
{
|
||||
std::size_t last = i;
|
||||
while (last + 1 < all.size() && all[last + 1].key == all[i].key)
|
||||
{
|
||||
++last;
|
||||
}
|
||||
unique.push_back(member{all[i].key, all[last].value, all[i].position});
|
||||
i = last + 1;
|
||||
}
|
||||
if (ordered)
|
||||
{
|
||||
std::sort(unique.begin(), unique.end(), [](const member & a, const member & b)
|
||||
{
|
||||
return a.position < b.position;
|
||||
});
|
||||
}
|
||||
out.reserve(unique.size());
|
||||
for (const member& m : unique)
|
||||
{
|
||||
out.emplace_back(m.key, view_side(m.value));
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
View m_view;
|
||||
};
|
||||
|
||||
/// the other side of a comparison: a basic_json value
|
||||
template<typename BasicJsonType, typename StringView>
|
||||
class json_side
|
||||
{
|
||||
public:
|
||||
using string_view_t = StringView;
|
||||
|
||||
explicit json_side(const BasicJsonType& j) noexcept
|
||||
: m_json(&j)
|
||||
{}
|
||||
|
||||
value_t type() const noexcept
|
||||
{
|
||||
return m_json->type();
|
||||
}
|
||||
|
||||
std::size_t size() const noexcept
|
||||
{
|
||||
return m_json->size();
|
||||
}
|
||||
|
||||
string_view_t string() const
|
||||
{
|
||||
const auto& s = m_json->template get_ref<const typename BasicJsonType::string_t&>();
|
||||
return string_view_t(s.data(), s.size());
|
||||
}
|
||||
|
||||
BasicJsonType scalar() const
|
||||
{
|
||||
return *m_json;
|
||||
}
|
||||
|
||||
void elements(std::vector<json_side>& out) const
|
||||
{
|
||||
out.reserve(m_json->size());
|
||||
for (const auto& e : *m_json)
|
||||
{
|
||||
out.emplace_back(e);
|
||||
}
|
||||
}
|
||||
|
||||
void members(std::vector<std::pair<string_view_t, json_side>>& out, bool ordered) const
|
||||
{
|
||||
out.reserve(m_json->size());
|
||||
for (auto it = m_json->cbegin(); it != m_json->cend(); ++it)
|
||||
{
|
||||
out.emplace_back(string_view_t(it.key().data(), it.key().size()), json_side(it.value()));
|
||||
}
|
||||
if (!ordered)
|
||||
{
|
||||
std::sort(out.begin(), out.end(), [](const std::pair<string_view_t, json_side>& a, const std::pair<string_view_t, json_side>& b)
|
||||
{
|
||||
return a.first < b.first;
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
const BasicJsonType* m_json;
|
||||
};
|
||||
|
||||
/// whether two sides are equal; iterative, so that the nesting depth is
|
||||
/// limited by memory only
|
||||
template<typename BasicJsonType, typename A, typename B>
|
||||
bool equal(const A& a0, const B& b0)
|
||||
{
|
||||
using string_view_t = typename A::string_view_t;
|
||||
const bool ordered = is_ordered_map<typename BasicJsonType::object_t>::value;
|
||||
|
||||
struct frame
|
||||
{
|
||||
std::vector<A> elements_a{};
|
||||
std::vector<B> elements_b{};
|
||||
std::vector<std::pair<string_view_t, A>> members_a{};
|
||||
std::vector<std::pair<string_view_t, B>> members_b{};
|
||||
bool object = false;
|
||||
std::size_t next = 0;
|
||||
};
|
||||
std::vector<frame> stack;
|
||||
A a = a0;
|
||||
B b = b0;
|
||||
for (;;)
|
||||
{
|
||||
const value_t ta = a.type();
|
||||
const value_t tb = b.type();
|
||||
const bool numbers = (ta == value_t::number_integer || ta == value_t::number_unsigned || ta == value_t::number_float)
|
||||
&& (tb == value_t::number_integer || tb == value_t::number_unsigned || tb == value_t::number_float);
|
||||
if (ta == value_t::discarded || tb == value_t::discarded)
|
||||
{
|
||||
// basic_json decides (JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON)
|
||||
if (ta != tb || !(BasicJsonType(value_t::discarded) == BasicJsonType(value_t::discarded)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if (!numbers && ta != tb)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if (ta == value_t::string)
|
||||
{
|
||||
if (!(a.string() == b.string()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else if (ta == value_t::array || ta == value_t::object)
|
||||
{
|
||||
if (a.size() != b.size() && ta == value_t::array)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
frame f;
|
||||
f.object = ta == value_t::object;
|
||||
if (f.object)
|
||||
{
|
||||
a.members(f.members_a, ordered);
|
||||
b.members(f.members_b, ordered);
|
||||
if (f.members_a.size() != f.members_b.size())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
a.elements(f.elements_a);
|
||||
b.elements(f.elements_b);
|
||||
}
|
||||
stack.push_back(std::move(f));
|
||||
}
|
||||
else if (!(a.scalar() == b.scalar())) // numbers (also of different types), null, boolean
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// the next pair of values
|
||||
for (;;)
|
||||
{
|
||||
if (stack.empty())
|
||||
{
|
||||
return true;
|
||||
}
|
||||
frame& f = stack.back();
|
||||
const std::size_t count = f.object ? f.members_a.size() : f.elements_a.size();
|
||||
if (f.next == count)
|
||||
{
|
||||
stack.pop_back();
|
||||
continue;
|
||||
}
|
||||
if (f.object)
|
||||
{
|
||||
if (!(f.members_a[f.next].first == f.members_b[f.next].first))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
a = f.members_a[f.next].second;
|
||||
b = f.members_b[f.next].second;
|
||||
}
|
||||
else
|
||||
{
|
||||
a = f.elements_a[f.next];
|
||||
b = f.elements_b[f.next];
|
||||
}
|
||||
++f.next;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
// #include <nlohmann/detail/view/document_data.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/errors.hpp>
|
||||
@@ -2203,14 +2523,23 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <cstring> // memcpy
|
||||
#include <string> // string
|
||||
#include <type_traits> // integral_constant, is_same
|
||||
|
||||
// #include <nlohmann/json.hpp>
|
||||
// #include <nlohmann/detail/view/document_data.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/macro_scope.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/node.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/scan.hpp>
|
||||
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -2265,6 +2594,96 @@ NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n)
|
||||
return v;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the double of a float token with at most 19 digits, from its layout
|
||||
|
||||
The digit layout recorded while parsing says where the integer digits, the
|
||||
fraction digits, and the exponent are, so the digits are read eight at a
|
||||
time without scanning. The result is correctly rounded (Clinger's fast path
|
||||
where both operands are exact, else the Eisel-Lemire algorithm, which needs
|
||||
no fallback for up to 19 digits), so it is the value parse() produces.
|
||||
|
||||
@param[in] p first character of the token
|
||||
@param[in] e end of the token
|
||||
@param[in] limit end of the readable memory (the source text)
|
||||
*/
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept
|
||||
{
|
||||
const bool negative = *p == '-';
|
||||
p += negative ? 1 : 0;
|
||||
std::uint64_t w = parse_upto19(p, int_digits, limit);
|
||||
p += int_digits;
|
||||
std::int64_t q = 0;
|
||||
if (frac_digits != 0)
|
||||
{
|
||||
w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit);
|
||||
p += 1 + frac_digits;
|
||||
q = -static_cast<std::int64_t>(frac_digits);
|
||||
}
|
||||
if (p != e)
|
||||
{
|
||||
// [eE][+-]digits; huge exponents saturate (the parser rejected overflow)
|
||||
++p;
|
||||
const bool exp_negative = *p == '-';
|
||||
p += (*p == '-' || *p == '+') ? 1 : 0;
|
||||
std::int64_t exp_value = 0;
|
||||
for (; p != e; ++p)
|
||||
{
|
||||
if (exp_value < 0x10000000)
|
||||
{
|
||||
exp_value = (exp_value * 10) + (*p - '0');
|
||||
}
|
||||
}
|
||||
q += exp_negative ? -exp_value : exp_value;
|
||||
}
|
||||
|
||||
double result = 0;
|
||||
if (w != 0)
|
||||
{
|
||||
#if !defined(FLT_EVAL_METHOD) || FLT_EVAL_METHOD == 0
|
||||
static const std::array<double, 23> pow10 = {{1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11, 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22}};
|
||||
if (q >= -22 && q <= 22 && w <= (std::uint64_t{1} << 53))
|
||||
{
|
||||
// Clinger's fast path: both operands exact, one rounding
|
||||
result = static_cast<double>(w);
|
||||
result = q < 0 ? result / pow10[static_cast<std::size_t>(-q)] : result * pow10[static_cast<std::size_t>(q)];
|
||||
return negative ? -result : result;
|
||||
}
|
||||
#endif
|
||||
const std::uint64_t bits = eisel_lemire(q, w);
|
||||
std::memcpy(&result, &bits, sizeof(result));
|
||||
}
|
||||
return negative ? -result : result;
|
||||
}
|
||||
|
||||
/// the value of the float token of a node, as parse() converts it; doubles
|
||||
/// with at most 19 digits are converted from the digit layout
|
||||
template<typename FloatType>
|
||||
FloatType float_value(const document_data& d, const node& n)
|
||||
{
|
||||
return float_value<FloatType>(d, n, std::is_same<FloatType, double> {});
|
||||
}
|
||||
|
||||
template<typename FloatType>
|
||||
FloatType float_value(const document_data& d, const node& n, std::true_type /*double*/)
|
||||
{
|
||||
const unsigned int_digits = n.extra & 0xFFu;
|
||||
const unsigned frac_digits = n.extra >> 8u;
|
||||
if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many")
|
||||
{
|
||||
// (a float token not written by an edit is in the text)
|
||||
const auto* const first = reinterpret_cast<const unsigned char*>(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
return layout_double(first, first + n.len, int_digits, frac_digits, reinterpret_cast<const unsigned char*>(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
}
|
||||
return float_value<FloatType>(d.str(n), n);
|
||||
}
|
||||
|
||||
template<typename FloatType>
|
||||
FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/)
|
||||
{
|
||||
return float_value<FloatType>(d.str(n), n);
|
||||
}
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -2333,7 +2752,7 @@ BasicJsonType materialize(const document_data& d, const node* n)
|
||||
++n;
|
||||
break;
|
||||
case value_t::number_float:
|
||||
sax.number_float(float_value<typename BasicJsonType::number_float_t>(d.str(*n), *n), no_token);
|
||||
sax.number_float(float_value<typename BasicJsonType::number_float_t>(d, *n), no_token);
|
||||
++n;
|
||||
break;
|
||||
case value_t::boolean:
|
||||
@@ -2822,7 +3241,7 @@ class view_serializer
|
||||
}
|
||||
else
|
||||
{
|
||||
write_float(float_value<number_float_t>(m_doc.str(n), n));
|
||||
write_float(float_value<number_float_t>(m_doc, n));
|
||||
}
|
||||
break;
|
||||
case value_t::object: // LCOV_EXCL_LINE (containers are written by dump())
|
||||
@@ -3155,7 +3574,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE T arithmetic_value(const document_data& d, const nod
|
||||
case value_t::number_integer:
|
||||
return static_cast<T>(static_cast<typename BasicJsonType::number_integer_t>(static_cast<std::int64_t>(integer_bits(n))));
|
||||
case value_t::number_float:
|
||||
return static_cast<T>(float_value<typename BasicJsonType::number_float_t>(d.str(n), n));
|
||||
return static_cast<T>(float_value<typename BasicJsonType::number_float_t>(d, n));
|
||||
case value_t::boolean:
|
||||
return static_cast<T>((n.flags & node_flags::is_true) != 0);
|
||||
case value_t::null:
|
||||
@@ -3757,6 +4176,44 @@ class basic_json_view
|
||||
}
|
||||
#endif
|
||||
|
||||
////////////////
|
||||
// comparison //
|
||||
////////////////
|
||||
|
||||
/// whether the values parse() would produce for two views are equal, as
|
||||
/// by BasicJsonType's operator== (numbers by value, objects by their
|
||||
/// members with duplicate keys resolved as parse() resolves them)
|
||||
friend bool operator==(const basic_json_view& a, const basic_json_view& b)
|
||||
{
|
||||
return detail::view::equal<BasicJsonType>(side(a), side(b));
|
||||
}
|
||||
|
||||
friend bool operator!=(const basic_json_view& a, const basic_json_view& b)
|
||||
{
|
||||
return !(a == b);
|
||||
}
|
||||
|
||||
/// whether the value parse() would produce for a view equals a value
|
||||
friend bool operator==(const basic_json_view& a, const BasicJsonType& j)
|
||||
{
|
||||
return detail::view::equal<BasicJsonType>(side(a), json_side_t(j));
|
||||
}
|
||||
|
||||
friend bool operator==(const BasicJsonType& j, const basic_json_view& a)
|
||||
{
|
||||
return a == j;
|
||||
}
|
||||
|
||||
friend bool operator!=(const basic_json_view& a, const BasicJsonType& j)
|
||||
{
|
||||
return !(a == j);
|
||||
}
|
||||
|
||||
friend bool operator!=(const BasicJsonType& j, const basic_json_view& a)
|
||||
{
|
||||
return !(a == j);
|
||||
}
|
||||
|
||||
/////////////////
|
||||
// materialize //
|
||||
/////////////////
|
||||
@@ -3789,6 +4246,13 @@ class basic_json_view
|
||||
: m_doc(d), m_node(n)
|
||||
{}
|
||||
|
||||
using json_side_t = detail::view::json_side<BasicJsonType, string_view_t>;
|
||||
|
||||
static detail::view::view_side<BasicJsonType, basic_json_view> side(const basic_json_view& v) noexcept
|
||||
{
|
||||
return detail::view::view_side<BasicJsonType, basic_json_view>(v);
|
||||
}
|
||||
|
||||
/// the number of source bytes of this value (estimated for values with
|
||||
/// decoded strings)
|
||||
std::size_t source_extent() const noexcept
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
build/
|
||||
@@ -0,0 +1,73 @@
|
||||
# json_view compared with other libraries
|
||||
|
||||
The in-tree benchmarks in [`tests/benchmarks`](../README.md) measure `json_document` against `json::parse` only. The
|
||||
programs here compare it with [yyjson](https://github.com/ibireme/yyjson),
|
||||
[simdjson](https://github.com/simdjson/simdjson), and [Boost.JSON](https://github.com/boostorg/json): the question
|
||||
users ask when they pick a library. They are not built by CMake or run by CI.
|
||||
|
||||
## Reproducing the numbers
|
||||
|
||||
`compare.py` builds both programs against `include/` of this checkout, runs them, and writes the results together with
|
||||
everything needed to reproduce them to `results/<date>-<host>.md` (and `.csv`): the date, the commit, the CPU, the
|
||||
OS, the compiler, the flags, and the versions of all libraries.
|
||||
|
||||
```sh
|
||||
python3 tests/benchmarks/json_view/compare.py --data <json_test_data directory> [--native] [--rounds 30]
|
||||
```
|
||||
|
||||
- `--data` is the downloaded [test data](https://github.com/nlohmann/json_test_data), e.g. the `test_files` directory
|
||||
of a CMake build directory. It needs `nativejson-benchmark/{twitter,citm_catalog,canada}.json` and
|
||||
`jeopardy/jeopardy.json`.
|
||||
- The other libraries come from the system: pkg-config, or Homebrew (`brew install yyjson simdjson boost`). With
|
||||
`--download`, pinned releases are downloaded instead and checked against their SHA-256. Without Boost headers (or
|
||||
with `--no-boost`), the Boost.JSON columns are skipped, and the results say so.
|
||||
- `--corpus file...` adds files to the corpus benchmark, e.g. those of
|
||||
[simdjson-data](https://github.com/simdjson/simdjson-data) or the
|
||||
[yyjson benchmark](https://github.com/ibireme/yyjson_benchmark).
|
||||
- Only the Python 3 standard library is used; a C++17 compiler is needed (`CXX` and `CC` are honored).
|
||||
|
||||
For numbers worth publishing, use a quiet machine (see [Getting stable numbers](../README.md#getting-stable-numbers)),
|
||||
the default 30 rounds or more, and `--native` only if the other libraries were built for the same CPU.
|
||||
|
||||
### On GitHub-hosted runners
|
||||
|
||||
The workflow [json_view benchmarks](../../../.github/workflows/json_view_benchmarks.yml) runs `compare.py --download`
|
||||
on demand: by hand (Actions → "json_view benchmarks" → "Run workflow"), on an x86-64 or AArch64 Ubuntu runner with GCC
|
||||
or Clang, or when a pull request gets the label `benchmark`, on both architectures with GCC. The results appear as the
|
||||
job summary and as an artifact. Shared runners are noisy, so these numbers show
|
||||
where `json_view` stands on another architecture; they are not meant for publication.
|
||||
|
||||
## What is measured
|
||||
|
||||
`bench_view.cpp` runs four workloads on twitter, citm_catalog, canada, jeopardy, a single tweet (`status`), and a
|
||||
JSON-RPC request (`rpc`):
|
||||
|
||||
| workload | what it does |
|
||||
|---|---|
|
||||
| parse | build and free a document |
|
||||
| traverse | parse, then visit every value, convert every number, touch every string and key |
|
||||
| select | parse, then read a few fields per record (e.g. id, user name, and retweet count of each tweet) |
|
||||
| dump | serialize a parsed document (compact) |
|
||||
|
||||
`bench_corpus.cpp` runs parse, traverse, and dump on any list of files, so that no library is tuned to a handful of
|
||||
documents.
|
||||
|
||||
Before anything is timed, all engines must accept each document and agree on the traversal: the number of values, the
|
||||
bytes of all strings and keys, and the sum of all numbers. All engines run interleaved in every round, and the best
|
||||
round is reported, as time and as a factor of the `json_view` time (below 1 means faster than `json_view`).
|
||||
|
||||
The engines do not all offer the same features, which the numbers should be read with:
|
||||
|
||||
| engine | document | random access | editable | notes |
|
||||
|---|---|---|---|---|
|
||||
| `json_view` | immutable index into the text | yes | no | a fresh document per parse; "reused" parses into the same document |
|
||||
| yyjson | immutable (`yyjson_read`) | yes | via a mutable copy | |
|
||||
| simdjson DOM | immutable, parser reused | yes | no | |
|
||||
| simdjson On-Demand | none: forward-only, lazy | no | no | only traverse and select |
|
||||
| Boost.JSON | owning, mutable DOM | yes | yes | monotonic resource |
|
||||
| `json::parse` | owning, mutable DOM | yes | yes | |
|
||||
|
||||
## Published results
|
||||
|
||||
Results are only published with the file `compare.py` wrote, which names the machine and the versions; see
|
||||
`results/`. Numbers from one machine and compiler do not carry over to another: rerun the script.
|
||||
@@ -0,0 +1,326 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
// Corpus benchmark: the read-only workloads of bench_view.cpp on any list of
|
||||
// JSON files (for example the benchmark sets of simdjson and yyjson).
|
||||
//
|
||||
// ./bench_corpus [--rounds N] file...
|
||||
//
|
||||
// For every file, all engines must accept it and agree on a traversal (value
|
||||
// count, string bytes, sum of numbers) before anything is timed. Workloads:
|
||||
// parse (build and free a document), traverse (visit every value, convert
|
||||
// every number), dump (compact), and for json_view also dump with the source
|
||||
// number text. Results go to bench_corpus.csv.
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
#include <boost/json.hpp>
|
||||
#include <boost/json/src.hpp>
|
||||
#endif
|
||||
#include <simdjson.h>
|
||||
#include <yyjson.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <chrono>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <functional>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
using nlohmann::json;
|
||||
using nlohmann::json_document;
|
||||
using nlohmann::json_view;
|
||||
|
||||
static volatile double g_sink;
|
||||
|
||||
struct stats
|
||||
{
|
||||
double num = 0;
|
||||
std::size_t str = 0, nodes = 0;
|
||||
};
|
||||
|
||||
static void walk(json_view v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (v.type())
|
||||
{
|
||||
case json::value_t::object:
|
||||
for (auto it = v.begin(); it != v.end(); ++it)
|
||||
{
|
||||
st.str += it.key().size();
|
||||
walk(*it, st);
|
||||
}
|
||||
break;
|
||||
case json::value_t::array:
|
||||
for (const json_view e : v)
|
||||
{
|
||||
walk(e, st);
|
||||
}
|
||||
break;
|
||||
case json::value_t::string:
|
||||
st.str += v.get_string().size();
|
||||
break;
|
||||
case json::value_t::number_integer:
|
||||
st.num += static_cast<double>(v.get<std::int64_t>());
|
||||
break;
|
||||
case json::value_t::number_unsigned:
|
||||
st.num += static_cast<double>(v.get<std::uint64_t>());
|
||||
break;
|
||||
case json::value_t::number_float:
|
||||
st.num += v.get<double>();
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void walk(yyjson_val* v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (yyjson_get_type(v))
|
||||
{
|
||||
case YYJSON_TYPE_OBJ:
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* k, * val;
|
||||
yyjson_obj_foreach(v, idx, max, k, val)
|
||||
{
|
||||
st.str += yyjson_get_len(k);
|
||||
walk(val, st);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case YYJSON_TYPE_ARR:
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* val;
|
||||
yyjson_arr_foreach(v, idx, max, val)
|
||||
{
|
||||
walk(val, st);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case YYJSON_TYPE_STR:
|
||||
st.str += yyjson_get_len(v);
|
||||
break;
|
||||
case YYJSON_TYPE_NUM:
|
||||
st.num += yyjson_is_sint(v) ? static_cast<double>(yyjson_get_sint(v)) : yyjson_is_uint(v) ? static_cast<double>(yyjson_get_uint(v)) : yyjson_get_real(v);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void walk(simdjson::dom::element e, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (e.type())
|
||||
{
|
||||
case simdjson::dom::element_type::OBJECT:
|
||||
for (auto f : simdjson::dom::object(e))
|
||||
{
|
||||
st.str += f.key.size();
|
||||
walk(f.value, st);
|
||||
}
|
||||
break;
|
||||
case simdjson::dom::element_type::ARRAY:
|
||||
for (auto c : simdjson::dom::array(e))
|
||||
{
|
||||
walk(c, st);
|
||||
}
|
||||
break;
|
||||
case simdjson::dom::element_type::STRING:
|
||||
st.str += std::string_view(e).size();
|
||||
break;
|
||||
case simdjson::dom::element_type::INT64:
|
||||
st.num += static_cast<double>(int64_t(e));
|
||||
break;
|
||||
case simdjson::dom::element_type::UINT64:
|
||||
st.num += static_cast<double>(uint64_t(e));
|
||||
break;
|
||||
case simdjson::dom::element_type::DOUBLE:
|
||||
st.num += double(e);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
static void walk(const boost::json::value& v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (v.kind())
|
||||
{
|
||||
case boost::json::kind::object:
|
||||
for (const auto& kv : v.get_object())
|
||||
{
|
||||
st.str += kv.key().size();
|
||||
walk(kv.value(), st);
|
||||
}
|
||||
break;
|
||||
case boost::json::kind::array:
|
||||
for (const auto& c : v.get_array())
|
||||
{
|
||||
walk(c, st);
|
||||
}
|
||||
break;
|
||||
case boost::json::kind::string:
|
||||
st.str += v.get_string().size();
|
||||
break;
|
||||
case boost::json::kind::int64:
|
||||
st.num += static_cast<double>(v.get_int64());
|
||||
break;
|
||||
case boost::json::kind::uint64:
|
||||
st.num += static_cast<double>(v.get_uint64());
|
||||
break;
|
||||
case boost::json::kind::double_:
|
||||
st.num += v.get_double();
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
static std::string slurp(const std::string& p)
|
||||
{
|
||||
std::ifstream f(p, std::ios::binary);
|
||||
std::stringstream ss;
|
||||
ss << f.rdbuf();
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
static bool same(const stats& a, const stats& b)
|
||||
{
|
||||
return a.nodes == b.nodes && a.str == b.str && (a.num == b.num || std::fabs(a.num - b.num) <= 1e-9 * std::fabs(a.num));
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
int rounds = 0; // 0: by size
|
||||
std::vector<std::string> files;
|
||||
for (int i = 1; i < argc; ++i)
|
||||
{
|
||||
if (std::strcmp(argv[i], "--rounds") == 0 && i + 1 < argc)
|
||||
{
|
||||
rounds = std::atoi(argv[++i]);
|
||||
}
|
||||
else
|
||||
{
|
||||
files.push_back(argv[i]);
|
||||
}
|
||||
}
|
||||
std::FILE* csv = std::fopen("bench_corpus.csv", "w");
|
||||
std::fprintf(csv, "file,bytes,workload,engine,ns\n");
|
||||
simdjson::dom::parser sj;
|
||||
for (const auto& path : files)
|
||||
{
|
||||
const std::string s = slurp(path);
|
||||
const std::string name = path.substr(path.rfind('/') + 1);
|
||||
const simdjson::padded_string ps(s);
|
||||
|
||||
// all engines must agree before timing
|
||||
stats a, b, c, d;
|
||||
const json_document doc = json_document::parse(s);
|
||||
walk(doc.root(), a);
|
||||
yyjson_doc* y = yyjson_read(s.data(), s.size(), 0);
|
||||
auto sjr = sj.parse(ps);
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
boost::json::parse_options opt;
|
||||
opt.numbers = boost::json::number_precision::precise;
|
||||
boost::json::monotonic_resource mr0;
|
||||
const boost::json::value bv = boost::json::parse(s, &mr0, opt);
|
||||
#endif
|
||||
if (y == nullptr || sjr.error())
|
||||
{
|
||||
std::printf("%-34s skipped (an engine rejects it)\n", name.c_str());
|
||||
yyjson_doc_free(y);
|
||||
continue;
|
||||
}
|
||||
walk(yyjson_doc_get_root(y), b);
|
||||
walk(sjr.value_unsafe(), c);
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
walk(bv, d);
|
||||
#else
|
||||
d = a;
|
||||
#endif
|
||||
yyjson_doc_free(y);
|
||||
const bool ok = same(a, b) && same(a, c) && same(a, d);
|
||||
|
||||
const int r = rounds > 0 ? rounds : static_cast<int>(std::max<std::size_t>(3, std::min<std::size_t>(60, 400000000 / (s.size() + 1))));
|
||||
struct engine
|
||||
{
|
||||
std::string name;
|
||||
std::function<void()> fn;
|
||||
};
|
||||
json_document vd = json_document::parse(s);
|
||||
yyjson_doc* yd = yyjson_read(s.data(), s.size(), 0);
|
||||
simdjson::dom::parser sjd;
|
||||
const simdjson::dom::element se = sjd.parse(ps).value_unsafe();
|
||||
const std::vector<std::pair<std::string, std::vector<engine>>> workloads =
|
||||
{
|
||||
{
|
||||
"parse", {
|
||||
{"json_view", [&] { auto x = json_document::parse(s); g_sink = static_cast<double>(x.node_count()); }},
|
||||
{"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast<double>(yyjson_doc_get_val_count(x)); yyjson_doc_free(x); }},
|
||||
{"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }},
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
{"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }},
|
||||
#endif
|
||||
}
|
||||
},
|
||||
{
|
||||
"traverse", {
|
||||
{"json_view", [&] { auto x = json_document::parse(s); stats st; walk(x.root(), st); g_sink = st.num; }},
|
||||
{"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(x), st); g_sink = st.num; yyjson_doc_free(x); }},
|
||||
{"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }},
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
{"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); stats st; walk(v, st); g_sink = st.num; }},
|
||||
#endif
|
||||
}
|
||||
},
|
||||
{
|
||||
"dump", {
|
||||
{"json_view", [&] { std::string o = vd.root().dump(); g_sink = static_cast<double>(o.size()); }},
|
||||
{"yyjson", [&] { std::size_t n = 0; char* o = yyjson_write(yd, 0, &n); g_sink = static_cast<double>(n); std::free(o); }},
|
||||
{"simdjson DOM", [&] { std::string o = simdjson::to_string(se); g_sink = static_cast<double>(o.size()); }},
|
||||
{"json_view (source numbers)", [&] { std::string o = vd.root().dump(-1, ' ', false, json_view::number_format::source); g_sink = static_cast<double>(o.size()); }},
|
||||
}
|
||||
},
|
||||
};
|
||||
std::printf("%-34s %9zu B%s\n", name.c_str(), s.size(), ok ? "" : " [ENGINES DISAGREE]");
|
||||
for (const auto& wl : workloads)
|
||||
{
|
||||
std::vector<double> best(wl.second.size(), 1e300);
|
||||
for (int i = 0; i < r; ++i)
|
||||
{
|
||||
for (std::size_t k = 0; k < wl.second.size(); ++k)
|
||||
{
|
||||
const auto t0 = std::chrono::steady_clock::now();
|
||||
wl.second[k].fn();
|
||||
best[k] = std::min(best[k], std::chrono::duration<double, std::nano>(std::chrono::steady_clock::now() - t0).count());
|
||||
}
|
||||
}
|
||||
std::printf(" %-9s", wl.first.c_str());
|
||||
for (std::size_t k = 0; k < wl.second.size(); ++k)
|
||||
{
|
||||
std::printf(" %s %.2f GB/s (%.2fx)", wl.second[k].name.c_str(), static_cast<double>(s.size()) / best[k], best[k] / best[0]);
|
||||
std::fprintf(csv, "%s,%zu,%s,%s,%.1f\n", name.c_str(), s.size(), wl.first.c_str(), wl.second[k].name.c_str(), best[k]);
|
||||
}
|
||||
std::printf("\n");
|
||||
std::fflush(stdout);
|
||||
}
|
||||
yyjson_doc_free(yd);
|
||||
}
|
||||
std::fclose(csv);
|
||||
}
|
||||
@@ -0,0 +1,735 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
// Same-feature-set benchmark: read-only JSON documents with random access.
|
||||
//
|
||||
// json_view nlohmann/json_view.hpp (fresh document per parse / reused)
|
||||
// yyjson yyjson_read(): immutable document, random access
|
||||
// simdjson DOM dom::parser (reused, as recommended): immutable, random access
|
||||
// references (different feature sets):
|
||||
// simdjson OD On-Demand: forward-only, lazy
|
||||
// Boost.JSON owning, mutable DOM (monotonic resource)
|
||||
// json::parse owning, mutable DOM (nlohmann today)
|
||||
//
|
||||
// Workloads: parse (build + free), traverse (visit everything, convert every
|
||||
// number, touch every string and key), select (a few fields per document),
|
||||
// dump (compact serialization of the parsed document).
|
||||
// All engines run interleaved in every round; the best round is reported.
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
#include <boost/json.hpp>
|
||||
#include <boost/json/src.hpp>
|
||||
#endif
|
||||
#include <simdjson.h>
|
||||
#include <yyjson.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <chrono>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <fstream>
|
||||
#include <functional>
|
||||
#include <map>
|
||||
#include <sstream>
|
||||
|
||||
using nlohmann::json;
|
||||
using nlohmann::json_document;
|
||||
using nlohmann::json_view;
|
||||
|
||||
static volatile double g_sink;
|
||||
|
||||
struct stats
|
||||
{
|
||||
double num = 0;
|
||||
std::size_t str = 0, nodes = 0;
|
||||
};
|
||||
|
||||
// ---------------- traversal ----------------
|
||||
|
||||
static void walk(json_view v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (v.type())
|
||||
{
|
||||
case json::value_t::object:
|
||||
for (auto it = v.begin(); it != v.end(); ++it)
|
||||
{
|
||||
st.str += it.key().size();
|
||||
walk(*it, st);
|
||||
}
|
||||
break;
|
||||
case json::value_t::array:
|
||||
for (const json_view e : v)
|
||||
{
|
||||
walk(e, st);
|
||||
}
|
||||
break;
|
||||
case json::value_t::string:
|
||||
st.str += v.get_string().size();
|
||||
break;
|
||||
case json::value_t::number_integer:
|
||||
st.num += static_cast<double>(v.get<std::int64_t>());
|
||||
break;
|
||||
case json::value_t::number_unsigned:
|
||||
st.num += static_cast<double>(v.get<std::uint64_t>());
|
||||
break;
|
||||
case json::value_t::number_float:
|
||||
st.num += v.get<double>();
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void walk(const json& j, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (j.type())
|
||||
{
|
||||
case json::value_t::object:
|
||||
for (const auto& kv : j.get_ref<const json::object_t&>())
|
||||
{
|
||||
st.str += kv.first.size();
|
||||
walk(kv.second, st);
|
||||
}
|
||||
break;
|
||||
case json::value_t::array:
|
||||
for (const auto& e : j.get_ref<const json::array_t&>())
|
||||
{
|
||||
walk(e, st);
|
||||
}
|
||||
break;
|
||||
case json::value_t::string:
|
||||
st.str += j.get_ref<const std::string&>().size();
|
||||
break;
|
||||
case json::value_t::number_integer:
|
||||
st.num += static_cast<double>(*j.get_ptr<const json::number_integer_t*>());
|
||||
break;
|
||||
case json::value_t::number_unsigned:
|
||||
st.num += static_cast<double>(*j.get_ptr<const json::number_unsigned_t*>());
|
||||
break;
|
||||
case json::value_t::number_float:
|
||||
st.num += *j.get_ptr<const json::number_float_t*>();
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void walk(yyjson_val* v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (yyjson_get_type(v))
|
||||
{
|
||||
case YYJSON_TYPE_OBJ:
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* k, * val;
|
||||
yyjson_obj_foreach(v, idx, max, k, val)
|
||||
{
|
||||
st.str += yyjson_get_len(k);
|
||||
walk(val, st);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case YYJSON_TYPE_ARR:
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* val;
|
||||
yyjson_arr_foreach(v, idx, max, val)
|
||||
{
|
||||
walk(val, st);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case YYJSON_TYPE_STR:
|
||||
st.str += yyjson_get_len(v);
|
||||
break;
|
||||
case YYJSON_TYPE_NUM:
|
||||
if (yyjson_is_sint(v))
|
||||
{
|
||||
st.num += static_cast<double>(yyjson_get_sint(v));
|
||||
}
|
||||
else if (yyjson_is_uint(v))
|
||||
{
|
||||
st.num += static_cast<double>(yyjson_get_uint(v));
|
||||
}
|
||||
else
|
||||
{
|
||||
st.num += yyjson_get_real(v);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void walk(simdjson::dom::element e, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (e.type())
|
||||
{
|
||||
case simdjson::dom::element_type::OBJECT:
|
||||
for (auto f : simdjson::dom::object(e))
|
||||
{
|
||||
st.str += f.key.size();
|
||||
walk(f.value, st);
|
||||
}
|
||||
break;
|
||||
case simdjson::dom::element_type::ARRAY:
|
||||
for (auto c : simdjson::dom::array(e))
|
||||
{
|
||||
walk(c, st);
|
||||
}
|
||||
break;
|
||||
case simdjson::dom::element_type::STRING:
|
||||
st.str += std::string_view(e).size();
|
||||
break;
|
||||
case simdjson::dom::element_type::INT64:
|
||||
st.num += static_cast<double>(int64_t(e));
|
||||
break;
|
||||
case simdjson::dom::element_type::UINT64:
|
||||
st.num += static_cast<double>(uint64_t(e));
|
||||
break;
|
||||
case simdjson::dom::element_type::DOUBLE:
|
||||
st.num += double(e);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void walk_od(simdjson::ondemand::value v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (v.type())
|
||||
{
|
||||
case simdjson::ondemand::json_type::object:
|
||||
for (auto f : v.get_object())
|
||||
{
|
||||
st.str += std::string_view(f.unescaped_key()).size();
|
||||
walk_od(f.value(), st);
|
||||
}
|
||||
break;
|
||||
case simdjson::ondemand::json_type::array:
|
||||
for (auto c : v.get_array())
|
||||
{
|
||||
walk_od(c.value(), st);
|
||||
}
|
||||
break;
|
||||
case simdjson::ondemand::json_type::string:
|
||||
st.str += std::string_view(v.get_string()).size();
|
||||
break;
|
||||
case simdjson::ondemand::json_type::number:
|
||||
{
|
||||
simdjson::ondemand::number n = v.get_number();
|
||||
switch (n.get_number_type())
|
||||
{
|
||||
case simdjson::ondemand::number_type::signed_integer:
|
||||
st.num += static_cast<double>(n.get_int64());
|
||||
break;
|
||||
case simdjson::ondemand::number_type::unsigned_integer:
|
||||
st.num += static_cast<double>(n.get_uint64());
|
||||
break;
|
||||
default:
|
||||
st.num += n.get_double();
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case simdjson::ondemand::json_type::boolean:
|
||||
(void)bool(v.get_bool());
|
||||
break;
|
||||
default:
|
||||
(void)v.is_null();
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
static void walk(const boost::json::value& v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (v.kind())
|
||||
{
|
||||
case boost::json::kind::object:
|
||||
for (const auto& kv : v.get_object())
|
||||
{
|
||||
st.str += kv.key().size();
|
||||
walk(kv.value(), st);
|
||||
}
|
||||
break;
|
||||
case boost::json::kind::array:
|
||||
for (const auto& c : v.get_array())
|
||||
{
|
||||
walk(c, st);
|
||||
}
|
||||
break;
|
||||
case boost::json::kind::string:
|
||||
st.str += v.get_string().size();
|
||||
break;
|
||||
case boost::json::kind::int64:
|
||||
st.num += static_cast<double>(v.get_int64());
|
||||
break;
|
||||
case boost::json::kind::uint64:
|
||||
st.num += static_cast<double>(v.get_uint64());
|
||||
break;
|
||||
case boost::json::kind::double_:
|
||||
st.num += v.get_double();
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
// ---------------- selective access ----------------
|
||||
// twitter: per status id, user.screen_name, retweet_count
|
||||
// citm: per performance id, eventId, #seatCategories; #events
|
||||
// canada: type, features[0].geometry.type, #coordinates
|
||||
// jeopardy: per question: round == "Final Jeopardy!", len(category)
|
||||
// status (one tweet): id, user.screen_name, retweet_count
|
||||
// rpc: method, params.minuend, id
|
||||
|
||||
static double pick(const std::string& name, json_view r)
|
||||
{
|
||||
double acc = 0;
|
||||
if (name == "twitter" || name == "status")
|
||||
{
|
||||
auto one = [&](json_view s)
|
||||
{
|
||||
acc += static_cast<double>(s["id"].get<std::uint64_t>());
|
||||
acc += static_cast<double>(s["user"]["screen_name"].get_string().size());
|
||||
acc += static_cast<double>(s["retweet_count"].get<std::int64_t>());
|
||||
};
|
||||
if (name == "twitter")
|
||||
{
|
||||
for (const json_view s : r["statuses"])
|
||||
{
|
||||
one(s);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
one(r);
|
||||
}
|
||||
}
|
||||
else if (name == "citm_catalog")
|
||||
{
|
||||
for (const json_view p : r["performances"])
|
||||
{
|
||||
acc += static_cast<double>(p["id"].get<std::uint64_t>() + p["eventId"].get<std::uint64_t>() + p["seatCategories"].size());
|
||||
}
|
||||
acc += static_cast<double>(r["events"].size());
|
||||
}
|
||||
else if (name == "canada")
|
||||
{
|
||||
const json_view g = r["features"][0]["geometry"];
|
||||
acc += static_cast<double>(r["type"].get_string().size() + g["type"].get_string().size() + g["coordinates"].size());
|
||||
}
|
||||
else if (name == "jeopardy")
|
||||
{
|
||||
for (const json_view q : r)
|
||||
{
|
||||
acc += q["round"].get_string() == "Final Jeopardy!" ? 1 : 0;
|
||||
acc += static_cast<double>(q["category"].get_string().size());
|
||||
}
|
||||
}
|
||||
else if (name == "rpc")
|
||||
{
|
||||
acc += static_cast<double>(r["method"].get_string().size());
|
||||
acc += static_cast<double>(r["params"]["minuend"].get<std::int64_t>() + r["id"].get<std::int64_t>());
|
||||
}
|
||||
return acc;
|
||||
}
|
||||
|
||||
static double pick(const std::string& name, const json& r)
|
||||
{
|
||||
double acc = 0;
|
||||
if (name == "twitter" || name == "status")
|
||||
{
|
||||
auto one = [&](const json & s)
|
||||
{
|
||||
acc += static_cast<double>(s["id"].get<std::uint64_t>());
|
||||
acc += static_cast<double>(s["user"]["screen_name"].get_ref<const std::string&>().size());
|
||||
acc += static_cast<double>(s["retweet_count"].get<std::int64_t>());
|
||||
};
|
||||
if (name == "twitter")
|
||||
{
|
||||
for (const auto& s : r["statuses"])
|
||||
{
|
||||
one(s);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
one(r);
|
||||
}
|
||||
}
|
||||
else if (name == "citm_catalog")
|
||||
{
|
||||
for (const auto& p : r["performances"])
|
||||
{
|
||||
acc += static_cast<double>(p["id"].get<std::uint64_t>() + p["eventId"].get<std::uint64_t>() + p["seatCategories"].size());
|
||||
}
|
||||
acc += static_cast<double>(r["events"].size());
|
||||
}
|
||||
else if (name == "canada")
|
||||
{
|
||||
const json& g = r["features"][0]["geometry"];
|
||||
acc += static_cast<double>(r["type"].get_ref<const std::string&>().size() + g["type"].get_ref<const std::string&>().size() + g["coordinates"].size());
|
||||
}
|
||||
else if (name == "jeopardy")
|
||||
{
|
||||
for (const auto& q : r)
|
||||
{
|
||||
acc += q["round"].get_ref<const std::string&>() == "Final Jeopardy!" ? 1 : 0;
|
||||
acc += static_cast<double>(q["category"].get_ref<const std::string&>().size());
|
||||
}
|
||||
}
|
||||
else if (name == "rpc")
|
||||
{
|
||||
acc += static_cast<double>(r["method"].get_ref<const std::string&>().size());
|
||||
acc += static_cast<double>(r["params"]["minuend"].get<std::int64_t>() + r["id"].get<std::int64_t>());
|
||||
}
|
||||
return acc;
|
||||
}
|
||||
|
||||
static double pick(const std::string& name, yyjson_val* r)
|
||||
{
|
||||
double acc = 0;
|
||||
auto get = [](yyjson_val * o, const char* k)
|
||||
{
|
||||
return yyjson_obj_get(o, k);
|
||||
};
|
||||
if (name == "twitter" || name == "status")
|
||||
{
|
||||
auto one = [&](yyjson_val * s)
|
||||
{
|
||||
acc += static_cast<double>(yyjson_get_uint(get(s, "id")));
|
||||
acc += static_cast<double>(yyjson_get_len(get(get(s, "user"), "screen_name")));
|
||||
acc += static_cast<double>(yyjson_get_sint(get(s, "retweet_count")));
|
||||
};
|
||||
if (name == "twitter")
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* s;
|
||||
yyjson_arr_foreach(get(r, "statuses"), idx, max, s)
|
||||
{
|
||||
one(s);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
one(r);
|
||||
}
|
||||
}
|
||||
else if (name == "citm_catalog")
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* p;
|
||||
yyjson_arr_foreach(get(r, "performances"), idx, max, p)
|
||||
{
|
||||
acc += static_cast<double>(yyjson_get_uint(get(p, "id")) + yyjson_get_uint(get(p, "eventId")) + yyjson_arr_size(get(p, "seatCategories")));
|
||||
}
|
||||
acc += static_cast<double>(yyjson_obj_size(get(r, "events")));
|
||||
}
|
||||
else if (name == "canada")
|
||||
{
|
||||
yyjson_val* g = get(yyjson_arr_get(get(r, "features"), 0), "geometry");
|
||||
acc += static_cast<double>(yyjson_get_len(get(r, "type")) + yyjson_get_len(get(g, "type")) + yyjson_arr_size(get(g, "coordinates")));
|
||||
}
|
||||
else if (name == "jeopardy")
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* q;
|
||||
yyjson_arr_foreach(r, idx, max, q)
|
||||
{
|
||||
acc += yyjson_equals_str(get(q, "round"), "Final Jeopardy!") ? 1 : 0;
|
||||
acc += static_cast<double>(yyjson_get_len(get(q, "category")));
|
||||
}
|
||||
}
|
||||
else if (name == "rpc")
|
||||
{
|
||||
acc += static_cast<double>(yyjson_get_len(get(r, "method")));
|
||||
acc += static_cast<double>(yyjson_get_sint(get(get(r, "params"), "minuend")) + yyjson_get_sint(get(r, "id")));
|
||||
}
|
||||
return acc;
|
||||
}
|
||||
|
||||
static double pick(const std::string& name, simdjson::dom::element r)
|
||||
{
|
||||
double acc = 0;
|
||||
if (name == "twitter" || name == "status")
|
||||
{
|
||||
auto one = [&](simdjson::dom::element s)
|
||||
{
|
||||
acc += static_cast<double>(uint64_t(s["id"]));
|
||||
acc += static_cast<double>(std::string_view(s["user"]["screen_name"]).size());
|
||||
acc += static_cast<double>(int64_t(s["retweet_count"]));
|
||||
};
|
||||
if (name == "twitter")
|
||||
{
|
||||
for (auto s : simdjson::dom::array(r["statuses"]))
|
||||
{
|
||||
one(s);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
one(r);
|
||||
}
|
||||
}
|
||||
else if (name == "citm_catalog")
|
||||
{
|
||||
for (auto p : simdjson::dom::array(r["performances"]))
|
||||
{
|
||||
acc += static_cast<double>(uint64_t(p["id"]) + uint64_t(p["eventId"]) + simdjson::dom::array(p["seatCategories"]).size());
|
||||
}
|
||||
acc += static_cast<double>(simdjson::dom::object(r["events"]).size());
|
||||
}
|
||||
else if (name == "canada")
|
||||
{
|
||||
auto g = r["features"].at(0)["geometry"];
|
||||
acc += static_cast<double>(std::string_view(r["type"]).size() + std::string_view(g["type"]).size() + simdjson::dom::array(g["coordinates"]).size());
|
||||
}
|
||||
else if (name == "jeopardy")
|
||||
{
|
||||
for (auto q : simdjson::dom::array(r))
|
||||
{
|
||||
acc += std::string_view(q["round"]) == "Final Jeopardy!" ? 1 : 0;
|
||||
acc += static_cast<double>(std::string_view(q["category"]).size());
|
||||
}
|
||||
}
|
||||
else if (name == "rpc")
|
||||
{
|
||||
acc += static_cast<double>(std::string_view(r["method"]).size());
|
||||
acc += static_cast<double>(int64_t(r["params"]["minuend"]) + int64_t(r["id"]));
|
||||
}
|
||||
return acc;
|
||||
}
|
||||
|
||||
static double pick_od(const std::string& name, simdjson::ondemand::document& d)
|
||||
{
|
||||
double acc = 0;
|
||||
if (name == "twitter" || name == "status")
|
||||
{
|
||||
auto one = [&](simdjson::ondemand::object s)
|
||||
{
|
||||
acc += static_cast<double>(uint64_t(s["id"]));
|
||||
acc += static_cast<double>(std::string_view(s["user"]["screen_name"]).size());
|
||||
acc += static_cast<double>(int64_t(s["retweet_count"]));
|
||||
};
|
||||
if (name == "twitter")
|
||||
{
|
||||
for (auto s : d["statuses"])
|
||||
{
|
||||
one(s.get_object());
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
one(d.get_object());
|
||||
}
|
||||
}
|
||||
else if (name == "citm_catalog")
|
||||
{
|
||||
simdjson::ondemand::object ev = d["events"].get_object();
|
||||
acc += static_cast<double>(ev.count_fields());
|
||||
for (auto p : d["performances"])
|
||||
{
|
||||
simdjson::ondemand::object o = p.get_object();
|
||||
const auto a = uint64_t(o["eventId"]) + uint64_t(o["id"]);
|
||||
simdjson::ondemand::array sc = o["seatCategories"].get_array();
|
||||
acc += static_cast<double>(a + sc.count_elements());
|
||||
}
|
||||
}
|
||||
else if (name == "canada")
|
||||
{
|
||||
acc += static_cast<double>(std::string_view(d["type"]).size());
|
||||
auto g = d["features"].at(0)["geometry"];
|
||||
acc += static_cast<double>(std::string_view(g["type"]).size());
|
||||
simdjson::ondemand::array co = g["coordinates"].get_array();
|
||||
acc += static_cast<double>(co.count_elements());
|
||||
}
|
||||
else if (name == "jeopardy")
|
||||
{
|
||||
for (auto q : d)
|
||||
{
|
||||
simdjson::ondemand::object o = q.get_object();
|
||||
acc += static_cast<double>(std::string_view(o["category"]).size());
|
||||
acc += std::string_view(o["round"]) == "Final Jeopardy!" ? 1 : 0;
|
||||
}
|
||||
}
|
||||
else if (name == "rpc")
|
||||
{
|
||||
acc += static_cast<double>(std::string_view(d["method"]).size());
|
||||
acc += static_cast<double>(int64_t(d["params"]["minuend"]));
|
||||
acc += static_cast<double>(int64_t(d["id"]));
|
||||
}
|
||||
return acc;
|
||||
}
|
||||
|
||||
// ---------------- harness ----------------
|
||||
|
||||
static std::string slurp(const std::string& p)
|
||||
{
|
||||
std::ifstream f(p, std::ios::binary);
|
||||
std::stringstream ss;
|
||||
ss << f.rdbuf();
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
struct engine
|
||||
{
|
||||
std::string name;
|
||||
std::function<void()> fn;
|
||||
};
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
if (argc < 2)
|
||||
{
|
||||
std::fprintf(stderr, "usage: %s <json_test_data directory> [rounds] [document]\n", argv[0]);
|
||||
return 1;
|
||||
}
|
||||
const std::string T = std::string(argv[1]) + "/";
|
||||
const int rounds = argc > 2 ? std::atoi(argv[2]) : 30;
|
||||
const std::string only = argc > 3 ? argv[3] : "";
|
||||
struct doc
|
||||
{
|
||||
std::string name, text;
|
||||
int batch;
|
||||
};
|
||||
std::vector<doc> docs;
|
||||
for (const char* f :
|
||||
{"nativejson-benchmark/twitter.json", "nativejson-benchmark/citm_catalog.json", "nativejson-benchmark/canada.json", "jeopardy/jeopardy.json"
|
||||
})
|
||||
{
|
||||
std::string n = std::string(f).substr(std::string(f).find('/') + 1);
|
||||
docs.push_back({n.substr(0, n.size() - 5), slurp(T + f), 1});
|
||||
}
|
||||
docs.push_back({"status", json::parse(docs[0].text)["statuses"][0].dump(), 200});
|
||||
docs.push_back({"rpc", R"({"jsonrpc": "2.0", "method": "subtract", "params": {"minuend": 42, "subtrahend": 23}, "id": 3})", 5000});
|
||||
|
||||
// correctness cross-check of the workloads
|
||||
for (const auto& dc : docs)
|
||||
{
|
||||
stats a, b, c, dd;
|
||||
walk(json::parse(dc.text), a);
|
||||
auto d = json_document::parse(dc.text);
|
||||
walk(d.root(), b);
|
||||
yyjson_doc* y = yyjson_read(dc.text.data(), dc.text.size(), 0);
|
||||
walk(yyjson_doc_get_root(y), c);
|
||||
simdjson::dom::parser p;
|
||||
walk(p.parse(dc.text).value(), dd);
|
||||
const bool ok = a.nodes == b.nodes && a.nodes == c.nodes && a.nodes == dd.nodes && a.str == b.str && a.str == c.str && a.str == dd.str
|
||||
&& std::fabs(a.num - b.num) <= 1e-9 * std::fabs(a.num) && std::fabs(a.num - c.num) <= 1e-9 * std::fabs(a.num);
|
||||
const double pa = pick(dc.name, json::parse(dc.text)), pb = pick(dc.name, d.root()), pc = pick(dc.name, yyjson_doc_get_root(y)), pd = pick(dc.name, p.parse(dc.text).value());
|
||||
std::printf("check %-13s traverse %s select %s\n", dc.name.c_str(), ok ? "OK" : "MISMATCH", (pa == pb && pa == pc && pa == pd) ? "OK" : "MISMATCH");
|
||||
yyjson_doc_free(y);
|
||||
}
|
||||
|
||||
std::FILE* csv = std::fopen("bench_view.csv", "w");
|
||||
std::fprintf(csv, "doc,bytes,workload,engine,ns\n");
|
||||
json_document reused;
|
||||
simdjson::dom::parser sj;
|
||||
simdjson::ondemand::parser od;
|
||||
for (const auto& dc : docs)
|
||||
{
|
||||
if (!only.empty() && dc.name != only)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
const std::string& s = dc.text;
|
||||
const simdjson::padded_string ps(s);
|
||||
const std::string name = dc.name;
|
||||
std::vector<std::pair<std::string, std::vector<engine>>> workloads;
|
||||
|
||||
workloads.push_back({"parse", {
|
||||
{"json_view", [&] { auto d = json_document::parse(s); g_sink = static_cast<double>(d.node_count()); }},
|
||||
{"json_view (reused)", [&] { reused.read(s); g_sink = static_cast<double>(reused.node_count()); }},
|
||||
{"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast<double>(yyjson_doc_get_val_count(d)); yyjson_doc_free(d); }},
|
||||
{"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }},
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
{"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }},
|
||||
#endif
|
||||
{"json::parse", [&] { json j = json::parse(s); g_sink = static_cast<double>(j.size()); }},
|
||||
}});
|
||||
workloads.push_back({"traverse", {
|
||||
{"json_view", [&] { auto d = json_document::parse(s); stats st; walk(d.root(), st); g_sink = st.num; }},
|
||||
{"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(d), st); g_sink = st.num; yyjson_doc_free(d); }},
|
||||
{"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }},
|
||||
{"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); stats st; walk_od(d.get_value().value_unsafe(), st); g_sink = st.num; }},
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
{"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); stats st; walk(v, st); g_sink = st.num; }},
|
||||
#endif
|
||||
{"json::parse", [&] { json j = json::parse(s); stats st; walk(j, st); g_sink = st.num; }},
|
||||
}});
|
||||
workloads.push_back({"select", {
|
||||
{"json_view", [&] { auto d = json_document::parse(s); g_sink = pick(name, d.root()); }},
|
||||
{"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = pick(name, yyjson_doc_get_root(d)); yyjson_doc_free(d); }},
|
||||
{"simdjson DOM", [&] { g_sink = pick(name, sj.parse(ps).value_unsafe()); }},
|
||||
{"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); g_sink = pick_od(name, d); }},
|
||||
{"json::parse", [&] { json j = json::parse(s); g_sink = pick(name, j); }},
|
||||
}});
|
||||
{
|
||||
// serialization of an already parsed document
|
||||
static json_document vd;
|
||||
vd.read(s);
|
||||
static yyjson_doc* yd = nullptr;
|
||||
if (yd)
|
||||
{
|
||||
yyjson_doc_free(yd);
|
||||
}
|
||||
yd = yyjson_read(s.data(), s.size(), 0);
|
||||
static simdjson::dom::parser sjd;
|
||||
static simdjson::dom::element se;
|
||||
se = sjd.parse(ps).value_unsafe();
|
||||
static json jd;
|
||||
jd = json::parse(s);
|
||||
workloads.push_back({"dump", {
|
||||
{"json_view", [&] { std::string o = vd.root().dump(); g_sink = static_cast<double>(o.size()); }},
|
||||
{"yyjson", [&] { std::size_t n = 0; char* o = yyjson_write(yd, 0, &n); g_sink = static_cast<double>(n); std::free(o); }},
|
||||
{"simdjson DOM", [&] { std::string o = simdjson::to_string(se); g_sink = static_cast<double>(o.size()); }},
|
||||
{"json::parse", [&] { std::string o = jd.dump(); g_sink = static_cast<double>(o.size()); }},
|
||||
}});
|
||||
}
|
||||
|
||||
for (auto& wl : workloads)
|
||||
{
|
||||
std::vector<double> best(wl.second.size(), 1e300);
|
||||
const int r = s.size() > 10000000 ? std::max(3, rounds / 5) : rounds;
|
||||
for (int i = 0; i < r; ++i)
|
||||
{
|
||||
for (std::size_t k = 0; k < wl.second.size(); ++k)
|
||||
{
|
||||
const auto t0 = std::chrono::steady_clock::now();
|
||||
for (int b = 0; b < dc.batch; ++b)
|
||||
{
|
||||
wl.second[k].fn();
|
||||
}
|
||||
const double ns = std::chrono::duration<double, std::nano>(std::chrono::steady_clock::now() - t0).count() / dc.batch;
|
||||
best[k] = std::min(best[k], ns);
|
||||
}
|
||||
}
|
||||
const double ref = best[0];
|
||||
std::printf("%-13s %-9s", dc.name.c_str(), wl.first.c_str());
|
||||
for (std::size_t k = 0; k < wl.second.size(); ++k)
|
||||
{
|
||||
const double us = best[k] / 1e3;
|
||||
std::printf(" %s %s%s (%.2fx)", wl.second[k].name.c_str(), us >= 100 ? "" : "", (us >= 1000 ? std::to_string(static_cast<long>(us)) + "us" : (std::to_string(us).substr(0, 5) + "us")).c_str(), best[k] / ref);
|
||||
std::fprintf(csv, "%s,%zu,%s,%s,%.1f\n", dc.name.c_str(), s.size(), wl.first.c_str(), wl.second[k].name.c_str(), best[k]);
|
||||
}
|
||||
std::printf("\n");
|
||||
std::fflush(stdout);
|
||||
}
|
||||
}
|
||||
std::fclose(csv);
|
||||
}
|
||||
Executable
+300
@@ -0,0 +1,300 @@
|
||||
#!/usr/bin/env python3
|
||||
# __ _____ _____ _____
|
||||
# __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
# | | |__ | | | | | | version 3.12.0
|
||||
# |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
#
|
||||
# SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""Compare json_view with yyjson, simdjson, Boost.JSON, and json::parse.
|
||||
|
||||
Builds bench_view.cpp and bench_corpus.cpp against the include/ directory of
|
||||
this checkout, runs them, and writes the results with everything needed to
|
||||
reproduce them (date, commit, CPU, OS, compiler, library versions, flags) to
|
||||
results/<date>-<host>.md and .csv next to this script.
|
||||
|
||||
The other libraries come from the system (--system, the default: pkg-config
|
||||
or Homebrew) or are downloaded as pinned releases and checked against their
|
||||
SHA-256 (--download). Boost.JSON is optional: without Boost headers, its
|
||||
columns are skipped, and the results say so.
|
||||
|
||||
Only the Python 3 standard library is used; a C++17 compiler is needed.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import datetime
|
||||
import hashlib
|
||||
import os
|
||||
import platform
|
||||
import re
|
||||
import shlex
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tarfile
|
||||
import urllib.request
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
REPO = os.path.abspath(os.path.join(HERE, '..', '..', '..'))
|
||||
|
||||
# pinned releases for --download; the hashes are those of the archives
|
||||
PINNED = {
|
||||
'yyjson': {
|
||||
'version': '0.13.0',
|
||||
'url': 'https://github.com/ibireme/yyjson/archive/refs/tags/0.13.0.tar.gz',
|
||||
'sha256': '34e0f62a2bc11ab20d601e8ca1cc2b2079503aa45119a19133d89d19b94a0fae',
|
||||
'dir': 'yyjson-0.13.0',
|
||||
},
|
||||
'simdjson': {
|
||||
'version': '4.6.11',
|
||||
'url': 'https://github.com/simdjson/simdjson/archive/refs/tags/v4.6.11.tar.gz',
|
||||
'sha256': '61d948fc24f0d793829ad658058e7597d064988a89b4607ea02e401a82df98ff',
|
||||
'dir': 'simdjson-4.6.11',
|
||||
},
|
||||
'boost': {
|
||||
'version': '1.92.0',
|
||||
'url': 'https://archives.boost.io/release/1.92.0/source/boost_1_92_0.tar.gz',
|
||||
'sha256': 'c4a3b310ddd2472416e091067166b0713be97c63f38c212c484ada022fd296ce',
|
||||
'dir': 'boost_1_92_0',
|
||||
},
|
||||
}
|
||||
|
||||
# the documents of bench_view.cpp, relative to the json_test_data directory
|
||||
DEFAULT_CORPUS = [
|
||||
'nativejson-benchmark/twitter.json',
|
||||
'nativejson-benchmark/citm_catalog.json',
|
||||
'nativejson-benchmark/canada.json',
|
||||
'jeopardy/jeopardy.json',
|
||||
]
|
||||
|
||||
|
||||
def run(cmd, **kwargs):
|
||||
print('+ ' + ' '.join(shlex.quote(c) for c in cmd), flush=True)
|
||||
return subprocess.run(cmd, check=True, **kwargs)
|
||||
|
||||
|
||||
def output(cmd):
|
||||
try:
|
||||
return subprocess.run(cmd, check=True, capture_output=True, text=True).stdout.strip()
|
||||
except (OSError, subprocess.CalledProcessError):
|
||||
return ''
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# libraries
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class Library:
|
||||
"""include directories, sources to compile, and linker flags of a library"""
|
||||
|
||||
def __init__(self, name, include=None, sources=None, link=None, version=''):
|
||||
self.name = name
|
||||
self.include = include or []
|
||||
self.sources = sources or []
|
||||
self.link = link or []
|
||||
self.version = version
|
||||
|
||||
|
||||
def header_version(path, pattern):
|
||||
try:
|
||||
with open(path, encoding='utf-8', errors='replace') as f:
|
||||
m = re.search(pattern, f.read())
|
||||
return m.group(1) if m else ''
|
||||
except OSError:
|
||||
return ''
|
||||
|
||||
|
||||
def library_version(name, include_dirs):
|
||||
patterns = {
|
||||
'yyjson': ('yyjson.h', r'#define\s+YYJSON_VERSION_STRING\s+"([^"]+)"'),
|
||||
'simdjson': ('simdjson.h', r'#define\s+SIMDJSON_VERSION\s+"?([0-9.]+)"?'),
|
||||
'boost': (os.path.join('boost', 'version.hpp'), r'#define\s+BOOST_LIB_VERSION\s+"([^"]+)"'),
|
||||
}
|
||||
header, pattern = patterns[name]
|
||||
for d in include_dirs:
|
||||
v = header_version(os.path.join(d, header), pattern)
|
||||
if v:
|
||||
return v.replace('_', '.')
|
||||
return ''
|
||||
|
||||
|
||||
def system_library(name):
|
||||
"""a library found with pkg-config or Homebrew, or None"""
|
||||
flags = output(['pkg-config', '--cflags', '--libs', name]).split()
|
||||
if flags:
|
||||
include = [f[2:] for f in flags if f.startswith('-I')]
|
||||
link = [f for f in flags if f.startswith('-L') or f.startswith('-l')]
|
||||
libdirs = [f[2:] for f in link if f.startswith('-L')]
|
||||
link += ['-Wl,-rpath,' + d for d in libdirs]
|
||||
return Library(name, include, [], link, library_version(name, include))
|
||||
prefix = output(['brew', '--prefix', name]) if shutil.which('brew') else ''
|
||||
if prefix and os.path.isdir(os.path.join(prefix, 'include')):
|
||||
include = [os.path.join(prefix, 'include')]
|
||||
link = []
|
||||
if name != 'boost':
|
||||
lib = os.path.join(prefix, 'lib')
|
||||
link = ['-L' + lib, '-l' + name, '-Wl,-rpath,' + lib]
|
||||
return Library(name, include, [], link, library_version(name, include))
|
||||
if name == 'boost':
|
||||
for d in ['/usr/include', '/usr/local/include']:
|
||||
if os.path.isfile(os.path.join(d, 'boost', 'json.hpp')):
|
||||
return Library(name, [d], [], [], library_version(name, [d]))
|
||||
return None
|
||||
|
||||
|
||||
def download_library(name, work):
|
||||
"""a pinned release, downloaded and checked, or an error"""
|
||||
pin = PINNED[name]
|
||||
archive = os.path.join(work, 'download', os.path.basename(pin['url']))
|
||||
os.makedirs(os.path.dirname(archive), exist_ok=True)
|
||||
if not os.path.isfile(archive):
|
||||
print(f'downloading {pin["url"]}', flush=True)
|
||||
urllib.request.urlretrieve(pin['url'], archive)
|
||||
with open(archive, 'rb') as f:
|
||||
digest = hashlib.sha256(f.read()).hexdigest()
|
||||
if digest != pin['sha256']:
|
||||
sys.exit(f'error: SHA-256 of {archive} is {digest}, expected {pin["sha256"]}')
|
||||
src = os.path.join(work, 'download', pin['dir'])
|
||||
if not os.path.isdir(src):
|
||||
with tarfile.open(archive) as t:
|
||||
# (the 'data' filter rejects links and paths outside the target where Python has it)
|
||||
kwargs = {'filter': 'data'} if hasattr(tarfile, 'data_filter') else {}
|
||||
t.extractall(os.path.join(work, 'download'), **kwargs) # noqa: S202 (checked archive)
|
||||
if name == 'yyjson':
|
||||
return Library(name, [os.path.join(src, 'src')], [os.path.join(src, 'src', 'yyjson.c')], [], pin['version'])
|
||||
if name == 'simdjson':
|
||||
single = os.path.join(src, 'singleheader')
|
||||
return Library(name, [single], [os.path.join(single, 'simdjson.cpp')], [], pin['version'])
|
||||
return Library(name, [src], [], [], pin['version'])
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# machine description
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def cpu_model():
|
||||
if sys.platform == 'darwin':
|
||||
return output(['sysctl', '-n', 'machdep.cpu.brand_string'])
|
||||
try:
|
||||
with open('/proc/cpuinfo', encoding='utf-8') as f:
|
||||
for line in f:
|
||||
if line.startswith('model name') or line.startswith('Model'):
|
||||
return line.split(':', 1)[1].strip()
|
||||
except OSError:
|
||||
pass
|
||||
# (AArch64 Linux: /proc/cpuinfo has no model name, lscpu knows it)
|
||||
for line in output(['lscpu']).splitlines():
|
||||
if line.startswith('Model name:'):
|
||||
return line.split(':', 1)[1].strip()
|
||||
return platform.processor() or platform.machine()
|
||||
|
||||
|
||||
def git_commit():
|
||||
commit = output(['git', '-C', REPO, 'rev-parse', '--short=12', 'HEAD'])
|
||||
dirty = output(['git', '-C', REPO, 'status', '--porcelain', '--untracked-files=no'])
|
||||
return commit + (' (with local changes)' if dirty else '')
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# main
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('--data', required=True, help='json_test_data directory (with nativejson-benchmark/ and jeopardy/)')
|
||||
ap.add_argument('--download', action='store_true', help='use pinned downloads instead of system libraries')
|
||||
ap.add_argument('--no-boost', action='store_true', help='skip Boost.JSON')
|
||||
ap.add_argument('--native', action='store_true', help='compile for this CPU (-march=native / -mcpu=native)')
|
||||
ap.add_argument('--rounds', type=int, default=30, help='rounds of bench_view (default: 30)')
|
||||
ap.add_argument('--corpus', nargs='*', default=[], help='more files for bench_corpus')
|
||||
ap.add_argument('--build-dir', default=os.path.join(HERE, 'build'), help='where to build (default: build/ next to this script)')
|
||||
args = ap.parse_args()
|
||||
|
||||
cxx = os.environ.get('CXX', 'c++')
|
||||
cc = os.environ.get('CC', 'cc')
|
||||
os.makedirs(args.build_dir, exist_ok=True)
|
||||
|
||||
libs = {}
|
||||
for name in ['yyjson', 'simdjson', 'boost']:
|
||||
if name == 'boost' and args.no_boost:
|
||||
continue
|
||||
lib = download_library(name, args.build_dir) if args.download else system_library(name)
|
||||
if lib is None and name != 'boost':
|
||||
sys.exit(f'error: {name} not found; install it, or use --download')
|
||||
if lib is not None:
|
||||
libs[name] = lib
|
||||
with_boost = 'boost' in libs
|
||||
if not with_boost:
|
||||
print('Boost.JSON not found: its columns are skipped', flush=True)
|
||||
|
||||
flags = ['-std=c++17', '-O3', '-DNDEBUG', f'-DJSON_VIEW_BENCH_BOOST={1 if with_boost else 0}']
|
||||
if args.native:
|
||||
flags.append('-mcpu=native' if platform.machine().lower() in ('arm64', 'aarch64') else '-march=native')
|
||||
include = ['-I' + os.path.join(REPO, 'include')] + ['-I' + d for lib in libs.values() for d in lib.include]
|
||||
link = [f for lib in libs.values() for f in lib.link]
|
||||
|
||||
# C sources of downloaded libraries are compiled once
|
||||
objects = []
|
||||
for lib in libs.values():
|
||||
for src in lib.sources:
|
||||
obj = os.path.join(args.build_dir, os.path.basename(src) + '.o')
|
||||
compiler = cc if src.endswith('.c') else cxx
|
||||
run([compiler] + (['-std=c++17'] if compiler == cxx else []) + ['-O3', '-DNDEBUG', '-c', src, '-o', obj]
|
||||
+ ['-I' + d for d in lib.include])
|
||||
objects.append(obj)
|
||||
|
||||
binaries = {}
|
||||
for bench in ['bench_view', 'bench_corpus']:
|
||||
exe = os.path.join(args.build_dir, bench)
|
||||
run([cxx] + flags + include + [os.path.join(HERE, bench + '.cpp')] + objects + link + ['-o', exe])
|
||||
binaries[bench] = exe
|
||||
|
||||
# run: bench_view on its documents, bench_corpus on those and the given files
|
||||
corpus = [os.path.join(args.data, f) for f in DEFAULT_CORPUS] + args.corpus
|
||||
outputs = {}
|
||||
outputs['bench_view'] = run([binaries['bench_view'], args.data, str(args.rounds)], cwd=args.build_dir,
|
||||
capture_output=True, text=True).stdout
|
||||
outputs['bench_corpus'] = run([binaries['bench_corpus']] + corpus, cwd=args.build_dir,
|
||||
capture_output=True, text=True).stdout
|
||||
for name, text in outputs.items():
|
||||
print(text)
|
||||
|
||||
# results with their metadata
|
||||
now = datetime.datetime.now()
|
||||
host = re.sub(r'[^A-Za-z0-9-]+', '-', platform.node().split('.')[0]) or 'host'
|
||||
stem = os.path.join(HERE, 'results', f'{now:%Y-%m-%d}-{host}')
|
||||
os.makedirs(os.path.dirname(stem), exist_ok=True)
|
||||
meta = [
|
||||
('date', f'{now:%Y-%m-%d %H:%M}'),
|
||||
('commit', git_commit()),
|
||||
('CPU', cpu_model()),
|
||||
('OS', f'{platform.system()} {platform.release()} ({platform.machine()})'),
|
||||
('compiler', output([cxx, '--version']).splitlines()[0] if output([cxx, '--version']) else cxx),
|
||||
('flags', ' '.join(flags)),
|
||||
('yyjson', libs['yyjson'].version),
|
||||
('simdjson', libs['simdjson'].version),
|
||||
('Boost.JSON', libs['boost'].version if with_boost else 'skipped (not found)'),
|
||||
('libraries from', 'pinned downloads' if args.download else 'the system'),
|
||||
('rounds', str(args.rounds)),
|
||||
]
|
||||
with open(stem + '.md', 'w', encoding='utf-8') as f:
|
||||
f.write(f'# json_view comparison, {now:%Y-%m-%d}\n\n')
|
||||
f.write('Generated by `tests/benchmarks/json_view/compare.py`; best of the interleaved rounds.\n\n')
|
||||
f.write('| | |\n|---|---|\n')
|
||||
for key, value in meta:
|
||||
f.write(f'| {key} | {value} |\n')
|
||||
for name, text in outputs.items():
|
||||
f.write(f'\n## {name}\n\n```\n{text.rstrip()}\n```\n')
|
||||
with open(stem + '.csv', 'w', encoding='utf-8') as out:
|
||||
out.write(''.join(f'# {key}: {value}\n' for key, value in meta))
|
||||
for name in ['bench_view', 'bench_corpus']:
|
||||
path = os.path.join(args.build_dir, name + '.csv')
|
||||
if os.path.isfile(path):
|
||||
with open(path, encoding='utf-8') as f:
|
||||
out.write(f'# {name}\n' + f.read())
|
||||
print(f'results: {stem}.md, {stem}.csv')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -826,7 +826,12 @@ TEST_CASE("json_view values")
|
||||
std::mt19937_64 rng(5295); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed)
|
||||
std::vector<std::string> tokens = {"0.1", "-0.0", "1e308", "1.7976931348623157e308", "2.2250738585072011e-308", "4.9e-324", "5e-324",
|
||||
"0.1000000000000000055511151231257827021181583404541015625", "123456789012345678901234567890",
|
||||
"9007199254740993", "1.00000000000000011102230246251565404236316680908203125", "7.2057594037927933e16"
|
||||
"9007199254740993", "1.00000000000000011102230246251565404236316680908203125", "7.2057594037927933e16",
|
||||
// around the limits of the conversion from the digit layout: 19 and 20
|
||||
// digits, and those of Clinger's fast path (2^53, 10^22)
|
||||
"1234567890.123456789", "1234567890.1234567891", "0.0000000000000000001", "123456789012345678.9",
|
||||
"9007199254740992.0", "9007199254740993.0", "9007199254740994.0", "1.5e22", "1.5e23", "15e-22", "15e-23",
|
||||
"1e-400", "0.0e0", "-0.0e-5", "12E+3", "12e-0"
|
||||
};
|
||||
for (int i = 0; i < 20000; ++i)
|
||||
{
|
||||
@@ -1130,3 +1135,87 @@ TEST_CASE("json_view dump")
|
||||
CHECK(json_view().dump() == json(json::value_t::discarded).dump());
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view comparison")
|
||||
{
|
||||
SECTION("equality of the values parse() produces")
|
||||
{
|
||||
generator g;
|
||||
std::vector<std::string> texts;
|
||||
for (int i = 0; i < 600; ++i)
|
||||
{
|
||||
std::string text;
|
||||
g.value(text, 0);
|
||||
texts.push_back(text);
|
||||
// the same value written differently: sorted keys, canonical numbers
|
||||
texts.push_back(json::parse(text).dump(1));
|
||||
}
|
||||
for (std::size_t i = 0; i + 2 < texts.size(); ++i)
|
||||
{
|
||||
for (std::size_t k = i; k < i + 3; ++k)
|
||||
{
|
||||
CAPTURE(texts[i]);
|
||||
CAPTURE(texts[k]);
|
||||
const json_document a = json_document::parse(texts[i]);
|
||||
const json_document b = json_document::parse(texts[k]);
|
||||
const json ja = json::parse(texts[i]);
|
||||
const json jb = json::parse(texts[k]);
|
||||
CHECK((a.root() == b.root()) == (ja == jb));
|
||||
CHECK((a.root() != b.root()) == (ja != jb));
|
||||
CHECK((a.root() == jb) == (ja == jb));
|
||||
CHECK((jb == a.root()) == (ja == jb));
|
||||
CHECK((a.root() != jb) == (ja != jb));
|
||||
CHECK((jb != a.root()) == (ja != jb));
|
||||
|
||||
// ordered_json compares members in order
|
||||
const ordered_json_document oa = ordered_json_document::parse(texts[i]);
|
||||
const ordered_json_document ob = ordered_json_document::parse(texts[k]);
|
||||
const ordered_json oja = ordered_json::parse(texts[i]);
|
||||
const ordered_json ojb = ordered_json::parse(texts[k]);
|
||||
CHECK((oa.root() == ob.root()) == (oja == ojb));
|
||||
CHECK((oa.root() == ojb) == (oja == ojb));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("numbers, duplicate keys, member order")
|
||||
{
|
||||
const auto same = [](const char* x, const char* y)
|
||||
{
|
||||
return json_document::parse(x).root() == json_document::parse(y).root();
|
||||
};
|
||||
CHECK(same("1", "1.0"));
|
||||
CHECK(same("[1, -1, 2.5]", "[1.0, -1.0, 25e-1]"));
|
||||
CHECK(!same("1", "1.5"));
|
||||
CHECK(same("18446744073709551615", "18446744073709551615"));
|
||||
CHECK(same(R"({"a": 1, "a": 2})", R"({"a": 2})"));
|
||||
CHECK(!same(R"({"a": 1, "a": 2})", R"({"a": 1})"));
|
||||
CHECK(same(R"({"a": 1, "b": 2})", R"({"b": 2, "a": 1})"));
|
||||
CHECK(!same(R"({"a": 1})", R"({"a": 1, "b": 2})"));
|
||||
CHECK(!same("[1, 2]", "[2, 1]"));
|
||||
CHECK(!same("\"a\"", "\"b\""));
|
||||
CHECK(same("\"\\u00e9\"", "\"\xc3\xa9\""));
|
||||
CHECK(!same("null", "false"));
|
||||
CHECK(!same("[]", "{}"));
|
||||
CHECK(ordered_json_document::parse(R"({"a": 1, "b": 2, "a": 3})").root() == ordered_json_document::parse(R"({"a": 3, "b": 2})").root());
|
||||
CHECK(ordered_json_document::parse(R"({"a": 1, "b": 2})").root() != ordered_json_document::parse(R"({"b": 2, "a": 1})").root());
|
||||
|
||||
// discarded values compare as basic_json's do
|
||||
const json discarded(json::value_t::discarded);
|
||||
CHECK((json_view() == json_view()) == (discarded == discarded)); // NOLINT(readability-container-size-empty): operator== is tested
|
||||
CHECK((json_view() == discarded) == (discarded == discarded));
|
||||
CHECK(!(json_view() == json_document::parse("null").root())); // NOLINT(readability-container-size-empty)
|
||||
CHECK(!(json_document::parse("null").root() == discarded));
|
||||
}
|
||||
|
||||
SECTION("deep nesting")
|
||||
{
|
||||
const std::string deep = std::string(100000, '[') + std::string(100000, ']');
|
||||
const json_document a = json_document::parse(deep);
|
||||
const json_document b = json_document::parse(deep);
|
||||
CHECK(a.root() == b.root());
|
||||
CHECK(a.root() == json::parse(deep));
|
||||
const std::string other = std::string(100000, '[') + "1" + std::string(100000, ']');
|
||||
CHECK(a.root() != json_document::parse(other).root());
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user