mirror of
https://github.com/nlohmann/json.git
synced 2026-09-02 22:47:14 +00:00
Compare commits
101
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
984893f753 | ||
|
|
1de8915313 | ||
|
|
85756b50cd | ||
|
|
12016a69a8 | ||
|
|
7db19dbf91 | ||
|
|
e73bfd0f25 | ||
|
|
c72895ab35 | ||
|
|
410ad9f402 | ||
|
|
fadcde4481 | ||
|
|
414ee33d88 | ||
|
|
1b84c1604f | ||
|
|
7c5146442a | ||
|
|
d1fcc0532f | ||
|
|
aebfb39f71 | ||
|
|
d4d4720a8c | ||
|
|
56fc1409b7 | ||
|
|
2f5573abc4 | ||
|
|
b3c6db6dc0 | ||
|
|
ec266c7b1d | ||
|
|
0386005cbd | ||
|
|
c9b9d7fc74 | ||
|
|
45f1f8f17a | ||
|
|
fbf3f28b41 | ||
|
|
3e30fc3bf1 | ||
|
|
ecd8003be1 | ||
|
|
f334d0e433 | ||
|
|
f800004f49 | ||
|
|
22e4061d03 | ||
|
|
3f7760ebba | ||
|
|
22d05f5ef5 | ||
|
|
0197a0e98b | ||
|
|
56d6b7b36e | ||
|
|
89334dd55f | ||
|
|
4170fe0b39 | ||
|
|
9e4b3b128e | ||
|
|
34c58fa8d2 | ||
|
|
4fc04afc8f | ||
|
|
27f5812bc5 | ||
|
|
079c297fa7 | ||
|
|
0f8724d646 | ||
|
|
21b6fec01f | ||
|
|
22688bf40d | ||
|
|
f7b068eff6 | ||
|
|
c74e7a23aa | ||
|
|
80dfbc3a30 | ||
|
|
ea3dd23f29 | ||
|
|
6a0c3ee9f1 | ||
|
|
4dfa01c07e | ||
|
|
980b5f344d | ||
|
|
7a522b00dc | ||
|
|
2bc08ada75 | ||
|
|
f2915a86ac | ||
|
|
1a6fb6a27b | ||
|
|
642c77f534 | ||
|
|
72cadadfc5 | ||
|
|
2c15b0617b | ||
|
|
4d606cdae5 | ||
|
|
a5895127b6 | ||
|
|
c864bb36da | ||
|
|
129d2891ed | ||
|
|
35705d79d8 | ||
|
|
892be68ca4 | ||
|
|
1ac268d409 | ||
|
|
3fa93dac65 | ||
|
|
1876493f87 | ||
|
|
01853ed6bc | ||
|
|
2f025f401e | ||
|
|
734fd305a1 | ||
|
|
36187cacfb | ||
|
|
b5378e8deb | ||
|
|
ce87157d4e | ||
|
|
cdf52ae9be | ||
|
|
146ba55453 | ||
|
|
e6978ba50c | ||
|
|
6285225fd0 | ||
|
|
21af527e75 | ||
|
|
23518f54fe | ||
|
|
1c136a66c4 | ||
|
|
c1c19a7bcd | ||
|
|
bacdabd176 | ||
|
|
d5647e6a3b | ||
|
|
9a091d2b82 | ||
|
|
b890b4cba3 | ||
|
|
dca9d49a33 | ||
|
|
acd87e2336 | ||
|
|
ad94fb01cc | ||
|
|
c2e1cc50e0 | ||
|
|
173f2a7407 | ||
|
|
1c63a120b6 | ||
|
|
85889e8843 | ||
|
|
3c0a9a99fd | ||
|
|
e82724d87f | ||
|
|
78821cd9c2 | ||
|
|
68f0722a19 | ||
|
|
5f121d8c50 | ||
|
|
585929bff9 | ||
|
|
31ba5208c8 | ||
|
|
2222d386c9 | ||
|
|
eaedec859a | ||
|
|
d94cbd99dc | ||
|
|
bc48951128 |
@@ -11,7 +11,7 @@ jobs:
|
|||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
@@ -34,7 +34,7 @@ jobs:
|
|||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
@@ -67,8 +67,18 @@ jobs:
|
|||||||
${{ github.workspace }}/venv/bin/astyle --project=tools/astyle/.astylerc --suffix=none --quiet \
|
${{ github.workspace }}/venv/bin/astyle --project=tools/astyle/.astylerc --suffix=none --quiet \
|
||||||
$INCLUDE_DIR/json.hpp $INCLUDE_DIR/json_fwd.hpp
|
$INCLUDE_DIR/json.hpp $INCLUDE_DIR/json_fwd.hpp
|
||||||
|
|
||||||
|
# fail loudly if a directory is renamed or removed: find would only warn
|
||||||
|
# about the missing path and silently drop its files from the check
|
||||||
|
SOURCE_DIRS="docs/mkdocs/docs/examples include tests"
|
||||||
|
for DIR in $SOURCE_DIRS; do
|
||||||
|
if [ ! -d "$DIR" ]; then
|
||||||
|
echo "::error::source directory '$DIR' does not exist"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
${{ github.workspace }}/venv/bin/astyle --project=tools/astyle/.astylerc --suffix=none --quiet \
|
${{ github.workspace }}/venv/bin/astyle --project=tools/astyle/.astylerc --suffix=none --quiet \
|
||||||
$(find docs/examples include tests -type f \( -name '*.hpp' -o -name '*.cpp' -o -name '*.cu' \) -not -path 'tests/thirdparty/*' -not -path 'tests/abi/include/nlohmann/*' | sort)
|
$(find $SOURCE_DIRS -type f \( -name '*.hpp' -o -name '*.cpp' -o -name '*.cu' \) -not -path 'tests/thirdparty/*' -not -path 'tests/abi/include/nlohmann/*' | sort)
|
||||||
|
|
||||||
- name: Build patch and check for differences
|
- name: Build patch and check for differences
|
||||||
id: diff
|
id: diff
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ jobs:
|
|||||||
runs-on: ubuntu-22.04
|
runs-on: ubuntu-22.04
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
|
|||||||
@@ -27,7 +27,7 @@ jobs:
|
|||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
@@ -38,14 +38,14 @@ jobs:
|
|||||||
|
|
||||||
# Initializes the CodeQL tools for scanning.
|
# Initializes the CodeQL tools for scanning.
|
||||||
- name: Initialize CodeQL
|
- name: Initialize CodeQL
|
||||||
uses: github/codeql-action/init@e0647621c2984b5ed2f768cb892365bf2a616ad1 # v4.37.2
|
uses: github/codeql-action/init@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4.37.8
|
||||||
with:
|
with:
|
||||||
languages: c-cpp
|
languages: c-cpp
|
||||||
|
|
||||||
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
|
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
|
||||||
# If this step fails, then you should remove it and run the build manually (see below)
|
# If this step fails, then you should remove it and run the build manually (see below)
|
||||||
- name: Autobuild
|
- name: Autobuild
|
||||||
uses: github/codeql-action/autobuild@e0647621c2984b5ed2f768cb892365bf2a616ad1 # v4.37.2
|
uses: github/codeql-action/autobuild@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4.37.8
|
||||||
|
|
||||||
- name: Perform CodeQL Analysis
|
- name: Perform CodeQL Analysis
|
||||||
uses: github/codeql-action/analyze@e0647621c2984b5ed2f768cb892365bf2a616ad1 # v4.37.2
|
uses: github/codeql-action/analyze@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4.37.8
|
||||||
|
|||||||
@@ -19,7 +19,7 @@ jobs:
|
|||||||
pull-requests: write
|
pull-requests: write
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
|
|||||||
@@ -17,7 +17,7 @@ jobs:
|
|||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
|
|||||||
@@ -27,7 +27,7 @@ jobs:
|
|||||||
security-events: write
|
security-events: write
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
@@ -43,6 +43,6 @@ jobs:
|
|||||||
output: 'flawfinder_results.sarif'
|
output: 'flawfinder_results.sarif'
|
||||||
|
|
||||||
- name: Upload analysis results to GitHub Security tab
|
- name: Upload analysis results to GitHub Security tab
|
||||||
uses: github/codeql-action/upload-sarif@e0647621c2984b5ed2f768cb892365bf2a616ad1 # v4.37.2
|
uses: github/codeql-action/upload-sarif@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4.37.8
|
||||||
with:
|
with:
|
||||||
sarif_file: ${{github.workspace}}/flawfinder_results.sarif
|
sarif_file: ${{github.workspace}}/flawfinder_results.sarif
|
||||||
|
|||||||
@@ -17,7 +17,7 @@ jobs:
|
|||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
|
|||||||
@@ -7,7 +7,6 @@ on:
|
|||||||
- develop
|
- develop
|
||||||
paths:
|
paths:
|
||||||
- docs/mkdocs/**
|
- docs/mkdocs/**
|
||||||
- docs/examples/**
|
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
|
|
||||||
# we don't want to have concurrent jobs, and we don't want to cancel running jobs to avoid broken publications
|
# we don't want to have concurrent jobs, and we don't want to cancel running jobs to avoid broken publications
|
||||||
@@ -27,7 +26,7 @@ jobs:
|
|||||||
runs-on: ubuntu-22.04
|
runs-on: ubuntu-22.04
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
|
|||||||
@@ -36,7 +36,7 @@ jobs:
|
|||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
@@ -76,6 +76,6 @@ jobs:
|
|||||||
|
|
||||||
# Upload the results to GitHub's code scanning dashboard.
|
# Upload the results to GitHub's code scanning dashboard.
|
||||||
- name: "Upload to code-scanning"
|
- name: "Upload to code-scanning"
|
||||||
uses: github/codeql-action/upload-sarif@e0647621c2984b5ed2f768cb892365bf2a616ad1 # v4.37.2
|
uses: github/codeql-action/upload-sarif@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4.37.8
|
||||||
with:
|
with:
|
||||||
sarif_file: results.sarif
|
sarif_file: results.sarif
|
||||||
|
|||||||
@@ -32,7 +32,7 @@ jobs:
|
|||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
@@ -61,7 +61,7 @@ jobs:
|
|||||||
|
|
||||||
# Upload SARIF file generated in previous step
|
# Upload SARIF file generated in previous step
|
||||||
- name: Upload SARIF file
|
- name: Upload SARIF file
|
||||||
uses: github/codeql-action/upload-sarif@e0647621c2984b5ed2f768cb892365bf2a616ad1 # v4.37.2
|
uses: github/codeql-action/upload-sarif@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4.37.8
|
||||||
with:
|
with:
|
||||||
sarif_file: semgrep.sarif
|
sarif_file: semgrep.sarif
|
||||||
if: always()
|
if: always()
|
||||||
|
|||||||
@@ -16,11 +16,11 @@ jobs:
|
|||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
- uses: actions/stale@1e223db275d687790206a7acac4d1a11bd6fe629 # v10.4.0
|
- uses: actions/stale@4391f3da665fdf50b6810c1a66712fb9ba21aa93 # v11.0.0
|
||||||
with:
|
with:
|
||||||
stale-issue-label: 'state: stale'
|
stale-issue-label: 'state: stale'
|
||||||
stale-pr-label: 'state: stale'
|
stale-pr-label: 'state: stale'
|
||||||
|
|||||||
@@ -25,7 +25,7 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DJSON_CI=On
|
run: cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -35,7 +35,7 @@ jobs:
|
|||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
@@ -47,7 +47,7 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DJSON_CI=On
|
run: cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -60,7 +60,7 @@ jobs:
|
|||||||
target: [ci_test_amalgamation, ci_test_single_header, ci_cppcheck, ci_cpplint, ci_reproducible_tests, ci_non_git_tests, ci_offline_testdata, ci_reuse_compliance, ci_test_valgrind]
|
target: [ci_test_amalgamation, ci_test_single_header, ci_cppcheck, ci_cpplint, ci_reproducible_tests, ci_non_git_tests, ci_offline_testdata, ci_reuse_compliance, ci_test_valgrind]
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
@@ -70,7 +70,7 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DJSON_CI=On
|
run: cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -89,7 +89,7 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DJSON_CI=On
|
run: cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -100,7 +100,7 @@ jobs:
|
|||||||
container: ubuntu:focal
|
container: ubuntu:focal
|
||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls]
|
target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_simdutf]
|
||||||
steps:
|
steps:
|
||||||
- name: Install build-essential
|
- name: Install build-essential
|
||||||
run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev
|
run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev
|
||||||
@@ -108,7 +108,7 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DJSON_CI=On
|
run: cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -118,7 +118,7 @@ jobs:
|
|||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
@@ -142,7 +142,7 @@ jobs:
|
|||||||
name: code-coverage-report
|
name: code-coverage-report
|
||||||
path: ${{ github.workspace }}/build/html
|
path: ${{ github.workspace }}/build/html
|
||||||
- name: Publish report to Coveralls
|
- name: Publish report to Coveralls
|
||||||
uses: coverallsapp/github-action@5cbfd81b66ca5d10c19b062c04de0199c215fb6e # v2.3.7
|
uses: coverallsapp/github-action@8d6379e14d29928660c4ba802d8e85393440b329 # v2.3.8
|
||||||
with:
|
with:
|
||||||
github-token: ${{ secrets.GITHUB_TOKEN }}
|
github-token: ${{ secrets.GITHUB_TOKEN }}
|
||||||
path-to-lcov: ${{ github.workspace }}/build/json.info.filtered.noexcept
|
path-to-lcov: ${{ github.workspace }}/build/json.info.filtered.noexcept
|
||||||
@@ -184,7 +184,7 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: CXX=g++-${{ matrix.compiler }} cmake -S . -B build -DJSON_CI=On
|
run: CXX=g++-${{ matrix.compiler }} cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -202,7 +202,7 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DJSON_CI=On
|
run: cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -212,14 +212,14 @@ jobs:
|
|||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
compiler: ['3.4', '3.5', '3.6', '3.7', '3.8', '3.9', '4', '5', '6', '7', '8', '9', '10', '11', '12', '13', '14', '15-bullseye', '16', '17', '18', '19', '20', 'latest']
|
compiler: ['3.4', '3.5', '3.6', '3.7', '3.8', '3.9', '4', '5', '6', '7', '8', '9', '10', '11', '12', '13', '14', '15-bullseye', '16', '17', '18', '19', '20', '21', '22', 'latest']
|
||||||
container: silkeh/clang:${{ matrix.compiler }}
|
container: silkeh/clang:${{ matrix.compiler }}
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Set env FORCE_STDCPPFS_FLAG for clang 7 / 8 / 9 / 10
|
- name: Set env FORCE_STDCPPFS_FLAG for clang 7 / 8 / 9 / 10
|
||||||
run: echo "JSON_FORCED_GLOBAL_COMPILE_OPTIONS=-DJSON_HAS_FILESYSTEM=0;-DJSON_HAS_EXPERIMENTAL_FILESYSTEM=0" >> "$GITHUB_ENV"
|
run: echo "JSON_FORCED_GLOBAL_COMPILE_OPTIONS=-DJSON_HAS_FILESYSTEM=0;-DJSON_HAS_EXPERIMENTAL_FILESYSTEM=0" >> "$GITHUB_ENV"
|
||||||
if: ${{ matrix.compiler == '7' || matrix.compiler == '8' || matrix.compiler == '9' || matrix.compiler == '10' }}
|
if: ${{ matrix.compiler == '7' || matrix.compiler == '8' || matrix.compiler == '9' || matrix.compiler == '10' }}
|
||||||
@@ -239,7 +239,7 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DJSON_CI=On
|
run: cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -259,7 +259,7 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DJSON_CI=On
|
run: cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build with libc++
|
- name: Build with libc++
|
||||||
@@ -286,7 +286,7 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DJSON_CI=On
|
run: cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -306,7 +306,7 @@ jobs:
|
|||||||
# import-std support. Its opt-in token is CMake-version-specific, so pin
|
# import-std support. Its opt-in token is CMake-version-specific, so pin
|
||||||
# CMake to the version whose token is set in tests/module_cpp20/CMakeLists.txt.
|
# CMake to the version whose token is set in tests/module_cpp20/CMakeLists.txt.
|
||||||
- name: Get pinned CMake and ninja
|
- name: Get pinned CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
with:
|
with:
|
||||||
cmakeVersion: 4.3.4
|
cmakeVersion: 4.3.4
|
||||||
# Clang: the std library module is provided by libc++ (the image's libstdc++
|
# Clang: the std library module is provided by libc++ (the image's libstdc++
|
||||||
@@ -332,7 +332,7 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DJSON_CI=On
|
run: cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -347,7 +347,7 @@ jobs:
|
|||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DJSON_CI=On
|
run: cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -359,7 +359,7 @@ jobs:
|
|||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DJSON_CI=On
|
run: cmake -S . -B build -DJSON_CI=On
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -369,7 +369,7 @@ jobs:
|
|||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
@@ -379,7 +379,7 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build -DCMAKE_TOOLCHAIN_FILE=$EMSDK/upstream/emscripten/cmake/Modules/Platform/Emscripten.cmake -GNinja
|
run: cmake -S . -B build -DCMAKE_TOOLCHAIN_FILE=$EMSDK/upstream/emscripten/cmake/Modules/Platform/Emscripten.cmake -GNinja
|
||||||
- name: Build
|
- name: Build
|
||||||
@@ -392,7 +392,7 @@ jobs:
|
|||||||
target: [ci_test_examples, ci_test_build_documentation]
|
target: [ci_test_examples, ci_test_build_documentation]
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
uses: step-security/harden-runner@05e31511f85b41b11d1cf0ef85d0992719546e2c # v2.21.0
|
||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
|
|||||||
@@ -88,7 +88,7 @@ jobs:
|
|||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||||
- name: Get latest CMake and ninja
|
- name: Get latest CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
- name: Set extra CXX_FLAGS for latest std_version
|
- name: Set extra CXX_FLAGS for latest std_version
|
||||||
# /wd5285 silences C5285 emitted by the bundled third-party doctest.h, which
|
# /wd5285 silences C5285 emitted by the bundled third-party doctest.h, which
|
||||||
# specializes std::tuple (newly diagnosed by the VS2026 v145 toolset)
|
# specializes std::tuple (newly diagnosed by the VS2026 v145 toolset)
|
||||||
@@ -153,10 +153,16 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
platform: x64
|
platform: x64
|
||||||
version: 12.2.0 # https://github.com/egor-tensin/setup-mingw/issues/14
|
version: 12.2.0 # https://github.com/egor-tensin/setup-mingw/issues/14
|
||||||
|
# CMAKE_CXX_FLAGS_DEBUG is overridden to drop the default -g: linking
|
||||||
|
# test-regression2_cpp20 intermittently fails with "relocation truncated
|
||||||
|
# to fit: IMAGE_REL_AMD64_SECREL against `.debug_line'" because the
|
||||||
|
# MinGW linker cannot relocate the debug sections this test produces.
|
||||||
|
# The tests are only built and run here, so the debug info is not used.
|
||||||
- name: Run CMake
|
- name: Run CMake
|
||||||
run: cmake -S . -B build ^
|
run: cmake -S . -B build ^
|
||||||
-DCMAKE_CXX_COMPILER="C:/Program Files/LLVM/bin/clang++.exe" ^
|
-DCMAKE_CXX_COMPILER="C:/Program Files/LLVM/bin/clang++.exe" ^
|
||||||
-DCMAKE_CXX_FLAGS="--target=x86_64-w64-mingw32 -stdlib=libstdc++ -pthread" ^
|
-DCMAKE_CXX_FLAGS="--target=x86_64-w64-mingw32 -stdlib=libstdc++ -pthread" ^
|
||||||
|
-DCMAKE_CXX_FLAGS_DEBUG="-g0" ^
|
||||||
-DCMAKE_EXE_LINKER_FLAGS="-lwinpthread" ^
|
-DCMAKE_EXE_LINKER_FLAGS="-lwinpthread" ^
|
||||||
-G"MinGW Makefiles" ^
|
-G"MinGW Makefiles" ^
|
||||||
-DCMAKE_BUILD_TYPE=Debug ^
|
-DCMAKE_BUILD_TYPE=Debug ^
|
||||||
@@ -193,7 +199,7 @@ jobs:
|
|||||||
# import-std support. Its opt-in token is CMake-version-specific, so pin
|
# import-std support. Its opt-in token is CMake-version-specific, so pin
|
||||||
# CMake to the version whose token is set in tests/module_cpp20/CMakeLists.txt.
|
# CMake to the version whose token is set in tests/module_cpp20/CMakeLists.txt.
|
||||||
- name: Get pinned CMake and ninja
|
- name: Get pinned CMake and ninja
|
||||||
uses: lukka/get-cmake@e6906078ebd1ccb8ce51ab4626ac46a1b5a517e3 # v4.4.0
|
uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2
|
||||||
with:
|
with:
|
||||||
cmakeVersion: 4.3.4
|
cmakeVersion: 4.3.4
|
||||||
- name: Run CMake (Debug)
|
- name: Run CMake (Debug)
|
||||||
|
|||||||
@@ -42,6 +42,7 @@
|
|||||||
- [Specializing enum conversion](#specializing-enum-conversion)
|
- [Specializing enum conversion](#specializing-enum-conversion)
|
||||||
- [Binary formats (BSON, CBOR, MessagePack, UBJSON, and BJData)](#binary-formats-bson-cbor-messagepack-ubjson-and-bjdata)
|
- [Binary formats (BSON, CBOR, MessagePack, UBJSON, and BJData)](#binary-formats-bson-cbor-messagepack-ubjson-and-bjdata)
|
||||||
- [Customers](#customers)
|
- [Customers](#customers)
|
||||||
|
- [Ecosystem](#ecosystem)
|
||||||
- [Supported compilers](#supported-compilers)
|
- [Supported compilers](#supported-compilers)
|
||||||
- [Integration](#integration)
|
- [Integration](#integration)
|
||||||
- [CMake](#cmake)
|
- [CMake](#cmake)
|
||||||
@@ -1186,6 +1187,11 @@ The library is used in multiple projects, applications, operating systems, etc.
|
|||||||
|
|
||||||
[](https://json.nlohmann.me/home/customers/)
|
[](https://json.nlohmann.me/home/customers/)
|
||||||
|
|
||||||
|
## Ecosystem
|
||||||
|
|
||||||
|
Beyond projects that use the library, there are third-party projects that build on top of it - schema validators,
|
||||||
|
language bindings, format converters, and the like. See the curated [Ecosystem](https://json.nlohmann.me/community/ecosystem/) page.
|
||||||
|
|
||||||
## Supported compilers
|
## Supported compilers
|
||||||
|
|
||||||
Though it's 2026 already, the support for C++11 is still a bit sparse. Currently, the following compilers are known to work:
|
Though it's 2026 already, the support for C++11 is still a bit sparse. Currently, the following compilers are known to work:
|
||||||
@@ -1801,13 +1807,13 @@ The library itself consists of a single header file licensed under the MIT licen
|
|||||||
- [**amalgamate.py - Amalgamate C source and header files**](https://github.com/edlund/amalgamate) to create a single header file
|
- [**amalgamate.py - Amalgamate C source and header files**](https://github.com/edlund/amalgamate) to create a single header file
|
||||||
- [**American fuzzy lop**](https://lcamtuf.coredump.cx/afl/) for fuzz testing
|
- [**American fuzzy lop**](https://lcamtuf.coredump.cx/afl/) for fuzz testing
|
||||||
- [**AppVeyor**](https://www.appveyor.com) for [continuous integration](https://ci.appveyor.com/project/nlohmann/json) on Windows
|
- [**AppVeyor**](https://www.appveyor.com) for [continuous integration](https://ci.appveyor.com/project/nlohmann/json) on Windows
|
||||||
- [**Artistic Style**](http://astyle.sourceforge.net) for automatic source code indentation
|
- [**Artistic Style**](https://astyle.sourceforge.net) for automatic source code indentation
|
||||||
- [**Clang**](https://clang.llvm.org) for compilation with code sanitizers
|
- [**Clang**](https://clang.llvm.org) for compilation with code sanitizers
|
||||||
- [**CMake**](https://cmake.org) for build automation
|
- [**CMake**](https://cmake.org) for build automation
|
||||||
- [**Codacy**](https://www.codacy.com) for further [code analysis](https://app.codacy.com/gh/nlohmann/json/dashboard)
|
- [**Codacy**](https://www.codacy.com) for further [code analysis](https://app.codacy.com/gh/nlohmann/json/dashboard)
|
||||||
- [**Coveralls**](https://coveralls.io) to measure [code coverage](https://coveralls.io/github/nlohmann/json)
|
- [**Coveralls**](https://coveralls.io) to measure [code coverage](https://coveralls.io/github/nlohmann/json)
|
||||||
- [**Coverity Scan**](https://scan.coverity.com) for [static analysis](https://scan.coverity.com/projects/nlohmann-json)
|
- [**Coverity Scan**](https://scan.coverity.com) for [static analysis](https://scan.coverity.com/projects/nlohmann-json)
|
||||||
- [**cppcheck**](http://cppcheck.sourceforge.net) for static analysis
|
- [**cppcheck**](https://cppcheck.sourceforge.io) for static analysis
|
||||||
- [**doctest**](https://github.com/onqtam/doctest) for the unit tests
|
- [**doctest**](https://github.com/onqtam/doctest) for the unit tests
|
||||||
- [**GitHub Changelog Generator**](https://github.com/skywinder/github-changelog-generator) to generate the [ChangeLog](https://github.com/nlohmann/json/blob/develop/ChangeLog.md)
|
- [**GitHub Changelog Generator**](https://github.com/skywinder/github-changelog-generator) to generate the [ChangeLog](https://github.com/nlohmann/json/blob/develop/ChangeLog.md)
|
||||||
- [**Google Benchmark**](https://github.com/google/benchmark) to implement the benchmarks
|
- [**Google Benchmark**](https://github.com/google/benchmark) to implement the benchmarks
|
||||||
@@ -1822,6 +1828,15 @@ The library itself consists of a single header file licensed under the MIT licen
|
|||||||
|
|
||||||
## Notes
|
## Notes
|
||||||
|
|
||||||
|
### Standards compliance
|
||||||
|
|
||||||
|
The library targets strict conformance with [RFC 8259](https://tools.ietf.org/html/rfc8259.html). Both the original [JSONTestSuite](https://github.com/nst/JSONTestSuite) and its updated revision are exercised in CI; their test data is downloaded from [`nlohmann/json_test_data`](https://github.com/nlohmann/json_test_data) at configure time rather than committed to this repository (see [`tests/src/unit-testsuites.cpp`](https://github.com/nlohmann/json/blob/develop/tests/src/unit-testsuites.cpp)):
|
||||||
|
|
||||||
|
- The updated revision runs all mandatory `y_` (must-accept) and `n_` (must-reject) cases through the strict [`parse()`](https://json.nlohmann.me/api/basic_json/parse/) entry point; the original suite runs its `n_` cases through `parse()` and its `y_` cases through [`operator>>`](https://json.nlohmann.me/api/operator_gtgt/).
|
||||||
|
- The `i_` (implementation-defined) cases are, by RFC 8259, free to be accepted *or* rejected, so "passing all `i_` cases" is not a meaningful conformance metric. The library makes deliberate, documented choices there: nesting depth is not artificially limited, a leading UTF-8 byte order mark is silently ignored, [Unicode noncharacters](https://www.unicode.org/faq/private_use.html#nonchar1) are forwarded unchanged, invalid UTF-8 and lone/unpaired UTF-16 surrogates are rejected (stricter than required), and a number that cannot be stored without becoming `NaN`/`INF` raises [`out_of_range.406`](https://json.nlohmann.me/home/exceptions/#jsonexceptionout_of_range406).
|
||||||
|
|
||||||
|
One behavioral nuance is worth calling out, because a superficial test often misreads it as non-compliance: [`parse()`](https://json.nlohmann.me/api/basic_json/parse/) is strict and rejects trailing data after a value, whereas [`operator>>`](https://json.nlohmann.me/api/operator_gtgt/) follows relaxed iostream semantics — it parses a single value and leaves the stream positioned right after it. Feeding "a valid document followed by trailing bytes" through `operator>>` reports success; the same input through `parse()` is rejected. This is a documented two-API design, not a conformance gap. See [**parsing**](https://json.nlohmann.me/features/parsing/) for details.
|
||||||
|
|
||||||
### Character encoding
|
### Character encoding
|
||||||
|
|
||||||
The library supports **Unicode input** as follows:
|
The library supports **Unicode input** as follows:
|
||||||
|
|||||||
+19
-1
@@ -212,6 +212,24 @@ add_custom_target(ci_test_legacycomparison
|
|||||||
COMMENT "Compile and test with legacy discarded value comparison enabled"
|
COMMENT "Compile and test with legacy discarded value comparison enabled"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
###############################################################################
|
||||||
|
# Validate UTF-8 with simdutf.
|
||||||
|
###############################################################################
|
||||||
|
|
||||||
|
add_custom_target(ci_test_simdutf
|
||||||
|
COMMAND ${CMAKE_COMMAND}
|
||||||
|
-DCMAKE_BUILD_TYPE=Debug -GNinja
|
||||||
|
-DJSON_BuildTests=ON -DJSON_TestSimdutf=ON
|
||||||
|
# simdutf needs C++17, so the library falls back to its scalar validator
|
||||||
|
# below that: build the suite at C++11 to cover the fallback with the macro
|
||||||
|
# defined, and at C++17 to run every test against simdutf itself
|
||||||
|
"-DJSON_TestStandards=11\;17"
|
||||||
|
-S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_simdutf
|
||||||
|
COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_simdutf
|
||||||
|
COMMAND cd ${PROJECT_BINARY_DIR}/build_simdutf && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure
|
||||||
|
COMMENT "Compile and test with simdutf UTF-8 validation enabled"
|
||||||
|
)
|
||||||
|
|
||||||
###############################################################################
|
###############################################################################
|
||||||
# Enable brace-init copy semantics.
|
# Enable brace-init copy semantics.
|
||||||
###############################################################################
|
###############################################################################
|
||||||
@@ -294,7 +312,7 @@ file(GLOB_RECURSE INDENT_FILES
|
|||||||
${PROJECT_SOURCE_DIR}/tests/src/*.cpp
|
${PROJECT_SOURCE_DIR}/tests/src/*.cpp
|
||||||
${PROJECT_SOURCE_DIR}/tests/src/*.hpp
|
${PROJECT_SOURCE_DIR}/tests/src/*.hpp
|
||||||
${PROJECT_SOURCE_DIR}/tests/benchmarks/src/benchmarks.cpp
|
${PROJECT_SOURCE_DIR}/tests/benchmarks/src/benchmarks.cpp
|
||||||
${PROJECT_SOURCE_DIR}/docs/examples/*.cpp
|
${PROJECT_SOURCE_DIR}/docs/mkdocs/docs/examples/*.cpp
|
||||||
)
|
)
|
||||||
|
|
||||||
set(include_dir ${PROJECT_SOURCE_DIR}/single_include/nlohmann)
|
set(include_dir ${PROJECT_SOURCE_DIR}/single_include/nlohmann)
|
||||||
|
|||||||
@@ -5,6 +5,9 @@
|
|||||||
# -Wno-extra-semi-stmt The library uses assert which triggers this warning.
|
# -Wno-extra-semi-stmt The library uses assert which triggers this warning.
|
||||||
# -Wno-padded We do not care about padding warnings.
|
# -Wno-padded We do not care about padding warnings.
|
||||||
# -Wno-covered-switch-default All switches list all cases and a default case.
|
# -Wno-covered-switch-default All switches list all cases and a default case.
|
||||||
|
# -Wno-c2y-extensions Clang 22.1 diagnoses __COUNTER__ as a C2y extension, also in
|
||||||
|
# C++ mode. The library does not use __COUNTER__; the warnings
|
||||||
|
# all come from vendored Doctest (SECTION/TEST_CASE macros).
|
||||||
# -Wno-unsafe-buffer-usage Pervasive: the library's own low-level numeric/buffer code
|
# -Wno-unsafe-buffer-usage Pervasive: the library's own low-level numeric/buffer code
|
||||||
# (to_chars, serializer, lexer, binary reader/writer, input
|
# (to_chars, serializer, lexer, binary reader/writer, input
|
||||||
# adapters, json_pointer) plus vendored Doctest itself (~208
|
# adapters, json_pointer) plus vendored Doctest itself (~208
|
||||||
@@ -20,5 +23,6 @@ set(CLANG_CXXFLAGS
|
|||||||
-Wno-extra-semi-stmt
|
-Wno-extra-semi-stmt
|
||||||
-Wno-padded
|
-Wno-padded
|
||||||
-Wno-covered-switch-default
|
-Wno-covered-switch-default
|
||||||
|
-Wno-c2y-extensions
|
||||||
-Wno-unsafe-buffer-usage
|
-Wno-unsafe-buffer-usage
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -29,7 +29,14 @@ Discarding a value (i.e., returning `#!cpp false`) has different effects dependi
|
|||||||
called:
|
called:
|
||||||
|
|
||||||
- Discarded values in structured types are skipped. That is, the parser will behave as if the discarded value was never
|
- Discarded values in structured types are skipped. That is, the parser will behave as if the discarded value was never
|
||||||
read.
|
read. This holds for every value type and for both kinds of parent: a discarded element is removed from the
|
||||||
|
surrounding array, and a discarded member is removed from the surrounding object together with its key.
|
||||||
|
- Arrays and objects can be discarded either at their `parse_event_t::array_start`/`parse_event_t::object_start` event
|
||||||
|
or at their `parse_event_t::array_end`/`parse_event_t::object_end` event, and both remove the whole value. Discarding
|
||||||
|
it at the start event also means the callback is called neither for the content of the value nor for its matching end
|
||||||
|
event.
|
||||||
|
- Discarding a `parse_event_t::key` event discards the whole object member. The callback is still called for the
|
||||||
|
associated value, but its return value has no further effect.
|
||||||
- In case a value outside a structured type is skipped, it is replaced with `null`. This case happens if the top-level
|
- In case a value outside a structured type is skipped, it is replaced with `null`. This case happens if the top-level
|
||||||
element is skipped.
|
element is skipped.
|
||||||
|
|
||||||
@@ -49,7 +56,7 @@ called:
|
|||||||
## Return value
|
## Return value
|
||||||
|
|
||||||
Whether the JSON value which called the function during parsing should be kept (`#!cpp true`) or not (`#!cpp false`). In
|
Whether the JSON value which called the function during parsing should be kept (`#!cpp true`) or not (`#!cpp false`). In
|
||||||
the latter case, it is either skipped completely or replaced by an empty discarded object.
|
the latter case, it is skipped completely, or replaced by `null` if it is the top-level value.
|
||||||
|
|
||||||
## Examples
|
## Examples
|
||||||
|
|
||||||
@@ -68,6 +75,21 @@ the latter case, it is either skipped completely or replaced by an empty discard
|
|||||||
--8<-- "examples/parse__string__parser_callback_t.output"
|
--8<-- "examples/parse__string__parser_callback_t.output"
|
||||||
```
|
```
|
||||||
|
|
||||||
|
??? example
|
||||||
|
|
||||||
|
The example below shows where discarded values are removed. The array and the number are discarded in different
|
||||||
|
ways, but in each case the parse result contains neither the value nor its key.
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
--8<-- "examples/parser_callback_t.cpp"
|
||||||
|
```
|
||||||
|
|
||||||
|
Output:
|
||||||
|
|
||||||
|
```json
|
||||||
|
--8<-- "examples/parser_callback_t.output"
|
||||||
|
```
|
||||||
|
|
||||||
## See also
|
## See also
|
||||||
|
|
||||||
- [parse](parse.md) deserialize from a compatible input
|
- [parse](parse.md) deserialize from a compatible input
|
||||||
@@ -76,3 +98,5 @@ the latter case, it is either skipped completely or replaced by an empty discard
|
|||||||
## Version history
|
## Version history
|
||||||
|
|
||||||
- Added in version 1.0.0.
|
- Added in version 1.0.0.
|
||||||
|
- Fixed in version 3.13.0 to also remove discarded values from a parent object; before, discarding an array or a value
|
||||||
|
stored under an object key left a discarded member behind, which made the parse result serialize to invalid JSON.
|
||||||
|
|||||||
@@ -52,6 +52,11 @@ optional, `#!cpp bjdata_version_t::draft2` by default.
|
|||||||
|
|
||||||
Strong guarantee: if an exception is thrown, there are no changes in the JSON value.
|
Strong guarantee: if an exception is thrown, there are no changes in the JSON value.
|
||||||
|
|
||||||
|
## Exceptions
|
||||||
|
|
||||||
|
- Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size`
|
||||||
|
is false.
|
||||||
|
|
||||||
## Complexity
|
## Complexity
|
||||||
|
|
||||||
Linear in the size of the JSON value `j`.
|
Linear in the size of the JSON value `j`.
|
||||||
|
|||||||
@@ -46,7 +46,9 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
|
|||||||
|
|
||||||
## Complexity
|
## Complexity
|
||||||
|
|
||||||
Linear in the size of the JSON value `j`.
|
Proportional to the size of the JSON value `j` multiplied by its maximum nesting
|
||||||
|
depth, `O(n × d)`. BSON length prefixes are computed recursively before nested
|
||||||
|
values are written.
|
||||||
|
|
||||||
## Examples
|
## Examples
|
||||||
|
|
||||||
|
|||||||
@@ -45,6 +45,11 @@ The exact mapping and its limitations are described on a [dedicated page](../../
|
|||||||
|
|
||||||
Strong guarantee: if an exception is thrown, there are no changes in the JSON value.
|
Strong guarantee: if an exception is thrown, there are no changes in the JSON value.
|
||||||
|
|
||||||
|
## Exceptions
|
||||||
|
|
||||||
|
- Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size`
|
||||||
|
is false.
|
||||||
|
|
||||||
## Complexity
|
## Complexity
|
||||||
|
|
||||||
Linear in the size of the JSON value `j`.
|
Linear in the size of the JSON value `j`.
|
||||||
|
|||||||
@@ -24,6 +24,7 @@ header. See also the [macro overview page](../../features/macros.md).
|
|||||||
- [**JSON_NO_IO**](json_no_io.md) - switch off functions relying on certain C++ I/O headers
|
- [**JSON_NO_IO**](json_no_io.md) - switch off functions relying on certain C++ I/O headers
|
||||||
- [**JSON_SKIP_UNSUPPORTED_COMPILER_CHECK**](json_skip_unsupported_compiler_check.md) - do not warn about unsupported compilers
|
- [**JSON_SKIP_UNSUPPORTED_COMPILER_CHECK**](json_skip_unsupported_compiler_check.md) - do not warn about unsupported compilers
|
||||||
- [**JSON_USE_GLOBAL_UDLS**](json_use_global_udls.md) - place user-defined string literals (UDLs) into the global namespace
|
- [**JSON_USE_GLOBAL_UDLS**](json_use_global_udls.md) - place user-defined string literals (UDLs) into the global namespace
|
||||||
|
- [**JSON_USE_SIMDUTF**](json_use_simdutf.md) - use the simdutf library to accelerate UTF-8 validation
|
||||||
|
|
||||||
## Library version
|
## Library version
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,71 @@
|
|||||||
|
# JSON_USE_SIMDUTF
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
#define JSON_USE_SIMDUTF
|
||||||
|
```
|
||||||
|
|
||||||
|
When defined, the parser validates the UTF-8 content of JSON strings that come from a **contiguous byte input**
|
||||||
|
(`std::string`, `std::vector<char>`/`<std::uint8_t>`, string literals, `const char*` ranges, …) using the
|
||||||
|
[simdutf](https://github.com/simdutf/simdutf) library instead of the built-in scalar validator. On text with many
|
||||||
|
non-ASCII characters (e.g. CJK or emoji) this can validate several times faster.
|
||||||
|
|
||||||
|
This is an **opt-in external dependency**. The library itself remains header-only and its behavior is unchanged: the
|
||||||
|
same input is accepted or rejected either way, and every parse error is reported at the same position with the same
|
||||||
|
message (simdutf is only used to fast-path *valid* runs; anything it flags falls back to the scalar path so the exact
|
||||||
|
diagnostic is preserved). Streaming inputs (files, `std::istream`, wide strings, user-defined adapters) always use the
|
||||||
|
scalar path.
|
||||||
|
|
||||||
|
When `JSON_USE_SIMDUTF` is defined you must make the `simdutf.h` header available on the include path and link the
|
||||||
|
simdutf library. When it is not defined, no simdutf header is included and there is no dependency.
|
||||||
|
|
||||||
|
!!! note "Requires C++17"
|
||||||
|
|
||||||
|
simdutf requires C++17 and its header rejects older standards with an `#!cpp #error`. The backend is therefore only
|
||||||
|
compiled in from C++17 on. In C++11 and C++14 the macro has no effect and the scalar validator is used, which
|
||||||
|
accepts and rejects exactly the same input -- only throughput differs. Setting the macro project-wide is therefore
|
||||||
|
safe even when some translation units are built with an older standard.
|
||||||
|
|
||||||
|
!!! warning "Define consistently"
|
||||||
|
|
||||||
|
The macro selects between two definitions of the same inline validation function. It must therefore be defined
|
||||||
|
identically for **every** translation unit that includes the library; mixing translation units that define it with
|
||||||
|
ones that do not is an ODR violation. Prefer setting it as a compile definition on the target rather than with
|
||||||
|
`#!cpp #define` in individual source files.
|
||||||
|
|
||||||
|
## Default definition
|
||||||
|
|
||||||
|
By default, `#!cpp JSON_USE_SIMDUTF` is not defined and the portable C++11 scalar validator is used.
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
#undef JSON_USE_SIMDUTF
|
||||||
|
```
|
||||||
|
|
||||||
|
## Examples
|
||||||
|
|
||||||
|
??? example
|
||||||
|
|
||||||
|
The code below enables the simdutf backend for UTF-8 validation.
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
#define JSON_USE_SIMDUTF 1
|
||||||
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
|
...
|
||||||
|
```
|
||||||
|
|
||||||
|
The project must also link against simdutf, e.g. with CMake:
|
||||||
|
|
||||||
|
```cmake
|
||||||
|
target_compile_definitions(your_target PRIVATE JSON_USE_SIMDUTF)
|
||||||
|
target_link_libraries(your_target PRIVATE simdutf::simdutf)
|
||||||
|
```
|
||||||
|
|
||||||
|
!!! hint "Testing this configuration"
|
||||||
|
|
||||||
|
The unit tests can be built against the simdutf backend with the CMake option `JSON_TestSimdutf` (`OFF` by
|
||||||
|
default), which fetches simdutf and defines `JSON_USE_SIMDUTF` for every test target. The `ci_test_simdutf` target
|
||||||
|
runs the whole test suite in that configuration.
|
||||||
|
|
||||||
|
## Version history
|
||||||
|
|
||||||
|
- Added in version 3.13.0.
|
||||||
@@ -33,17 +33,44 @@ A UTF-8 byte order mark is silently ignored.
|
|||||||
Invalid Unicode escapes and unpaired surrogates in the input are reported as
|
Invalid Unicode escapes and unpaired surrogates in the input are reported as
|
||||||
[`parse_error.101`](../home/exceptions.md#jsonexceptionparse_error101) with a detailed message.
|
[`parse_error.101`](../home/exceptions.md#jsonexceptionparse_error101) with a detailed message.
|
||||||
|
|
||||||
`operator>>` parses exactly one JSON value and leaves the stream positioned right after it, so it can be called
|
`operator>>` parses exactly one JSON value, so it can be called repeatedly to read a sequence of concatenated JSON
|
||||||
repeatedly to read a sequence of concatenated JSON values from the same stream:
|
values from the same stream:
|
||||||
|
|
||||||
```cpp
|
```cpp
|
||||||
json j1, j2;
|
json j1, j2;
|
||||||
input >> j1; // parses the first value, stream now positioned right after it
|
input >> j1; // parses the first value
|
||||||
input >> j2; // parses the next value
|
input >> j2; // parses the next value
|
||||||
```
|
```
|
||||||
|
|
||||||
Note this does **not** work for [JSON Lines](../features/parsing/json_lines.md) (newline-delimited JSON) input --
|
!!! warning "A number must be followed by whitespace"
|
||||||
see that page for why and for the recommended alternative.
|
|
||||||
|
A number is only terminated by the character that follows it. That character is read from the stream to detect the
|
||||||
|
end of the number, and it is **not** put back. When a value that is a number is immediately followed by the next
|
||||||
|
value, the first character of that next value is lost:
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
std::istringstream input("1true");
|
||||||
|
json j1, j2;
|
||||||
|
input >> j1; // j1 == 1
|
||||||
|
input >> j2; // throws parse_error.101: the stream now starts at "rue"
|
||||||
|
```
|
||||||
|
|
||||||
|
Separating the values with whitespace avoids this, because the character that is eaten is then the separator:
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
std::istringstream input("1 true");
|
||||||
|
json j1, j2;
|
||||||
|
input >> j1; // j1 == 1
|
||||||
|
input >> j2; // j2 == true
|
||||||
|
```
|
||||||
|
|
||||||
|
Only numbers are affected. Values ending in a self-delimiting character do not read past themselves, so
|
||||||
|
`truefalse`, `[1][2]`, `{"a":1}{"b":2}`, and `"a""b"` can be read back to back without a separator.
|
||||||
|
|
||||||
|
This is tracked in [#5340](https://github.com/nlohmann/json/issues/5340).
|
||||||
|
|
||||||
|
Note that reading concatenated values does **not** work for [JSON Lines](../features/parsing/json_lines.md)
|
||||||
|
(newline-delimited JSON) input -- see that page for why and for the recommended alternative.
|
||||||
|
|
||||||
!!! warning "Deprecation"
|
!!! warning "Deprecation"
|
||||||
|
|
||||||
|
|||||||
@@ -13,6 +13,12 @@ Therefore, adding object elements can yield a reallocation in which case all ite
|
|||||||
[`end()`](basic_json/end.md) iterator) and all references to the elements are invalidated. Also, any iterator or
|
[`end()`](basic_json/end.md) iterator) and all references to the elements are invalidated. Also, any iterator or
|
||||||
reference after the insertion point will point to the same index, which is now a different value.
|
reference after the insertion point will point to the same index, which is now a different value.
|
||||||
|
|
||||||
|
## Complexity
|
||||||
|
|
||||||
|
[`ordered_map`](ordered_map.md) has no lookup index: every key-based object operation is a linear scan, so building or
|
||||||
|
parsing an object of `n` keys costs O(n²) rather than O(n log n). See
|
||||||
|
[`ordered_map` complexity](ordered_map.md#complexity) for the per-operation table and for measured numbers.
|
||||||
|
|
||||||
## Examples
|
## Examples
|
||||||
|
|
||||||
??? example
|
??? example
|
||||||
|
|||||||
@@ -56,6 +56,48 @@ std::equal_to<> // since C++14
|
|||||||
- **find**
|
- **find**
|
||||||
- **insert**
|
- **insert**
|
||||||
|
|
||||||
|
## Complexity
|
||||||
|
|
||||||
|
Because the elements are stored in a `std::vector` in insertion order, there is no index to look a key up by. Every
|
||||||
|
key-based operation performs a **linear scan** over the stored elements. With `n` denoting the number of elements in the
|
||||||
|
container:
|
||||||
|
|
||||||
|
| Operation | Complexity | Note |
|
||||||
|
|----------------------------------------|----------------|----------------------------------------------------------|
|
||||||
|
| **emplace** | O(n) | scans for an existing key, then appends (amortized O(1)) |
|
||||||
|
| **operator\[\]** | O(n) | delegates to **emplace** (non-const) or **at** (const) |
|
||||||
|
| **at** | O(n) | throws `#!cpp std::out_of_range` if the key is not found |
|
||||||
|
| **find** | O(n) | |
|
||||||
|
| **count** | O(n) | the result is always 0 or 1 |
|
||||||
|
| **erase(key)** | O(n) | scan, then move the remaining elements one position down |
|
||||||
|
| **erase(pos)**, **erase(first, last)** | O(n) | moves all elements after the erased range |
|
||||||
|
| **insert(value)** | O(n) | equivalent to **emplace** |
|
||||||
|
| **insert(first, last)** | O((n + m) * m) | for `m` inserted elements |
|
||||||
|
|
||||||
|
This differs from `#!cpp std::map`, where the same operations are O(log n).
|
||||||
|
|
||||||
|
!!! warning "Quadratic cost of building large objects"
|
||||||
|
|
||||||
|
Because every insertion scans all elements inserted so far, building an object of `n` distinct keys costs
|
||||||
|
**O(n²)** in total. This applies to filling an [`ordered_json`](ordered_json.md) object key by key as well as to
|
||||||
|
parsing one, since the parser inserts each key as it is read.
|
||||||
|
|
||||||
|
The cost is negligible for the object sizes typically found in configuration files or API payloads, but it grows
|
||||||
|
steeply for machine-generated objects with many thousands of keys. Measured with `-O2 -DNDEBUG` for parsing a flat
|
||||||
|
object of `n` keys, relative to `#!cpp nlohmann::json` (which uses `#!cpp std::map`):
|
||||||
|
|
||||||
|
| `n` | `json` | `ordered_json` | factor |
|
||||||
|
|--------|--------|----------------|--------|
|
||||||
|
| 2000 | 0.7 ms | 3.6 ms | 5× |
|
||||||
|
| 4000 | 0.8 ms | 14.0 ms | 19× |
|
||||||
|
| 8000 | 1.6 ms | 67.8 ms | 43× |
|
||||||
|
| 16 000 | 3.3 ms | 181.6 ms | 54× |
|
||||||
|
|
||||||
|
If key order matters for objects of that size, consider a container with a lookup index, such as
|
||||||
|
[`tsl::ordered_map`](https://github.com/Tessil/ordered-map)
|
||||||
|
([integration](https://github.com/nlohmann/json/issues/546#issuecomment-304447518)), as the object type -- see
|
||||||
|
[object order](../features/object_order.md).
|
||||||
|
|
||||||
## Examples
|
## Examples
|
||||||
|
|
||||||
??? example
|
??? example
|
||||||
|
|||||||
@@ -0,0 +1,40 @@
|
|||||||
|
# Ecosystem
|
||||||
|
|
||||||
|
The projects below build on top of `nlohmann::json` rather than merely using it - schema validators, language
|
||||||
|
bindings, format converters, and similar building blocks. The list is not exhaustive, and is curated rather than
|
||||||
|
automatically generated. If you maintain or know of a project that belongs here,
|
||||||
|
[please let me know](mailto:mail@nlohmann.me).
|
||||||
|
|
||||||
|
For products, applications, and organizations that use the library, see [Customers](../home/customers.md) instead.
|
||||||
|
|
||||||
|
## Schema validation
|
||||||
|
|
||||||
|
- [**json-schema-validator**](https://github.com/pboettch/json-schema-validator), a JSON Schema (draft 7) validator
|
||||||
|
with human-readable error messages
|
||||||
|
|
||||||
|
## Serialization and reflection
|
||||||
|
|
||||||
|
- [**nlohmann_json_reflect**](https://github.com/1261385937/nlohmann_json_reflect), a reflection extension for
|
||||||
|
(de)serializing nested containers-in-structs-in-containers
|
||||||
|
|
||||||
|
## Encodings
|
||||||
|
|
||||||
|
- [**base-encode-decode**](https://github.com/saxonnicholls/base-encode-decode), a header-only Base64/32/16/8/4/2
|
||||||
|
(and DNA/RNA) encoding library, with an adapter that serializes binary data through `nlohmann::json`
|
||||||
|
|
||||||
|
## Language bindings and interop
|
||||||
|
|
||||||
|
- [**pybind11_json**](https://github.com/pybind/pybind11_json), a bidirectional type caster between
|
||||||
|
`nlohmann::json` and Python objects for [pybind11](https://github.com/pybind/pybind11) bindings
|
||||||
|
- [**nanobind_json**](https://github.com/ianhbell/nanobind_json), the same idea for
|
||||||
|
[nanobind](https://github.com/wjakob/nanobind) bindings
|
||||||
|
- [**nlohmann_json_qt**](https://github.com/dpurgin/nlohmann_json_qt), deserialization helpers for Qt types
|
||||||
|
(`QString`, `QUrl`, `QDateTime`, `QVector`, ...) from `nlohmann::json`
|
||||||
|
- [**vulkan2json**](https://github.com/Fadis/vulkan2json), serialization and deserialization of Vulkan API structs
|
||||||
|
|
||||||
|
## Format converters
|
||||||
|
|
||||||
|
- [**tojson**](https://github.com/mircodz/tojson), a header-only converter between YAML/XML documents and
|
||||||
|
`nlohmann::json`
|
||||||
|
- [**json2xml**](https://github.com/testillano/json2xml), a header-only converter from `nlohmann::json` to XML for
|
||||||
|
simple configuration documents
|
||||||
@@ -1,5 +1,6 @@
|
|||||||
# Community
|
# Community
|
||||||
|
|
||||||
|
- [Ecosystem](ecosystem.md) - third-party projects built on top of this library
|
||||||
- [Code of Conduct](code_of_conduct.md) - the rules and norms of this project
|
- [Code of Conduct](code_of_conduct.md) - the rules and norms of this project
|
||||||
- [Contribution Guidelines](contribution_guidelines.md) - guidelines how to contribute to this project
|
- [Contribution Guidelines](contribution_guidelines.md) - guidelines how to contribute to this project
|
||||||
- [Governance](governance.md) - the governance model of this project
|
- [Governance](governance.md) - the governance model of this project
|
||||||
|
|||||||
@@ -66,6 +66,7 @@ Note: Some modern features (like C++20 ranges or filesystem support) may be disa
|
|||||||
| Clang 20.1.1 | x86_64 | Ubuntu 22.04.1 LTS | GitHub |
|
| Clang 20.1.1 | x86_64 | Ubuntu 22.04.1 LTS | GitHub |
|
||||||
| Clang 20.1.8 with GNU-like command-line | x86_64 | Windows Server 2022 (Build 20348) | GitHub |
|
| Clang 20.1.8 with GNU-like command-line | x86_64 | Windows Server 2022 (Build 20348) | GitHub |
|
||||||
| Clang 21.1.8 | x86_64 | Ubuntu 22.04.1 LTS | GitHub |
|
| Clang 21.1.8 | x86_64 | Ubuntu 22.04.1 LTS | GitHub |
|
||||||
|
| Clang 22.1.8 | x86_64 | Ubuntu 22.04.1 LTS | GitHub |
|
||||||
| CUDA 11.8.0 (nvcc) | x86_64 | Ubuntu 22.04 LTS | GitHub |
|
| CUDA 11.8.0 (nvcc) | x86_64 | Ubuntu 22.04 LTS | GitHub |
|
||||||
| CUDA 12.1.1 (nvcc) | x86_64 | Ubuntu 22.04 LTS | GitHub |
|
| CUDA 12.1.1 (nvcc) | x86_64 | Ubuntu 22.04 LTS | GitHub |
|
||||||
| CUDA 12.6.3 (nvcc) | x86_64 | Ubuntu 22.04 LTS | GitHub |
|
| CUDA 12.6.3 (nvcc) | x86_64 | Ubuntu 22.04 LTS | GitHub |
|
||||||
|
|||||||
@@ -0,0 +1,47 @@
|
|||||||
|
#include <iostream>
|
||||||
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
|
using json = nlohmann::json;
|
||||||
|
|
||||||
|
int main()
|
||||||
|
{
|
||||||
|
// a JSON text with an array and a number inside an object
|
||||||
|
auto text = R"({"IDs": [116, 943], "Width": 800})";
|
||||||
|
|
||||||
|
// discard the array when the parser reads its opening bracket
|
||||||
|
json j_array_start = json::parse(text, [](int /*depth*/, json::parse_event_t event, json& /*parsed*/)
|
||||||
|
{
|
||||||
|
return event != json::parse_event_t::array_start;
|
||||||
|
});
|
||||||
|
|
||||||
|
// discard the same array when the parser reads its closing bracket
|
||||||
|
json j_array_end = json::parse(text, [](int /*depth*/, json::parse_event_t event, json& /*parsed*/)
|
||||||
|
{
|
||||||
|
return event != json::parse_event_t::array_end;
|
||||||
|
});
|
||||||
|
|
||||||
|
// discard the number, but keep its key
|
||||||
|
json j_value = json::parse(text, [](int /*depth*/, json::parse_event_t event, json & parsed)
|
||||||
|
{
|
||||||
|
return !(event == json::parse_event_t::value && parsed == json(800));
|
||||||
|
});
|
||||||
|
|
||||||
|
// discard the key of the number
|
||||||
|
json j_key = json::parse(text, [](int /*depth*/, json::parse_event_t event, json & parsed)
|
||||||
|
{
|
||||||
|
return !(event == json::parse_event_t::key && parsed == json("Width"));
|
||||||
|
});
|
||||||
|
|
||||||
|
// discard the top-level object
|
||||||
|
json j_root = json::parse(text, [](int /*depth*/, json::parse_event_t event, json& /*parsed*/)
|
||||||
|
{
|
||||||
|
return event != json::parse_event_t::object_end;
|
||||||
|
});
|
||||||
|
|
||||||
|
// in every case, the discarded value is removed together with its key
|
||||||
|
std::cout << j_array_start << '\n'
|
||||||
|
<< j_array_end << '\n'
|
||||||
|
<< j_value << '\n'
|
||||||
|
<< j_key << '\n'
|
||||||
|
<< j_root << '\n';
|
||||||
|
}
|
||||||
@@ -0,0 +1,5 @@
|
|||||||
|
{"Width":800}
|
||||||
|
{"Width":800}
|
||||||
|
{"IDs":[116,943]}
|
||||||
|
{"IDs":[116,943]}
|
||||||
|
null
|
||||||
@@ -116,9 +116,19 @@ The library uses the following mapping from JSON values types to BJData types ac
|
|||||||
```
|
```
|
||||||
|
|
||||||
Likewise, when a JSON object in the above form is serialized using
|
Likewise, when a JSON object in the above form is serialized using
|
||||||
[`to_bjdata`](../../api/basic_json/to_bjdata.md), it is automatically converted into a compact BJData ND-array. The
|
[`to_bjdata`](../../api/basic_json/to_bjdata.md), it is automatically converted into a compact BJData ND-array. When
|
||||||
only exception is, that when the 1-dimensional vector stored in `"_ArraySize_"` contains a single integer or two
|
the 1-dimensional vector stored in `"_ArraySize_"` contains a single integer or two integers with one being 1, a
|
||||||
integers with one being 1, a regular 1-D optimized array is generated.
|
regular 1-D optimized array is generated instead.
|
||||||
|
|
||||||
|
An object is only converted if the annotation actually describes a packed array; otherwise it is serialized as a
|
||||||
|
regular JSON object. This requires all of the following:
|
||||||
|
|
||||||
|
- `"_ArrayType_"` is one of `uint8`, `int8`, `uint16`, `int16`, `uint32`, `int32`, `uint64`, `int64`, `single`,
|
||||||
|
`double`, `char`, or `byte`,
|
||||||
|
- every entry of `"_ArraySize_"` is a non-negative integer, and their product is representable as a `std::size_t`,
|
||||||
|
- `"_ArrayData_"` holds exactly that many elements, and
|
||||||
|
- every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"` (a floating-point number for
|
||||||
|
`single` and `double`, an integer otherwise).
|
||||||
|
|
||||||
The current version of this library does not yet support automatic detection of and conversion from a nested JSON
|
The current version of this library does not yet support automatic detection of and conversion from a nested JSON
|
||||||
array input to a BJData ND-array.
|
array input to a BJData ND-array.
|
||||||
|
|||||||
@@ -98,6 +98,17 @@ The library maps BSON record types to JSON value types as follows:
|
|||||||
This library deserializes BSON type `0x11` (Timestamp) as a `number_unsigned` value. The 64-bit value is preserved,
|
This library deserializes BSON type `0x11` (Timestamp) as a `number_unsigned` value. The 64-bit value is preserved,
|
||||||
but the Timestamp type information is not.
|
but the Timestamp type information is not.
|
||||||
|
|
||||||
|
!!! warning "Lenient BSON input handling"
|
||||||
|
|
||||||
|
The BSON reader is lenient in a few areas where the BSON specification is more restrictive:
|
||||||
|
|
||||||
|
- array element keys are not checked against the required decimal sequence (`0`, `1`, `2`, ...),
|
||||||
|
- any non-zero byte is accepted as `true` for the boolean type, and
|
||||||
|
- the payload for binary subtype `0x02` is returned as-is, including its inner length prefix.
|
||||||
|
|
||||||
|
If BSON input must be validated for strict specification compliance, validate it separately before passing it to
|
||||||
|
`from_bson()`.
|
||||||
|
|
||||||
??? example
|
??? example
|
||||||
|
|
||||||
```cpp
|
```cpp
|
||||||
|
|||||||
@@ -160,14 +160,11 @@ The library maps CBOR types to JSON value types as follows:
|
|||||||
|
|
||||||
The mapping is **incomplete** in the sense that not all CBOR types can be converted to a JSON value. The following CBOR types are not supported and will yield parse errors:
|
The mapping is **incomplete** in the sense that not all CBOR types can be converted to a JSON value. The following CBOR types are not supported and will yield parse errors:
|
||||||
|
|
||||||
- date/time (0xC0..0xC1)
|
|
||||||
- bignum (0xC2..0xC3)
|
|
||||||
- decimal fraction (0xC4)
|
|
||||||
- bigfloat (0xC5)
|
|
||||||
- expected conversions (0xD5..0xD7)
|
|
||||||
- simple values (0xE0..0xF3, 0xF8)
|
- simple values (0xE0..0xF3, 0xF8)
|
||||||
- undefined (0xF7)
|
- undefined (0xF7)
|
||||||
|
|
||||||
|
Tagged items (0xC0..0xDB) are not interpreted either; see the note on tagged items below.
|
||||||
|
|
||||||
!!! warning "Negative integer overflow"
|
!!! warning "Negative integer overflow"
|
||||||
|
|
||||||
CBOR negative integers (major type 1) are decoded as `-1 - n`. If the encoded magnitude `n` is too large for the
|
CBOR negative integers (major type 1) are decoded as `-1 - n`. If the encoded magnitude `n` is too large for the
|
||||||
@@ -181,7 +178,7 @@ The library maps CBOR types to JSON value types as follows:
|
|||||||
|
|
||||||
!!! warning "Tagged items"
|
!!! warning "Tagged items"
|
||||||
|
|
||||||
Tagged items will throw a parse error by default. They can be ignored by passing `cbor_tag_handler_t::ignore` to function `from_cbor`. They can be stored by passing `cbor_tag_handler_t::store` to function `from_cbor`.
|
Tagged items (0xC0..0xDB) will throw a parse error by default. They can be ignored by passing `cbor_tag_handler_t::ignore` to function `from_cbor`, in which case the tag is skipped and the enclosed data item is parsed on its own. They can be stored by passing `cbor_tag_handler_t::store` to function `from_cbor`. Note that no tag is ever interpreted: for instance, a text string tagged with tag 0 (date/time) stays a string.
|
||||||
|
|
||||||
??? example
|
??? example
|
||||||
|
|
||||||
|
|||||||
@@ -54,6 +54,30 @@ json j = {1.0, "hello", 42};
|
|||||||
auto t = j.get<std::tuple<double, std::string, int>>(); // {1.0, "hello", 42}
|
auto t = j.get<std::tuple<double, std::string, int>>(); // {1.0, "hello", 42}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
!!! warning "Serializing a `std::pair`/`std::tuple` whose every element is a string-keyed pair"
|
||||||
|
|
||||||
|
When *every* element of a `#!cpp std::pair` or `#!cpp std::tuple` is itself a two-element array whose first
|
||||||
|
element is a string (for example `#!cpp std::pair<std::string, int>`), serializing it produces a JSON **object**
|
||||||
|
instead of the expected array:
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
using kv = std::pair<std::string, int>;
|
||||||
|
json j = std::pair<kv, kv>{{"a", 1}, {"b", 2}}; // {"a":1,"b":2}, not [["a",1],["b",2]]
|
||||||
|
```
|
||||||
|
|
||||||
|
This is a consequence of the [brace-initializer object-detection rule](creating_values.md): the same rule that
|
||||||
|
lets `#!cpp json{{"a", 1}, {"b", 2}}` create an object also fires here. The resulting object cannot be read back
|
||||||
|
into the original type (`#!cpp get<std::pair<kv, kv>>()` throws [`type_error.302`](../home/exceptions.md#jsonexceptiontype_error302)),
|
||||||
|
and duplicate keys collapse into one, losing elements. This only affects `#!cpp std::pair`/`#!cpp std::tuple`
|
||||||
|
themselves; a `#!cpp std::vector<std::pair<std::string, int>>`, or a pair/tuple with at least one element that is
|
||||||
|
not a string-keyed pair, serializes to an array as expected. To force an array, build one explicitly from the
|
||||||
|
elements with [`array`](../api/basic_json/array.md):
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
std::pair<kv, kv> p{{"a", 1}, {"b", 2}};
|
||||||
|
json a = json::array({p.first, p.second}); // [["a",1],["b",2]]
|
||||||
|
```
|
||||||
|
|
||||||
!!! info "Extracting references into a tuple"
|
!!! info "Extracting references into a tuple"
|
||||||
|
|
||||||
A tuple type may also hold references (e.g. `#!cpp std::tuple<double&, std::string&>`) to avoid copying: `get`
|
A tuple type may also hold references (e.g. `#!cpp std::tuple<double&, std::string&>`) to avoid copying: `get`
|
||||||
@@ -66,8 +90,14 @@ auto t = j.get<std::tuple<double, std::string, int>>(); // {1.0, "hello", 42}
|
|||||||
std::get<1>(refs) = "world"; // modifies j[1] in place
|
std::get<1>(refs) = "world"; // modifies j[1] in place
|
||||||
```
|
```
|
||||||
|
|
||||||
A referenced type must be one the library actually stores (or an arithmetic type it can convert to/from);
|
A referenced element must name the type the library actually *stores* — one of [`boolean_t`](../api/basic_json/boolean_t.md),
|
||||||
otherwise this is a compile error.
|
[`number_integer_t`](../api/basic_json/number_integer_t.md), [`number_unsigned_t`](../api/basic_json/number_unsigned_t.md),
|
||||||
|
[`number_float_t`](../api/basic_json/number_float_t.md), [`string_t`](../api/basic_json/string_t.md),
|
||||||
|
[`binary_t`](../api/basic_json/binary_t.md), [`array_t`](../api/basic_json/array_t.md), or
|
||||||
|
[`object_t`](../api/basic_json/object_t.md). There is nothing else to refer to, so a reference to any other type is a
|
||||||
|
compile error even when a conversion would exist: `#!cpp std::tuple<int&>` is rejected, because the library stores a
|
||||||
|
`#!cpp number_integer_t` (`#!cpp std::int64_t` by default) and not an `#!cpp int`. This restriction applies only to
|
||||||
|
reference elements — a plain `#!cpp std::tuple<int>` converts by value as usual.
|
||||||
|
|
||||||
## Implicit conversions
|
## Implicit conversions
|
||||||
|
|
||||||
@@ -116,17 +146,34 @@ which forces the explicit `get` form and can catch unintended conversions at com
|
|||||||
with a custom `adl_serializer<std::optional<T>>` specialization. Prefer `get<std::optional<T>>()`/`get_to()`
|
with a custom `adl_serializer<std::optional<T>>` specialization. Prefer `get<std::optional<T>>()`/`get_to()`
|
||||||
over `static_cast` for optional types.
|
over `static_cast` for optional types.
|
||||||
|
|
||||||
!!! warning "Converting to a fixed-size `std::array` does not check length"
|
!!! warning "Converting to a fixed-size destination does not check the array size"
|
||||||
|
|
||||||
Converting a JSON array to `#!cpp std::array<T, N>` does not check that the JSON array's size matches `N`:
|
Some destination types have a size that is fixed by their C++ type rather than by the JSON value:
|
||||||
if the JSON array is longer, the extra elements are silently dropped; if it is shorter, the remaining
|
`#!cpp std::pair<A, B>`, `#!cpp std::tuple<Ts...>`, `#!cpp std::array<T, N>`, C arrays `#!cpp T[N]`, and
|
||||||
`std::array` elements are left default-constructed. No exception is thrown in either case.
|
`#!cpp std::map`/`#!cpp std::unordered_map` with a non-string key type (which is read from an array of
|
||||||
|
two-element arrays). All of them read exactly as many elements as they need via
|
||||||
|
[`at`](../api/basic_json/at.md) and **never compare the JSON array's size to that number**. The two
|
||||||
|
mismatch directions therefore behave differently:
|
||||||
|
|
||||||
|
- The JSON array has **too many** elements: the surplus is **silently discarded**, and no exception is
|
||||||
|
thrown.
|
||||||
|
- The JSON array has **too few** elements: `at` throws
|
||||||
|
[`out_of_range.401`](../home/exceptions.md#jsonexceptionout_of_range401) for the first missing index --
|
||||||
|
an out-of-range error, not a [`type_error`](../home/exceptions.md#type-errors), even though the cause
|
||||||
|
is a shape mismatch.
|
||||||
|
|
||||||
```cpp
|
```cpp
|
||||||
json j = {1, 2, 3, 4, 5};
|
json j = {1, 2, 3, 4, 5};
|
||||||
auto a = j.get<std::array<int, 3>>(); // {1, 2, 3} -- elements 4 and 5 silently dropped
|
|
||||||
|
auto a = j.get<std::array<int, 3>>(); // {1, 2, 3} -- elements 4 and 5 silently dropped
|
||||||
|
auto p = j.get<std::pair<int, int>>(); // (1, 2) -- elements 3, 4, and 5 silently dropped
|
||||||
|
|
||||||
|
json k = {1};
|
||||||
|
auto q = k.get<std::pair<int, int>>(); // ❌ throws out_of_range.401
|
||||||
```
|
```
|
||||||
|
|
||||||
|
If a size mismatch is an error in your application, check the size yourself before converting.
|
||||||
|
|
||||||
## Omitting a field when serializing `std::optional`
|
## Omitting a field when serializing `std::optional`
|
||||||
|
|
||||||
By default, `to_json` for `std::optional<T>` writes either the value or `#!json null` -- there is no built-in way
|
By default, `to_json` for `std::optional<T>` writes either the value or `#!json null` -- there is no built-in way
|
||||||
|
|||||||
@@ -137,6 +137,14 @@ behavior is deprecated and switched off (`0`) by default.
|
|||||||
|
|
||||||
See [full documentation of `JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md).
|
See [full documentation of `JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md).
|
||||||
|
|
||||||
|
## `JSON_USE_SIMDUTF`
|
||||||
|
|
||||||
|
When defined, UTF-8 validation of JSON strings read from contiguous byte input is delegated to the
|
||||||
|
[simdutf](https://github.com/simdutf/simdutf) library instead of the built-in scalar validator. This is an opt-in
|
||||||
|
external dependency and is not defined by default.
|
||||||
|
|
||||||
|
See [full documentation of `JSON_USE_SIMDUTF`](../api/macros/json_use_simdutf.md).
|
||||||
|
|
||||||
## `NLOHMANN_DEFINE_TYPE_*(...)`, `NLOHMANN_DEFINE_DERIVED_TYPE_*(...)`
|
## `NLOHMANN_DEFINE_TYPE_*(...)`, `NLOHMANN_DEFINE_DERIVED_TYPE_*(...)`
|
||||||
|
|
||||||
The library defines 12 macros to simplify the serialization/deserialization of types. See the page on
|
The library defines 12 macros to simplify the serialization/deserialization of types. See the page on
|
||||||
|
|||||||
@@ -53,6 +53,12 @@ If you do want to preserve the **insertion order**, you can use the type [`nlohm
|
|||||||
|
|
||||||
Alternatively, you can use a more sophisticated ordered map like [`tsl::ordered_map`](https://github.com/Tessil/ordered-map) ([integration](https://github.com/nlohmann/json/issues/546#issuecomment-304447518)) or [`nlohmann::fifo_map`](https://github.com/nlohmann/fifo_map) ([integration](https://github.com/nlohmann/json/issues/485#issuecomment-333652309)).
|
Alternatively, you can use a more sophisticated ordered map like [`tsl::ordered_map`](https://github.com/Tessil/ordered-map) ([integration](https://github.com/nlohmann/json/issues/546#issuecomment-304447518)) or [`nlohmann::fifo_map`](https://github.com/nlohmann/fifo_map) ([integration](https://github.com/nlohmann/json/issues/485#issuecomment-333652309)).
|
||||||
|
|
||||||
|
The [`ordered_map`](../api/ordered_map.md) behind `nlohmann::ordered_json` is deliberately minimal and has no lookup
|
||||||
|
index, so every key access is a linear scan and building an object of `n` keys costs O(n²). This is unnoticeable at
|
||||||
|
typical object sizes but becomes significant for objects with many thousands of keys; see
|
||||||
|
[`ordered_map` complexity](../api/ordered_map.md#complexity). The alternatives above keep a lookup index and do not
|
||||||
|
have this cost.
|
||||||
|
|
||||||
### Notes on parsing
|
### Notes on parsing
|
||||||
|
|
||||||
Note that you also need to call the right [`parse`](../api/basic_json/parse.md) function when reading from a file.
|
Note that you also need to call the right [`parse`](../api/basic_json/parse.md) function when reading from a file.
|
||||||
|
|||||||
@@ -28,6 +28,22 @@ Inputs consisting of multiple values separated by newlines are handled by the [J
|
|||||||
By default, the library rejects comments and trailing commas. Both can be enabled with parameters of the `parse`
|
By default, the library rejects comments and trailing commas. Both can be enabled with parameters of the `parse`
|
||||||
function — see [comments](../comments.md) and [trailing commas](../trailing_commas.md).
|
function — see [comments](../comments.md) and [trailing commas](../trailing_commas.md).
|
||||||
|
|
||||||
|
## Strictness and trailing data
|
||||||
|
|
||||||
|
[`parse`](../../api/basic_json/parse.md) reads a single JSON value and requires the whole input to be consumed: any
|
||||||
|
non-whitespace data after the value is reported as a parse error. Use it when you want to guarantee that an input is
|
||||||
|
exactly one complete JSON document.
|
||||||
|
|
||||||
|
[`operator>>`](../../api/operator_gtgt.md) follows relaxed `#!cpp std::istream` semantics instead: it parses one JSON
|
||||||
|
value and leaves the stream positioned right after it, without requiring the rest of the stream to be consumed. This is
|
||||||
|
what makes it possible to read several concatenated values from the same stream, but it also means that "a valid
|
||||||
|
document followed by trailing bytes" is accepted rather than rejected. If you are validating conformance, or need to
|
||||||
|
reject any input that is not exactly one JSON document, prefer `parse`.
|
||||||
|
|
||||||
|
When using `operator>>` to read several concatenated values this way, a value that is a number must be followed by
|
||||||
|
whitespace, because `operator>>` consumes the character that terminates a number — see the
|
||||||
|
[`operator>>` notes](../../api/operator_gtgt.md#notes) for details and examples.
|
||||||
|
|
||||||
## SAX vs. DOM parsing
|
## SAX vs. DOM parsing
|
||||||
|
|
||||||
The library offers two parsing models:
|
The library offers two parsing models:
|
||||||
|
|||||||
@@ -49,4 +49,5 @@ JSON Lines input with more than one value is treated as invalid JSON by the [`pa
|
|||||||
with a JSON Lines input does not work, because the parser will try to parse one value after the last one.
|
with a JSON Lines input does not work, because the parser will try to parse one value after the last one.
|
||||||
|
|
||||||
This is different from parsing a stream of *concatenated* (non-newline-delimited) JSON values, for which
|
This is different from parsing a stream of *concatenated* (non-newline-delimited) JSON values, for which
|
||||||
`operator>>` does work -- see its [notes](../../api/operator_gtgt.md#notes) for details.
|
`operator>>` does work, provided that a value that is a number is followed by whitespace -- see its
|
||||||
|
[notes](../../api/operator_gtgt.md#notes) for details.
|
||||||
|
|||||||
@@ -291,9 +291,10 @@ A JSON Pointer array index must be a number.
|
|||||||
|
|
||||||
### json.exception.parse_error.110
|
### json.exception.parse_error.110
|
||||||
|
|
||||||
When parsing CBOR or MessagePack, the byte vector ends before the complete value has been read.
|
When parsing a [binary format](../features/binary_formats/index.md), the byte vector ends before the complete value has
|
||||||
|
been read.
|
||||||
|
|
||||||
!!! failure "Example message"
|
!!! failure "Example messages"
|
||||||
|
|
||||||
```
|
```
|
||||||
[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing CBOR string: unexpected end of input
|
[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing CBOR string: unexpected end of input
|
||||||
@@ -301,6 +302,9 @@ When parsing CBOR or MessagePack, the byte vector ends before the complete value
|
|||||||
```
|
```
|
||||||
[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing UBJSON value: expected end of input; last byte: 0x5A
|
[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing UBJSON value: expected end of input; last byte: 0x5A
|
||||||
```
|
```
|
||||||
|
```
|
||||||
|
[json.exception.parse_error.110] parse error at byte 8: syntax error while parsing BSON number: unexpected end of input
|
||||||
|
```
|
||||||
|
|
||||||
### json.exception.parse_error.112
|
### json.exception.parse_error.112
|
||||||
|
|
||||||
@@ -329,10 +333,14 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde
|
|||||||
```
|
```
|
||||||
[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow
|
[json.exception.parse_error.112] parse error at byte 9: syntax error while parsing CBOR value: negative integer overflow
|
||||||
```
|
```
|
||||||
|
```
|
||||||
|
[json.exception.parse_error.112] parse error at byte 5: syntax error while parsing BSON document: document size 6 does not match the number of bytes read (5)
|
||||||
|
```
|
||||||
|
|
||||||
### json.exception.parse_error.113
|
### json.exception.parse_error.113
|
||||||
|
|
||||||
While parsing a map key, a value that is not a string has been read.
|
A string could not be read from a [binary format](../features/binary_formats/index.md): either a value that is not a
|
||||||
|
string was read where one was required (for instance as a map key), or the string's length specification is invalid.
|
||||||
|
|
||||||
!!! failure "Example messages"
|
!!! failure "Example messages"
|
||||||
|
|
||||||
@@ -345,6 +353,9 @@ While parsing a map key, a value that is not a string has been read.
|
|||||||
```
|
```
|
||||||
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing UBJSON char: byte after 'C' must be in range 0x00..0x7F; last byte: 0x82
|
[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing UBJSON char: byte after 'C' must be in range 0x00..0x7F; last byte: 0x82
|
||||||
```
|
```
|
||||||
|
```
|
||||||
|
[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData string: string length must not be negative
|
||||||
|
```
|
||||||
|
|
||||||
### json.exception.parse_error.114
|
### json.exception.parse_error.114
|
||||||
|
|
||||||
@@ -853,13 +864,21 @@ and this exception no longer occurs.
|
|||||||
|
|
||||||
### json.exception.out_of_range.408
|
### json.exception.out_of_range.408
|
||||||
|
|
||||||
The size (following `#`) of an UBJSON array or object exceeds the maximal capacity.
|
The size of an array or object in a [binary format](../features/binary_formats/index.md) exceeds the maximal capacity:
|
||||||
|
the size following `#` for [UBJSON](../features/binary_formats/ubjson.md)/[BJData](../features/binary_formats/bjdata.md),
|
||||||
|
or the encoded length for [CBOR](../features/binary_formats/cbor.md).
|
||||||
|
|
||||||
!!! failure "Example message"
|
!!! failure "Example messages"
|
||||||
|
|
||||||
```
|
```
|
||||||
excessive array size: 8658170730974374167
|
excessive array size: 8658170730974374167
|
||||||
```
|
```
|
||||||
|
```
|
||||||
|
[json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive array size
|
||||||
|
```
|
||||||
|
```
|
||||||
|
[json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive map size
|
||||||
|
```
|
||||||
|
|
||||||
### json.exception.out_of_range.409
|
### json.exception.out_of_range.409
|
||||||
|
|
||||||
@@ -946,3 +965,19 @@ A JSON Patch operation 'test' failed. The unsuccessful operation is also printed
|
|||||||
```
|
```
|
||||||
[json.exception.other_error.501] unsuccessful: {"op":"test","path":"/baz","value":"bar"}
|
[json.exception.other_error.501] unsuccessful: {"op":"test","path":"/baz","value":"bar"}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### json.exception.other_error.502
|
||||||
|
|
||||||
|
[`to_ubjson`](../api/basic_json/to_ubjson.md) and [`to_bjdata`](../api/basic_json/to_bjdata.md) were called with
|
||||||
|
`use_type = true` but `use_size = false`. UBJSON requires a size marker (`#`) after a type marker (`$`).
|
||||||
|
|
||||||
|
!!! failure "Example message"
|
||||||
|
|
||||||
|
```
|
||||||
|
[json.exception.other_error.502] use_type requires use_size = true
|
||||||
|
```
|
||||||
|
|
||||||
|
!!! note
|
||||||
|
|
||||||
|
This exception was added in version 3.13.0. Before that, debug builds aborted on an assertion and release builds
|
||||||
|
wrote a `$` marker without `#`, which [`from_ubjson`](../api/basic_json/from_ubjson.md) then rejected.
|
||||||
|
|||||||
@@ -296,6 +296,7 @@ nav:
|
|||||||
- 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md
|
- 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md
|
||||||
- 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md
|
- 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md
|
||||||
- 'JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON': api/macros/json_use_legacy_discarded_value_comparison.md
|
- 'JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON': api/macros/json_use_legacy_discarded_value_comparison.md
|
||||||
|
- 'JSON_USE_SIMDUTF': api/macros/json_use_simdutf.md
|
||||||
- 'NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_derived_type.md
|
- 'NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_derived_type.md
|
||||||
- 'NLOHMANN_DEFINE_TYPE_INTRUSIVE, NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_intrusive.md
|
- 'NLOHMANN_DEFINE_TYPE_INTRUSIVE, NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_intrusive.md
|
||||||
- 'NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_non_intrusive.md
|
- 'NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_non_intrusive.md
|
||||||
@@ -308,6 +309,7 @@ nav:
|
|||||||
- 'NLOHMANN_JSON_VERSION_MAJOR, NLOHMANN_JSON_VERSION_MINOR, NLOHMANN_JSON_VERSION_PATCH': api/macros/nlohmann_json_version_major.md
|
- 'NLOHMANN_JSON_VERSION_MAJOR, NLOHMANN_JSON_VERSION_MINOR, NLOHMANN_JSON_VERSION_PATCH': api/macros/nlohmann_json_version_major.md
|
||||||
- Community:
|
- Community:
|
||||||
- community/index.md
|
- community/index.md
|
||||||
|
- community/ecosystem.md
|
||||||
- "Code of Conduct": community/code_of_conduct.md
|
- "Code of Conduct": community/code_of_conduct.md
|
||||||
- community/contribution_guidelines.md
|
- community/contribution_guidelines.md
|
||||||
- community/quality_assurance.md
|
- community/quality_assurance.md
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
wheel==0.47.0
|
wheel==0.48.0
|
||||||
|
|
||||||
mkdocs==1.6.1 # documentation framework
|
mkdocs==1.6.1 # documentation framework
|
||||||
mkdocs-git-revision-date-localized-plugin==1.5.3 # plugin "git-revision-date-localized"
|
mkdocs-git-revision-date-localized-plugin==1.5.4 # plugin "git-revision-date-localized"
|
||||||
mkdocs-material==9.7.7 # theme for mkdocs
|
mkdocs-material==9.7.7 # theme for mkdocs
|
||||||
mkdocs-material-extensions==1.3.1 # extensions
|
mkdocs-material-extensions==1.3.1 # extensions
|
||||||
mkdocs-minify-plugin==0.8.0 # plugin "minify"
|
mkdocs-minify-plugin==0.8.0 # plugin "minify"
|
||||||
|
|||||||
@@ -465,15 +465,6 @@ class binary_reader
|
|||||||
// CBOR //
|
// CBOR //
|
||||||
//////////
|
//////////
|
||||||
|
|
||||||
/*!
|
|
||||||
@param[in] get_char whether a new character should be retrieved from the
|
|
||||||
input (true) or whether the last read character should
|
|
||||||
be considered instead (false)
|
|
||||||
@param[in] tag_handler how CBOR tags should be treated
|
|
||||||
|
|
||||||
@return whether a valid CBOR value was passed to the SAX parser
|
|
||||||
*/
|
|
||||||
|
|
||||||
template<typename NumberType>
|
template<typename NumberType>
|
||||||
bool get_cbor_negative_integer()
|
bool get_cbor_negative_integer()
|
||||||
{
|
{
|
||||||
@@ -492,6 +483,14 @@ class binary_reader
|
|||||||
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@param[in] get_char whether a new character should be retrieved from the
|
||||||
|
input (true) or whether the last read character should
|
||||||
|
be considered instead (false)
|
||||||
|
@param[in] tag_handler how CBOR tags should be treated
|
||||||
|
|
||||||
|
@return whether a valid CBOR value was passed to the SAX parser
|
||||||
|
*/
|
||||||
bool parse_cbor_internal(const bool get_char,
|
bool parse_cbor_internal(const bool get_char,
|
||||||
const cbor_tag_handler_t tag_handler)
|
const cbor_tag_handler_t tag_handler)
|
||||||
{
|
{
|
||||||
@@ -704,13 +703,15 @@ class binary_reader
|
|||||||
case 0x9A: // array (four-byte uint32_t for n follow)
|
case 0x9A: // array (four-byte uint32_t for n follow)
|
||||||
{
|
{
|
||||||
std::uint32_t len{};
|
std::uint32_t len{};
|
||||||
return get_number(input_format_t::cbor, len) && get_cbor_array(conditional_static_cast<std::size_t>(len), tag_handler);
|
std::size_t size{};
|
||||||
|
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && get_cbor_array(size, tag_handler);
|
||||||
}
|
}
|
||||||
|
|
||||||
case 0x9B: // array (eight-byte uint64_t for n follow)
|
case 0x9B: // array (eight-byte uint64_t for n follow)
|
||||||
{
|
{
|
||||||
std::uint64_t len{};
|
std::uint64_t len{};
|
||||||
return get_number(input_format_t::cbor, len) && get_cbor_array(conditional_static_cast<std::size_t>(len), tag_handler);
|
std::size_t size{};
|
||||||
|
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && get_cbor_array(size, tag_handler);
|
||||||
}
|
}
|
||||||
|
|
||||||
case 0x9F: // array (indefinite length)
|
case 0x9F: // array (indefinite length)
|
||||||
@@ -758,19 +759,27 @@ class binary_reader
|
|||||||
case 0xBA: // map (four-byte uint32_t for n follow)
|
case 0xBA: // map (four-byte uint32_t for n follow)
|
||||||
{
|
{
|
||||||
std::uint32_t len{};
|
std::uint32_t len{};
|
||||||
return get_number(input_format_t::cbor, len) && get_cbor_object(conditional_static_cast<std::size_t>(len), tag_handler);
|
std::size_t size{};
|
||||||
|
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && get_cbor_object(size, tag_handler);
|
||||||
}
|
}
|
||||||
|
|
||||||
case 0xBB: // map (eight-byte uint64_t for n follow)
|
case 0xBB: // map (eight-byte uint64_t for n follow)
|
||||||
{
|
{
|
||||||
std::uint64_t len{};
|
std::uint64_t len{};
|
||||||
return get_number(input_format_t::cbor, len) && get_cbor_object(conditional_static_cast<std::size_t>(len), tag_handler);
|
std::size_t size{};
|
||||||
|
return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && get_cbor_object(size, tag_handler);
|
||||||
}
|
}
|
||||||
|
|
||||||
case 0xBF: // map (indefinite length)
|
case 0xBF: // map (indefinite length)
|
||||||
return get_cbor_object(detail::unknown_size(), tag_handler);
|
return get_cbor_object(detail::unknown_size(), tag_handler);
|
||||||
|
|
||||||
case 0xC6: // tagged item
|
case 0xC0: // tagged item
|
||||||
|
case 0xC1:
|
||||||
|
case 0xC2:
|
||||||
|
case 0xC3:
|
||||||
|
case 0xC4:
|
||||||
|
case 0xC5:
|
||||||
|
case 0xC6:
|
||||||
case 0xC7:
|
case 0xC7:
|
||||||
case 0xC8:
|
case 0xC8:
|
||||||
case 0xC9:
|
case 0xC9:
|
||||||
@@ -785,6 +794,9 @@ class binary_reader
|
|||||||
case 0xD2:
|
case 0xD2:
|
||||||
case 0xD3:
|
case 0xD3:
|
||||||
case 0xD4:
|
case 0xD4:
|
||||||
|
case 0xD5:
|
||||||
|
case 0xD6:
|
||||||
|
case 0xD7:
|
||||||
case 0xD8: // tagged item (1 byte follows)
|
case 0xD8: // tagged item (1 byte follows)
|
||||||
case 0xD9: // tagged item (2 bytes follow)
|
case 0xD9: // tagged item (2 bytes follow)
|
||||||
case 0xDA: // tagged item (4 bytes follow)
|
case 0xDA: // tagged item (4 bytes follow)
|
||||||
@@ -807,25 +819,37 @@ class binary_reader
|
|||||||
case 0xD8:
|
case 0xD8:
|
||||||
{
|
{
|
||||||
std::uint8_t subtype_to_ignore{};
|
std::uint8_t subtype_to_ignore{};
|
||||||
get_number(input_format_t::cbor, subtype_to_ignore);
|
if (!get_number(input_format_t::cbor, subtype_to_ignore))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
case 0xD9:
|
case 0xD9:
|
||||||
{
|
{
|
||||||
std::uint16_t subtype_to_ignore{};
|
std::uint16_t subtype_to_ignore{};
|
||||||
get_number(input_format_t::cbor, subtype_to_ignore);
|
if (!get_number(input_format_t::cbor, subtype_to_ignore))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
case 0xDA:
|
case 0xDA:
|
||||||
{
|
{
|
||||||
std::uint32_t subtype_to_ignore{};
|
std::uint32_t subtype_to_ignore{};
|
||||||
get_number(input_format_t::cbor, subtype_to_ignore);
|
if (!get_number(input_format_t::cbor, subtype_to_ignore))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
case 0xDB:
|
case 0xDB:
|
||||||
{
|
{
|
||||||
std::uint64_t subtype_to_ignore{};
|
std::uint64_t subtype_to_ignore{};
|
||||||
get_number(input_format_t::cbor, subtype_to_ignore);
|
if (!get_number(input_format_t::cbor, subtype_to_ignore))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
default:
|
default:
|
||||||
@@ -843,28 +867,40 @@ class binary_reader
|
|||||||
case 0xD8:
|
case 0xD8:
|
||||||
{
|
{
|
||||||
std::uint8_t subtype{};
|
std::uint8_t subtype{};
|
||||||
get_number(input_format_t::cbor, subtype);
|
if (!get_number(input_format_t::cbor, subtype))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
|
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
case 0xD9:
|
case 0xD9:
|
||||||
{
|
{
|
||||||
std::uint16_t subtype{};
|
std::uint16_t subtype{};
|
||||||
get_number(input_format_t::cbor, subtype);
|
if (!get_number(input_format_t::cbor, subtype))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
|
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
case 0xDA:
|
case 0xDA:
|
||||||
{
|
{
|
||||||
std::uint32_t subtype{};
|
std::uint32_t subtype{};
|
||||||
get_number(input_format_t::cbor, subtype);
|
if (!get_number(input_format_t::cbor, subtype))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
|
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
case 0xDB:
|
case 0xDB:
|
||||||
{
|
{
|
||||||
std::uint64_t subtype{};
|
std::uint64_t subtype{};
|
||||||
get_number(input_format_t::cbor, subtype);
|
if (!get_number(input_format_t::cbor, subtype))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
|
b.set_subtype(detail::conditional_static_cast<typename binary_t::subtype_type>(subtype));
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
@@ -1155,6 +1191,31 @@ class binary_reader
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief narrow a definite CBOR array/map length to std::size_t
|
||||||
|
|
||||||
|
A definite length is rejected if it does not fit in std::size_t or if it
|
||||||
|
equals detail::unknown_size(), which is reserved to mark an indefinite-
|
||||||
|
length container and would otherwise make the length read as indefinite.
|
||||||
|
Both cases exceed any container's max_size(), so no representable input
|
||||||
|
is affected.
|
||||||
|
|
||||||
|
@param[in] len the declared length
|
||||||
|
@param[out] result the length narrowed to std::size_t
|
||||||
|
@param[in] context "array" or "map", for the error message
|
||||||
|
@return whether the length is usable
|
||||||
|
*/
|
||||||
|
bool get_cbor_container_size(const std::uint64_t len, std::size_t& result, const char* context)
|
||||||
|
{
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!value_in_range_of<std::size_t>(len) || len == detail::unknown_size()))
|
||||||
|
{
|
||||||
|
return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408,
|
||||||
|
exception_message(input_format_t::cbor, concat("excessive ", context, " size"), "size"), nullptr));
|
||||||
|
}
|
||||||
|
result = conditional_static_cast<std::size_t>(len);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@param[in] len the length of the array or detail::unknown_size() for an
|
@param[in] len the length of the array or detail::unknown_size() for an
|
||||||
array of indefinite size
|
array of indefinite size
|
||||||
@@ -1935,7 +1996,11 @@ class binary_reader
|
|||||||
{
|
{
|
||||||
if (get_char)
|
if (get_char)
|
||||||
{
|
{
|
||||||
get(); // TODO(niels): may we ignore N here?
|
// no get_ignore_noop() here: the byte read next must be a string
|
||||||
|
// length type specification, and a no-op ('N') is not valid in
|
||||||
|
// that position. No-ops at positions where a value may appear are
|
||||||
|
// already consumed by the callers via get_ignore_noop().
|
||||||
|
get();
|
||||||
}
|
}
|
||||||
|
|
||||||
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "value")))
|
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "value")))
|
||||||
@@ -2825,7 +2890,17 @@ class binary_reader
|
|||||||
case token_type::value_unsigned:
|
case token_type::value_unsigned:
|
||||||
return sax->number_unsigned(number_lexer.get_number_unsigned());
|
return sax->number_unsigned(number_lexer.get_number_unsigned());
|
||||||
case token_type::value_float:
|
case token_type::value_float:
|
||||||
return sax->number_float(number_lexer.get_number_float(), std::move(number_string));
|
{
|
||||||
|
const auto parsed_float = number_lexer.get_number_float();
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!std::isfinite(parsed_float)))
|
||||||
|
{
|
||||||
|
return sax->parse_error(
|
||||||
|
chars_read,
|
||||||
|
number_string,
|
||||||
|
out_of_range::create(406, concat("number overflow parsing '", number_string, '\''), nullptr));
|
||||||
|
}
|
||||||
|
return sax->number_float(parsed_float, std::move(number_string));
|
||||||
|
}
|
||||||
case token_type::uninitialized:
|
case token_type::uninitialized:
|
||||||
case token_type::literal_true:
|
case token_type::literal_true:
|
||||||
case token_type::literal_false:
|
case token_type::literal_false:
|
||||||
|
|||||||
@@ -155,11 +155,31 @@ class input_stream_adapter
|
|||||||
|
|
||||||
// General-purpose iterator-based adapter. It might not be as fast as
|
// General-purpose iterator-based adapter. It might not be as fast as
|
||||||
// theoretically possible for some containers, but it is extremely versatile.
|
// theoretically possible for some containers, but it is extremely versatile.
|
||||||
// SentinelType defaults to IteratorType for backward compatibility, but may
|
// SentinelType defaults to IteratorType for backward compatibility, but may be
|
||||||
// be a different type (e.g., a C++20 sentinel or counted_iterator).
|
// a different type, e.g. a C++20 sentinel such as std::default_sentinel_t when
|
||||||
|
// IteratorType is a std::counted_iterator.
|
||||||
template<typename IteratorType, typename SentinelType = IteratorType>
|
template<typename IteratorType, typename SentinelType = IteratorType>
|
||||||
class iterator_input_adapter
|
class iterator_input_adapter
|
||||||
{
|
{
|
||||||
|
// Whether the number of elements between two positions can be computed in
|
||||||
|
// O(1): either the iterator and the sentinel have the same type (plain
|
||||||
|
// std::distance) or, in C++20, the sentinel is a sized sentinel for the
|
||||||
|
// iterator (std::ranges::distance), e.g. std::default_sentinel_t paired
|
||||||
|
// with std::counted_iterator.
|
||||||
|
//
|
||||||
|
// JSON_HAS_RANGES gates the C++20 branch: on standard libraries with an
|
||||||
|
// incomplete <ranges> (libstdc++ < 11, see #4440) evaluating
|
||||||
|
// std::contiguous_iterator on a std::counted_iterator is a hard error
|
||||||
|
// instead of yielding false, and these traits are instantiated for every
|
||||||
|
// adapter. Such toolchains fall back to the pointer-only test and simply
|
||||||
|
// use the byte-at-a-time scanner.
|
||||||
|
static constexpr bool sentinel_is_sized =
|
||||||
|
#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||||
|
std::is_same<IteratorType, SentinelType>::value || std::sized_sentinel_for<SentinelType, IteratorType>;
|
||||||
|
#else
|
||||||
|
std::is_same<IteratorType, SentinelType>::value;
|
||||||
|
#endif
|
||||||
|
|
||||||
public:
|
public:
|
||||||
using char_type = typename std::iterator_traits<IteratorType>::value_type;
|
using char_type = typename std::iterator_traits<IteratorType>::value_type;
|
||||||
|
|
||||||
@@ -171,7 +191,7 @@ class iterator_input_adapter
|
|||||||
// in wide_string_input_adapter, which does not expose this).
|
// in wide_string_input_adapter, which does not expose this).
|
||||||
static constexpr bool supports_seek =
|
static constexpr bool supports_seek =
|
||||||
std::is_same<typename std::iterator_traits<IteratorType>::iterator_category, std::random_access_iterator_tag>::value
|
std::is_same<typename std::iterator_traits<IteratorType>::iterator_category, std::random_access_iterator_tag>::value
|
||||||
&& std::is_same<IteratorType, SentinelType>::value
|
&& sentinel_is_sized
|
||||||
&& sizeof(char_type) == 1;
|
&& sizeof(char_type) == 1;
|
||||||
|
|
||||||
iterator_input_adapter(IteratorType first, SentinelType last)
|
iterator_input_adapter(IteratorType first, SentinelType last)
|
||||||
@@ -219,30 +239,60 @@ class iterator_input_adapter
|
|||||||
private:
|
private:
|
||||||
// whether IteratorType refers to a contiguous range and therefore supports
|
// whether IteratorType refers to a contiguous range and therefore supports
|
||||||
// a std::memcpy fast path (pointers always do; in C++20 we can also detect
|
// a std::memcpy fast path (pointers always do; in C++20 we can also detect
|
||||||
// library iterators such as those of std::vector and std::string).
|
// library iterators such as those of std::vector and std::string). The
|
||||||
// Computing the available element count needs either same-type iterators
|
// available element count must also be computable in O(1), hence
|
||||||
// (plain std::distance) or, in C++20, a sized sentinel (std::ranges::distance),
|
// sentinel_is_sized.
|
||||||
// e.g. std::counted_iterator paired with std::default_sentinel_t.
|
static constexpr bool iterator_is_contiguous = sentinel_is_sized &&
|
||||||
static constexpr bool iterator_is_contiguous =
|
#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||||
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
(std::contiguous_iterator<IteratorType> || std::is_pointer<IteratorType>::value);
|
||||||
(std::is_same<IteratorType, SentinelType>::value || std::sized_sentinel_for<SentinelType, IteratorType>)
|
|
||||||
&& (std::contiguous_iterator<IteratorType> || std::is_pointer<IteratorType>::value);
|
|
||||||
#else
|
#else
|
||||||
std::is_same<IteratorType, SentinelType>::value && std::is_pointer<IteratorType>::value;
|
std::is_pointer<IteratorType>::value;
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
// number of unread elements in [current, end)
|
||||||
|
std::size_t remaining_count() const
|
||||||
|
{
|
||||||
|
#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||||
|
// std::ranges::distance also supports sized sentinels of a different
|
||||||
|
// type (e.g. std::counted_iterator + std::default_sentinel_t)
|
||||||
|
return static_cast<std::size_t>(std::ranges::distance(current, end));
|
||||||
|
#else
|
||||||
|
return static_cast<std::size_t>(std::distance(current, end));
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
public:
|
||||||
|
// Whether the remaining input is a single contiguous block of 1-byte
|
||||||
|
// elements that the lexer can inspect directly (used for the SWAR string
|
||||||
|
// fast path).
|
||||||
|
static constexpr bool supports_bulk_scan =
|
||||||
|
iterator_is_contiguous && sizeof(char_type) == 1;
|
||||||
|
|
||||||
|
// Pointer to the next unread element; only valid when bulk_remaining() > 0.
|
||||||
|
const char_type* bulk_data() const
|
||||||
|
{
|
||||||
|
return &*current;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Number of unread elements available as one contiguous block.
|
||||||
|
std::size_t bulk_remaining() const
|
||||||
|
{
|
||||||
|
return remaining_count();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Consume @a n elements previously inspected via bulk_data().
|
||||||
|
void bulk_skip(std::size_t n)
|
||||||
|
{
|
||||||
|
std::advance(current, static_cast<typename std::iterator_traits<IteratorType>::difference_type>(n));
|
||||||
|
}
|
||||||
|
|
||||||
|
private:
|
||||||
// contiguous fast path: bulk copy the remaining range with std::memcpy
|
// contiguous fast path: bulk copy the remaining range with std::memcpy
|
||||||
template<class T>
|
template<class T>
|
||||||
std::size_t get_elements_impl(T* dest, std::size_t count, std::true_type /*contiguous*/)
|
std::size_t get_elements_impl(T* dest, std::size_t count, std::true_type /*contiguous*/)
|
||||||
{
|
{
|
||||||
const std::size_t wanted = count * sizeof(T);
|
const std::size_t wanted = count * sizeof(T);
|
||||||
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
const std::size_t available = remaining_count() * sizeof(char_type);
|
||||||
// std::ranges::distance also supports sized sentinels of a different
|
|
||||||
// type (e.g. std::counted_iterator + std::default_sentinel_t)
|
|
||||||
const std::size_t available = static_cast<std::size_t>(std::ranges::distance(current, end)) * sizeof(char_type);
|
|
||||||
#else
|
|
||||||
const std::size_t available = static_cast<std::size_t>(std::distance(current, end)) * sizeof(char_type);
|
|
||||||
#endif
|
|
||||||
const std::size_t copied = (std::min)(wanted, available);
|
const std::size_t copied = (std::min)(wanted, available);
|
||||||
if (JSON_HEDLEY_LIKELY(copied != 0))
|
if (JSON_HEDLEY_LIKELY(copied != 0))
|
||||||
{
|
{
|
||||||
@@ -345,8 +395,12 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
|
|||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
// unknown character
|
// A code point above U+10FFFF has no UTF-8 encoding. Passing the
|
||||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
|
// unit through would narrow it to int, where 0xFFFFFFFF becomes
|
||||||
|
// char_traits<char>::eof() and would end the input silently, so
|
||||||
|
// emit a byte that is never valid UTF-8 and let the decoder
|
||||||
|
// reject it.
|
||||||
|
utf8_bytes[0] = 0xFF;
|
||||||
utf8_bytes_filled = 1;
|
utf8_bytes_filled = 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -566,6 +620,46 @@ typename iterator_input_adapter_factory<IteratorType, SentinelType>::adapter_typ
|
|||||||
return factory_type::create(first, last);
|
return factory_type::create(first, last);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The element type a container's data() points at, cv-qualifiers removed.
|
||||||
|
// Ill-formed - and therefore SFINAE-friendly - for types without data().
|
||||||
|
template<typename ContainerType>
|
||||||
|
using container_data_t = typename std::remove_cv<typename std::remove_pointer <
|
||||||
|
decltype(std::declval<const ContainerType&>().data()) >::type >::type;
|
||||||
|
|
||||||
|
// The container's own element type, cv-qualifiers removed. It is looked up on
|
||||||
|
// the bare type so it is also found when ContainerType is deduced as a
|
||||||
|
// reference by the forwarding-reference overload below.
|
||||||
|
template<typename ContainerType>
|
||||||
|
using container_value_t = typename std::remove_cv <
|
||||||
|
typename std::remove_cv<typename std::remove_reference<ContainerType>::type>::type::value_type >::type;
|
||||||
|
|
||||||
|
// Detect a container that stores its elements contiguously as single bytes
|
||||||
|
// (std::string, std::vector<char/unsigned char>, std::array<char, N>,
|
||||||
|
// std::string_view, ...). Such inputs are wrapped in a pointer-based adapter so
|
||||||
|
// they benefit from the contiguous fast paths (bulk string scanning, memcpy for
|
||||||
|
// binary formats) in every C++ standard - not only in C++20, where the standard
|
||||||
|
// library iterators model std::contiguous_iterator and are detected directly.
|
||||||
|
//
|
||||||
|
// data() and size() on their own would be duck typing: they say nothing about
|
||||||
|
// size() counting the units data() points at, and reading [data(), data() +
|
||||||
|
// size()) as bytes would be wrong for a type where it does not. Requiring the
|
||||||
|
// container's own value_type to be that same single-byte element ties the two
|
||||||
|
// together; every contiguous standard container satisfies it. Anything else
|
||||||
|
// keeps the iterator-based adapter, which is always correct - only slower.
|
||||||
|
template<typename ContainerType, typename = void>
|
||||||
|
struct is_contiguous_byte_container : std::false_type {};
|
||||||
|
|
||||||
|
template<typename ContainerType>
|
||||||
|
struct is_contiguous_byte_container < ContainerType, void_t <
|
||||||
|
container_data_t<ContainerType>,
|
||||||
|
container_value_t<ContainerType>,
|
||||||
|
decltype(std::declval<const ContainerType&>().size()) >>
|
||||||
|
: std::integral_constant < bool,
|
||||||
|
std::is_pointer<decltype(std::declval<const ContainerType&>().data())>::value&&
|
||||||
|
std::is_integral<container_data_t<ContainerType>>::value&&
|
||||||
|
sizeof(container_data_t<ContainerType>) == 1 &&
|
||||||
|
std::is_same<container_data_t<ContainerType>, container_value_t<ContainerType>>::value > {};
|
||||||
|
|
||||||
// Convenience shorthand from container to iterator
|
// Convenience shorthand from container to iterator
|
||||||
// Enables ADL on begin(container) and end(container)
|
// Enables ADL on begin(container) and end(container)
|
||||||
// Encloses the using declarations in namespace for not to leak them to outside scope
|
// Encloses the using declarations in namespace for not to leak them to outside scope
|
||||||
@@ -593,12 +687,32 @@ struct container_input_adapter_factory< ContainerType,
|
|||||||
|
|
||||||
} // namespace container_input_adapter_factory_impl
|
} // namespace container_input_adapter_factory_impl
|
||||||
|
|
||||||
template<typename ContainerType>
|
// General container path (iterator-based). Contiguous single-byte containers
|
||||||
typename container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::adapter_type input_adapter(ContainerType&& container)
|
// are excluded here and routed through the pointer-based overload below.
|
||||||
|
template < typename ContainerType,
|
||||||
|
enable_if_t < !is_contiguous_byte_container<ContainerType>::value, int > = 0 >
|
||||||
|
typename container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::adapter_type input_adapter(ContainerType && container)
|
||||||
{
|
{
|
||||||
return container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::create(std::forward<ContainerType>(container));
|
return container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::create(std::forward<ContainerType>(container));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Contiguous single-byte containers (std::string, std::vector<char>, ...) are
|
||||||
|
// wrapped in a pointer-based adapter so the contiguous fast paths apply in every
|
||||||
|
// standard. The pointer keeps the container's own element type (const char* for
|
||||||
|
// std::string, const std::uint8_t* for std::vector<std::uint8_t>, ...), so the
|
||||||
|
// resulting char_type - and therefore the parsing behavior - is byte-for-byte
|
||||||
|
// identical to the iterator-based path; only the raw pointer additionally
|
||||||
|
// enables the bulk fast paths. The container outlives the adapter for the whole
|
||||||
|
// parse (temporaries live until the end of the full expression), exactly as the
|
||||||
|
// iterators it replaces did.
|
||||||
|
template < typename ContainerType,
|
||||||
|
enable_if_t < is_contiguous_byte_container<ContainerType>::value, int > = 0 >
|
||||||
|
auto input_adapter(const ContainerType& container)
|
||||||
|
-> decltype(input_adapter(container.data(), container.data() + container.size()))
|
||||||
|
{
|
||||||
|
return input_adapter(container.data(), container.data() + container.size());
|
||||||
|
}
|
||||||
|
|
||||||
// specialization for std::string
|
// specialization for std::string
|
||||||
using string_input_adapter_type = decltype(input_adapter(std::declval<std::string>()));
|
using string_input_adapter_type = decltype(input_adapter(std::declval<std::string>()));
|
||||||
|
|
||||||
|
|||||||
@@ -370,8 +370,10 @@ class json_sax_dom_parser
|
|||||||
|
|
||||||
case value_t::string:
|
case value_t::string:
|
||||||
{
|
{
|
||||||
// include the length of the quotes, which is 2
|
// escape sequences make the token longer than the value it
|
||||||
v.start_position = v.end_position - v.m_data.m_value.string->size() - 2;
|
// parses to, so the start position cannot be derived from
|
||||||
|
// the value; use the offset the lexer recorded instead
|
||||||
|
v.start_position = m_lexer_ref->get_token_start_position();
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -626,14 +628,7 @@ class json_sax_dom_callback_parser
|
|||||||
if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_structured())
|
if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_structured())
|
||||||
{
|
{
|
||||||
// remove discarded value
|
// remove discarded value
|
||||||
for (auto it = ref_stack.back()->begin(); it != ref_stack.back()->end(); ++it)
|
remove_discarded_value(*ref_stack.back());
|
||||||
{
|
|
||||||
if (it->is_discarded())
|
|
||||||
{
|
|
||||||
ref_stack.back()->erase(it);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
return true;
|
return true;
|
||||||
@@ -674,8 +669,9 @@ class json_sax_dom_callback_parser
|
|||||||
bool end_array()
|
bool end_array()
|
||||||
{
|
{
|
||||||
bool keep = true;
|
bool keep = true;
|
||||||
|
const bool stored = ref_stack.back() != nullptr;
|
||||||
|
|
||||||
if (ref_stack.back())
|
if (stored)
|
||||||
{
|
{
|
||||||
keep = callback(static_cast<int>(ref_stack.size()) - 1, parse_event_t::array_end, *ref_stack.back());
|
keep = callback(static_cast<int>(ref_stack.size()) - 1, parse_event_t::array_end, *ref_stack.back());
|
||||||
if (keep)
|
if (keep)
|
||||||
@@ -709,9 +705,19 @@ class json_sax_dom_callback_parser
|
|||||||
keep_stack.pop_back();
|
keep_stack.pop_back();
|
||||||
|
|
||||||
// remove discarded value
|
// remove discarded value
|
||||||
if (!keep && !ref_stack.empty() && ref_stack.back()->is_array())
|
if (!ref_stack.empty() && ref_stack.back())
|
||||||
{
|
{
|
||||||
ref_stack.back()->m_data.m_value.array->pop_back();
|
if (!keep && ref_stack.back()->is_array())
|
||||||
|
{
|
||||||
|
ref_stack.back()->m_data.m_value.array->pop_back();
|
||||||
|
}
|
||||||
|
else if ((!keep || !stored) && ref_stack.back()->is_object())
|
||||||
|
{
|
||||||
|
// the array is either still stored under its key or was never
|
||||||
|
// stored, leaving the placeholder key() wrote; both show up as
|
||||||
|
// a discarded member of the parent object
|
||||||
|
remove_discarded_value(*ref_stack.back());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return true;
|
return true;
|
||||||
@@ -765,8 +771,10 @@ class json_sax_dom_callback_parser
|
|||||||
|
|
||||||
case value_t::string:
|
case value_t::string:
|
||||||
{
|
{
|
||||||
// include the length of the quotes, which is 2
|
// escape sequences make the token longer than the value it
|
||||||
v.start_position = v.end_position - v.m_data.m_value.string->size() - 2;
|
// parses to, so the start position cannot be derived from
|
||||||
|
// the value; use the offset the lexer recorded instead
|
||||||
|
v.start_position = m_lexer_ref->get_token_start_position();
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -801,6 +809,19 @@ class json_sax_dom_callback_parser
|
|||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
/// remove the discarded value the callback rejected from its parent
|
||||||
|
static void remove_discarded_value(BasicJsonType& parent)
|
||||||
|
{
|
||||||
|
for (auto it = parent.begin(); it != parent.end(); ++it)
|
||||||
|
{
|
||||||
|
if (it->is_discarded())
|
||||||
|
{
|
||||||
|
parent.erase(it);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@param[in] v value to add to the JSON value we build during parsing
|
@param[in] v value to add to the JSON value we build during parsing
|
||||||
@param[in] skip_callback whether we should skip calling the callback
|
@param[in] skip_callback whether we should skip calling the callback
|
||||||
@@ -841,6 +862,18 @@ class json_sax_dom_callback_parser
|
|||||||
// do not handle this value if we just learnt it shall be discarded
|
// do not handle this value if we just learnt it shall be discarded
|
||||||
if (!keep)
|
if (!keep)
|
||||||
{
|
{
|
||||||
|
// if the value was to become an object member, key() already
|
||||||
|
// stored a placeholder for it that has to be removed again
|
||||||
|
if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_object())
|
||||||
|
{
|
||||||
|
JSON_ASSERT(!key_keep_stack.empty());
|
||||||
|
const bool placeholder_stored = key_keep_stack.back();
|
||||||
|
key_keep_stack.pop_back();
|
||||||
|
if (placeholder_stored)
|
||||||
|
{
|
||||||
|
remove_discarded_value(*ref_stack.back());
|
||||||
|
}
|
||||||
|
}
|
||||||
return {false, nullptr};
|
return {false, nullptr};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -19,7 +19,9 @@
|
|||||||
#include <vector> // vector
|
#include <vector> // vector
|
||||||
|
|
||||||
#include <nlohmann/detail/input/input_adapters.hpp>
|
#include <nlohmann/detail/input/input_adapters.hpp>
|
||||||
|
#include <nlohmann/detail/input/number_parse.hpp>
|
||||||
#include <nlohmann/detail/input/position_t.hpp>
|
#include <nlohmann/detail/input/position_t.hpp>
|
||||||
|
#include <nlohmann/detail/input/string_scan.hpp>
|
||||||
#include <nlohmann/detail/macro_scope.hpp>
|
#include <nlohmann/detail/macro_scope.hpp>
|
||||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||||
|
|
||||||
@@ -125,6 +127,25 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/)
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Detect whether an input adapter exposes a contiguous byte block that the
|
||||||
|
// lexer can scan directly (see iterator_input_adapter::supports_bulk_scan).
|
||||||
|
// Adapters without the flag - file, stream, wide-string, user-defined - fall
|
||||||
|
// back to the character-at-a-time string scanner.
|
||||||
|
template<typename InputAdapterType>
|
||||||
|
using detect_supports_bulk_scan = decltype(InputAdapterType::supports_bulk_scan);
|
||||||
|
|
||||||
|
template<typename InputAdapterType>
|
||||||
|
constexpr bool input_adapter_supports_bulk_scan(std::true_type /*detected*/)
|
||||||
|
{
|
||||||
|
return InputAdapterType::supports_bulk_scan;
|
||||||
|
}
|
||||||
|
|
||||||
|
template<typename InputAdapterType>
|
||||||
|
constexpr bool input_adapter_supports_bulk_scan(std::false_type /*detected*/)
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief lexical analysis
|
@brief lexical analysis
|
||||||
|
|
||||||
@@ -146,6 +167,14 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
static constexpr bool lazy_token_string =
|
static constexpr bool lazy_token_string =
|
||||||
input_adapter_supports_seek<InputAdapterType>(is_detected<detect_supports_seek, InputAdapterType> {});
|
input_adapter_supports_seek<InputAdapterType>(is_detected<detect_supports_seek, InputAdapterType> {});
|
||||||
|
|
||||||
|
/// whether string scanning may bulk-consume runs of ordinary characters
|
||||||
|
/// directly from a contiguous input buffer (SWAR fast path). This requires
|
||||||
|
/// the token to be reconstructible lazily (lazy_token_string), so bypassing
|
||||||
|
/// the per-character capture in get() cannot lose error diagnostics.
|
||||||
|
static constexpr bool bulk_scan =
|
||||||
|
lazy_token_string
|
||||||
|
&& input_adapter_supports_bulk_scan<InputAdapterType>(is_detected<detect_supports_bulk_scan, InputAdapterType> {});
|
||||||
|
|
||||||
public:
|
public:
|
||||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||||
|
|
||||||
@@ -265,6 +294,40 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// contiguous input: bulk-append the run of ordinary characters and complete
|
||||||
|
/// well-formed UTF-8 sequences starting at the current read position, leaving
|
||||||
|
/// the first byte that needs individual handling (the closing quote, an
|
||||||
|
/// escape, a control character, or an ill-formed UTF-8 byte) for get()
|
||||||
|
void scan_string_bulk(std::true_type /*bulk*/)
|
||||||
|
{
|
||||||
|
// a pending unget must be consumed through the normal path first
|
||||||
|
if (next_unget)
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const std::size_t remaining = ia.bulk_remaining();
|
||||||
|
if (remaining == 0)
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const auto* const data = reinterpret_cast<const unsigned char*>(ia.bulk_data());
|
||||||
|
|
||||||
|
const std::size_t pos = string_bulk_run(data, remaining);
|
||||||
|
if (pos == 0)
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
token_buffer.append(reinterpret_cast<const typename string_t::value_type*>(data), pos);
|
||||||
|
ia.bulk_skip(pos);
|
||||||
|
// the run contains no newline (all bytes < 0x20 are treated as special),
|
||||||
|
// so only the flat character counters advance
|
||||||
|
position.chars_read_total += pos;
|
||||||
|
position.chars_read_current_line += pos;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// streaming input: no bulk fast path
|
||||||
|
void scan_string_bulk(std::false_type /*bulk*/) const noexcept {}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@brief scan a string literal
|
@brief scan a string literal
|
||||||
|
|
||||||
@@ -290,6 +353,10 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
|
|
||||||
while (true)
|
while (true)
|
||||||
{
|
{
|
||||||
|
// bulk-consume ordinary characters from contiguous input, then
|
||||||
|
// handle the next special byte through the switch below
|
||||||
|
scan_string_bulk(std::integral_constant<bool, bulk_scan> {});
|
||||||
|
|
||||||
// get the next character
|
// get the next character
|
||||||
switch (get())
|
switch (get())
|
||||||
{
|
{
|
||||||
@@ -1008,6 +1075,12 @@ class lexer : public lexer_base<BasicJsonType>
|
|||||||
// changed if minus sign, decimal point, or exponent is read
|
// changed if minus sign, decimal point, or exponent is read
|
||||||
token_type number_type = token_type::value_unsigned;
|
token_type number_type = token_type::value_unsigned;
|
||||||
|
|
||||||
|
// offset just past the last mantissa byte in token_buffer (i.e. the
|
||||||
|
// index of 'e'/'E', or the whole token when there is no exponent).
|
||||||
|
// convert_number() uses it to count significant digits; npos means
|
||||||
|
// "not seen an exponent yet" and is resolved at scan_number_done
|
||||||
|
std::size_t mantissa_end = std::string::npos;
|
||||||
|
|
||||||
// state (init): we just found out we need to scan a number
|
// state (init): we just found out we need to scan a number
|
||||||
switch (current)
|
switch (current)
|
||||||
{
|
{
|
||||||
@@ -1193,6 +1266,9 @@ scan_number_decimal2:
|
|||||||
scan_number_exponent:
|
scan_number_exponent:
|
||||||
// we just parsed an exponent
|
// we just parsed an exponent
|
||||||
number_type = token_type::value_float;
|
number_type = token_type::value_float;
|
||||||
|
// this label is reached only right after the 'e'/'E' was appended (from
|
||||||
|
// the zero, any1, and decimal2 states), so the mantissa ends before it
|
||||||
|
mantissa_end = token_buffer.size() - 1;
|
||||||
switch (get())
|
switch (get())
|
||||||
{
|
{
|
||||||
case '+':
|
case '+':
|
||||||
@@ -1279,45 +1355,147 @@ scan_number_done:
|
|||||||
// we are done scanning a number)
|
// we are done scanning a number)
|
||||||
unget();
|
unget();
|
||||||
|
|
||||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
// no exponent was scanned: the mantissa spans the whole token
|
||||||
errno = 0;
|
if (mantissa_end == std::string::npos)
|
||||||
|
{
|
||||||
|
mantissa_end = token_buffer.size();
|
||||||
|
}
|
||||||
|
|
||||||
// try to parse integers first and fall back to floats
|
return convert_number(number_type, mantissa_end);
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief convert an already-validated integer token to its value
|
||||||
|
|
||||||
|
The digit sequence in [first, last) has been validated by the caller, so a
|
||||||
|
dedicated parser can avoid the locale/errno overhead of std::strtoull.
|
||||||
|
|
||||||
|
@return the token type on success; token_type::uninitialized if @a
|
||||||
|
number_type is not an integer type or the value does not fit, in
|
||||||
|
which case the caller falls back to the floating-point conversion
|
||||||
|
(matching the previous std::strtoull/std::strtoll behavior)
|
||||||
|
*/
|
||||||
|
token_type convert_integer(token_type number_type, const char* first, const char* last)
|
||||||
|
{
|
||||||
if (number_type == token_type::value_unsigned)
|
if (number_type == token_type::value_unsigned)
|
||||||
{
|
{
|
||||||
const auto x = std::strtoull(token_buffer.data(), &endptr, 10);
|
if (parse_integer_unsigned(first, last, value_unsigned))
|
||||||
|
|
||||||
// we checked the number format before
|
|
||||||
JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size());
|
|
||||||
|
|
||||||
if (errno != ERANGE)
|
|
||||||
{
|
{
|
||||||
value_unsigned = static_cast<number_unsigned_t>(x);
|
return token_type::value_unsigned;
|
||||||
if (value_unsigned == x)
|
|
||||||
{
|
|
||||||
return token_type::value_unsigned;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
else if (number_type == token_type::value_integer)
|
else if (number_type == token_type::value_integer)
|
||||||
{
|
{
|
||||||
const auto x = std::strtoll(token_buffer.data(), &endptr, 10);
|
if (parse_integer_signed(first, last, value_integer))
|
||||||
|
|
||||||
// we checked the number format before
|
|
||||||
JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size());
|
|
||||||
|
|
||||||
if (errno != ERANGE)
|
|
||||||
{
|
{
|
||||||
value_integer = static_cast<number_integer_t>(x);
|
return token_type::value_integer;
|
||||||
if (value_integer == x)
|
}
|
||||||
{
|
}
|
||||||
return token_type::value_integer;
|
|
||||||
}
|
return token_type::uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief check whether Clinger's fast path can still succeed for this token
|
||||||
|
|
||||||
|
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
|
||||||
|
more significant digits is at least 10^16 and therefore always exceeds it,
|
||||||
|
so calling the fast path would walk the token one extra time only to
|
||||||
|
decline before strtod has to run anyway.
|
||||||
|
|
||||||
|
Significant digits are the mantissa's digits from the first nonzero one on;
|
||||||
|
the sign, the decimal point, leading zeros, and the exponent do not count.
|
||||||
|
The answer is derived from indices - the digits are not scanned again - so
|
||||||
|
this stays off the hot path of the number scanners.
|
||||||
|
|
||||||
|
@param[in] mantissa_end offset just past the last mantissa byte in
|
||||||
|
token_buffer
|
||||||
|
@return false if parse_float_fast() is guaranteed to decline
|
||||||
|
*/
|
||||||
|
bool mantissa_fits_clinger(std::size_t mantissa_end) const
|
||||||
|
{
|
||||||
|
// 10^16 already exceeds 2^53, so 17 digits can never fit
|
||||||
|
constexpr std::size_t limit = 17;
|
||||||
|
|
||||||
|
const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u;
|
||||||
|
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
|
||||||
|
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
|
||||||
|
// a leading zero can only be a lone "0", which is not significant
|
||||||
|
const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u;
|
||||||
|
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
|
||||||
|
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
|
||||||
|
|
||||||
|
if (JSON_HEDLEY_LIKELY(digits < limit))
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Only a number below 1 can carry further insignificant zeros, and only
|
||||||
|
// while the count stays at the limit does removing them change the
|
||||||
|
// answer - so this loop is skipped for all but a few tokens. Note
|
||||||
|
// token_buffer holds the locale's decimal point, so the fraction is
|
||||||
|
// located through decimal_point_position rather than by searching '.'.
|
||||||
|
if (lead_zero != 0)
|
||||||
|
{
|
||||||
|
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
|
||||||
|
for (std::size_t i = decimal_point_position + 1;
|
||||||
|
digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i)
|
||||||
|
{
|
||||||
|
--digits;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return digits < limit;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief convert the number text in token_buffer to its value and token type
|
||||||
|
|
||||||
|
The digit sequence in token_buffer has already been validated (by the
|
||||||
|
scan_number() state machine or by the contiguous fast path) and holds the
|
||||||
|
locale decimal point in place of '.'. Integers are parsed first and fall
|
||||||
|
back to floating point on overflow. This is shared so both scanners produce
|
||||||
|
identical results.
|
||||||
|
|
||||||
|
@param[in] mantissa_end offset just past the last mantissa byte in
|
||||||
|
token_buffer (the index of 'e'/'E', or
|
||||||
|
token_buffer.size() when there is no exponent);
|
||||||
|
used to skip Clinger's fast path when it cannot
|
||||||
|
possibly succeed - see mantissa_fits_clinger()
|
||||||
|
*/
|
||||||
|
token_type convert_number(token_type number_type, std::size_t mantissa_end)
|
||||||
|
{
|
||||||
|
const char* const num_begin = token_buffer.data();
|
||||||
|
const char* const num_end = num_begin + token_buffer.size();
|
||||||
|
|
||||||
|
if (number_type != token_type::value_float)
|
||||||
|
{
|
||||||
|
const token_type integer_result = convert_integer(number_type, num_begin, num_end);
|
||||||
|
if (integer_result != token_type::uninitialized)
|
||||||
|
{
|
||||||
|
return integer_result;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// this code is reached if we parse a floating-point number or if an
|
// this code is reached if we parse a floating-point number or if an
|
||||||
// integer conversion above failed
|
// integer conversion above overflowed. Prefer std::from_chars
|
||||||
|
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
|
||||||
|
// otherwise the exact Clinger fast path (double only); otherwise the
|
||||||
|
// locale-aware strtof/strtod.
|
||||||
|
if (parse_float_from_chars(num_begin, num_end, value_float))
|
||||||
|
{
|
||||||
|
return token_type::value_float;
|
||||||
|
}
|
||||||
|
// Skipping a fast path that cannot succeed is lossless and saves a full
|
||||||
|
// extra pass over the token's bytes, which otherwise shows up on
|
||||||
|
// high-precision inputs such as canada.json
|
||||||
|
if (mantissa_fits_clinger(mantissa_end)
|
||||||
|
&& parse_float_fast(num_begin, num_end, decimal_point_char, value_float))
|
||||||
|
{
|
||||||
|
return token_type::value_float;
|
||||||
|
}
|
||||||
|
|
||||||
|
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||||
strtof(value_float, token_buffer.data(), &endptr);
|
strtof(value_float, token_buffer.data(), &endptr);
|
||||||
|
|
||||||
// we checked the number format before
|
// we checked the number format before
|
||||||
@@ -1326,6 +1504,158 @@ scan_number_done:
|
|||||||
return token_type::value_float;
|
return token_type::value_float;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief contiguous fast path for scanning a number
|
||||||
|
|
||||||
|
Parses the whole number token straight from the input buffer, avoiding the
|
||||||
|
per-character get()/add() of scan_number(). On success it fills token_buffer
|
||||||
|
(with the locale decimal point substituted, as scan_number() does) and
|
||||||
|
returns the token type. On anything it does not fully recognize as a
|
||||||
|
well-formed number it makes no state change and returns
|
||||||
|
token_type::uninitialized, so the caller falls back to scan_number(), which
|
||||||
|
then produces the exact diagnostic. @a current is the first digit or the
|
||||||
|
leading minus (already read); the remaining bytes are taken from the adapter.
|
||||||
|
*/
|
||||||
|
token_type scan_number_bulk_contiguous()
|
||||||
|
{
|
||||||
|
// a pending unget offsets the buffer position from current; fall back
|
||||||
|
if (next_unget)
|
||||||
|
{
|
||||||
|
return token_type::uninitialized;
|
||||||
|
}
|
||||||
|
const std::size_t rem = ia.bulk_remaining();
|
||||||
|
if (rem == 0)
|
||||||
|
{
|
||||||
|
// the first digit is the last input byte; let scan_number() finish
|
||||||
|
return token_type::uninitialized;
|
||||||
|
}
|
||||||
|
// the byte before the next unread one is current (contiguous input)
|
||||||
|
const char* const data = reinterpret_cast<const char*>(ia.bulk_data()) - 1;
|
||||||
|
const std::size_t avail = rem + 1;
|
||||||
|
|
||||||
|
// validate + classify the number extent (mirrors scan_number()'s grammar)
|
||||||
|
std::size_t i = 0;
|
||||||
|
std::size_t dot_index = std::string::npos;
|
||||||
|
token_type number_type = token_type::value_unsigned;
|
||||||
|
if (data[0] == '-')
|
||||||
|
{
|
||||||
|
number_type = token_type::value_integer;
|
||||||
|
i = 1;
|
||||||
|
if (i >= avail)
|
||||||
|
{
|
||||||
|
return token_type::uninitialized;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (data[i] == '0')
|
||||||
|
{
|
||||||
|
++i;
|
||||||
|
}
|
||||||
|
else if (data[i] >= '1' && data[i] <= '9')
|
||||||
|
{
|
||||||
|
++i;
|
||||||
|
while (i < avail && data[i] >= '0' && data[i] <= '9')
|
||||||
|
{
|
||||||
|
++i;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
return token_type::uninitialized;
|
||||||
|
}
|
||||||
|
if (i < avail && data[i] == '.')
|
||||||
|
{
|
||||||
|
number_type = token_type::value_float;
|
||||||
|
dot_index = i;
|
||||||
|
++i;
|
||||||
|
if (i >= avail || !(data[i] >= '0' && data[i] <= '9'))
|
||||||
|
{
|
||||||
|
return token_type::uninitialized;
|
||||||
|
}
|
||||||
|
while (i < avail && data[i] >= '0' && data[i] <= '9')
|
||||||
|
{
|
||||||
|
++i;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// the mantissa ends here, whether or not an exponent part follows
|
||||||
|
const std::size_t mantissa_end = i;
|
||||||
|
if (i < avail && (data[i] == 'e' || data[i] == 'E'))
|
||||||
|
{
|
||||||
|
number_type = token_type::value_float;
|
||||||
|
++i;
|
||||||
|
if (i < avail && (data[i] == '+' || data[i] == '-'))
|
||||||
|
{
|
||||||
|
++i;
|
||||||
|
}
|
||||||
|
if (i >= avail || !(data[i] >= '0' && data[i] <= '9'))
|
||||||
|
{
|
||||||
|
return token_type::uninitialized;
|
||||||
|
}
|
||||||
|
while (i < avail && data[i] >= '0' && data[i] <= '9')
|
||||||
|
{
|
||||||
|
++i;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const std::size_t len = i;
|
||||||
|
|
||||||
|
// reset() records where this token starts (for diagnostics), so it has
|
||||||
|
// to run before the input position advances below
|
||||||
|
reset();
|
||||||
|
|
||||||
|
// An integer token needs no token_buffer: the SAX callbacks for
|
||||||
|
// number_integer/number_unsigned take only the value, and the overflow
|
||||||
|
// diagnostic rebuilds the text from the input. Convert straight from the
|
||||||
|
// input buffer and leave token_buffer empty. (JSON_DIAGNOSTIC_POSITIONS
|
||||||
|
// derives a number's start position from get_string().size(), so there
|
||||||
|
// the token still has to be materialized.)
|
||||||
|
#if !JSON_DIAGNOSTIC_POSITIONS
|
||||||
|
if (number_type != token_type::value_float)
|
||||||
|
{
|
||||||
|
const token_type integer_result = convert_integer(number_type, data, data + len);
|
||||||
|
if (JSON_HEDLEY_LIKELY(integer_result != token_type::uninitialized))
|
||||||
|
{
|
||||||
|
ia.bulk_skip(len - 1);
|
||||||
|
position.chars_read_total += (len - 1);
|
||||||
|
position.chars_read_current_line += (len - 1);
|
||||||
|
return integer_result;
|
||||||
|
}
|
||||||
|
// The value does not fit an integer, so this token converts as a
|
||||||
|
// float. Recording that here keeps convert_number() below from
|
||||||
|
// repeating the integer attempt that just failed.
|
||||||
|
number_type = token_type::value_float;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// materialize the token exactly as scan_number() would, substituting the
|
||||||
|
// locale decimal point so convert_number()'s strtof fallback stays valid.
|
||||||
|
// reset() already cleared token_buffer, so append() fills it (assign() is
|
||||||
|
// avoided because custom string_t types need not provide it)
|
||||||
|
token_buffer.append(reinterpret_cast<const typename string_t::value_type*>(data), len);
|
||||||
|
if (dot_index != std::string::npos)
|
||||||
|
{
|
||||||
|
token_buffer[dot_index] = static_cast<typename string_t::value_type>(decimal_point_char);
|
||||||
|
decimal_point_position = dot_index;
|
||||||
|
}
|
||||||
|
|
||||||
|
ia.bulk_skip(len - 1);
|
||||||
|
position.chars_read_total += (len - 1);
|
||||||
|
position.chars_read_current_line += (len - 1);
|
||||||
|
|
||||||
|
return convert_number(number_type, mantissa_end);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// contiguous input: try the number fast path, else the byte-path scanner
|
||||||
|
token_type scan_number_dispatch(std::true_type /*bulk*/)
|
||||||
|
{
|
||||||
|
const token_type t = scan_number_bulk_contiguous();
|
||||||
|
return (t != token_type::uninitialized) ? t : scan_number();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// streaming input: always use the byte-path scanner
|
||||||
|
token_type scan_number_dispatch(std::false_type /*bulk*/)
|
||||||
|
{
|
||||||
|
return scan_number();
|
||||||
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@param[in] literal_text the literal text to expect
|
@param[in] literal_text the literal text to expect
|
||||||
@param[in] length the length of the passed literal text
|
@param[in] length the length of the passed literal text
|
||||||
@@ -1357,6 +1687,11 @@ scan_number_done:
|
|||||||
token_buffer.clear();
|
token_buffer.clear();
|
||||||
decimal_point_position = std::string::npos;
|
decimal_point_position = std::string::npos;
|
||||||
|
|
||||||
|
#if JSON_DIAGNOSTIC_POSITIONS
|
||||||
|
// the first character of the token has already been read, hence the -1
|
||||||
|
token_start_position = position.chars_read_total - 1;
|
||||||
|
#endif
|
||||||
|
|
||||||
note_token_start(std::integral_constant<bool, lazy_token_string> {});
|
note_token_start(std::integral_constant<bool, lazy_token_string> {});
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1408,6 +1743,9 @@ scan_number_done:
|
|||||||
if (current == '\n')
|
if (current == '\n')
|
||||||
{
|
{
|
||||||
++position.lines_read;
|
++position.lines_read;
|
||||||
|
// remember the column the newline was read at: chars_read_current_line
|
||||||
|
// is about to be cleared, and a matching unget() cannot reconstruct it
|
||||||
|
chars_read_before_newline = position.chars_read_current_line;
|
||||||
position.chars_read_current_line = 0;
|
position.chars_read_current_line = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1441,12 +1779,20 @@ scan_number_done:
|
|||||||
--position.chars_read_total;
|
--position.chars_read_total;
|
||||||
|
|
||||||
// in case we "unget" a newline, we have to also decrement the lines_read
|
// in case we "unget" a newline, we have to also decrement the lines_read
|
||||||
|
// and restore the column that get() cleared when it saw the newline;
|
||||||
|
// chars_read_current_line == 0 can only mean the last get() read one
|
||||||
if (position.chars_read_current_line == 0)
|
if (position.chars_read_current_line == 0)
|
||||||
{
|
{
|
||||||
if (position.lines_read > 0)
|
if (position.lines_read > 0)
|
||||||
{
|
{
|
||||||
--position.lines_read;
|
--position.lines_read;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// chars_read_before_newline counts the newline itself, which is the
|
||||||
|
// character being ungotten, hence the -1
|
||||||
|
position.chars_read_current_line = (chars_read_before_newline > 0)
|
||||||
|
? chars_read_before_newline - 1
|
||||||
|
: 0;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -1519,6 +1865,15 @@ scan_number_done:
|
|||||||
return position;
|
return position;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#if JSON_DIAGNOSTIC_POSITIONS
|
||||||
|
/// return the offset of the first character of the last read token; unlike
|
||||||
|
/// the token's parsed value, this accounts for escape sequences
|
||||||
|
constexpr std::size_t get_token_start_position() const noexcept
|
||||||
|
{
|
||||||
|
return token_start_position;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
/// seekable adapter: rebuild the last read token from the input on demand
|
/// seekable adapter: rebuild the last read token from the input on demand
|
||||||
const std::vector<char_type>& collect_token_chars(std::vector<char_type>& out, std::true_type /*lazy*/) const
|
const std::vector<char_type>& collect_token_chars(std::vector<char_type>& out, std::true_type /*lazy*/) const
|
||||||
{
|
{
|
||||||
@@ -1680,7 +2035,7 @@ scan_number_done:
|
|||||||
case '7':
|
case '7':
|
||||||
case '8':
|
case '8':
|
||||||
case '9':
|
case '9':
|
||||||
return scan_number();
|
return scan_number_dispatch(std::integral_constant<bool, bulk_scan> {});
|
||||||
|
|
||||||
// end of input (the null byte is needed when parsing from
|
// end of input (the null byte is needed when parsing from
|
||||||
// string literals)
|
// string literals)
|
||||||
@@ -1711,6 +2066,10 @@ scan_number_done:
|
|||||||
/// the start position of the current token
|
/// the start position of the current token
|
||||||
position_t position {};
|
position_t position {};
|
||||||
|
|
||||||
|
/// the value chars_read_current_line had when the last newline was read, so
|
||||||
|
/// that unget() can restore the column instead of leaving it at 0
|
||||||
|
std::size_t chars_read_before_newline = 0;
|
||||||
|
|
||||||
/// raw input token string for error messages; only populated for streaming
|
/// raw input token string for error messages; only populated for streaming
|
||||||
/// adapters (seekable adapters reconstruct it lazily via token_string_start)
|
/// adapters (seekable adapters reconstruct it lazily via token_string_start)
|
||||||
std::vector<char_type> token_string {};
|
std::vector<char_type> token_string {};
|
||||||
@@ -1719,6 +2078,12 @@ scan_number_done:
|
|||||||
/// the last read token on error for seekable adapters (see collect_token_chars)
|
/// the last read token on error for seekable adapters (see collect_token_chars)
|
||||||
std::size_t token_string_start = 0;
|
std::size_t token_string_start = 0;
|
||||||
|
|
||||||
|
#if JSON_DIAGNOSTIC_POSITIONS
|
||||||
|
/// start offset of the current token within the input, used to report
|
||||||
|
/// diagnostic positions (see reset())
|
||||||
|
std::size_t token_start_position = 0;
|
||||||
|
#endif
|
||||||
|
|
||||||
/// buffer for variable-length tokens (numbers, strings)
|
/// buffer for variable-length tokens (numbers, strings)
|
||||||
string_t token_buffer {};
|
string_t token_buffer {};
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,302 @@
|
|||||||
|
// __ _____ _____ _____
|
||||||
|
// __| | __| | | | JSON for Modern C++
|
||||||
|
// | | |__ | | | | | | version 3.12.0
|
||||||
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||||
|
//
|
||||||
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||||
|
// SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include <array> // array
|
||||||
|
#include <cfloat> // FLT_EVAL_METHOD
|
||||||
|
#include <cstddef> // size_t
|
||||||
|
#include <cstdint> // int64_t, uint64_t
|
||||||
|
#include <limits> // numeric_limits
|
||||||
|
|
||||||
|
#include <nlohmann/detail/macro_scope.hpp>
|
||||||
|
|
||||||
|
// std::from_chars lives in <charconv>, but being in C++17 mode does not
|
||||||
|
// guarantee the header exists: GCC 7 sets __cplusplus to C++17 yet ships no
|
||||||
|
// <charconv> (added in GCC 8; floating-point support in GCC 11). Guard the
|
||||||
|
// include with __has_include so such toolchains fall back to the scalar path.
|
||||||
|
#if defined(JSON_HAS_CPP_17) && defined(__has_include)
|
||||||
|
#if __has_include(<charconv>)
|
||||||
|
#include <charconv> // from_chars (only used when __cpp_lib_to_chars is defined)
|
||||||
|
#include <system_error> // errc
|
||||||
|
#endif
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// This file contains the value-conversion helpers used by the lexer to turn an
|
||||||
|
// already-validated number token into a value, without the locale/errno
|
||||||
|
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
|
||||||
|
// stays focused on scanning; see lexer::convert_number().
|
||||||
|
|
||||||
|
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||||
|
namespace detail
|
||||||
|
{
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief fast integer parser for an already-validated unsigned integer
|
||||||
|
|
||||||
|
The number scanner has already checked that [first, last) is a valid JSON
|
||||||
|
integer, so this only needs to accumulate the digits and detect overflow. This
|
||||||
|
avoids the locale/errno machinery of std::strtoull, which dominates
|
||||||
|
integer-heavy inputs.
|
||||||
|
|
||||||
|
@param[in] first pointer to the first character (a digit)
|
||||||
|
@param[in] last pointer past the last character
|
||||||
|
@param[out] value the parsed value on success
|
||||||
|
@return true if the value fit into @a NumberUnsignedType; false on overflow, in
|
||||||
|
which case the caller falls back to floating-point parsing (matching the
|
||||||
|
previous std::strtoull behavior)
|
||||||
|
*/
|
||||||
|
template<typename NumberUnsignedType>
|
||||||
|
bool parse_integer_unsigned(const char* first, const char* last, NumberUnsignedType& value) noexcept
|
||||||
|
{
|
||||||
|
// accumulate in the widest unsigned type used by the previous strtoull
|
||||||
|
// path so the overflow behavior is unchanged for custom number types
|
||||||
|
std::uint64_t x = 0;
|
||||||
|
constexpr std::uint64_t cutoff = (std::numeric_limits<std::uint64_t>::max)() / 10u;
|
||||||
|
constexpr std::uint64_t cutlim = (std::numeric_limits<std::uint64_t>::max)() % 10u;
|
||||||
|
for (const char* p = first; p != last; ++p)
|
||||||
|
{
|
||||||
|
const auto digit = static_cast<std::uint64_t>(static_cast<unsigned char>(*p) - static_cast<unsigned char>('0'));
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(x > cutoff || (x == cutoff && digit > cutlim)))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
x = (x * 10u) + digit;
|
||||||
|
}
|
||||||
|
value = static_cast<NumberUnsignedType>(x);
|
||||||
|
// reject values that do not round-trip into a narrower NumberUnsignedType
|
||||||
|
return static_cast<std::uint64_t>(value) == x;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief fast integer parser for an already-validated negative integer
|
||||||
|
|
||||||
|
@param[in] first pointer to the leading '-'
|
||||||
|
@param[in] last pointer past the last character
|
||||||
|
@param[out] value the parsed (negative) value on success
|
||||||
|
@return true on success; false on overflow (caller falls back to float)
|
||||||
|
*/
|
||||||
|
template<typename NumberIntegerType>
|
||||||
|
bool parse_integer_signed(const char* first, const char* last, NumberIntegerType& value) noexcept
|
||||||
|
{
|
||||||
|
// the state machine only reaches the signed path via a leading '-'
|
||||||
|
JSON_ASSERT(first != last && *first == '-');
|
||||||
|
std::uint64_t magnitude = 0;
|
||||||
|
// |INT64_MIN| == INT64_MAX + 1; this is the largest admissible magnitude
|
||||||
|
constexpr std::uint64_t limit = static_cast<std::uint64_t>((std::numeric_limits<std::int64_t>::max)()) + 1u;
|
||||||
|
for (const char* p = first + 1; p != last; ++p)
|
||||||
|
{
|
||||||
|
const auto digit = static_cast<std::uint64_t>(static_cast<unsigned char>(*p) - static_cast<unsigned char>('0'));
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(magnitude > (limit - digit) / 10u))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
magnitude = (magnitude * 10u) + digit;
|
||||||
|
}
|
||||||
|
const std::int64_t x = (magnitude == limit)
|
||||||
|
? (std::numeric_limits<std::int64_t>::min)()
|
||||||
|
: -static_cast<std::int64_t>(magnitude);
|
||||||
|
value = static_cast<NumberIntegerType>(x);
|
||||||
|
// reject values that do not round-trip into a narrower NumberIntegerType
|
||||||
|
return static_cast<std::int64_t>(value) == x;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief exact fast path for parsing a `double` (Clinger's algorithm)
|
||||||
|
|
||||||
|
For the common case - at most 19 significant digits, a decimal exponent in
|
||||||
|
[-22, 22], and a significand below 2^53 - the value equals significand *
|
||||||
|
10^exp computed in IEEE-754 double arithmetic, which is exact under
|
||||||
|
round-to-nearest because both operands are exactly representable. This is the
|
||||||
|
same fast path used by fast_float/simdjson; the general cases are left to
|
||||||
|
std::strtod. The parser only activates for number_float_t == double; float and
|
||||||
|
long double keep the std::strtof/std::strtold paths (see the templated overload
|
||||||
|
below).
|
||||||
|
|
||||||
|
@param[in] first pointer to the first character of the number
|
||||||
|
@param[in] last pointer past the last character
|
||||||
|
@param[in] decimal_point the (locale-dependent) decimal point character
|
||||||
|
@param[out] out the parsed value on success
|
||||||
|
@return true if the value was parsed exactly; false to fall back to strtod
|
||||||
|
*/
|
||||||
|
template<typename DecimalPointType>
|
||||||
|
bool parse_float_fast(const char* first, const char* last, DecimalPointType decimal_point, double& out) noexcept
|
||||||
|
{
|
||||||
|
#if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0
|
||||||
|
// Clinger's fast path is only exact when double operations are evaluated in
|
||||||
|
// true double precision. On platforms that keep intermediates in extended
|
||||||
|
// precision (e.g. the x87 FPU on 32-bit x86, where FLT_EVAL_METHOD == 2) the
|
||||||
|
// single significand * 10^scale step is double-rounded and can be 1 ULP off,
|
||||||
|
// so decline and let the caller fall back to the correctly-rounded
|
||||||
|
// std::from_chars / std::strtod path.
|
||||||
|
static_cast<void>(first);
|
||||||
|
static_cast<void>(last);
|
||||||
|
static_cast<void>(decimal_point);
|
||||||
|
static_cast<void>(out);
|
||||||
|
return false;
|
||||||
|
#else
|
||||||
|
static const std::array<double, 23> powers_of_ten =
|
||||||
|
{
|
||||||
|
{
|
||||||
|
1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11,
|
||||||
|
1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const char* p = first;
|
||||||
|
bool negative = false;
|
||||||
|
if (p != last && (*p == '-' || *p == '+'))
|
||||||
|
{
|
||||||
|
negative = (*p == '-');
|
||||||
|
++p;
|
||||||
|
}
|
||||||
|
|
||||||
|
std::uint64_t significand = 0;
|
||||||
|
int num_digits = 0;
|
||||||
|
int fractional_digits = 0;
|
||||||
|
bool seen_dot = false;
|
||||||
|
bool any_digit = false;
|
||||||
|
for (; p != last; ++p)
|
||||||
|
{
|
||||||
|
const char c = *p;
|
||||||
|
if (c >= '0' && c <= '9')
|
||||||
|
{
|
||||||
|
any_digit = true;
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(num_digits >= 19))
|
||||||
|
{
|
||||||
|
return false; // significand may not fit into uint64_t
|
||||||
|
}
|
||||||
|
significand = (significand * 10u) + static_cast<std::uint64_t>(c - '0');
|
||||||
|
++num_digits;
|
||||||
|
fractional_digits += static_cast<int>(seen_dot);
|
||||||
|
}
|
||||||
|
else if (static_cast<DecimalPointType>(c) == decimal_point)
|
||||||
|
{
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(seen_dot))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
seen_dot = true;
|
||||||
|
}
|
||||||
|
else if (c == 'e' || c == 'E')
|
||||||
|
{
|
||||||
|
++p;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!any_digit))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
int exponent = 0;
|
||||||
|
if (p != last) // an exponent part remains
|
||||||
|
{
|
||||||
|
bool exp_negative = false;
|
||||||
|
if (p != last && (*p == '-' || *p == '+'))
|
||||||
|
{
|
||||||
|
exp_negative = (*p == '-');
|
||||||
|
++p;
|
||||||
|
}
|
||||||
|
bool any_exp_digit = false;
|
||||||
|
for (; p != last; ++p)
|
||||||
|
{
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(*p < '0' || *p > '9'))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
exponent = (exponent * 10) + (*p - '0');
|
||||||
|
any_exp_digit = true;
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(exponent > 9999))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!any_exp_digit))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (exp_negative)
|
||||||
|
{
|
||||||
|
exponent = -exponent;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const int scale = exponent - fractional_digits;
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(significand >= (static_cast<std::uint64_t>(1) << 53)))
|
||||||
|
{
|
||||||
|
return false; // significand not exactly representable as double
|
||||||
|
}
|
||||||
|
|
||||||
|
auto result = static_cast<double>(significand);
|
||||||
|
if (scale >= 0)
|
||||||
|
{
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(scale > 22))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
result *= powers_of_ten[static_cast<std::size_t>(scale)];
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(-scale > 22))
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
result /= powers_of_ten[static_cast<std::size_t>(-scale)];
|
||||||
|
}
|
||||||
|
out = negative ? -result : result;
|
||||||
|
return true;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
/// fast float path is only exact for `double`; decline for float/long double
|
||||||
|
template<typename DecimalPointType, typename FloatType>
|
||||||
|
bool parse_float_fast(const char* /*first*/, const char* /*last*/, DecimalPointType /*decimal_point*/, FloatType& /*out*/) noexcept
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief parse a float with std::from_chars (Eisel-Lemire) when available
|
||||||
|
|
||||||
|
std::from_chars is locale-independent, correctly rounded, and - via the
|
||||||
|
Eisel-Lemire algorithm in modern standard libraries - much faster than strtod
|
||||||
|
over the whole value range (not just the Clinger subset). It is used only when
|
||||||
|
__cpp_lib_to_chars indicates full floating-point support and only when it
|
||||||
|
consumes the entire token ([first, last)); a partial parse means the buffer
|
||||||
|
uses a non-'.' locale decimal point, in which case the caller falls back to the
|
||||||
|
locale-aware path. An under-/overflow (result_out_of_range) also declines, so
|
||||||
|
the caller's strtod fallback supplies the well-defined ±inf/0 result the parser
|
||||||
|
expects (side-stepping the P4168 divergence between implementations).
|
||||||
|
|
||||||
|
@return true if the value was parsed exactly and fully; false to fall back
|
||||||
|
*/
|
||||||
|
template<typename FloatType>
|
||||||
|
bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept
|
||||||
|
{
|
||||||
|
// JSON_HAS_CPP_17 must gate the use as well as the <charconv> include above:
|
||||||
|
// some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even
|
||||||
|
// in C++14 mode, where <charconv> is not included.
|
||||||
|
#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars)
|
||||||
|
const auto result = std::from_chars(first, last, out);
|
||||||
|
return result.ec == std::errc() && result.ptr == last;
|
||||||
|
#else
|
||||||
|
static_cast<void>(first);
|
||||||
|
static_cast<void>(last);
|
||||||
|
static_cast<void>(out);
|
||||||
|
return false;
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
} // namespace detail
|
||||||
|
NLOHMANN_JSON_NAMESPACE_END
|
||||||
@@ -0,0 +1,293 @@
|
|||||||
|
// __ _____ _____ _____
|
||||||
|
// __| | __| | | | JSON for Modern C++
|
||||||
|
// | | |__ | | | | | | version 3.12.0
|
||||||
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||||
|
//
|
||||||
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||||
|
// SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include <cstddef> // size_t
|
||||||
|
#include <cstdint> // uint64_t
|
||||||
|
#include <cstring> // memcpy
|
||||||
|
|
||||||
|
#include <nlohmann/detail/macro_scope.hpp>
|
||||||
|
|
||||||
|
// Optional SIMD backend for bulk UTF-8 validation. This is an opt-in external
|
||||||
|
// dependency: nlohmann/json itself stays header-only and the C++11 scalar
|
||||||
|
// validator below is always available; defining JSON_USE_SIMDUTF additionally
|
||||||
|
// requires the simdutf headers on the include path and linking the simdutf
|
||||||
|
// library. See string_bulk_run().
|
||||||
|
//
|
||||||
|
// simdutf.h itself requires C++17 - it rejects older standards with an #error -
|
||||||
|
// so the backend is only compiled in from C++17 on. Below that the macro has no
|
||||||
|
// effect and the scalar validator is used; it accepts and rejects exactly the
|
||||||
|
// same input, so only throughput differs. macro_scope.hpp is included above to
|
||||||
|
// have JSON_HAS_CPP_17 available for this test.
|
||||||
|
#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17)
|
||||||
|
#include <simdutf.h>
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// This file contains the byte-level string-scanning helpers used by the lexer's
|
||||||
|
// contiguous fast path. They operate purely on raw bytes (no dependency on the
|
||||||
|
// lexer's template parameters) so they are free functions, keeping the lexer
|
||||||
|
// itself focused on the state machine; see lexer::scan_string_bulk().
|
||||||
|
|
||||||
|
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||||
|
namespace detail
|
||||||
|
{
|
||||||
|
|
||||||
|
// classify a single byte as needing individual string handling: the closing
|
||||||
|
// quote, an escape, a control character, or a non-ASCII (UTF-8)
|
||||||
|
// lead/continuation byte. Ordinary bytes (0x20..0x7F except '"' and '\\') are
|
||||||
|
// copied verbatim, which the bulk scanner does 8 bytes at a time.
|
||||||
|
inline bool is_string_special(unsigned char c) noexcept
|
||||||
|
{
|
||||||
|
return c == '\"' || c == '\\' || c < 0x20u || c >= 0x80u;
|
||||||
|
}
|
||||||
|
|
||||||
|
// SWAR helper: return a word whose high bit is set in every byte of @a v that
|
||||||
|
// is_string_special(); zero if the 8 bytes are all ordinary.
|
||||||
|
inline std::uint64_t swar_string_special(std::uint64_t v) noexcept
|
||||||
|
{
|
||||||
|
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
||||||
|
constexpr std::uint64_t high = 0x8080808080808080ull;
|
||||||
|
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
|
||||||
|
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
|
||||||
|
const std::uint64_t has_quote = (q - ones) & ~q & high;
|
||||||
|
const std::uint64_t has_backslash = (b - ones) & ~b & high;
|
||||||
|
const std::uint64_t has_control = (v - 0x2020202020202020ull) & ~v & high; // < 0x20
|
||||||
|
const std::uint64_t has_non_ascii = v & high; // >= 0x80
|
||||||
|
return has_quote | has_backslash | has_control | has_non_ascii;
|
||||||
|
}
|
||||||
|
|
||||||
|
// return the index of the first is_string_special() byte in [data, data+n), or
|
||||||
|
// n if every byte is ordinary; scans 8 bytes at a time
|
||||||
|
inline std::size_t find_string_special(const unsigned char* data, std::size_t n) noexcept
|
||||||
|
{
|
||||||
|
std::size_t i = 0;
|
||||||
|
for (; i + 8 <= n; i += 8)
|
||||||
|
{
|
||||||
|
std::uint64_t word = 0;
|
||||||
|
std::memcpy(&word, data + i, sizeof(word));
|
||||||
|
if (swar_string_special(word) != 0)
|
||||||
|
{
|
||||||
|
// a special byte is in this word; locate it (endian-agnostic)
|
||||||
|
for (std::size_t j = 0; j < 8; ++j)
|
||||||
|
{
|
||||||
|
if (is_string_special(data[i + j]))
|
||||||
|
{
|
||||||
|
return i + j;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (; i < n; ++i)
|
||||||
|
{
|
||||||
|
if (is_string_special(data[i]))
|
||||||
|
{
|
||||||
|
return i;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
|
// classify a byte as one the serializer must NOT copy verbatim when
|
||||||
|
// ensure_ascii is requested: the closing quote, an escape, a control character
|
||||||
|
// (< 0x20), DEL (0x7F), or any non-ASCII byte (>= 0x80). Everything else -
|
||||||
|
// printable ASCII except '"' and '\\' - is emitted unchanged. Note this differs
|
||||||
|
// from is_string_special() only in that 0x7F is also a stop (it is escaped as
|
||||||
|
// \u007f under ensure_ascii).
|
||||||
|
inline bool is_ascii_copyable(unsigned char c) noexcept
|
||||||
|
{
|
||||||
|
return c >= 0x20u && c < 0x7Fu && c != '\"' && c != '\\';
|
||||||
|
}
|
||||||
|
|
||||||
|
// return the index of the first byte in [data, data+n) that is NOT
|
||||||
|
// is_ascii_copyable(), or n if every byte can be copied verbatim; scans 8 bytes
|
||||||
|
// at a time. Used by the serializer's ensure_ascii fast path.
|
||||||
|
inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_t n) noexcept
|
||||||
|
{
|
||||||
|
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
||||||
|
constexpr std::uint64_t high = 0x8080808080808080ull;
|
||||||
|
std::size_t i = 0;
|
||||||
|
for (; i + 8 <= n; i += 8)
|
||||||
|
{
|
||||||
|
std::uint64_t v = 0;
|
||||||
|
std::memcpy(&v, data + i, sizeof(v));
|
||||||
|
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
|
||||||
|
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
|
||||||
|
const std::uint64_t d = v ^ 0x7F7F7F7F7F7F7F7Full; // DEL (0x7F)
|
||||||
|
const std::uint64_t stop = ((q - ones) & ~q & high) // == '"'
|
||||||
|
| ((b - ones) & ~b & high) // == '\\'
|
||||||
|
| ((d - ones) & ~d & high) // == 0x7F
|
||||||
|
| ((v - 0x2020202020202020ull) & ~v & high) // < 0x20
|
||||||
|
| (v & high); // >= 0x80
|
||||||
|
if (stop != 0)
|
||||||
|
{
|
||||||
|
for (std::size_t j = 0; j < 8; ++j)
|
||||||
|
{
|
||||||
|
if (!is_ascii_copyable(data[i + j]))
|
||||||
|
{
|
||||||
|
return i + j;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (; i < n; ++i)
|
||||||
|
{
|
||||||
|
if (!is_ascii_copyable(data[i]))
|
||||||
|
{
|
||||||
|
return i;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its
|
||||||
|
// length (2..4) only when the bytes form a *well-formed* sequence using exactly
|
||||||
|
// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts
|
||||||
|
// precisely what the byte path accepts. Returns 0 for anything that is invalid,
|
||||||
|
// incomplete, or that the byte path must diagnose (the caller then defers to
|
||||||
|
// that path, keeping error messages unchanged). Lead bytes < 0x80 are handled
|
||||||
|
// by the caller and never passed here.
|
||||||
|
inline std::size_t validate_one_utf8(const unsigned char* data, std::size_t avail) noexcept
|
||||||
|
{
|
||||||
|
const unsigned char c0 = data[0];
|
||||||
|
if (c0 >= 0xC2 && c0 <= 0xDF) // U+0080..U+07FF
|
||||||
|
{
|
||||||
|
if (avail >= 2 && data[1] >= 0x80 && data[1] <= 0xBF)
|
||||||
|
{
|
||||||
|
return 2;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else if (c0 == 0xE0) // U+0800..U+0FFF
|
||||||
|
{
|
||||||
|
if (avail >= 3 && data[1] >= 0xA0 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF)
|
||||||
|
{
|
||||||
|
return 3;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else if ((c0 >= 0xE1 && c0 <= 0xEC) || c0 == 0xEE || c0 == 0xEF) // U+1000..U+CFFF, U+E000..U+FFFF
|
||||||
|
{
|
||||||
|
if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF)
|
||||||
|
{
|
||||||
|
return 3;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else if (c0 == 0xED) // U+D000..U+D7FF (excludes surrogates)
|
||||||
|
{
|
||||||
|
if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0x9F && data[2] >= 0x80 && data[2] <= 0xBF)
|
||||||
|
{
|
||||||
|
return 3;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else if (c0 == 0xF0) // U+10000..U+3FFFF
|
||||||
|
{
|
||||||
|
if (avail >= 4 && data[1] >= 0x90 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
||||||
|
{
|
||||||
|
return 4;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else if (c0 >= 0xF1 && c0 <= 0xF3) // U+40000..U+FFFFF
|
||||||
|
{
|
||||||
|
if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
||||||
|
{
|
||||||
|
return 4;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else if (c0 == 0xF4) // U+100000..U+10FFFF
|
||||||
|
{
|
||||||
|
if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0x8F && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
||||||
|
{
|
||||||
|
return 4;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return 0; // invalid, incomplete, or must be diagnosed by the byte path
|
||||||
|
}
|
||||||
|
|
||||||
|
// Scalar (C++11) computation of the bulk run length: the number of leading
|
||||||
|
// bytes in [data, data+n) that are ordinary ASCII or complete well-formed UTF-8
|
||||||
|
// sequences, stopping before the first byte that needs individual handling (the
|
||||||
|
// closing quote, an escape, a control character, or an ill-formed/truncated
|
||||||
|
// sequence). ASCII is skipped 8 bytes at a time.
|
||||||
|
inline std::size_t scalar_string_bulk_run(const unsigned char* data, std::size_t n) noexcept
|
||||||
|
{
|
||||||
|
std::size_t pos = 0;
|
||||||
|
while (pos < n)
|
||||||
|
{
|
||||||
|
pos += find_string_special(data + pos, n - pos);
|
||||||
|
if (pos >= n || data[pos] < 0x80u)
|
||||||
|
{
|
||||||
|
break; // end of buffer, or a quote/escape/control byte
|
||||||
|
}
|
||||||
|
const std::size_t seq = validate_one_utf8(data + pos, n - pos);
|
||||||
|
if (seq == 0)
|
||||||
|
{
|
||||||
|
break; // ill-formed or truncated: let the byte path diagnose it
|
||||||
|
}
|
||||||
|
pos += seq;
|
||||||
|
}
|
||||||
|
return pos;
|
||||||
|
}
|
||||||
|
|
||||||
|
#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17)
|
||||||
|
// Index of the first quote/escape/control byte in [data, data+n) (non-ASCII
|
||||||
|
// bytes are *not* stops here - the whole run is handed to simdutf), or n.
|
||||||
|
inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t n) noexcept
|
||||||
|
{
|
||||||
|
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
||||||
|
constexpr std::uint64_t high = 0x8080808080808080ull;
|
||||||
|
std::size_t i = 0;
|
||||||
|
for (; i + 8 <= n; i += 8)
|
||||||
|
{
|
||||||
|
std::uint64_t v = 0;
|
||||||
|
std::memcpy(&v, data + i, sizeof(v));
|
||||||
|
const std::uint64_t q = v ^ 0x2222222222222222ull;
|
||||||
|
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull;
|
||||||
|
const std::uint64_t hit = ((q - ones) & ~q & high)
|
||||||
|
| ((b - ones) & ~b & high)
|
||||||
|
| ((v - 0x2020202020202020ull) & ~v & high);
|
||||||
|
if (hit != 0)
|
||||||
|
{
|
||||||
|
for (std::size_t j = 0; j < 8; ++j)
|
||||||
|
{
|
||||||
|
const unsigned char c = data[i + j];
|
||||||
|
if (c == '\"' || c == '\\' || c < 0x20u)
|
||||||
|
{
|
||||||
|
return i + j;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (; i < n; ++i)
|
||||||
|
{
|
||||||
|
const unsigned char c = data[i];
|
||||||
|
if (c == '\"' || c == '\\' || c < 0x20u)
|
||||||
|
{
|
||||||
|
return i;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// Backend-dispatched bulk run length. With JSON_USE_SIMDUTF the run up to the
|
||||||
|
// next delimiter is validated in one shot by simdutf; on the rare failure the
|
||||||
|
// scalar helper recomputes the exact valid prefix so the byte path still
|
||||||
|
// produces the precise diagnostic. Without it, the pure scalar path is used.
|
||||||
|
inline std::size_t string_bulk_run(const unsigned char* data, std::size_t n) noexcept
|
||||||
|
{
|
||||||
|
#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17)
|
||||||
|
const std::size_t run = find_string_delimiter(data, n);
|
||||||
|
if (run != 0 && simdutf::validate_utf8(reinterpret_cast<const char*>(data), run))
|
||||||
|
{
|
||||||
|
return run;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
return scalar_string_bulk_run(data, n);
|
||||||
|
}
|
||||||
|
|
||||||
|
} // namespace detail
|
||||||
|
NLOHMANN_JSON_NAMESPACE_END
|
||||||
@@ -231,7 +231,9 @@ struct char_traits<signed char> : std::char_traits<char>
|
|||||||
// Redefine to_int_type function
|
// Redefine to_int_type function
|
||||||
static int_type to_int_type(char_type c) noexcept
|
static int_type to_int_type(char_type c) noexcept
|
||||||
{
|
{
|
||||||
return static_cast<int_type>(c);
|
// cast via unsigned char: sign-extending a negative char_type would make
|
||||||
|
// byte 0xFF indistinguishable from eof()
|
||||||
|
return static_cast<int_type>(static_cast<unsigned char>(c));
|
||||||
}
|
}
|
||||||
|
|
||||||
static char_type to_char_type(int_type i) noexcept
|
static char_type to_char_type(int_type i) noexcept
|
||||||
@@ -699,21 +701,35 @@ struct is_json_pointer_of<A, ::nlohmann::json_pointer<A>> : std::true_type {};
|
|||||||
template <typename A>
|
template <typename A>
|
||||||
struct is_json_pointer_of<A, ::nlohmann::json_pointer<A>&> : std::true_type {};
|
struct is_json_pointer_of<A, ::nlohmann::json_pointer<A>&> : std::true_type {};
|
||||||
|
|
||||||
// checks if A and B are comparable using Compare functor
|
// checks if A and B are comparable using Compare functor, assuming that
|
||||||
|
// neither A nor B is a json_pointer type (that case is handled by
|
||||||
|
// is_comparable below, which never instantiates this helper otherwise)
|
||||||
template<typename Compare, typename A, typename B, typename = void>
|
template<typename Compare, typename A, typename B, typename = void>
|
||||||
struct is_comparable : std::false_type {};
|
struct is_comparable_no_json_pointer : std::false_type {};
|
||||||
|
|
||||||
// We exclude json_pointer here, because the checks using Compare(A, B) will
|
|
||||||
// use json_pointer::operator string_t() which triggers a deprecation warning
|
|
||||||
// for GCC. See https://github.com/nlohmann/json/issues/4621. The call to
|
|
||||||
// is_json_pointer_of can be removed once the deprecated function has been
|
|
||||||
// removed.
|
|
||||||
template<typename Compare, typename A, typename B>
|
template<typename Compare, typename A, typename B>
|
||||||
struct is_comparable < Compare, A, B, enable_if_t < !is_json_pointer_of<A, B>::value
|
struct is_comparable_no_json_pointer < Compare, A, B, enable_if_t <
|
||||||
&& std::is_constructible <decltype(std::declval<Compare>()(std::declval<A>(), std::declval<B>()))>::value
|
std::is_constructible <decltype(std::declval<Compare>()(std::declval<A>(), std::declval<B>()))>::value
|
||||||
&& std::is_constructible <decltype(std::declval<Compare>()(std::declval<B>(), std::declval<A>()))>::value
|
&& std::is_constructible <decltype(std::declval<Compare>()(std::declval<B>(), std::declval<A>()))>::value
|
||||||
>> : std::true_type {};
|
>> : std::true_type {};
|
||||||
|
|
||||||
|
// checks if A and B are comparable using Compare functor
|
||||||
|
// We dispatch on is_json_pointer_of as a plain bool (rather than folding it
|
||||||
|
// into a single enable_if_t condition together with the checks below) so
|
||||||
|
// that the Compare(A, B) checks are only ever written - and thus only ever
|
||||||
|
// instantiated - when A/B are not a json_pointer/string pair. Those checks
|
||||||
|
// use json_pointer::operator string_t() (GCC, see #4621) resp. the
|
||||||
|
// deprecated json_pointer/string operator== (Clang, see #5288), and merely
|
||||||
|
// naming them as later operands of a plain && chain is not sufficient to
|
||||||
|
// avoid their instantiation on all compilers, even when the first operand
|
||||||
|
// is false. The dispatch on is_json_pointer_of can be removed once the
|
||||||
|
// deprecated json_pointer comparison operators have been removed.
|
||||||
|
template<typename Compare, typename A, typename B, bool = is_json_pointer_of<A, B>::value>
|
||||||
|
struct is_comparable : std::false_type {};
|
||||||
|
|
||||||
|
template<typename Compare, typename A, typename B>
|
||||||
|
struct is_comparable<Compare, A, B, false> : is_comparable_no_json_pointer<Compare, A, B> {};
|
||||||
|
|
||||||
template<typename T>
|
template<typename T>
|
||||||
using detect_is_transparent = typename T::is_transparent;
|
using detect_is_transparent = typename T::is_transparent;
|
||||||
|
|
||||||
|
|||||||
@@ -813,7 +813,10 @@ class binary_writer
|
|||||||
bool prefix_required = true;
|
bool prefix_required = true;
|
||||||
if (use_type && !j.m_data.m_value.array->empty())
|
if (use_type && !j.m_data.m_value.array->empty())
|
||||||
{
|
{
|
||||||
JSON_ASSERT(use_count);
|
if (!use_count)
|
||||||
|
{
|
||||||
|
JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j));
|
||||||
|
}
|
||||||
const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata);
|
const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata);
|
||||||
const bool same_prefix = std::all_of(j.begin() + 1, j.end(),
|
const bool same_prefix = std::all_of(j.begin() + 1, j.end(),
|
||||||
[this, first_prefix, use_bjdata](const BasicJsonType & v)
|
[this, first_prefix, use_bjdata](const BasicJsonType & v)
|
||||||
@@ -859,7 +862,10 @@ class binary_writer
|
|||||||
|
|
||||||
if (use_type && (bjdata_draft3 || !j.m_data.m_value.binary->empty()))
|
if (use_type && (bjdata_draft3 || !j.m_data.m_value.binary->empty()))
|
||||||
{
|
{
|
||||||
JSON_ASSERT(use_count);
|
if (!use_count)
|
||||||
|
{
|
||||||
|
JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j));
|
||||||
|
}
|
||||||
oa->write_character(to_char_type('$'));
|
oa->write_character(to_char_type('$'));
|
||||||
oa->write_character(bjdata_draft3 ? 'B' : 'U');
|
oa->write_character(bjdata_draft3 ? 'B' : 'U');
|
||||||
}
|
}
|
||||||
@@ -911,7 +917,10 @@ class binary_writer
|
|||||||
bool prefix_required = true;
|
bool prefix_required = true;
|
||||||
if (use_type && !j.m_data.m_value.object->empty())
|
if (use_type && !j.m_data.m_value.object->empty())
|
||||||
{
|
{
|
||||||
JSON_ASSERT(use_count);
|
if (!use_count)
|
||||||
|
{
|
||||||
|
JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j));
|
||||||
|
}
|
||||||
const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata);
|
const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata);
|
||||||
const bool same_prefix = std::all_of(j.begin(), j.end(),
|
const bool same_prefix = std::all_of(j.begin(), j.end(),
|
||||||
[this, first_prefix, use_bjdata](const BasicJsonType & v)
|
[this, first_prefix, use_bjdata](const BasicJsonType & v)
|
||||||
@@ -1670,7 +1679,23 @@ class binary_writer
|
|||||||
{
|
{
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
len *= static_cast<std::size_t>(el.template get<std::uint64_t>());
|
|
||||||
|
// a dimension that does not fit into std::size_t, or a product that
|
||||||
|
// overflows it, would wrap around and could match the size of
|
||||||
|
// _ArrayData_ by accident; the resulting header announces an
|
||||||
|
// element count that no reader can honor (the binary reader rejects
|
||||||
|
// it with out_of_range.408), so encode as a plain object instead
|
||||||
|
const auto dim = el.template get<std::uint64_t>();
|
||||||
|
if (!value_in_range_of<std::size_t>(dim))
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
const auto dim_size = static_cast<std::size_t>(dim);
|
||||||
|
if (dim_size != 0 && len > (std::numeric_limits<std::size_t>::max)() / dim_size)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
len *= dim_size;
|
||||||
}
|
}
|
||||||
|
|
||||||
key = "_ArrayData_";
|
key = "_ArrayData_";
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -1341,11 +1341,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
const error_handler_t error_handler = error_handler_t::strict) const
|
const error_handler_t error_handler = error_handler_t::strict) const
|
||||||
{
|
{
|
||||||
string_t result;
|
string_t result;
|
||||||
serializer s(detail::output_adapter<char, string_t>(result), indent_char, error_handler);
|
detail::output_string_adapter<char, string_t> string_adapter(result);
|
||||||
|
serializer s(string_adapter, indent_char, error_handler);
|
||||||
|
|
||||||
if (indent >= 0)
|
if (indent >= 0)
|
||||||
{
|
{
|
||||||
s.dump(*this, true, ensure_ascii, static_cast<unsigned int>(indent));
|
s.dump(*this, true, ensure_ascii, static_cast<std::size_t>(indent));
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -3516,7 +3517,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
if (merge_objects && it.value().is_object())
|
if (merge_objects && it.value().is_object())
|
||||||
{
|
{
|
||||||
auto it2 = m_data.m_value.object->find(it.key());
|
auto it2 = m_data.m_value.object->find(it.key());
|
||||||
if (it2 != m_data.m_value.object->end())
|
// Only recurse when the existing value is itself an object.
|
||||||
|
// Otherwise overwrite, matching the documented "all other values
|
||||||
|
// are overwritten as usual" behavior (see #5402).
|
||||||
|
if (it2 != m_data.m_value.object->end() && it2->second.is_object())
|
||||||
{
|
{
|
||||||
it2->second.update(it.value(), true);
|
it2->second.update(it.value(), true);
|
||||||
#if JSON_DIAGNOSTICS
|
#if JSON_DIAGNOSTICS
|
||||||
@@ -3652,6 +3656,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
|
|
||||||
// note parentheses around operands are necessary; see
|
// note parentheses around operands are necessary; see
|
||||||
// https://github.com/nlohmann/json/issues/1530
|
// https://github.com/nlohmann/json/issues/1530
|
||||||
|
// Mixed signed/unsigned integer comparisons check whether the signed value
|
||||||
|
// is negative before casting. If it is, the comparison is performed with
|
||||||
|
// the fixed values -1 and 1, which preserves the ordering relationship
|
||||||
|
// because any negative signed value is smaller than any unsigned value.
|
||||||
|
// Otherwise, the non-negative signed value is cast to unsigned before the
|
||||||
|
// comparison to avoid wraparound.
|
||||||
#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result) \
|
#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result) \
|
||||||
const auto lhs_type = lhs.type(); \
|
const auto lhs_type = lhs.type(); \
|
||||||
const auto rhs_type = rhs.type(); \
|
const auto rhs_type = rhs.type(); \
|
||||||
@@ -3710,12 +3720,16 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
} \
|
} \
|
||||||
else if (lhs_type == value_t::number_unsigned && rhs_type == value_t::number_integer) \
|
else if (lhs_type == value_t::number_unsigned && rhs_type == value_t::number_integer) \
|
||||||
{ \
|
{ \
|
||||||
return static_cast<number_integer_t>(lhs.m_data.m_value.number_unsigned) op rhs.m_data.m_value.number_integer; \
|
return (rhs.m_data.m_value.number_integer < 0) \
|
||||||
|
? (number_integer_t(1) op number_integer_t(-1)) \
|
||||||
|
: (lhs.m_data.m_value.number_unsigned op static_cast<number_unsigned_t>(rhs.m_data.m_value.number_integer)); \
|
||||||
} \
|
} \
|
||||||
else if (lhs_type == value_t::number_integer && rhs_type == value_t::number_unsigned) \
|
else if (lhs_type == value_t::number_integer && rhs_type == value_t::number_unsigned) \
|
||||||
{ \
|
{ \
|
||||||
return lhs.m_data.m_value.number_integer op static_cast<number_integer_t>(rhs.m_data.m_value.number_unsigned); \
|
return (lhs.m_data.m_value.number_integer < 0) \
|
||||||
} \
|
? (number_integer_t(-1) op number_integer_t(1)) \
|
||||||
|
: (static_cast<number_unsigned_t>(lhs.m_data.m_value.number_integer) op rhs.m_data.m_value.number_unsigned); \
|
||||||
|
} \
|
||||||
else if(compares_unordered(lhs, rhs))\
|
else if(compares_unordered(lhs, rhs))\
|
||||||
{\
|
{\
|
||||||
return (unordered_result);\
|
return (unordered_result);\
|
||||||
@@ -4042,7 +4056,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
|||||||
o.width(0);
|
o.width(0);
|
||||||
|
|
||||||
// do the actual serialization
|
// do the actual serialization
|
||||||
serializer s(detail::output_adapter<char>(o), o.fill());
|
detail::output_stream_adapter<char> stream_adapter(o);
|
||||||
|
serializer s(stream_adapter, o.fill());
|
||||||
s.dump(j, pretty_print, false, static_cast<unsigned int>(indentation));
|
s.dump(j, pretty_print, false, static_cast<unsigned int>(indentation));
|
||||||
return o;
|
return o;
|
||||||
}
|
}
|
||||||
|
|||||||
+2165
-220
File diff suppressed because it is too large
Load Diff
@@ -2,6 +2,9 @@ cmake_minimum_required(VERSION 3.13...4.0)
|
|||||||
|
|
||||||
option(JSON_Valgrind "Execute test suite with Valgrind." OFF)
|
option(JSON_Valgrind "Execute test suite with Valgrind." OFF)
|
||||||
option(JSON_FastTests "Skip expensive/slow tests." OFF)
|
option(JSON_FastTests "Skip expensive/slow tests." OFF)
|
||||||
|
option(JSON_TestSimdutf "Build the unit tests against the simdutf UTF-8 validation backend." OFF)
|
||||||
|
|
||||||
|
set(JSON_SIMDUTF_VERSION 9.1.0 CACHE STRING "The simdutf version used by JSON_TestSimdutf.")
|
||||||
|
|
||||||
set(JSON_32bitTest AUTO CACHE STRING "Enable the 32bit unit test (ON/OFF/AUTO/ONLY).")
|
set(JSON_32bitTest AUTO CACHE STRING "Enable the 32bit unit test (ON/OFF/AUTO/ONLY).")
|
||||||
set(JSON_TestStandards "" CACHE STRING "The list of standards to test explicitly.")
|
set(JSON_TestStandards "" CACHE STRING "The list of standards to test explicitly.")
|
||||||
@@ -149,6 +152,71 @@ if(test_force)
|
|||||||
endif()
|
endif()
|
||||||
message(STATUS "${msg}")
|
message(STATUS "${msg}")
|
||||||
|
|
||||||
|
#############################################################################
|
||||||
|
# optionally validate UTF-8 with simdutf (JSON_USE_SIMDUTF)
|
||||||
|
#############################################################################
|
||||||
|
|
||||||
|
# The simdutf backend is opt-in and not vendored, so it is fetched here rather
|
||||||
|
# than being a checked-in dependency. Everything below hangs off test_main,
|
||||||
|
# whose usage requirements every test target inherits; the library target and
|
||||||
|
# the installed CMake package are deliberately left untouched.
|
||||||
|
if (JSON_TestSimdutf)
|
||||||
|
# simdutf requires C++17, both to compile itself and to be reachable from
|
||||||
|
# the library, which keeps its scalar validator below that. Find a tested
|
||||||
|
# standard that satisfies it.
|
||||||
|
set(simdutf_standard "")
|
||||||
|
foreach(cxx_standard ${test_cxx_standards})
|
||||||
|
if(NOT cxx_standard LESS 17 AND compiler_supports_cpp_${cxx_standard})
|
||||||
|
set(simdutf_standard ${cxx_standard})
|
||||||
|
break()
|
||||||
|
endif()
|
||||||
|
endforeach()
|
||||||
|
|
||||||
|
if("${simdutf_standard}" STREQUAL "")
|
||||||
|
# Building simdutf would fail outright without a C++17 compiler, and
|
||||||
|
# even with one it would go unused if no C++17-or-later standard is
|
||||||
|
# tested. Say so and fall back to the scalar validator rather than
|
||||||
|
# failing the build.
|
||||||
|
if(NOT compiler_supports_cpp_17)
|
||||||
|
set(simdutf_reason "the compiler does not support C++17")
|
||||||
|
else()
|
||||||
|
set(simdutf_reason "no tested standard is C++17 or later (testing ${msg_standards})")
|
||||||
|
endif()
|
||||||
|
message(WARNING
|
||||||
|
"JSON_TestSimdutf is enabled, but ${simdutf_reason}. simdutf requires C++17, so it "
|
||||||
|
"is not fetched and JSON_USE_SIMDUTF is not defined: the tests run against the "
|
||||||
|
"built-in scalar UTF-8 validator instead. Set JSON_TestStandards to include 17 or "
|
||||||
|
"later, or build with a compiler that supports C++17.")
|
||||||
|
else()
|
||||||
|
if (CMAKE_VERSION VERSION_LESS 3.18)
|
||||||
|
message(FATAL_ERROR "JSON_TestSimdutf requires CMake 3.18 or later (simdutf's minimum).")
|
||||||
|
endif()
|
||||||
|
|
||||||
|
include(FetchContent)
|
||||||
|
|
||||||
|
# simdutf builds its tests and tools by default, and its tests pull
|
||||||
|
# further dependencies of their own; only the library is needed here
|
||||||
|
set(SIMDUTF_TESTS OFF CACHE BOOL "" FORCE)
|
||||||
|
set(SIMDUTF_TOOLS OFF CACHE BOOL "" FORCE)
|
||||||
|
set(SIMDUTF_BENCHMARKS OFF CACHE BOOL "" FORCE)
|
||||||
|
set(SIMDUTF_ICONV OFF CACHE BOOL "" FORCE)
|
||||||
|
|
||||||
|
FetchContent_Declare(simdutf
|
||||||
|
URL https://github.com/simdutf/simdutf/archive/refs/tags/v${JSON_SIMDUTF_VERSION}.tar.gz
|
||||||
|
DOWNLOAD_EXTRACT_TIMESTAMP TRUE
|
||||||
|
)
|
||||||
|
FetchContent_MakeAvailable(simdutf)
|
||||||
|
|
||||||
|
target_compile_definitions(test_main PUBLIC JSON_USE_SIMDUTF)
|
||||||
|
target_link_libraries(test_main PUBLIC simdutf::simdutf)
|
||||||
|
|
||||||
|
# simdutf.h requires C++17; below that the library keeps its scalar
|
||||||
|
# validator, so any C++11/14 test targets exercise the fallback and the
|
||||||
|
# C++17-and-later ones exercise simdutf. Both must agree.
|
||||||
|
message(STATUS "UTF-8 validation delegated to simdutf ${JSON_SIMDUTF_VERSION} for C++17 and later (JSON_USE_SIMDUTF)")
|
||||||
|
endif()
|
||||||
|
endif()
|
||||||
|
|
||||||
# *DO* use json_test_set_test_options() above this line
|
# *DO* use json_test_set_test_options() above this line
|
||||||
|
|
||||||
json_test_should_build_32bit_test(json_32bit_test json_32bit_test_only "${JSON_32bitTest}")
|
json_test_should_build_32bit_test(json_32bit_test json_32bit_test_only "${JSON_32bitTest}")
|
||||||
|
|||||||
@@ -81,6 +81,44 @@ BENCHMARK_CAPTURE(ParseString, signed_ints, TEST_DATA_DIRECTORY "/regressi
|
|||||||
BENCHMARK_CAPTURE(ParseString, unsigned_ints, TEST_DATA_DIRECTORY "/regression/unsigned_ints.json");
|
BENCHMARK_CAPTURE(ParseString, unsigned_ints, TEST_DATA_DIRECTORY "/regression/unsigned_ints.json");
|
||||||
BENCHMARK_CAPTURE(ParseString, small_signed_ints, TEST_DATA_DIRECTORY "/regression/small_signed_ints.json");
|
BENCHMARK_CAPTURE(ParseString, small_signed_ints, TEST_DATA_DIRECTORY "/regression/small_signed_ints.json");
|
||||||
|
|
||||||
|
//////////////////////////////////////////////////////////////////////////////
|
||||||
|
// parse pretty-printed JSON from string
|
||||||
|
//
|
||||||
|
// Every file in the corpus above is minified or only lightly spaced, so none of
|
||||||
|
// them exercise the lexer's whitespace handling. Real-world JSON is frequently
|
||||||
|
// indented - configuration files, pretty-printed API responses, anything kept
|
||||||
|
// under version control - where insignificant whitespace can outweigh the data.
|
||||||
|
// Re-serializing a document with an indentation and parsing that keeps the
|
||||||
|
// content identical to the ParseString row above, so the pair isolates the cost
|
||||||
|
// of the whitespace alone.
|
||||||
|
//////////////////////////////////////////////////////////////////////////////
|
||||||
|
|
||||||
|
static void ParseIndented(benchmark::State& state, const char* filename, int indent)
|
||||||
|
{
|
||||||
|
std::ifstream f(filename);
|
||||||
|
std::string str((std::istreambuf_iterator<char>(f)), std::istreambuf_iterator<char>());
|
||||||
|
const std::string indented = json::parse(str).dump(indent);
|
||||||
|
|
||||||
|
while (state.KeepRunning())
|
||||||
|
{
|
||||||
|
state.PauseTiming();
|
||||||
|
auto* j = new json();
|
||||||
|
state.ResumeTiming();
|
||||||
|
|
||||||
|
*j = json::parse(indented);
|
||||||
|
|
||||||
|
state.PauseTiming();
|
||||||
|
delete j;
|
||||||
|
state.ResumeTiming();
|
||||||
|
}
|
||||||
|
|
||||||
|
state.SetBytesProcessed(state.iterations() * indented.size());
|
||||||
|
}
|
||||||
|
BENCHMARK_CAPTURE(ParseIndented, jeopardy / 4, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", 4);
|
||||||
|
BENCHMARK_CAPTURE(ParseIndented, canada / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", 4);
|
||||||
|
BENCHMARK_CAPTURE(ParseIndented, citm_catalog / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", 4);
|
||||||
|
BENCHMARK_CAPTURE(ParseIndented, twitter / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", 4);
|
||||||
|
|
||||||
//////////////////////////////////////////////////////////////////////////////
|
//////////////////////////////////////////////////////////////////////////////
|
||||||
// serialize JSON
|
// serialize JSON
|
||||||
//////////////////////////////////////////////////////////////////////////////
|
//////////////////////////////////////////////////////////////////////////////
|
||||||
|
|||||||
@@ -132,3 +132,37 @@ TEST_CASE("BJData")
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("CBOR")
|
||||||
|
{
|
||||||
|
SECTION("parse errors")
|
||||||
|
{
|
||||||
|
SECTION("array/map size larger than std::size_t")
|
||||||
|
{
|
||||||
|
// declared lengths do not fit in a 32-bit std::size_t and must not be truncated
|
||||||
|
std::vector<uint8_t> const varr = {0x9B, 0x00, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00, 0x05};
|
||||||
|
std::vector<uint8_t> const vmap = {0xBB, 0x00, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00, 0x05};
|
||||||
|
|
||||||
|
json _;
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(varr), "[json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive array size", json::out_of_range&);
|
||||||
|
CHECK(json::from_cbor(varr, true, false).is_discarded());
|
||||||
|
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vmap), "[json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive map size", json::out_of_range&);
|
||||||
|
CHECK(json::from_cbor(vmap, true, false).is_discarded());
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("array/map size equal to the indefinite-length sentinel")
|
||||||
|
{
|
||||||
|
// on 32-bit platforms a four-byte length of 0xFFFFFFFF aliases unknown_size()
|
||||||
|
std::vector<uint8_t> const varr = {0x9A, 0xFF, 0xFF, 0xFF, 0xFF};
|
||||||
|
std::vector<uint8_t> const vmap = {0xBA, 0xFF, 0xFF, 0xFF, 0xFF};
|
||||||
|
|
||||||
|
json _;
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(varr), "[json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive array size", json::out_of_range&);
|
||||||
|
CHECK(json::from_cbor(varr, true, false).is_discarded());
|
||||||
|
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(vmap), "[json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive map size", json::out_of_range&);
|
||||||
|
CHECK(json::from_cbor(vmap, true, false).is_discarded());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -1347,6 +1347,8 @@ TEST_CASE("BJData")
|
|||||||
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vec2), "[json.exception.parse_error.115] parse error at byte 5: syntax error while parsing BJData high-precision number: invalid number text: 1A", json::parse_error);
|
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vec2), "[json.exception.parse_error.115] parse error at byte 5: syntax error while parsing BJData high-precision number: invalid number text: 1A", json::parse_error);
|
||||||
std::vector<uint8_t> const vec3 = {'H', 'i', 2, '1', '.'};
|
std::vector<uint8_t> const vec3 = {'H', 'i', 2, '1', '.'};
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vec3), "[json.exception.parse_error.115] parse error at byte 5: syntax error while parsing BJData high-precision number: invalid number text: 1.", json::parse_error);
|
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vec3), "[json.exception.parse_error.115] parse error at byte 5: syntax error while parsing BJData high-precision number: invalid number text: 1.", json::parse_error);
|
||||||
|
std::vector<uint8_t> const vec_overflow = {'H', 'i', 5, '1', 'e', '4', '0', '0'};
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vec_overflow), "[json.exception.out_of_range.406] number overflow parsing '1e400'", json::out_of_range);
|
||||||
std::vector<uint8_t> const vec4 = {'H', 2, '1', '0'};
|
std::vector<uint8_t> const vec4 = {'H', 2, '1', '0'};
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vec4), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing BJData size: expected length type specification (U, i, u, I, m, l, M, L) after '#'; last byte: 0x02", json::parse_error);
|
CHECK_THROWS_WITH_AS(_ = json::from_bjdata(vec4), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing BJData size: expected length type specification (U, i, u, I, m, l, M, L) after '#'; last byte: 0x02", json::parse_error);
|
||||||
}
|
}
|
||||||
@@ -2728,6 +2730,27 @@ TEST_CASE("BJData")
|
|||||||
CHECK(json::from_bjdata(json::to_bjdata(j_type), true, true) == j_type);
|
CHECK(json::from_bjdata(json::to_bjdata(j_type), true, true) == j_type);
|
||||||
CHECK(json::from_bjdata(json::to_bjdata(j_size), true, true) == j_size);
|
CHECK(json::from_bjdata(json::to_bjdata(j_size), true, true) == j_size);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SECTION("ndarray whose dimensions overflow stays as object")
|
||||||
|
{
|
||||||
|
// the product of the dimensions wraps around std::size_t to 0
|
||||||
|
// and so matches the size of the empty _ArrayData_; writing this
|
||||||
|
// as an ndarray would announce an element count no reader can
|
||||||
|
// honor, so it has to stay a plain object
|
||||||
|
json j_overflow = json({{"_ArrayData_", json::array()}, {"_ArraySize_", {9223372036854775808ull, 2}}, {"_ArrayType_", "uint8"}});
|
||||||
|
CHECK(json::from_bjdata(json::to_bjdata(j_overflow), true, true) == j_overflow);
|
||||||
|
|
||||||
|
// a single dimension that does not fit into std::size_t is
|
||||||
|
// rejected for the same reason (only observable where
|
||||||
|
// std::size_t is narrower than 64 bit)
|
||||||
|
json j_huge = json({{"_ArrayData_", json::array()}, {"_ArraySize_", {18446744073709551615ull}}, {"_ArrayType_", "uint8"}});
|
||||||
|
CHECK(json::from_bjdata(json::to_bjdata(j_huge), true, true) == j_huge);
|
||||||
|
|
||||||
|
// a well-formed ndarray is still encoded as one
|
||||||
|
json j_ok = json({{"_ArrayData_", {1, 2, 3, 4, 5, 6}}, {"_ArraySize_", {2, 3}}, {"_ArrayType_", "uint8"}});
|
||||||
|
CHECK(json::to_bjdata(j_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, 'i', 3, ']', 1, 2, 3, 4, 5, 6}));
|
||||||
|
CHECK(json::from_bjdata(json::to_bjdata(j_ok), true, true) == j_ok);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -3820,6 +3843,48 @@ TEST_CASE("all BJData first bytes")
|
|||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
TEST_CASE("BJData use_type requires use_size")
|
||||||
|
{
|
||||||
|
SECTION("non-empty object throws other_error.502")
|
||||||
|
{
|
||||||
|
const json j = {{"a", 1}, {"b", 2}};
|
||||||
|
CHECK_THROWS_WITH_AS(json::to_bjdata(j, false, true),
|
||||||
|
"[json.exception.other_error.502] use_type requires use_size = true",
|
||||||
|
json::other_error&);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("non-empty array throws other_error.502")
|
||||||
|
{
|
||||||
|
const json j = {1, 2, 3};
|
||||||
|
CHECK_THROWS_WITH_AS(json::to_bjdata(j, false, true),
|
||||||
|
"[json.exception.other_error.502] use_type requires use_size = true",
|
||||||
|
json::other_error&);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("scalars do not throw with use_type=true, use_count=false")
|
||||||
|
{
|
||||||
|
CHECK_NOTHROW(json::to_bjdata(42, false, true));
|
||||||
|
CHECK_NOTHROW(json::to_bjdata(3.14, false, true));
|
||||||
|
CHECK_NOTHROW(json::to_bjdata("hello", false, true));
|
||||||
|
CHECK_NOTHROW(json::to_bjdata(true, false, true));
|
||||||
|
CHECK_NOTHROW(json::to_bjdata(nullptr, false, true));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("empty containers do not throw with use_type=true, use_count=false")
|
||||||
|
{
|
||||||
|
CHECK_NOTHROW(json::to_bjdata(json::array(), false, true));
|
||||||
|
CHECK_NOTHROW(json::to_bjdata(json::object(), false, true));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("valid combinations on non-empty containers")
|
||||||
|
{
|
||||||
|
const json j = {{"a", 1}, {"b", 2}};
|
||||||
|
CHECK_NOTHROW(json::to_bjdata(j, false, false));
|
||||||
|
CHECK_NOTHROW(json::to_bjdata(j, true, false));
|
||||||
|
CHECK_NOTHROW(json::to_bjdata(j, true, true));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("BJData roundtrips" * doctest::skip())
|
TEST_CASE("BJData roundtrips" * doctest::skip())
|
||||||
{
|
{
|
||||||
SECTION("input from self-generated BJData files")
|
SECTION("input from self-generated BJData files")
|
||||||
|
|||||||
+49
-2
@@ -1999,6 +1999,42 @@ TEST_CASE("CBOR regressions")
|
|||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
TEST_CASE("CBOR definite length equal to the indefinite-length sentinel")
|
||||||
|
{
|
||||||
|
// A definite-length array or map whose declared element count equals the
|
||||||
|
// reserved unknown_size() sentinel (SIZE_MAX) must be rejected. Otherwise
|
||||||
|
// it is read as an indefinite-length container and the following bytes are
|
||||||
|
// silently accepted instead of the (impossible) count being reported.
|
||||||
|
json _;
|
||||||
|
|
||||||
|
SECTION("array")
|
||||||
|
{
|
||||||
|
// 0x9B: array with eight-byte length; length = 0xFFFFFFFFFFFFFFFF
|
||||||
|
const std::vector<uint8_t> input = {0x9B, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0x01, 0x02, 0xFF};
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive array size", json::out_of_range&);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("map")
|
||||||
|
{
|
||||||
|
// 0xBB: map with eight-byte length; length = 0xFFFFFFFFFFFFFFFF
|
||||||
|
const std::vector<uint8_t> input = {0xBB, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0x61, 0x61, 0x01, 0xFF};
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(input), "[json.exception.out_of_range.408] syntax error while parsing CBOR size: excessive map size", json::out_of_range&);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("indefinite-length containers are unaffected")
|
||||||
|
{
|
||||||
|
CHECK(json::from_cbor(std::vector<uint8_t>({0x9F, 0x01, 0x02, 0xFF})) == json({1, 2}));
|
||||||
|
CHECK(json::from_cbor(std::vector<uint8_t>({0xBF, 0x61, 0x61, 0x01, 0xFF})) == json({{"a", 1}}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("ordinary four-byte length containers are unaffected")
|
||||||
|
{
|
||||||
|
// 0x9A/0xBA carry a four-byte length; a normal count still parses
|
||||||
|
CHECK(json::from_cbor(std::vector<uint8_t>({0x9A, 0x00, 0x00, 0x00, 0x02, 0x01, 0x02})) == json({1, 2}));
|
||||||
|
CHECK(json::from_cbor(std::vector<uint8_t>({0xBA, 0x00, 0x00, 0x00, 0x01, 0x61, 0x61, 0x01})) == json({{"a", 1}}));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("CBOR roundtrips" * doctest::skip())
|
TEST_CASE("CBOR roundtrips" * doctest::skip())
|
||||||
{
|
{
|
||||||
SECTION("input from flynn")
|
SECTION("input from flynn")
|
||||||
@@ -2529,11 +2565,16 @@ TEST_CASE("Tagged values")
|
|||||||
const json j = "s";
|
const json j = "s";
|
||||||
auto v = json::to_cbor(j);
|
auto v = json::to_cbor(j);
|
||||||
|
|
||||||
SECTION("0xC6..0xD4")
|
const json j_bin_payload = json::binary(std::vector<std::uint8_t> {0x01, 0x02, 0x03});
|
||||||
|
auto v_bin_payload = json::to_cbor(j_bin_payload);
|
||||||
|
|
||||||
|
SECTION("0xC0..0xD7")
|
||||||
{
|
{
|
||||||
for (const auto b : std::vector<std::uint8_t>
|
for (const auto b : std::vector<std::uint8_t>
|
||||||
{
|
{
|
||||||
0xC6, 0xC7, 0xC8, 0xC9, 0xCA, 0xCB, 0xCC, 0xCD, 0xCE, 0xCF, 0xD0, 0xD1, 0xD2, 0xD3, 0xD4
|
0xC0, 0xC1, 0xC2, 0xC3, 0xC4, 0xC5,
|
||||||
|
0xC6, 0xC7, 0xC8, 0xC9, 0xCA, 0xCB, 0xCC, 0xCD, 0xCE, 0xCF, 0xD0, 0xD1, 0xD2, 0xD3, 0xD4,
|
||||||
|
0xD5, 0xD6, 0xD7
|
||||||
})
|
})
|
||||||
{
|
{
|
||||||
CAPTURE(b);
|
CAPTURE(b);
|
||||||
@@ -2553,6 +2594,12 @@ TEST_CASE("Tagged values")
|
|||||||
|
|
||||||
auto j_tagged_stored = json::from_cbor(v_tagged, true, true, json::cbor_tag_handler_t::store);
|
auto j_tagged_stored = json::from_cbor(v_tagged, true, true, json::cbor_tag_handler_t::store);
|
||||||
CHECK(j_tagged_stored == j);
|
CHECK(j_tagged_stored == j);
|
||||||
|
|
||||||
|
auto v_binary_tagged = v_bin_payload;
|
||||||
|
v_binary_tagged.insert(v_binary_tagged.begin(), b);
|
||||||
|
auto j_binary_tagged_stored = json::from_cbor(v_binary_tagged, true, true, json::cbor_tag_handler_t::store);
|
||||||
|
CHECK(j_binary_tagged_stored == j_bin_payload);
|
||||||
|
CHECK(!j_binary_tagged_stored.get_binary().has_subtype());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -12,6 +12,11 @@
|
|||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
using nlohmann::json;
|
using nlohmann::json;
|
||||||
|
|
||||||
|
#include <cstdlib> // strtod
|
||||||
|
#include <sstream> // stringstream
|
||||||
|
#include <string> // string
|
||||||
|
#include <vector> // vector
|
||||||
|
|
||||||
namespace
|
namespace
|
||||||
{
|
{
|
||||||
// shortcut to scan a string literal
|
// shortcut to scan a string literal
|
||||||
@@ -224,3 +229,431 @@ TEST_CASE("lexer class")
|
|||||||
CHECK((scan_string("/**//**//**/", true) == json::lexer::token_type::end_of_input));
|
CHECK((scan_string("/**//**//**/", true) == json::lexer::token_type::end_of_input));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("lexer number fast path")
|
||||||
|
{
|
||||||
|
// The contiguous fast path (used for pointer/string input) must agree with
|
||||||
|
// the streaming byte path (used for std::istream) on token type, numeric
|
||||||
|
// value, and round-trip text for every well-formed number, and reject the
|
||||||
|
// same malformed numbers with the same message.
|
||||||
|
SECTION("contiguous vs streaming parity")
|
||||||
|
{
|
||||||
|
const std::vector<std::string> numbers =
|
||||||
|
{
|
||||||
|
"0", "-0", "1", "-1", "42", "-42", "10", "100", "1234567890",
|
||||||
|
"0.0", "-0.0", "3.14", "-3.14", "0.5", "-0.001", "123.456789",
|
||||||
|
"1e0", "1E0", "1e10", "1e-10", "1e+10", "1.5e3", "-2.5E-4",
|
||||||
|
"9223372036854775807", // INT64_MAX -> unsigned
|
||||||
|
"9223372036854775808", // INT64_MAX + 1 -> unsigned
|
||||||
|
"18446744073709551615", // UINT64_MAX -> unsigned
|
||||||
|
"18446744073709551616", // UINT64_MAX + 1 -> float
|
||||||
|
"-9223372036854775808", // INT64_MIN -> integer
|
||||||
|
"-9223372036854775809", // INT64_MIN - 1 -> float
|
||||||
|
"123456789012345678901234567890", // huge -> float
|
||||||
|
"0.30000000000000004", "2.2250738585072014e-308", "1e308",
|
||||||
|
// high-precision / wide-exponent values that exercise the
|
||||||
|
// std::from_chars (Eisel-Lemire) path beyond the Clinger subset
|
||||||
|
"1.7976931348623157e308", "1.2345678901234567e-250",
|
||||||
|
"9007199254740993", "5e-324", "1e-320"
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const auto& n : numbers)
|
||||||
|
{
|
||||||
|
const std::string doc = "[" + n + "]";
|
||||||
|
|
||||||
|
// contiguous fast path
|
||||||
|
const json a = json::parse(doc);
|
||||||
|
// streaming byte path
|
||||||
|
std::stringstream ss(doc);
|
||||||
|
const json b = json::parse(ss);
|
||||||
|
|
||||||
|
CAPTURE(n);
|
||||||
|
CHECK(a == b);
|
||||||
|
CHECK(a.dump() == b.dump());
|
||||||
|
CHECK(a[0].type() == b[0].type());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("significant-digit gate for the Clinger fast path")
|
||||||
|
{
|
||||||
|
// Clinger's fast path needs a significand below 2^53, so it cannot
|
||||||
|
// succeed once the mantissa has 17 or more significant digits (the
|
||||||
|
// significand would be at least 10^16). The lexer skips the attempt
|
||||||
|
// there. That is only allowed to save work: every value must still come
|
||||||
|
// out bit-exactly, and both scanners must agree. In particular the gate
|
||||||
|
// must not fire for tokens whose leading zeros merely look like extra
|
||||||
|
// digits - "0.1234567890123456" has 16 significant digits, not 17.
|
||||||
|
const std::vector<std::string> numbers =
|
||||||
|
{
|
||||||
|
"1234567890123456", // 16 significant digits
|
||||||
|
"12345678901234567", // 17 -> attempt skipped
|
||||||
|
"123456789012345678", // 18 -> attempt skipped
|
||||||
|
"0.1234567890123456", // 16: the leading "0" is not significant
|
||||||
|
"0.12345678901234567", // 17
|
||||||
|
"0.00000000000000001", // 1, in a long token
|
||||||
|
"0.000000000000000012345678901234", // 14, in a long token
|
||||||
|
"-0.0000000000000000000001", // 1, negative
|
||||||
|
"1.0000000000000000", // 17: trailing zeros are significant here
|
||||||
|
"10000000000000000", // 17
|
||||||
|
"9007199254740992", // 2^53
|
||||||
|
"9007199254740993", // 2^53 + 1
|
||||||
|
"-65.613616999999977", // canada.json shape
|
||||||
|
"1.2345678901234567e-250", // 17 with an exponent
|
||||||
|
"1.234567890123456e-250", // 16 with an exponent
|
||||||
|
"1e10", "0.0", "-0.0", "0e0", "0.000123"
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const auto& n : numbers)
|
||||||
|
{
|
||||||
|
CAPTURE(n);
|
||||||
|
const std::string doc = "[" + n + "]";
|
||||||
|
|
||||||
|
const json a = json::parse(doc); // contiguous fast path
|
||||||
|
std::stringstream ss(doc);
|
||||||
|
const json b = json::parse(ss); // streaming byte path
|
||||||
|
|
||||||
|
CHECK(a[0].type() == b[0].type());
|
||||||
|
CHECK(a == b);
|
||||||
|
|
||||||
|
if (a[0].is_number_float())
|
||||||
|
{
|
||||||
|
const double expected = std::strtod(n.c_str(), nullptr);
|
||||||
|
CHECK(a[0].get<double>() == expected);
|
||||||
|
CHECK(b[0].get<double>() == expected);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("token type classification")
|
||||||
|
{
|
||||||
|
CHECK((scan_string("0") == json::lexer::token_type::value_unsigned));
|
||||||
|
CHECK((scan_string("-1") == json::lexer::token_type::value_integer));
|
||||||
|
CHECK((scan_string("1.5") == json::lexer::token_type::value_float));
|
||||||
|
CHECK((scan_string("1e5") == json::lexer::token_type::value_float));
|
||||||
|
CHECK((scan_string("18446744073709551615") == json::lexer::token_type::value_unsigned));
|
||||||
|
CHECK((scan_string("18446744073709551616") == json::lexer::token_type::value_float));
|
||||||
|
CHECK((scan_string("-9223372036854775808") == json::lexer::token_type::value_integer));
|
||||||
|
CHECK((scan_string("-9223372036854775809") == json::lexer::token_type::value_float));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("malformed numbers are rejected identically")
|
||||||
|
{
|
||||||
|
for (const char* bad :
|
||||||
|
{"-", "1.", "1e", "1e+", "1.2e", "01", "-01", "1..2", "1.2.3"
|
||||||
|
})
|
||||||
|
{
|
||||||
|
CAPTURE(bad);
|
||||||
|
// the contiguous fast path must decline and let the byte path report
|
||||||
|
const std::string doc = std::string("[") + bad + "]";
|
||||||
|
CHECK_FALSE(json::accept(doc));
|
||||||
|
std::stringstream ss(doc);
|
||||||
|
CHECK_FALSE(json::accept(ss));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#if !defined(JSON_NOEXCEPTION)
|
||||||
|
// these sections parse invalid input, which aborts when exceptions are off
|
||||||
|
SECTION("exhaustive grammar parity with the streaming path")
|
||||||
|
{
|
||||||
|
// The JSON number grammar is encoded twice: once as the scan_number()
|
||||||
|
// state machine and once as the contiguous fast path. Enumerate every
|
||||||
|
// short string over the number alphabet and require the two encodings to
|
||||||
|
// agree exactly - on acceptance, on the reported error, and on the parsed
|
||||||
|
// value - so they cannot drift apart.
|
||||||
|
const std::string alphabet = "01.eE+-";
|
||||||
|
|
||||||
|
// full outcome of parsing @a doc, so a mismatch in type, value, or error
|
||||||
|
// message is caught, not just a mismatch in acceptance
|
||||||
|
const auto outcome = [](const std::string & doc, bool streaming) -> std::string
|
||||||
|
{
|
||||||
|
try
|
||||||
|
{
|
||||||
|
if (streaming)
|
||||||
|
{
|
||||||
|
std::stringstream ss(doc);
|
||||||
|
const json j = json::parse(ss);
|
||||||
|
return std::string(j[0].type_name()) + '|' + j.dump();
|
||||||
|
}
|
||||||
|
const json j = json::parse(doc);
|
||||||
|
return std::string(j[0].type_name()) + '|' + j.dump();
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
return {e.what()};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
std::vector<std::string> mismatches;
|
||||||
|
std::vector<std::string> tokens{""};
|
||||||
|
for (std::size_t length = 1; length <= 4; ++length)
|
||||||
|
{
|
||||||
|
std::vector<std::string> next;
|
||||||
|
next.reserve(tokens.size() * alphabet.size());
|
||||||
|
for (const auto& prefix : tokens)
|
||||||
|
{
|
||||||
|
for (const char c : alphabet)
|
||||||
|
{
|
||||||
|
next.push_back(prefix + c);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
tokens = next;
|
||||||
|
|
||||||
|
for (const auto& token : tokens)
|
||||||
|
{
|
||||||
|
const std::string doc = "[" + token + "]";
|
||||||
|
if (outcome(doc, false) != outcome(doc, true))
|
||||||
|
{
|
||||||
|
mismatches.push_back(doc);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// 7 + 49 + 343 + 2401 tokens
|
||||||
|
CHECK(tokens.size() == 2401);
|
||||||
|
CAPTURE(mismatches);
|
||||||
|
CHECK(mismatches.empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("error positions match the streaming path")
|
||||||
|
{
|
||||||
|
// Rejecting identically is not enough: the fast path must also report the
|
||||||
|
// error at the same position as the byte path. A number directly followed
|
||||||
|
// by a newline is the interesting case, because the byte path reaches the
|
||||||
|
// newline (which resets the column) and then ungets it.
|
||||||
|
// returns the parse_error message, or "" if the document parsed
|
||||||
|
const auto contiguous_error = [](const std::string & doc) -> std::string
|
||||||
|
{
|
||||||
|
try
|
||||||
|
{
|
||||||
|
const json j = json::parse(doc);
|
||||||
|
static_cast<void>(j);
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
return {e.what()};
|
||||||
|
}
|
||||||
|
return {};
|
||||||
|
};
|
||||||
|
const auto streaming_error = [](const std::string & doc) -> std::string
|
||||||
|
{
|
||||||
|
try
|
||||||
|
{
|
||||||
|
std::stringstream ss(doc);
|
||||||
|
const json j = json::parse(ss);
|
||||||
|
static_cast<void>(j);
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
return {e.what()};
|
||||||
|
}
|
||||||
|
return {};
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const char* bad :
|
||||||
|
{"[01\n]", "[00\n]", "[-01\n]", "{1\n}", "[1\n2]", "[1.2.3\n]",
|
||||||
|
"[1 \n2]", "[\n1\n2]", "1\n2", "[01\r\n]", "[1e\n]", "[-\n]"
|
||||||
|
})
|
||||||
|
{
|
||||||
|
CAPTURE(bad);
|
||||||
|
const std::string doc = bad;
|
||||||
|
const std::string contiguous_what = contiguous_error(doc);
|
||||||
|
|
||||||
|
CHECK_FALSE(contiguous_what.empty());
|
||||||
|
CHECK(contiguous_what == streaming_error(doc));
|
||||||
|
}
|
||||||
|
|
||||||
|
// A number terminated by a newline must report the same position as the
|
||||||
|
// same number terminated by anything else: scan_number() reads the
|
||||||
|
// terminator and ungets it, so the reported column is the one reached
|
||||||
|
// after the number's last character - not the 0 that an unget() across
|
||||||
|
// the newline used to leave behind.
|
||||||
|
CHECK(contiguous_error("[01\n]") == contiguous_error("[01 ]"));
|
||||||
|
CHECK(contiguous_error("[01\n]") ==
|
||||||
|
"[json.exception.parse_error.101] parse error at line 1, column 3: "
|
||||||
|
"syntax error while parsing array - unexpected number literal; expected ']'");
|
||||||
|
|
||||||
|
// the same for a multi-character token, where the column of the last
|
||||||
|
// character (the '3' of "-2.5e3") differs from the column it starts at
|
||||||
|
CHECK(contiguous_error("null -2.5e3\nfalse") == contiguous_error("null -2.5e3 false"));
|
||||||
|
CHECK(contiguous_error("null -2.5e3\nfalse") ==
|
||||||
|
"[json.exception.parse_error.101] parse error at line 1, column 11: "
|
||||||
|
"syntax error while parsing value - unexpected number literal; expected end of input");
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_CASE("lexer string fast path")
|
||||||
|
{
|
||||||
|
// Build a byte string from explicit values: a hex escape in a string
|
||||||
|
// literal swallows every following hex digit, which makes sequences like
|
||||||
|
// "\xC3\xA9b" mean something other than they look like.
|
||||||
|
const auto bytes = [](std::initializer_list<int> values)
|
||||||
|
{
|
||||||
|
std::string result;
|
||||||
|
for (const int value : values)
|
||||||
|
{
|
||||||
|
result.push_back(static_cast<char>(value));
|
||||||
|
}
|
||||||
|
return result;
|
||||||
|
};
|
||||||
|
|
||||||
|
#if !defined(JSON_NOEXCEPTION)
|
||||||
|
// the full outcome of parsing @a doc: the parsed value, or the exact error
|
||||||
|
// message, so a mismatch in either is caught. Only usable with exceptions
|
||||||
|
// on: parsing invalid input aborts when they are off.
|
||||||
|
const auto outcome = [](const std::string & doc, bool streaming) -> std::string
|
||||||
|
{
|
||||||
|
try
|
||||||
|
{
|
||||||
|
if (streaming)
|
||||||
|
{
|
||||||
|
std::stringstream ss(doc);
|
||||||
|
const json j = json::parse(ss);
|
||||||
|
return j.dump();
|
||||||
|
}
|
||||||
|
const json j = json::parse(doc);
|
||||||
|
return j.dump();
|
||||||
|
}
|
||||||
|
// not just parse_error: if a bulk scanner ever let ill-formed UTF-8
|
||||||
|
// through, dump() would throw type_error.316, and that has to surface
|
||||||
|
// as a reported mismatch rather than as an uncaught exception
|
||||||
|
catch (const json::exception& e)
|
||||||
|
{
|
||||||
|
return {e.what()};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// once at the start of the string, once past the first 8-byte SWAR word, so
|
||||||
|
// the bulk scanner sees each case with and without a run behind it
|
||||||
|
const std::vector<std::size_t> offsets{0, 9};
|
||||||
|
|
||||||
|
#if !defined(JSON_NOEXCEPTION)
|
||||||
|
SECTION("exhaustive contiguous vs streaming parity")
|
||||||
|
{
|
||||||
|
// ordinary ASCII, both specials, a control byte, characters that make
|
||||||
|
// the preceding backslash a valid escape, a UTF-8 lead byte of each
|
||||||
|
// length, a continuation byte, and a byte that is never valid
|
||||||
|
const std::vector<std::string> alphabet =
|
||||||
|
{
|
||||||
|
"a", "\"", "\\", "n", "u", "0", bytes({0x01}),
|
||||||
|
bytes({0xC3}), bytes({0xA9}), bytes({0xE4}), bytes({0xF0}),
|
||||||
|
bytes({0x80}), bytes({0xFF})
|
||||||
|
};
|
||||||
|
|
||||||
|
std::vector<std::string> mismatches;
|
||||||
|
std::vector<std::string> tokens{""};
|
||||||
|
for (std::size_t length = 1; length <= 3; ++length)
|
||||||
|
{
|
||||||
|
std::vector<std::string> next;
|
||||||
|
next.reserve(tokens.size() * alphabet.size());
|
||||||
|
for (const auto& prefix : tokens)
|
||||||
|
{
|
||||||
|
for (const auto& symbol : alphabet)
|
||||||
|
{
|
||||||
|
next.push_back(prefix + symbol);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
tokens = next;
|
||||||
|
|
||||||
|
for (const auto& token : tokens)
|
||||||
|
{
|
||||||
|
for (const std::size_t offset : offsets)
|
||||||
|
{
|
||||||
|
const std::string doc = "[\"" + std::string(offset, 'a') + token + "\"]";
|
||||||
|
if (outcome(doc, false) != outcome(doc, true))
|
||||||
|
{
|
||||||
|
mismatches.push_back(doc);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// 13 + 169 + 2197 tokens, each at two offsets
|
||||||
|
CHECK(tokens.size() == 2197);
|
||||||
|
CAPTURE(mismatches);
|
||||||
|
CHECK(mismatches.empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("special bytes at every offset of the SWAR stride")
|
||||||
|
{
|
||||||
|
// The bulk scanner consumes 8 bytes at a time and then a tail; place
|
||||||
|
// every kind of byte that ends a run at each offset across two words,
|
||||||
|
// so multibyte sequences also straddle the word boundary.
|
||||||
|
const std::vector<std::string> specials =
|
||||||
|
{
|
||||||
|
"\"", "\\", bytes({0x01}), bytes({0x1F}), bytes({0x7F}),
|
||||||
|
bytes({0xC3, 0xA9}), bytes({0xE4, 0xB8, 0xAD}), bytes({0xF0, 0x9F, 0x98, 0x80}),
|
||||||
|
bytes({0xFF}), bytes({0xC3}), bytes({0xE4, 0xB8})
|
||||||
|
};
|
||||||
|
|
||||||
|
std::vector<std::string> mismatches;
|
||||||
|
for (std::size_t offset = 0; offset <= 17; ++offset)
|
||||||
|
{
|
||||||
|
for (const auto& special : specials)
|
||||||
|
{
|
||||||
|
const std::string doc = "[\"" + std::string(offset, 'a') + special + "\"]";
|
||||||
|
if (outcome(doc, false) != outcome(doc, true))
|
||||||
|
{
|
||||||
|
mismatches.push_back(doc);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
CAPTURE(mismatches);
|
||||||
|
CHECK(mismatches.empty());
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// json::accept() never throws, so the ranges stay covered without exceptions
|
||||||
|
SECTION("UTF-8 ranges are accepted and rejected as documented")
|
||||||
|
{
|
||||||
|
// The bulk validator must accept exactly what the byte-at-a-time
|
||||||
|
// scanner accepts, so pin the boundaries of every range it recognizes.
|
||||||
|
// aggregate, only ever brace-initialized below; default member
|
||||||
|
// initializers would stop it being an aggregate in C++11
|
||||||
|
struct utf8_case // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init)
|
||||||
|
{
|
||||||
|
std::string sequence;
|
||||||
|
bool valid;
|
||||||
|
const char* description;
|
||||||
|
};
|
||||||
|
const std::vector<utf8_case> cases =
|
||||||
|
{
|
||||||
|
{bytes({0xC2, 0x80}), true, "U+0080, shortest two-byte"},
|
||||||
|
{bytes({0xDF, 0xBF}), true, "U+07FF, longest two-byte"},
|
||||||
|
{bytes({0xC1, 0xBF}), false, "overlong two-byte"},
|
||||||
|
{bytes({0xC2, 0x7F}), false, "two-byte with bad continuation"},
|
||||||
|
{bytes({0xE0, 0xA0, 0x80}), true, "U+0800, shortest three-byte"},
|
||||||
|
{bytes({0xE0, 0x9F, 0xBF}), false, "overlong three-byte"},
|
||||||
|
{bytes({0xED, 0x9F, 0xBF}), true, "U+D7FF, just below the surrogates"},
|
||||||
|
{bytes({0xED, 0xA0, 0x80}), false, "surrogate U+D800"},
|
||||||
|
{bytes({0xED, 0xBF, 0xBF}), false, "surrogate U+DFFF"},
|
||||||
|
{bytes({0xEE, 0x80, 0x80}), true, "U+E000, just above the surrogates"},
|
||||||
|
{bytes({0xEF, 0xBF, 0xBF}), true, "U+FFFF"},
|
||||||
|
{bytes({0xF0, 0x90, 0x80, 0x80}), true, "U+10000, shortest four-byte"},
|
||||||
|
{bytes({0xF0, 0x8F, 0xBF, 0xBF}), false, "overlong four-byte"},
|
||||||
|
{bytes({0xF4, 0x8F, 0xBF, 0xBF}), true, "U+10FFFF, highest code point"},
|
||||||
|
{bytes({0xF4, 0x90, 0x80, 0x80}), false, "above U+10FFFF"},
|
||||||
|
{bytes({0xF5, 0x80, 0x80, 0x80}), false, "lead byte out of range"},
|
||||||
|
{bytes({0x80}), false, "bare continuation byte"},
|
||||||
|
{bytes({0xFF}), false, "byte that never appears in UTF-8"},
|
||||||
|
{bytes({0xC3}), false, "truncated two-byte"},
|
||||||
|
{bytes({0xE4, 0xB8}), false, "truncated three-byte"},
|
||||||
|
{bytes({0xF0, 0x9F, 0x98}), false, "truncated four-byte"}
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const auto& test_case : cases)
|
||||||
|
{
|
||||||
|
CAPTURE(test_case.description);
|
||||||
|
for (const std::size_t offset : offsets)
|
||||||
|
{
|
||||||
|
CAPTURE(offset);
|
||||||
|
const std::string doc = "[\"" + std::string(offset, 'a') + test_case.sequence + "\"]";
|
||||||
|
CHECK(json::accept(doc) == test_case.valid);
|
||||||
|
#if !defined(JSON_NOEXCEPTION)
|
||||||
|
CHECK(outcome(doc, false) == outcome(doc, true));
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -1440,6 +1440,13 @@ TEST_CASE("parser class")
|
|||||||
]
|
]
|
||||||
)";
|
)";
|
||||||
|
|
||||||
|
const auto* structured_object = R"(
|
||||||
|
{
|
||||||
|
"foo": [1, 2],
|
||||||
|
"bar": 3
|
||||||
|
}
|
||||||
|
)";
|
||||||
|
|
||||||
SECTION("filter nothing")
|
SECTION("filter nothing")
|
||||||
{
|
{
|
||||||
const json j_object = json::parse(s_object, [](int /*unused*/, json::parse_event_t /*unused*/, const json& /*unused*/) noexcept
|
const json j_object = json::parse(s_object, [](int /*unused*/, json::parse_event_t /*unused*/, const json& /*unused*/) noexcept
|
||||||
@@ -1515,6 +1522,48 @@ TEST_CASE("parser class")
|
|||||||
CHECK (j_filtered2 == json({1}));
|
CHECK (j_filtered2 == json({1}));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SECTION("filter array in object")
|
||||||
|
{
|
||||||
|
// the array is discarded once it is already stored under its key
|
||||||
|
const json j_filtered1 = json::parse(structured_object, [](int /*unused*/, json::parse_event_t e, const json& /*parsed*/) noexcept
|
||||||
|
{
|
||||||
|
return e != json::parse_event_t::array_end;
|
||||||
|
});
|
||||||
|
|
||||||
|
CHECK (j_filtered1 == json({{"bar", 3}}));
|
||||||
|
|
||||||
|
// the array is discarded before it is stored, leaving the
|
||||||
|
// placeholder the key event wrote
|
||||||
|
const json j_filtered2 = json::parse(structured_object, [](int /*unused*/, json::parse_event_t e, const json& /*parsed*/) noexcept
|
||||||
|
{
|
||||||
|
return e != json::parse_event_t::array_start;
|
||||||
|
});
|
||||||
|
|
||||||
|
CHECK (j_filtered2 == json({{"bar", 3}}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("filter value in object")
|
||||||
|
{
|
||||||
|
// the value is discarded after its key was kept, leaving the
|
||||||
|
// placeholder the key event wrote
|
||||||
|
const json j_filtered1 = json::parse(structured_object, [](int /*unused*/, json::parse_event_t e, const json & parsed) noexcept
|
||||||
|
{
|
||||||
|
return !(e == json::parse_event_t::value && parsed == json(3));
|
||||||
|
});
|
||||||
|
|
||||||
|
CHECK (j_filtered1 == json({{"foo", {1, 2}}}));
|
||||||
|
|
||||||
|
// the same value is discarded together with its key, so no
|
||||||
|
// placeholder was stored for it
|
||||||
|
const json j_filtered2 = json::parse(structured_object, [](int /*unused*/, json::parse_event_t e, const json & parsed) noexcept
|
||||||
|
{
|
||||||
|
return !((e == json::parse_event_t::key && parsed == json("bar")) ||
|
||||||
|
(e == json::parse_event_t::value && parsed == json(3)));
|
||||||
|
});
|
||||||
|
|
||||||
|
CHECK (j_filtered2 == json({{"foo", {1, 2}}}));
|
||||||
|
}
|
||||||
|
|
||||||
SECTION("filter specific events")
|
SECTION("filter specific events")
|
||||||
{
|
{
|
||||||
SECTION("first closing event")
|
SECTION("first closing event")
|
||||||
|
|||||||
@@ -15,6 +15,8 @@
|
|||||||
|
|
||||||
#include "doctest_compatibility.h"
|
#include "doctest_compatibility.h"
|
||||||
|
|
||||||
|
#include <cstdint>
|
||||||
|
|
||||||
#define JSON_TESTS_PRIVATE
|
#define JSON_TESTS_PRIVATE
|
||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
using nlohmann::json;
|
using nlohmann::json;
|
||||||
@@ -255,6 +257,75 @@ TEST_CASE("lexicographical comparison operators")
|
|||||||
{f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_}, // 21
|
{f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_, f_}, // 21
|
||||||
};
|
};
|
||||||
|
|
||||||
|
SECTION("signed/unsigned mixed comparison above INT64_MAX")
|
||||||
|
{
|
||||||
|
const json above_int64_max = static_cast<std::uint64_t>((std::numeric_limits<std::int64_t>::max)()) + 1ULL;
|
||||||
|
const json max_uint64 = (std::numeric_limits<std::uint64_t>::max)();
|
||||||
|
const json negative_one = -1;
|
||||||
|
const json one = 1;
|
||||||
|
const json max_int64 = (std::numeric_limits<std::int64_t>::max)();
|
||||||
|
|
||||||
|
CHECK_FALSE(above_int64_max == negative_one);
|
||||||
|
CHECK(above_int64_max != negative_one);
|
||||||
|
CHECK(negative_one < above_int64_max);
|
||||||
|
CHECK(negative_one <= above_int64_max);
|
||||||
|
CHECK_FALSE(negative_one > above_int64_max);
|
||||||
|
CHECK_FALSE(negative_one >= above_int64_max);
|
||||||
|
CHECK_FALSE(above_int64_max < negative_one);
|
||||||
|
CHECK_FALSE(above_int64_max <= negative_one);
|
||||||
|
CHECK(above_int64_max > negative_one);
|
||||||
|
CHECK(above_int64_max >= negative_one);
|
||||||
|
CHECK(negative_one != above_int64_max);
|
||||||
|
CHECK_FALSE(negative_one == above_int64_max);
|
||||||
|
|
||||||
|
CHECK_FALSE(max_uint64 == negative_one);
|
||||||
|
CHECK(max_uint64 != negative_one);
|
||||||
|
CHECK(negative_one < max_uint64);
|
||||||
|
CHECK(negative_one <= max_uint64);
|
||||||
|
CHECK_FALSE(negative_one > max_uint64);
|
||||||
|
CHECK_FALSE(negative_one >= max_uint64);
|
||||||
|
CHECK_FALSE(max_uint64 < negative_one);
|
||||||
|
CHECK_FALSE(max_uint64 <= negative_one);
|
||||||
|
CHECK(max_uint64 > negative_one);
|
||||||
|
CHECK(max_uint64 >= negative_one);
|
||||||
|
CHECK(negative_one != max_uint64);
|
||||||
|
CHECK_FALSE(negative_one == max_uint64);
|
||||||
|
|
||||||
|
CHECK_FALSE(one == above_int64_max);
|
||||||
|
CHECK(one != above_int64_max);
|
||||||
|
CHECK(one < above_int64_max);
|
||||||
|
CHECK(one <= above_int64_max);
|
||||||
|
CHECK_FALSE(one > above_int64_max);
|
||||||
|
CHECK_FALSE(one >= above_int64_max);
|
||||||
|
CHECK_FALSE(above_int64_max < one);
|
||||||
|
CHECK_FALSE(above_int64_max <= one);
|
||||||
|
CHECK(above_int64_max > one);
|
||||||
|
CHECK(above_int64_max >= one);
|
||||||
|
|
||||||
|
CHECK_FALSE(max_int64 == above_int64_max);
|
||||||
|
CHECK(max_int64 != above_int64_max);
|
||||||
|
CHECK(max_int64 < above_int64_max);
|
||||||
|
CHECK(max_int64 <= above_int64_max);
|
||||||
|
CHECK_FALSE(max_int64 > above_int64_max);
|
||||||
|
CHECK_FALSE(max_int64 >= above_int64_max);
|
||||||
|
CHECK_FALSE(above_int64_max < max_int64);
|
||||||
|
CHECK_FALSE(above_int64_max <= max_int64);
|
||||||
|
CHECK(above_int64_max > max_int64);
|
||||||
|
CHECK(above_int64_max >= max_int64);
|
||||||
|
|
||||||
|
#if JSON_HAS_THREE_WAY_COMPARISON
|
||||||
|
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
|
||||||
|
CHECK((negative_one <=> above_int64_max) == std::partial_ordering::less); // *NOPAD*
|
||||||
|
CHECK((above_int64_max <=> negative_one) == std::partial_ordering::greater); // *NOPAD*
|
||||||
|
CHECK((negative_one <=> max_uint64) == std::partial_ordering::less); // *NOPAD*
|
||||||
|
CHECK((max_uint64 <=> negative_one) == std::partial_ordering::greater); // *NOPAD*
|
||||||
|
CHECK((one <=> above_int64_max) == std::partial_ordering::less); // *NOPAD*
|
||||||
|
CHECK((above_int64_max <=> one) == std::partial_ordering::greater); // *NOPAD*
|
||||||
|
CHECK((max_int64 <=> above_int64_max) == std::partial_ordering::less); // *NOPAD*
|
||||||
|
CHECK((above_int64_max <=> max_int64) == std::partial_ordering::greater); // *NOPAD*
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
SECTION("compares unordered")
|
SECTION("compares unordered")
|
||||||
{
|
{
|
||||||
std::vector<std::vector<bool>> expected =
|
std::vector<std::vector<bool>> expected =
|
||||||
|
|||||||
@@ -98,8 +98,10 @@ void check_escaped(const char* original, const char* escaped = "", bool ensure_a
|
|||||||
void check_escaped(const char* original, const char* escaped, const bool ensure_ascii)
|
void check_escaped(const char* original, const char* escaped, const bool ensure_ascii)
|
||||||
{
|
{
|
||||||
std::stringstream ss;
|
std::stringstream ss;
|
||||||
json::serializer s(nlohmann::detail::output_adapter<char>(ss), ' ');
|
nlohmann::detail::output_stream_adapter<char> adapter(ss);
|
||||||
|
json::serializer s(adapter, ' ');
|
||||||
s.dump_escaped(original, ensure_ascii);
|
s.dump_escaped(original, ensure_ascii);
|
||||||
|
s.flush(); // dump_escaped writes into the serializer's internal buffer
|
||||||
CHECK(ss.str() == escaped);
|
CHECK(ss.str() == escaped);
|
||||||
}
|
}
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|||||||
@@ -427,6 +427,30 @@ TEST_CASE("deserialization")
|
|||||||
CHECK(l.events == std::vector<std::string>({"boolean(true)"}));
|
CHECK(l.events == std::vector<std::string>({"boolean(true)"}));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SECTION("from std::vector<signed char>")
|
||||||
|
{
|
||||||
|
std::vector<signed char> const v = {'t', 'r', 'u', 'e'};
|
||||||
|
CHECK(json::parse(v) == json(true));
|
||||||
|
CHECK(json::accept(v));
|
||||||
|
|
||||||
|
SaxEventLogger l;
|
||||||
|
CHECK(json::sax_parse(v, &l));
|
||||||
|
CHECK(l.events.size() == 1);
|
||||||
|
CHECK(l.events == std::vector<std::string>({"boolean(true)"}));
|
||||||
|
|
||||||
|
// bytes outside ASCII are negative here and must not be sign-extended;
|
||||||
|
// 0xC3 and 0xA9 do not fit in signed char (MSVC C4309), so spell them as negative values
|
||||||
|
std::vector<signed char> const umlaut = {'"', static_cast<signed char>(0xC3 - 0x100), static_cast<signed char>(0xA9 - 0x100), '"'};
|
||||||
|
CHECK(json::parse(umlaut) == json("\xC3\xA9"));
|
||||||
|
CHECK(json::accept(umlaut));
|
||||||
|
|
||||||
|
// 0xFF (spelled as -1 to stay in range) must not be reported as end of input
|
||||||
|
std::vector<signed char> const trailing = {'t', 'r', 'u', 'e', static_cast<signed char>(0xFF - 0x100)};
|
||||||
|
json _;
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'true\xFF'; expected end of input", json::parse_error&);
|
||||||
|
CHECK(!json::accept(trailing));
|
||||||
|
}
|
||||||
|
|
||||||
SECTION("from std::array")
|
SECTION("from std::array")
|
||||||
{
|
{
|
||||||
std::array<uint8_t, 5> const v { {'t', 'r', 'u', 'e'} };
|
std::array<uint8_t, 5> const v { {'t', 'r', 'u', 'e'} };
|
||||||
|
|||||||
@@ -38,6 +38,36 @@ TEST_CASE("Better diagnostics with positions")
|
|||||||
"[json.exception.type_error.302] type must be number, but is string", json::type_error);
|
"[json.exception.type_error.302] type must be number, but is string", json::type_error);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SECTION("positions of strings containing escape sequences")
|
||||||
|
{
|
||||||
|
// escape sequences make the token longer than the string it parses to,
|
||||||
|
// so the positions must not be derived from the parsed value's length
|
||||||
|
const auto check = [](const std::string & text, const std::string & token)
|
||||||
|
{
|
||||||
|
CAPTURE(text)
|
||||||
|
CAPTURE(token)
|
||||||
|
const json j = json::parse(text);
|
||||||
|
const json& v = j.at("a");
|
||||||
|
CHECK(text.substr(v.start_pos(), v.end_pos() - v.start_pos()) == token);
|
||||||
|
};
|
||||||
|
|
||||||
|
check(R"({"a":"plain"})", R"("plain")");
|
||||||
|
check(R"({"a":"tab\there"})", R"("tab\there")");
|
||||||
|
check(R"({"a":"\n\n\n\n\n\n"})", R"("\n\n\n\n\n\n")");
|
||||||
|
check(R"({"a":"\""})", R"("\"")");
|
||||||
|
check(R"({"a":"\\"})", R"("\\")");
|
||||||
|
check(R"({"a":"é"})", R"("é")");
|
||||||
|
check(R"({"a":"🌞"})", R"("🌞")");
|
||||||
|
check("{\"a\":\"\xc3\xa9\"}", "\"\xc3\xa9\""); // multi-byte UTF-8, no escapes
|
||||||
|
|
||||||
|
// a string at the root, where an escape would otherwise push the
|
||||||
|
// reported start position past the opening quote
|
||||||
|
const std::string root = R"("a\tb")";
|
||||||
|
const json j = json::parse(root);
|
||||||
|
CHECK(j.start_pos() == 0);
|
||||||
|
CHECK(j.end_pos() == root.size());
|
||||||
|
}
|
||||||
|
|
||||||
SECTION("JSON patch add to primitive parent (#4292)")
|
SECTION("JSON patch add to primitive parent (#4292)")
|
||||||
{
|
{
|
||||||
// the JSON Patch "add" target /foo/bar/baz has a string parent
|
// the JSON Patch "add" target /foo/bar/baz has a string parent
|
||||||
|
|||||||
@@ -801,6 +801,30 @@ TEST_CASE("modifiers")
|
|||||||
j1.update(j2, true);
|
j1.update(j2, true);
|
||||||
CHECK(j1 == json({{"string", "t"}, {"numbers", 1}}));
|
CHECK(j1 == json({{"string", "t"}, {"numbers", 1}}));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SECTION("overwrite primitive with object")
|
||||||
|
{
|
||||||
|
json j1 = {{"k", 1}};
|
||||||
|
json const j2 = {{"k", {{"x", 2}}}};
|
||||||
|
j1.update(j2, true);
|
||||||
|
CHECK(j1 == json({{"k", {{"x", 2}}}}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("overwrite array with object")
|
||||||
|
{
|
||||||
|
json j1 = {{"k", {1, 2}}};
|
||||||
|
json const j2 = {{"k", {{"x", 2}}}};
|
||||||
|
j1.update(j2, true);
|
||||||
|
CHECK(j1 == json({{"k", {{"x", 2}}}}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("overwrite nested primitive with object")
|
||||||
|
{
|
||||||
|
json j1 = {{"k", {{"inner", 1}}}};
|
||||||
|
json const j2 = {{"k", {{"inner", {{"x", 2}}}}}};
|
||||||
|
j1.update(j2, true);
|
||||||
|
CHECK(j1 == json({{"k", {{"inner", {{"x", 2}}}}}}));
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1530,4 +1530,40 @@ TEST_CASE("issue #4320 - custom base class must not leak nlohmann::detail into A
|
|||||||
CHECK(j == json({{"x", 1.0}, {"y", 2.0}, {"z", 3.0}}));
|
CHECK(j == json({{"x", 1.0}, {"y", 2.0}, {"z", 3.0}}));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("issue #5338 - truncated CBOR tagged binary subtype is rejected")
|
||||||
|
{
|
||||||
|
const std::vector<std::vector<std::uint8_t>> truncated_tags =
|
||||||
|
{
|
||||||
|
{0xD8},
|
||||||
|
{0xD9, 0x00},
|
||||||
|
{0xDA, 0x00, 0x00, 0x00},
|
||||||
|
{0xDB, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const auto& data : truncated_tags)
|
||||||
|
{
|
||||||
|
CAPTURE(data);
|
||||||
|
for (const auto tag_handler :
|
||||||
|
{
|
||||||
|
json::cbor_tag_handler_t::ignore, json::cbor_tag_handler_t::store
|
||||||
|
})
|
||||||
|
{
|
||||||
|
CAPTURE(tag_handler);
|
||||||
|
const auto result = json::from_cbor(data, true, false, tag_handler);
|
||||||
|
CHECK(result.is_discarded());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_CASE("issue #5402 - update(merge_objects=true) overwrites a primitive with an object")
|
||||||
|
{
|
||||||
|
json t = {{"k", 1}};
|
||||||
|
t.update(json{{"k", {{"x", 2}}}}, true);
|
||||||
|
CHECK(t == json({{"k", {{"x", 2}}}}));
|
||||||
|
|
||||||
|
json mixed = {{"keep", {{"a", 1}}}, {"replace", 1}};
|
||||||
|
mixed.update(json{{"keep", {{"b", 2}}}, {"replace", {{"x", 2}}}}, true);
|
||||||
|
CHECK(mixed == json({{"keep", {{"a", 1}, {"b", 2}}}, {"replace", {{"x", 2}}}}));
|
||||||
|
}
|
||||||
|
|
||||||
DOCTEST_CLANG_SUPPRESS_WARNING_POP
|
DOCTEST_CLANG_SUPPRESS_WARNING_POP
|
||||||
|
|||||||
@@ -382,3 +382,232 @@ TEST_CASE("dump for basic_json with long double number_float_t")
|
|||||||
check_same(100.0L, 100.0);
|
check_same(100.0L, 100.0);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("serialization of strings (bulk fast path)")
|
||||||
|
{
|
||||||
|
// These cases exercise the SWAR bulk-copy fast path in dump_escaped and the
|
||||||
|
// internal write buffer: long runs, escapes interrupting runs, 0x7F/DEL,
|
||||||
|
// multibyte UTF-8 under both ensure_ascii settings, and payloads larger than
|
||||||
|
// the write buffer.
|
||||||
|
|
||||||
|
SECTION("long unescaped ASCII exceeds the write buffer")
|
||||||
|
{
|
||||||
|
const std::string big(3000, 'a');
|
||||||
|
const json j = big;
|
||||||
|
CHECK(j.dump() == '"' + big + '"');
|
||||||
|
CHECK(j.dump(-1, ' ', true) == '"' + big + '"');
|
||||||
|
// round-trips
|
||||||
|
CHECK(json::parse(j.dump()) == j);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("runs interrupted by escapes")
|
||||||
|
{
|
||||||
|
const json j = std::string(500, 'x') + "\n\"\\" + std::string(500, 'y');
|
||||||
|
const std::string out = j.dump();
|
||||||
|
CHECK(out == '"' + std::string(500, 'x') + "\\n\\\"\\\\" + std::string(500, 'y') + '"');
|
||||||
|
CHECK(json::parse(out) == j);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("DEL (0x7F) depends on ensure_ascii")
|
||||||
|
{
|
||||||
|
const json j = std::string("a\x7f" "b");
|
||||||
|
CHECK(j.dump(-1, ' ', false) == "\"a\x7f" "b\""); // copied verbatim
|
||||||
|
CHECK(j.dump(-1, ' ', true) == "\"a\\u007fb\""); // escaped
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("multibyte UTF-8 under both ensure_ascii settings")
|
||||||
|
{
|
||||||
|
const json j = std::string("A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z"); // A é 你 😀 Z
|
||||||
|
// not escaping non-ASCII: bytes are copied through the bulk validator
|
||||||
|
CHECK(j.dump(-1, ' ', false) == "\"A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z\"");
|
||||||
|
// ensure_ascii: escaped (with a surrogate pair for the emoji)
|
||||||
|
CHECK(j.dump(-1, ' ', true) == "\"A\\u00e9\\u4f60\\ud83d\\ude00Z\"");
|
||||||
|
CHECK(json::parse(j.dump(-1, ' ', true)) == j);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("many small structural writes exceed the write buffer")
|
||||||
|
{
|
||||||
|
json arr = json::array();
|
||||||
|
for (int i = 0; i < 2000; ++i)
|
||||||
|
{
|
||||||
|
arr.push_back(i);
|
||||||
|
}
|
||||||
|
const std::string out = arr.dump();
|
||||||
|
CHECK(out.front() == '[');
|
||||||
|
CHECK(out.back() == ']');
|
||||||
|
CHECK(json::parse(out) == arr);
|
||||||
|
|
||||||
|
json obj = json::object();
|
||||||
|
for (int i = 0; i < 500; ++i)
|
||||||
|
{
|
||||||
|
obj["key" + std::to_string(i)] = i;
|
||||||
|
}
|
||||||
|
CHECK(json::parse(obj.dump()) == obj);
|
||||||
|
CHECK(json::parse(obj.dump(2)) == obj);
|
||||||
|
|
||||||
|
// an array of many empty strings emits a long run of single-character
|
||||||
|
// writes ('"', '"', ',') at shallow nesting depth, so the write buffer
|
||||||
|
// fills and flushes mid-run without the deep recursion that would
|
||||||
|
// overflow the stack on some debug builds
|
||||||
|
json many_empty = json::array();
|
||||||
|
for (int i = 0; i < 500; ++i)
|
||||||
|
{
|
||||||
|
many_empty.push_back("");
|
||||||
|
}
|
||||||
|
const std::string out2 = many_empty.dump();
|
||||||
|
CHECK(out2.size() > 1024); // spans multiple write-buffer flushes
|
||||||
|
CHECK(out2.front() == '[');
|
||||||
|
CHECK(out2.back() == ']');
|
||||||
|
CHECK(json::parse(out2) == many_empty);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("invalid UTF-8 handling is unaffected by the fast path")
|
||||||
|
{
|
||||||
|
const json j = std::string("valid\xff" "more");
|
||||||
|
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0xFF", json::type_error&);
|
||||||
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"valid\xef\xbf\xbd" "more\"");
|
||||||
|
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"valid\\ufffdmore\"");
|
||||||
|
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"validmore\"");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_CASE("indentation is written straight into the write buffer")
|
||||||
|
{
|
||||||
|
// put_indent() memsets the indentation into the write buffer instead of
|
||||||
|
// copying it out of a pre-grown indentation string. These cases cover an
|
||||||
|
// indentation wider than the buffer, a non-space indentation character, and
|
||||||
|
// nesting deep enough that the accumulated indentation spans several
|
||||||
|
// buffer-fulls - the situations the old grow-a-string approach got wrong.
|
||||||
|
|
||||||
|
SECTION("indent_step wider than the write buffer")
|
||||||
|
{
|
||||||
|
const json j = {{"a", 1}};
|
||||||
|
// 2000 > the 1024-byte write buffer, and > the 512 the indentation
|
||||||
|
// string used to start at
|
||||||
|
CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"a\": 1\n}");
|
||||||
|
// several whole buffer-fulls, so the buffer is refilled once and then
|
||||||
|
// flushed repeatedly
|
||||||
|
CHECK(j.dump(5000) == "{\n" + std::string(5000, ' ') + "\"a\": 1\n}");
|
||||||
|
CHECK(j.dump(5000, '\t') == "{\n" + std::string(5000, '\t') + "\"a\": 1\n}");
|
||||||
|
// an exact multiple of the buffer size
|
||||||
|
CHECK(j.dump(4096) == "{\n" + std::string(4096, ' ') + "\"a\": 1\n}");
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("a non-space indentation character is used throughout")
|
||||||
|
{
|
||||||
|
const json j = {{"a", 1}};
|
||||||
|
// 600 is past the point where the indentation used to be grown, which
|
||||||
|
// is where a hard-coded space would have shown up
|
||||||
|
CHECK(j.dump(600, '\t') == "{\n" + std::string(600, '\t') + "\"a\": 1\n}");
|
||||||
|
CHECK(j.dump(3, '.') == "{\n...\"a\": 1\n}");
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("accumulated indentation spans several buffer-fulls")
|
||||||
|
{
|
||||||
|
// five levels deep at 400 per level: the innermost value is indented by
|
||||||
|
// 2000 characters, reached in steps that each straddle the buffer end
|
||||||
|
json j = json::array({1});
|
||||||
|
for (int i = 0; i < 4; ++i)
|
||||||
|
{
|
||||||
|
j = json::array({j});
|
||||||
|
}
|
||||||
|
|
||||||
|
const std::string out = j.dump(400);
|
||||||
|
CHECK(out.find(std::string("\n") + std::string(2000, ' ') + "1\n") != std::string::npos);
|
||||||
|
CHECK(json::parse(out) == j);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("indentation is unchanged for ordinary widths")
|
||||||
|
{
|
||||||
|
const json j = {{"a", {1, 2}}, {"b", nullptr}};
|
||||||
|
CHECK(j.dump(2) == "{\n \"a\": [\n 1,\n 2\n ],\n \"b\": null\n}");
|
||||||
|
CHECK(j.dump(0) == "{\n\"a\": [\n1,\n2\n],\n\"b\": null\n}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_CASE("serialization of deeply nested values")
|
||||||
|
{
|
||||||
|
// dump() descends into a bounded number of levels and writes out whatever
|
||||||
|
// is nested deeper than that without the call stack; see
|
||||||
|
// https://github.com/nlohmann/json/issues/5387
|
||||||
|
|
||||||
|
SECTION("nested deeper than the call stack could follow")
|
||||||
|
{
|
||||||
|
// parsing is iterative, so building these costs little
|
||||||
|
const std::size_t depth = 100000;
|
||||||
|
|
||||||
|
const std::string array_text = std::string(depth, '[') + '0' + std::string(depth, ']');
|
||||||
|
CHECK(json::parse(array_text).dump() == array_text);
|
||||||
|
|
||||||
|
std::string object_text;
|
||||||
|
object_text.reserve((6 * depth) + 1);
|
||||||
|
for (std::size_t i = 0; i < depth; ++i)
|
||||||
|
{
|
||||||
|
object_text += "{\"a\":";
|
||||||
|
}
|
||||||
|
object_text += '1';
|
||||||
|
object_text.append(depth, '}');
|
||||||
|
CHECK(json::parse(object_text).dump() == object_text);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("depths around the bound of the descent")
|
||||||
|
{
|
||||||
|
// Cover every depth around the bound, so that the two ways of writing a
|
||||||
|
// value are known to meet cleanly - wherever the bound is set.
|
||||||
|
for (std::size_t d = 1; d <= 300; ++d)
|
||||||
|
{
|
||||||
|
CAPTURE(d);
|
||||||
|
|
||||||
|
const std::string array_text = std::string(d, '[') + '7' + std::string(d, ']');
|
||||||
|
CHECK(json::parse(array_text).dump() == array_text);
|
||||||
|
|
||||||
|
std::string object_text;
|
||||||
|
for (std::size_t i = 0; i < d; ++i)
|
||||||
|
{
|
||||||
|
object_text += "{\"k\":";
|
||||||
|
}
|
||||||
|
object_text += '7';
|
||||||
|
object_text.append(d, '}');
|
||||||
|
CHECK(json::parse(object_text).dump() == object_text);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("pretty-printing across the bound")
|
||||||
|
{
|
||||||
|
for (std::size_t d = 120; d <= 140; ++d)
|
||||||
|
{
|
||||||
|
CAPTURE(d);
|
||||||
|
|
||||||
|
const json j = json::parse(std::string(d, '[') + '7' + std::string(d, ']'));
|
||||||
|
|
||||||
|
std::string expected;
|
||||||
|
for (std::size_t i = 0; i < d; ++i)
|
||||||
|
{
|
||||||
|
expected += std::string(2 * i, ' ') + "[\n";
|
||||||
|
}
|
||||||
|
expected += std::string(2 * d, ' ') + '7';
|
||||||
|
for (std::size_t i = d; i > 0; --i)
|
||||||
|
{
|
||||||
|
expected += '\n' + std::string(2 * (i - 1), ' ') + ']';
|
||||||
|
}
|
||||||
|
|
||||||
|
CHECK(j.dump(2) == expected);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("an empty container below the bound")
|
||||||
|
{
|
||||||
|
// an empty container is written out in full and never descended into,
|
||||||
|
// so it must not gain a newline when it is reached iteratively
|
||||||
|
for (std::size_t d = 125; d <= 135; ++d)
|
||||||
|
{
|
||||||
|
CAPTURE(d);
|
||||||
|
|
||||||
|
const std::string compact = std::string(d, '[') + "[]" + std::string(d, ']');
|
||||||
|
CHECK(json::parse(compact).dump() == compact);
|
||||||
|
|
||||||
|
const std::string with_object = std::string(d, '[') + "{}" + std::string(d, ']');
|
||||||
|
CHECK(json::parse(with_object).dump() == with_object);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -53,4 +53,34 @@ TEST_CASE("type traits")
|
|||||||
// NOLINTEND(hicpp-avoid-c-arrays,modernize-avoid-c-arrays,cppcoreguidelines-avoid-c-arrays)
|
// NOLINTEND(hicpp-avoid-c-arrays,modernize-avoid-c-arrays,cppcoreguidelines-avoid-c-arrays)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SECTION("char_traits")
|
||||||
|
{
|
||||||
|
SECTION("to_int_type does not sign-extend")
|
||||||
|
{
|
||||||
|
using unsigned_traits = nlohmann::detail::char_traits<unsigned char>;
|
||||||
|
using signed_traits = nlohmann::detail::char_traits<signed char>;
|
||||||
|
|
||||||
|
CHECK(unsigned_traits::to_int_type(static_cast<unsigned char>(0x7F)) == 0x7F);
|
||||||
|
CHECK(unsigned_traits::to_int_type(static_cast<unsigned char>(0x80)) == 0x80);
|
||||||
|
CHECK(unsigned_traits::to_int_type(static_cast<unsigned char>(0xFF)) == 0xFF);
|
||||||
|
|
||||||
|
CHECK(signed_traits::to_int_type(static_cast<signed char>(0x7F)) == 0x7F);
|
||||||
|
// 0x80 and 0xFF do not fit in signed char (MSVC C4309), so spell them as negative values
|
||||||
|
CHECK(signed_traits::to_int_type(static_cast<signed char>(0x80 - 0x100)) == 0x80);
|
||||||
|
CHECK(signed_traits::to_int_type(static_cast<signed char>(0xFF - 0x100)) == 0xFF);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("no byte value collides with eof")
|
||||||
|
{
|
||||||
|
using unsigned_traits = nlohmann::detail::char_traits<unsigned char>;
|
||||||
|
using signed_traits = nlohmann::detail::char_traits<signed char>;
|
||||||
|
|
||||||
|
for (int i = 0; i < 256; ++i)
|
||||||
|
{
|
||||||
|
CHECK(unsigned_traits::to_int_type(static_cast<unsigned char>(i)) != unsigned_traits::eof());
|
||||||
|
CHECK(signed_traits::to_int_type(static_cast<signed char>(i)) != signed_traits::eof());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -819,6 +819,8 @@ TEST_CASE("UBJSON")
|
|||||||
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(vec2), "[json.exception.parse_error.115] parse error at byte 5: syntax error while parsing UBJSON high-precision number: invalid number text: 1A", json::parse_error);
|
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(vec2), "[json.exception.parse_error.115] parse error at byte 5: syntax error while parsing UBJSON high-precision number: invalid number text: 1A", json::parse_error);
|
||||||
std::vector<uint8_t> const vec3 = {'H', 'i', 2, '1', '.'};
|
std::vector<uint8_t> const vec3 = {'H', 'i', 2, '1', '.'};
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(vec3), "[json.exception.parse_error.115] parse error at byte 5: syntax error while parsing UBJSON high-precision number: invalid number text: 1.", json::parse_error);
|
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(vec3), "[json.exception.parse_error.115] parse error at byte 5: syntax error while parsing UBJSON high-precision number: invalid number text: 1.", json::parse_error);
|
||||||
|
std::vector<uint8_t> const vec_overflow = {'H', 'i', 5, '1', 'e', '4', '0', '0'};
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(vec_overflow), "[json.exception.out_of_range.406] number overflow parsing '1e400'", json::out_of_range&);
|
||||||
std::vector<uint8_t> const vec4 = {'H', 2, '1', '0'};
|
std::vector<uint8_t> const vec4 = {'H', 2, '1', '0'};
|
||||||
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(vec4), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing UBJSON size: expected length type specification (U, i, I, l, L) after '#'; last byte: 0x02", json::parse_error);
|
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(vec4), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing UBJSON size: expected length type specification (U, i, I, l, L) after '#'; last byte: 0x02", json::parse_error);
|
||||||
}
|
}
|
||||||
@@ -1711,6 +1713,44 @@ TEST_CASE("UBJSON")
|
|||||||
CHECK(json::to_ubjson(json::from_ubjson(s_L)) == s_i);
|
CHECK(json::to_ubjson(json::from_ubjson(s_L)) == s_i);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SECTION("no-op markers")
|
||||||
|
{
|
||||||
|
// A no-op ('N') is valid wherever a value may start; it is consumed
|
||||||
|
// by get_ignore_noop() before the value is read. It is not valid
|
||||||
|
// where a string length type specification is expected.
|
||||||
|
|
||||||
|
SECTION("accepted where a value may start")
|
||||||
|
{
|
||||||
|
// at top level, also repeated
|
||||||
|
CHECK(json::from_ubjson(std::vector<uint8_t>({'N', 'i', 1})) == json(1));
|
||||||
|
CHECK(json::from_ubjson(std::vector<uint8_t>({'N', 'N', 'N', 'i', 1})) == json(1));
|
||||||
|
|
||||||
|
// inside an array of unknown size, before and after an element
|
||||||
|
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', 'N', 'i', 1, ']'})) == json({1}));
|
||||||
|
CHECK(json::from_ubjson(std::vector<uint8_t>({'[', 'i', 1, 'N', ']'})) == json({1}));
|
||||||
|
|
||||||
|
// inside an object of unknown size: before a key, between key
|
||||||
|
// and value, and before the closing '}'
|
||||||
|
CHECK(json::from_ubjson(std::vector<uint8_t>({'{', 'N', 'U', 1, 'a', 'i', 1, '}'})) == json({{"a", 1}}));
|
||||||
|
CHECK(json::from_ubjson(std::vector<uint8_t>({'{', 'U', 1, 'a', 'N', 'i', 1, '}'})) == json({{"a", 1}}));
|
||||||
|
CHECK(json::from_ubjson(std::vector<uint8_t>({'{', 'U', 1, 'a', 'i', 1, 'N', '}'})) == json({{"a", 1}}));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("rejected where a length type specification is expected")
|
||||||
|
{
|
||||||
|
json _;
|
||||||
|
|
||||||
|
// after the 'S' marker of a string value
|
||||||
|
std::vector<uint8_t> const v_S = {'S', 'N', 'U', 1, 'a'};
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(v_S), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing UBJSON string: expected length type specification (U, i, I, l, L); last byte: 0x4E", json::parse_error&);
|
||||||
|
|
||||||
|
// as the key length of an object with a known size, where
|
||||||
|
// no-ops are not permitted in the first place
|
||||||
|
std::vector<uint8_t> const v_key = {'{', '#', 'i', 1, 'N', 'U', 1, 'a', 'i', 1};
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::from_ubjson(v_key), "[json.exception.parse_error.113] parse error at byte 5: syntax error while parsing UBJSON string: expected length type specification (U, i, I, l, L); last byte: 0x4E", json::parse_error&);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
SECTION("number")
|
SECTION("number")
|
||||||
{
|
{
|
||||||
SECTION("float")
|
SECTION("float")
|
||||||
@@ -2463,6 +2503,48 @@ TEST_CASE("all UBJSON first bytes")
|
|||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
TEST_CASE("UBJSON use_type requires use_size")
|
||||||
|
{
|
||||||
|
SECTION("non-empty array throws other_error.502")
|
||||||
|
{
|
||||||
|
const json j = {1, 2, 3};
|
||||||
|
CHECK_THROWS_WITH_AS(json::to_ubjson(j, false, true),
|
||||||
|
"[json.exception.other_error.502] use_type requires use_size = true",
|
||||||
|
json::other_error&);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("non-empty object throws other_error.502")
|
||||||
|
{
|
||||||
|
const json j = {{"a", 1}, {"b", 2}};
|
||||||
|
CHECK_THROWS_WITH_AS(json::to_ubjson(j, false, true),
|
||||||
|
"[json.exception.other_error.502] use_type requires use_size = true",
|
||||||
|
json::other_error&);
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("scalars do not throw with use_type=true, use_count=false")
|
||||||
|
{
|
||||||
|
CHECK_NOTHROW(json::to_ubjson(42, false, true));
|
||||||
|
CHECK_NOTHROW(json::to_ubjson(3.14, false, true));
|
||||||
|
CHECK_NOTHROW(json::to_ubjson("hello", false, true));
|
||||||
|
CHECK_NOTHROW(json::to_ubjson(true, false, true));
|
||||||
|
CHECK_NOTHROW(json::to_ubjson(nullptr, false, true));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("empty containers do not throw with use_type=true, use_count=false")
|
||||||
|
{
|
||||||
|
CHECK_NOTHROW(json::to_ubjson(json::array(), false, true));
|
||||||
|
CHECK_NOTHROW(json::to_ubjson(json::object(), false, true));
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTION("valid combinations on non-empty containers")
|
||||||
|
{
|
||||||
|
const json j = {1, 2, 3};
|
||||||
|
CHECK_NOTHROW(json::to_ubjson(j, false, false));
|
||||||
|
CHECK_NOTHROW(json::to_ubjson(j, true, false));
|
||||||
|
CHECK_NOTHROW(json::to_ubjson(j, true, true));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("UBJSON roundtrips" * doctest::skip())
|
TEST_CASE("UBJSON roundtrips" * doctest::skip())
|
||||||
{
|
{
|
||||||
SECTION("input from self-generated UBJSON files")
|
SECTION("input from self-generated UBJSON files")
|
||||||
|
|||||||
@@ -18,7 +18,12 @@
|
|||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
using nlohmann::json;
|
using nlohmann::json;
|
||||||
|
|
||||||
|
#include <array> // array
|
||||||
|
#include <cstddef> // size_t
|
||||||
|
#include <cstdint> // uint8_t
|
||||||
#include <list>
|
#include <list>
|
||||||
|
#include <string> // string
|
||||||
|
#include <vector> // vector
|
||||||
|
|
||||||
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||||
#include <iterator>
|
#include <iterator>
|
||||||
@@ -212,6 +217,66 @@ TEST_CASE("Parse with heterogeneous iterator and sentinel types")
|
|||||||
CHECK(j2.at(0) == 1);
|
CHECK(j2.at(0) == 1);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A type whose data() hands out raw bytes but whose size() counts something
|
||||||
|
// else - here fixed-size records. Reading [data(), data() + size()) as bytes
|
||||||
|
// would silently truncate the input, so data() and size() alone must not be
|
||||||
|
// taken as evidence of contiguous byte storage.
|
||||||
|
struct record_buffer
|
||||||
|
{
|
||||||
|
using value_type = std::array<char, 4>;
|
||||||
|
|
||||||
|
std::string bytes;
|
||||||
|
|
||||||
|
const char* data() const noexcept
|
||||||
|
{
|
||||||
|
return bytes.data();
|
||||||
|
}
|
||||||
|
std::size_t size() const noexcept
|
||||||
|
{
|
||||||
|
return bytes.size() / sizeof(value_type);
|
||||||
|
}
|
||||||
|
const char* begin() const noexcept
|
||||||
|
{
|
||||||
|
return bytes.data();
|
||||||
|
}
|
||||||
|
const char* end() const noexcept
|
||||||
|
{
|
||||||
|
return bytes.data() + bytes.size();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
TEST_CASE("Contiguous byte containers take the pointer adapter")
|
||||||
|
{
|
||||||
|
// Containers with contiguous single-byte storage are routed through the
|
||||||
|
// pointer-based adapter so the bulk fast paths apply in every standard, not
|
||||||
|
// only in C++20 where the library iterators model std::contiguous_iterator.
|
||||||
|
CHECK(nlohmann::detail::is_contiguous_byte_container<std::string>::value);
|
||||||
|
CHECK(nlohmann::detail::is_contiguous_byte_container<std::vector<char>>::value);
|
||||||
|
CHECK(nlohmann::detail::is_contiguous_byte_container<std::vector<std::uint8_t>>::value);
|
||||||
|
CHECK(nlohmann::detail::is_contiguous_byte_container<std::array<char, 4>>::value);
|
||||||
|
|
||||||
|
// input_adapter() takes its container by forwarding reference, so the trait
|
||||||
|
// is also asked about reference types
|
||||||
|
CHECK(nlohmann::detail::is_contiguous_byte_container<std::string&>::value);
|
||||||
|
CHECK(nlohmann::detail::is_contiguous_byte_container<const std::string&>::value);
|
||||||
|
|
||||||
|
// everything else keeps the iterator-based adapter
|
||||||
|
CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container<std::list<char>>::value);
|
||||||
|
CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container<std::vector<int>>::value);
|
||||||
|
CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container<const char*>::value);
|
||||||
|
|
||||||
|
// including a type that has data() and size() but whose size() does not
|
||||||
|
// count the units data() points at: its value_type says so
|
||||||
|
CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container<record_buffer>::value);
|
||||||
|
|
||||||
|
// and such a container still parses through its iterators, in full - taking
|
||||||
|
// it for a byte container would stop after data() + size() bytes
|
||||||
|
const record_buffer buffer{"[1,2,3,4,5]"};
|
||||||
|
CHECK(buffer.data() == buffer.bytes.data());
|
||||||
|
CHECK(buffer.size() * sizeof(record_buffer::value_type) < buffer.bytes.size());
|
||||||
|
CHECK(json::parse(buffer) == json({1, 2, 3, 4, 5}));
|
||||||
|
}
|
||||||
|
|
||||||
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||||
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
|
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
|
||||||
TEST_CASE("Parse with std::counted_iterator and std::default_sentinel_t")
|
TEST_CASE("Parse with std::counted_iterator and std::default_sentinel_t")
|
||||||
@@ -228,6 +293,180 @@ TEST_CASE("Parse with std::counted_iterator and std::default_sentinel_t")
|
|||||||
const std::counted_iterator<iterator_type> first2(json_str.begin(), len);
|
const std::counted_iterator<iterator_type> first2(json_str.begin(), len);
|
||||||
CHECK(json::accept(first2, std::default_sentinel));
|
CHECK(json::accept(first2, std::default_sentinel));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("std::counted_iterator reaches the contiguous fast paths")
|
||||||
|
{
|
||||||
|
// A sized sentinel makes the remaining element count computable in O(1), so
|
||||||
|
// std::counted_iterator over a contiguous iterator must reach the same bulk
|
||||||
|
// string/number scanners as a plain pointer - not just the byte-at-a-time
|
||||||
|
// fallback (see #5268 for the equivalent memcpy fast path).
|
||||||
|
#if JSON_HAS_RANGES
|
||||||
|
// JSON_HAS_RANGES is 0 on standard libraries with an incomplete <ranges>
|
||||||
|
// (libstdc++ < 11, libc++ < 16), where the adapter deliberately falls back
|
||||||
|
// to the byte-at-a-time scanner; everything below still has to work there.
|
||||||
|
using adapter_type = nlohmann::detail::iterator_input_adapter<std::counted_iterator<const char*>, std::default_sentinel_t>;
|
||||||
|
CHECK(adapter_type::supports_bulk_scan);
|
||||||
|
CHECK(adapter_type::supports_seek);
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// exercise every fast path: long ASCII run, multibyte UTF-8, escapes, and
|
||||||
|
// integer/floating-point numbers
|
||||||
|
const std::string json_str =
|
||||||
|
R"({"ascii":"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",)"
|
||||||
|
"\"utf8\":\"\xe4\xb8\xad\xe6\x96\x87\xf0\x9f\x98\x80\xc3\xa9\","
|
||||||
|
R"("escaped":"aéb\n\\","ints":[0,-1,18446744073709551615,-9223372036854775808],)"
|
||||||
|
R"("floats":[1.5,-2.25e3,0.30000000000000004]})";
|
||||||
|
const auto len = static_cast<std::iter_difference_t<const char*>>(json_str.size());
|
||||||
|
|
||||||
|
const std::counted_iterator<const char*> first(json_str.data(), len);
|
||||||
|
const json j = json::parse(first, std::default_sentinel);
|
||||||
|
|
||||||
|
// parsing through the pointer adapter must give exactly the same result
|
||||||
|
CHECK(j == json::parse(json_str));
|
||||||
|
|
||||||
|
#if !defined(JSON_NOEXCEPTION)
|
||||||
|
// Diagnostics that quote the offending token are reconstructed from the
|
||||||
|
// already-consumed input (supports_seek), a path a sized sentinel only
|
||||||
|
// reaches now; check a few that include the "last read" text. Parsing
|
||||||
|
// invalid input aborts when exceptions are off, hence the guard.
|
||||||
|
// Raw strings and explicit bytes: an escaped literal and two literals
|
||||||
|
// written next to each other both read as mistakes to static analysis.
|
||||||
|
const auto byte = [](int value)
|
||||||
|
{
|
||||||
|
return std::string(1, static_cast<char>(value));
|
||||||
|
};
|
||||||
|
const std::vector<std::string> diagnostic_docs =
|
||||||
|
{
|
||||||
|
"1\nx",
|
||||||
|
"truX",
|
||||||
|
"[tru]",
|
||||||
|
R"("abc)",
|
||||||
|
R"(["\ud834"])",
|
||||||
|
R"(["a)" + byte(0x01) + R"(b"])",
|
||||||
|
R"([")" + byte(0xC3) + byte(0x28) + R"("])",
|
||||||
|
"[1e]",
|
||||||
|
R"(["aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaX)"
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const auto& text : diagnostic_docs)
|
||||||
|
{
|
||||||
|
CAPTURE(text);
|
||||||
|
const std::counted_iterator<const char*> it(text.data(), static_cast<std::iter_difference_t<const char*>>(text.size()));
|
||||||
|
std::string counted_message;
|
||||||
|
std::string string_message;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
const json counted_result = json::parse(it, std::default_sentinel);
|
||||||
|
static_cast<void>(counted_result);
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
counted_message = e.what();
|
||||||
|
}
|
||||||
|
try
|
||||||
|
{
|
||||||
|
const json string_result = json::parse(text);
|
||||||
|
static_cast<void>(string_result);
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
string_message = e.what();
|
||||||
|
}
|
||||||
|
CHECK_FALSE(counted_message.empty());
|
||||||
|
CHECK(counted_message == string_message);
|
||||||
|
}
|
||||||
|
|
||||||
|
// and errors must still be reported identically
|
||||||
|
const std::string bad = "[01\n]";
|
||||||
|
const std::counted_iterator<const char*> bad_first(bad.data(), static_cast<std::iter_difference_t<const char*>>(bad.size()));
|
||||||
|
std::string counted_what;
|
||||||
|
std::string string_what;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
const json counted_result = json::parse(bad_first, std::default_sentinel);
|
||||||
|
static_cast<void>(counted_result);
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
counted_what = e.what();
|
||||||
|
}
|
||||||
|
try
|
||||||
|
{
|
||||||
|
const json string_result = json::parse(bad);
|
||||||
|
static_cast<void>(string_result);
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
string_what = e.what();
|
||||||
|
}
|
||||||
|
CHECK_FALSE(counted_what.empty());
|
||||||
|
CHECK(counted_what == string_what);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
#if !defined(JSON_NOEXCEPTION)
|
||||||
|
// several cases below are truncated on purpose, and parsing invalid input
|
||||||
|
// aborts when exceptions are off
|
||||||
|
TEST_CASE("std::counted_iterator bulk scanning stops at the counted end")
|
||||||
|
{
|
||||||
|
// The count, not the size of the underlying buffer, is the end of the
|
||||||
|
// input: the bulk scanners must never look at the bytes behind it, even
|
||||||
|
// though they are readable. Each case is compared against parsing the
|
||||||
|
// equivalent prefix as a std::string.
|
||||||
|
const auto via_counted = [](const std::string & buf, std::size_t n) -> std::string
|
||||||
|
{
|
||||||
|
const std::counted_iterator<const char*> first(buf.data(), static_cast<std::iter_difference_t<const char*>>(n));
|
||||||
|
try
|
||||||
|
{
|
||||||
|
const json j = json::parse(first, std::default_sentinel);
|
||||||
|
return "OK|" + j.dump();
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
return {e.what()};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
const auto via_prefix = [](const std::string & buf, std::size_t n) -> std::string
|
||||||
|
{
|
||||||
|
try
|
||||||
|
{
|
||||||
|
const json j = json::parse(buf.substr(0, n));
|
||||||
|
return "OK|" + j.dump();
|
||||||
|
}
|
||||||
|
catch (const json::parse_error& e)
|
||||||
|
{
|
||||||
|
return {e.what()};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
struct testcase // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init)
|
||||||
|
{
|
||||||
|
const char* buffer;
|
||||||
|
std::size_t count;
|
||||||
|
};
|
||||||
|
const std::vector<testcase> cases =
|
||||||
|
{
|
||||||
|
{"[\"abc\"]____TRAILING____", 7}, // exact fit, tail hidden
|
||||||
|
{"[\"abcdefghijklmnop\"]____", 8}, // cut inside a string
|
||||||
|
{"[\"abc\"]____", 6}, // cut just before the closing quote
|
||||||
|
{"[12345]xxxxx", 4}, // cut inside a number
|
||||||
|
{"[123]999999", 5}, // number ends exactly at the count
|
||||||
|
{"[\"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\"]", 12}, // closing quote only behind the count
|
||||||
|
{"[\"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\"]", 19}, // cut inside an 8-byte SWAR stride
|
||||||
|
{"[\"\xe4\xb8\xad\xe6\x96\x87\"]", 5}, // cut inside a UTF-8 sequence
|
||||||
|
{"[\"\xe4\xb8\xad\xe6\x96\x87\"]____", 10}, // complete UTF-8, tail hidden
|
||||||
|
{"[1.25e3]TRAILINGDIGITS999", 7}, // number token reaches the count
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const auto& tc : cases)
|
||||||
|
{
|
||||||
|
CAPTURE(tc.buffer);
|
||||||
|
CAPTURE(tc.count);
|
||||||
|
const std::string buffer = tc.buffer;
|
||||||
|
CHECK(via_counted(buffer, tc.count) == via_prefix(buffer, tc.count));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
#endif
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|||||||
@@ -125,6 +125,16 @@ TEST_CASE("wide strings")
|
|||||||
std::u32string const w = U"\"\x110000";
|
std::u32string const w = U"\"\x110000";
|
||||||
json _;
|
json _;
|
||||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||||
|
|
||||||
|
// a code unit above U+10FFFF must not be narrowed onto the EOF
|
||||||
|
// sentinel: 0xFFFFFFFF would otherwise end the document silently and
|
||||||
|
// let everything following it pass the strict end-of-input check
|
||||||
|
std::u32string const trailing{U'[', U'1', U']', static_cast<char32_t>(0xFFFFFFFF), U'x'};
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::parse(trailing), "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
|
||||||
|
CHECK(!json::accept(trailing));
|
||||||
|
|
||||||
|
// the same unit inside a string is reported as an ill-formed byte
|
||||||
|
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user