diff --git a/.github/workflows/check_amalgamation.yml b/.github/workflows/check_amalgamation.yml index 35f3a57f7..d81a80b18 100644 --- a/.github/workflows/check_amalgamation.yml +++ b/.github/workflows/check_amalgamation.yml @@ -35,6 +35,7 @@ jobs: MAIN_DIR: ${{ github.workspace }}/main INCLUDE_DIR: ${{ github.workspace }}/main/single_include/nlohmann TOOL_DIR: ${{ github.workspace }}/tools/tools/amalgamate + NATVIS_TOOL_DIR: ${{ github.workspace }}/tools/tools/generate_natvis steps: - name: Harden Runner @@ -61,6 +62,48 @@ jobs: python3 -mvenv venv venv/bin/pip3 install -r $MAIN_DIR/tools/astyle/requirements.txt + - name: Install generate_natvis dependencies + run: pip3 install -r $NATVIS_TOOL_DIR/requirements.txt + + - name: Regenerate the tools/macro_builder tables in macro_scope.hpp + run: | + cd $MAIN_DIR + + # Built from this PR's own tools/macro_builder/main.cpp, not a + # develop checkout: unlike amalgamate.py and generate_natvis.py, + # this tool has no other source of truth to check against (its + # own README documents that it must reproduce macro_scope.hpp + # byte for byte), so there is nothing to gain from checking out a + # separate copy, and doing so would make this step fail on a PR + # that adds support for a new dispatch table until that PR itself + # merges to develop, the same way generate_natvis.py's --version + # requirement briefly did. + TMPDIR=$(mktemp -d ./macro_builder_check.XXXXXX) + c++ -std=c++11 tools/macro_builder/main.cpp -o "$TMPDIR/macro_builder" + "$TMPDIR/macro_builder" > "$TMPDIR/paste.hpp" + "$TMPDIR/macro_builder" type_body > "$TMPDIR/type_body.hpp" + + # Splice the (still unindented) generated blocks back into + # macro_scope.hpp; the astyle pass below indents their + # continuation lines the same way it does for the rest of + # include/, so a correctly regenerated file comes out unchanged. + awk -v newfile="$TMPDIR/paste.hpp" ' + BEGIN { while ((getline line < newfile) > 0) { new = new line "\n" } } + /^#define NLOHMANN_JSON_EXPAND\( x \) x$/ { printf "%s", new; skip=1 } + skip && /^#define NLOHMANN_JSON_DOUBLE_PASTE63\(/ { skip=0; next } + skip { next } + { print } + ' include/nlohmann/detail/macro_scope.hpp > "$TMPDIR/macro_scope_1.hpp" + awk -v newfile="$TMPDIR/type_body.hpp" ' + BEGIN { while ((getline line < newfile) > 0) { new = new line "\n" } } + /^#define NLOHMANN_JSON_TYPE_BODY\(Prefix, \.\.\.\)/ { printf "%s", new; skip=1 } + skip && /^[[:space:]]*NLOHMANN_JSON_TYPE_BODY_SENTINEL\)\)$/ { skip=0; next } + skip { next } + { print } + ' "$TMPDIR/macro_scope_1.hpp" > "$TMPDIR/macro_scope_2.hpp" + mv "$TMPDIR/macro_scope_2.hpp" include/nlohmann/detail/macro_scope.hpp + rm -rf "$TMPDIR" + - name: Regenerate amalgamation, formatting, and BUILD.bazel run: | cd $MAIN_DIR @@ -88,6 +131,19 @@ jobs: ${{ github.workspace }}/venv/bin/astyle --project=tools/astyle/.astylerc --suffix=none --quiet \ $(find $SOURCE_DIRS -type f \( -name '*.hpp' -o -name '*.cpp' -o -name '*.cu' \) -not -path 'tests/thirdparty/*' -not -path 'tests/abi/include/nlohmann/*' | sort) + - name: Regenerate nlohmann_json.natvis + run: | + cd $MAIN_DIR + # Pass --version explicitly so this step also works with the tool + # copy from develop before this repository's own generate_natvis.py + # learns to derive the version itself: the older script requires + # --version, and the newer one accepts it as an explicit override. + ABI_MACROS=include/nlohmann/detail/abi_macros.hpp + VERSION_MAJOR=$(grep -m1 'define NLOHMANN_JSON_VERSION_MAJOR' $ABI_MACROS | grep -o '[0-9]\+') + VERSION_MINOR=$(grep -m1 'define NLOHMANN_JSON_VERSION_MINOR' $ABI_MACROS | grep -o '[0-9]\+') + VERSION_PATCH=$(grep -m1 'define NLOHMANN_JSON_VERSION_PATCH' $ABI_MACROS | grep -o '[0-9]\+') + python3 $NATVIS_TOOL_DIR/generate_natvis.py --version "$VERSION_MAJOR.$VERSION_MINOR.$VERSION_PATCH" $MAIN_DIR + - name: Build patch and check for differences id: diff run: | diff --git a/.github/workflows/publish_documentation.yml b/.github/workflows/publish_documentation.yml index 189d95419..f65d63a92 100644 --- a/.github/workflows/publish_documentation.yml +++ b/.github/workflows/publish_documentation.yml @@ -7,6 +7,16 @@ on: - develop paths: - docs/mkdocs/** + # the site also embeds these files via pymdownx.snippets + # (mkdocs.yml sets restrict_base_path: false for this) + - .clang-tidy + - .github/CODE_OF_CONDUCT.md + - .github/CONTRIBUTING.md + - .github/SECURITY.md + - cmake/clang_flags.cmake + - cmake/gcc_flags.cmake + - tests/fmt_formatter/project/main.cpp + - tools/astyle/.astylerc workflow_dispatch: # we don't want to have concurrent jobs, and we don't want to cancel running jobs to avoid broken publications @@ -23,7 +33,7 @@ jobs: contents: write if: github.repository == 'nlohmann/json' - runs-on: ubuntu-22.04 + runs-on: ubuntu-latest steps: - name: Harden Runner uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 @@ -31,6 +41,8 @@ jobs: egress-policy: audit - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Install virtual environment run: make install_venv -C docs/mkdocs diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index d8c27c621..fb3cd4148 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -100,7 +100,7 @@ jobs: container: ubuntu:focal strategy: matrix: - target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_skiplibraryversioncheck, ci_test_simdutf, ci_test_strict_nul_handling, ci_test_no_thread_local] + target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_disabletuplereferenceconversion, ci_test_skiplibraryversioncheck, ci_test_simdutf, ci_test_strict_nul_handling, ci_test_no_thread_local] steps: - name: Install build-essential run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev @@ -346,6 +346,8 @@ jobs: container: intel/oneapi-hpckit:latest steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Get latest CMake and ninja uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2 - name: Run CMake @@ -358,6 +360,8 @@ jobs: container: nvcr.io/nvidia/nvhpc:25.5-devel-cuda12.9-ubuntu22.04 steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Get latest CMake and ninja uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2 - name: Run CMake diff --git a/.github/workflows/windows.yml b/.github/workflows/windows.yml index ea742773f..65169eac4 100644 --- a/.github/workflows/windows.yml +++ b/.github/workflows/windows.yml @@ -87,6 +87,8 @@ jobs: steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Get latest CMake and ninja uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2 - name: Set extra CXX_FLAGS for latest std_version @@ -123,6 +125,8 @@ jobs: steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Run CMake (Release) run: cmake -S . -B build -G "Visual Studio 18 2026" -A ARM64 -DJSON_BuildTests=On -DCMAKE_CXX_FLAGS="/W4 /WX" if: matrix.build_type == 'Release' diff --git a/CMakeLists.txt b/CMakeLists.txt index 4c43c23ff..f0a6771dc 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -55,6 +55,7 @@ option(JSON_Diagnostic_Positions "Enable diagnostic positions." OFF) option(JSON_GlobalUDLs "Place user-defined string literals in the global namespace." ON) option(JSON_ImplicitConversions "Enable implicit conversions." ON) option(JSON_DisableEnumSerialization "Disable default integer enum serialization." OFF) +option(JSON_DisableTupleReferenceConversion "Disable conversion from a one-element tuple of a JSON reference." OFF) option(JSON_LegacyDiscardedValueComparison "Enable legacy discarded value comparison." OFF) option(JSON_Install "Install CMake targets during install step." ${MAIN_PROJECT}) option(JSON_MultipleHeaders "Use non-amalgamated version of the library." ON) @@ -101,6 +102,10 @@ if (JSON_DisableEnumSerialization) message(STATUS "Enum integer serialization is disabled (JSON_DISABLE_ENUM_SERIALIZATION=1)") endif() +if (JSON_DisableTupleReferenceConversion) + message(STATUS "Tuple reference conversion is disabled (JSON_DISABLE_TUPLE_REFERENCE_CONVERSION=1)") +endif() + if (JSON_LegacyDiscardedValueComparison) message(STATUS "Legacy discarded value comparison enabled (JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1)") endif() @@ -143,6 +148,7 @@ target_compile_definitions( $<$>:JSON_USE_GLOBAL_UDLS=0> $<$>:JSON_USE_IMPLICIT_CONVERSIONS=0> $<$:JSON_DISABLE_ENUM_SERIALIZATION=1> + $<$:JSON_DISABLE_TUPLE_REFERENCE_CONVERSION=1> $<$:JSON_DIAGNOSTICS=1> $<$:JSON_DIAGNOSTIC_POSITIONS=1> $<$:JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1> diff --git a/Makefile b/Makefile index ccea759af..fec936343 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: pretty clean ChangeLog.md release update_hedley update_hedley_undef BUILD.bazel +.PHONY: pretty clean ChangeLog.md release update_hedley update_hedley_undef BUILD.bazel natvis macro_builder_check ########################################################################## # configuration @@ -24,6 +24,9 @@ AMALGAMATED_FWD_FILE=single_include/nlohmann/json_fwd.hpp # json_literals.hpp only includes , so it is copied verbatim AMALGAMATED_LITERALS_FILE=single_include/nlohmann/json_literals.hpp +# the header with the argument-counting macros generated by tools/macro_builder +MACRO_SCOPE_HPP=include/nlohmann/detail/macro_scope.hpp + ########################################################################## # documentation of the Makefile's targets @@ -36,13 +39,9 @@ all: @echo "ChangeLog.md - generate ChangeLog file" @echo "check-amalgamation - check whether sources have been amalgamated and BUILD.bazel is up to date" @echo "clean - remove built files" - @echo "doctest - compile example files and check their output" - @echo "fuzz_testing - prepare fuzz testing of the JSON parser" - @echo "fuzz_testing_bon8 - prepare fuzz testing of the BON8 parser" - @echo "fuzz_testing_bson - prepare fuzz testing of the BSON parser" - @echo "fuzz_testing_cbor - prepare fuzz testing of the CBOR parser" - @echo "fuzz_testing_msgpack - prepare fuzz testing of the MessagePack parser" - @echo "fuzz_testing_ubjson - prepare fuzz testing of the UBJSON parser" + @echo "fuzzing - see tests/fuzzing.md for how to build and run the fuzzers" + @echo "macro_builder_check - check that macro_scope.hpp matches tools/macro_builder's output" + @echo "natvis - regenerate nlohmann_json.natvis from the current ABI tags and version" @echo "pretty - beautify code with Artistic Style" @echo "run_benchmarks - build and run benchmarks" @echo "update_hedley - download Hedley and regenerate hedley.hpp / hedley_undef.hpp" @@ -61,74 +60,6 @@ run_benchmarks: cd cmake-build-benchmarks ; ./json_benchmarks -########################################################################## -# fuzzing -########################################################################## - -# the overall fuzz testing target -fuzz_testing: - rm -fr fuzz-testing - mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out - $(MAKE) parse_afl_fuzzer -C tests CXX=afl-clang++ - mv tests/parse_afl_fuzzer fuzz-testing/fuzzer - find tests/data/json_tests -size -5k -name *json | xargs -I{} cp "{}" fuzz-testing/testcases - @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" - -fuzz_testing_bon8: - rm -fr fuzz-testing - mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out - $(MAKE) parse_bon8_fuzzer -C tests CXX=afl-clang++ - mv tests/parse_bon8_fuzzer fuzz-testing/fuzzer - find tests/data -size -5k -name *.bon8 | xargs -I{} cp "{}" fuzz-testing/testcases - @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" - -fuzz_testing_bson: - rm -fr fuzz-testing - mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out - $(MAKE) parse_bson_fuzzer -C tests CXX=afl-clang++ - mv tests/parse_bson_fuzzer fuzz-testing/fuzzer - find tests/data -size -5k -name *.bson | xargs -I{} cp "{}" fuzz-testing/testcases - @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" - -fuzz_testing_cbor: - rm -fr fuzz-testing - mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out - $(MAKE) parse_cbor_fuzzer -C tests CXX=afl-clang++ - mv tests/parse_cbor_fuzzer fuzz-testing/fuzzer - find tests/data -size -5k -name *.cbor | xargs -I{} cp "{}" fuzz-testing/testcases - @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" - -fuzz_testing_msgpack: - rm -fr fuzz-testing - mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out - $(MAKE) parse_msgpack_fuzzer -C tests CXX=afl-clang++ - mv tests/parse_msgpack_fuzzer fuzz-testing/fuzzer - find tests/data -size -5k -name *.msgpack | xargs -I{} cp "{}" fuzz-testing/testcases - @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" - -fuzz_testing_ubjson: - rm -fr fuzz-testing - mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out - $(MAKE) parse_ubjson_fuzzer -C tests CXX=afl-clang++ - mv tests/parse_ubjson_fuzzer fuzz-testing/fuzzer - find tests/data -size -5k -name *.ubjson | xargs -I{} cp "{}" fuzz-testing/testcases - @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" - -fuzzing-start: - afl-fuzz -S fuzzer1 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -S fuzzer2 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -S fuzzer3 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -S fuzzer4 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -S fuzzer5 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -S fuzzer6 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -S fuzzer7 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -M fuzzer0 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer - -fuzzing-stop: - -killall fuzzer - -killall afl-fuzz - - ########################################################################## # Static analysis ########################################################################## @@ -158,10 +89,6 @@ install_astyle: pretty: install_astyle $(ASTYLE) --project=tools/astyle/.astylerc $(SRCS) $(TESTS_SRCS) $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) docs/mkdocs/docs/examples/*.cpp -# call the Clang-Format on all source files -pretty_format: - for FILE in $(SRCS) $(TESTS_SRCS) $(AMALGAMATED_FILE) docs/mkdocs/docs/examples/*.cpp; do echo $$FILE; clang-format -i $$FILE; done - # create single header files and pretty print amalgamate: $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) $(MAKE) pretty @@ -178,8 +105,26 @@ $(AMALGAMATED_FWD_FILE): $(SRCS) $(AMALGAMATED_LITERALS_FILE): include/nlohmann/json_literals.hpp cp include/nlohmann/json_literals.hpp $(AMALGAMATED_LITERALS_FILE) +# regenerate nlohmann_json.natvis from the ABI tags and version in include/nlohmann/detail/abi_macros.hpp +natvis: + python3 tools/generate_natvis/generate_natvis.py . + +# regenerate the two tools/macro_builder blocks of $(MACRO_SCOPE_HPP) (see its README.md) and diff against the +# checked-in header; phony, because it never writes $(MACRO_SCOPE_HPP) itself +macro_builder_check: + @set -e; \ + TMPDIR=$$(mktemp -d ./macro_builder_check.XXXXXX); \ + trap 'rm -rf "$$TMPDIR"' EXIT; \ + $(CXX) -std=c++11 tools/macro_builder/main.cpp -o "$$TMPDIR/macro_builder"; \ + "$$TMPDIR/macro_builder" > "$$TMPDIR/paste.hpp"; \ + "$$TMPDIR/macro_builder" type_body > "$$TMPDIR/type_body.hpp"; \ + $(ASTYLE) --project=tools/astyle/.astylerc --suffix=none --quiet "$$TMPDIR/paste.hpp" "$$TMPDIR/type_body.hpp"; \ + sed -n '/^#define NLOHMANN_JSON_EXPAND( x ) x$$/,/^#define NLOHMANN_JSON_DOUBLE_PASTE63(/p' $(MACRO_SCOPE_HPP) > "$$TMPDIR/paste_actual.hpp"; \ + sed -n '/^#define NLOHMANN_JSON_TYPE_BODY(Prefix, \.\.\.)/,/^ NLOHMANN_JSON_TYPE_BODY_SENTINEL))$$/p' $(MACRO_SCOPE_HPP) > "$$TMPDIR/type_body_actual.hpp"; \ + diff "$$TMPDIR/paste.hpp" "$$TMPDIR/paste_actual.hpp" || (echo "===================================================================\n $(MACRO_SCOPE_HPP) (NLOHMANN_JSON_EXPAND..NLOHMANN_JSON_DOUBLE_PASTE63) is out of date!\n Regenerate it, see tools/macro_builder/README.md.\n===================================================================" ; exit 1); \ + diff "$$TMPDIR/type_body.hpp" "$$TMPDIR/type_body_actual.hpp" || (echo "===================================================================\n $(MACRO_SCOPE_HPP) (NLOHMANN_JSON_TYPE_BODY) is out of date!\n Regenerate it, see tools/macro_builder/README.md.\n===================================================================" ; exit 1) + # check if file single_include/nlohmann/json.hpp has been amalgamated from the nlohmann sources -# Note: this target is called by Travis check-amalgamation: @mv $(AMALGAMATED_FILE) $(AMALGAMATED_FILE)~ @mv $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_FWD_FILE)~ @@ -195,6 +140,11 @@ check-amalgamation: @$(MAKE) BUILD.bazel @diff BUILD.bazel BUILD.bazel~ || (echo "===================================================================\n BUILD.bazel is out of date! Please run 'make BUILD.bazel'.\n===================================================================" ; mv BUILD.bazel~ BUILD.bazel ; false) @mv BUILD.bazel~ BUILD.bazel + @mv nlohmann_json.natvis nlohmann_json.natvis~ + @$(MAKE) natvis + @diff nlohmann_json.natvis nlohmann_json.natvis~ || (echo "===================================================================\n nlohmann_json.natvis is out of date! Please run 'make natvis'.\n===================================================================" ; mv nlohmann_json.natvis~ nlohmann_json.natvis ; false) + @mv nlohmann_json.natvis~ nlohmann_json.natvis + @$(MAKE) macro_builder_check # generate the Bazel BUILD file; phony, because a removed header would not trigger a rebuild BUILD.bazel: @@ -246,7 +196,7 @@ release: include.zip json.tar.xz cp $(AMALGAMATED_FWD_FILE) release_files cp $(AMALGAMATED_LITERALS_FILE) release_files mv $(AMALGAMATED_FILE).asc $(AMALGAMATED_FWD_FILE).asc $(AMALGAMATED_LITERALS_FILE).asc json.tar.xz json.tar.xz.asc include.zip include.zip.asc release_files - cd release_files ; shasum -a 256 json.hpp include.zip json.tar.xz > hashes.txt + cd release_files ; shasum -a 256 $$(find . -type f -not -name '*.asc' | sed 's|^\./||' | sort) > hashes.txt ########################################################################## @@ -256,7 +206,6 @@ release: include.zip json.tar.xz # clean up clean: rm -fr fuzz fuzz-testing *.dSYM tests/*.dSYM - rm -fr benchmarks/files/numbers/*.json rm -fr cmake-build-benchmarks fuzz-testing cmake-build-pvs-studio release_files $(MAKE) clean -Cdocs diff --git a/cmake/ci.cmake b/cmake/ci.cmake index a001a05ae..27187f235 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -276,6 +276,20 @@ add_custom_target(ci_test_disableenumserialization COMMENT "Compile and test with enum serialization disabled" ) +############################################################################### +# Disable conversion from a one-element tuple of a JSON reference. +############################################################################### + +add_custom_target(ci_test_disabletuplereferenceconversion + COMMAND ${CMAKE_COMMAND} + -DCMAKE_BUILD_TYPE=Debug -GNinja + -DJSON_BuildTests=ON -DJSON_FastTests=ON -DJSON_DisableTupleReferenceConversion=ON + -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_disabletuplereferenceconversion + COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_disabletuplereferenceconversion + COMMAND cd ${PROJECT_BINARY_DIR}/build_disabletuplereferenceconversion && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure + COMMENT "Compile and test with tuple reference conversion disabled" +) + ############################################################################### # Skip the multiple-inclusion library version check. ############################################################################### diff --git a/docs/Makefile b/docs/Makefile index 0412fb90a..a61c30182 100644 --- a/docs/Makefile +++ b/docs/Makefile @@ -11,20 +11,31 @@ EXAMPLES = $(wildcard mkdocs/docs/examples/*.cpp) cxx_standard = $(lastword c++11 $(filter c++%, $(subst ., ,$1))) +# common compile flags for the stand-alone example files +EXAMPLE_CPPFLAGS = -I $(SRCDIR) -DJSON_USE_GLOBAL_UDLS=0 +EXAMPLE_WARNFLAGS = -Werror=deprecated-declarations + +# examples that document deprecated API and are allowed to use it +DEPRECATED_EXAMPLES = $(addprefix mkdocs/docs/examples/, \ + json_pointer__operator__equal_stringtype \ + json_pointer__operator__notequal_stringtype \ + json_pointer__operator_string_t) +$(DEPRECATED_EXAMPLES:=.output) $(DEPRECATED_EXAMPLES:=.test): EXAMPLE_WARNFLAGS = -Wno-deprecated-declarations + # create output from a stand-alone example file %.output: %.cpp - @echo "standard $(call cxx_standard $(<:.cpp=))" + @echo "standard $(call cxx_standard,$(<:.cpp=))" $(MAKE) $(<:.cpp=) \ - CPPFLAGS="-I $(SRCDIR) -DJSON_USE_GLOBAL_UDLS=0" \ - CXXFLAGS="-std=$(call cxx_standard,$(<:.cpp=)) -Wno-deprecated-declarations" + CPPFLAGS="$(EXAMPLE_CPPFLAGS)" \ + CXXFLAGS="-std=$(call cxx_standard,$(<:.cpp=)) $(EXAMPLE_WARNFLAGS)" ./$(<:.cpp=) > $@ rm $(<:.cpp=) # compare created output with current output of the example files %.test: %.cpp $(MAKE) $(<:.cpp=) \ - CPPFLAGS="-I $(SRCDIR) -DJSON_USE_GLOBAL_UDLS=0" \ - CXXFLAGS="-std=$(call cxx_standard,$(<:.cpp=)) -Wno-deprecated-declarations" + CPPFLAGS="$(EXAMPLE_CPPFLAGS)" \ + CXXFLAGS="-std=$(call cxx_standard,$(<:.cpp=)) $(EXAMPLE_WARNFLAGS)" ./$(<:.cpp=) > $@ diff $@ $(<:.cpp=.output) rm $(<:.cpp=) $@ diff --git a/docs/docset/Info.plist b/docs/docset/Info.plist index 772ec08af..8f63bc5dd 100644 --- a/docs/docset/Info.plist +++ b/docs/docset/Info.plist @@ -13,7 +13,7 @@ dashIndexFilePath index.html DashDocSetFallbackURL - https://nlohmann.github.io/json/ + https://json.nlohmann.me/ isJavaScriptEnabled diff --git a/docs/docset/Makefile b/docs/docset/Makefile index 9f4363462..b638390db 100644 --- a/docs/docset/Makefile +++ b/docs/docset/Makefile @@ -52,35 +52,25 @@ install_docset_zeal: JSON_for_Modern_C++.docset mkdir -p $$docset_root; \ cp -r JSON_for_Modern_C++.docset $$docset_root/ +# both targets below compare the docset search index with the mkdocs page +# set. They share the same normalization (docs/foo/index.md and +# docs/foo.md both become foo/index.html, the URL mkdocs itself would +# give the page; the top-level index.md is excluded, as it is not part +# of the hand-curated docSet.sql) and use comm(1) on two sorted lists +# instead of running a sqlite3 query, or an O(n*m) nested shell loop, +# once per page. +DOCSET_INDEX_PATHS=$(shell sqlite3 docSet.dsidx "SELECT DISTINCT path FROM searchIndex" | sort) +DOCSET_PAGE_PATHS=$(shell echo '$(MKDOCS_PAGES)' | tr ' ' '\n' | grep -v '^index\.md$$' | $(SED) -E 's@/index\.md$$@/index.html@; s@\.md$$@/index.html@' | sort) + # list mkdocs pages missing from the docset index .PHONY: list_missing_pages list_missing_pages: docSet.dsidx - @for page in $(MKDOCS_PAGES); do \ - case "$$page" in \ - */index.md) path=$${page/\/index.md/} ;; \ - *) path=$${page/.md/} ;; \ - esac; \ - if [ "x$$page" != "xindex.md" -a "x$$(sqlite3 docSet.dsidx "SELECT COUNT(*) FROM searchIndex WHERE path='$$path/index.html'")" = "x0" ]; then \ - echo $$page; \ - fi \ - done + @comm -23 <(echo '$(DOCSET_PAGE_PATHS)' | tr ' ' '\n') <(echo '$(DOCSET_INDEX_PATHS)' | tr ' ' '\n') # list paths in the docset index without a corresponding mkdocs page .PHONY: list_removed_paths list_removed_paths: docSet.dsidx - @for path in $$(sqlite3 docSet.dsidx "SELECT path FROM searchIndex"); do \ - page=$${path/\/index.html/.md}; \ - page_index=$${path/index.html/index.md}; \ - page_found=0; \ - for p in $(MKDOCS_PAGES); do \ - if [ "x$$p" = "x$$page" -o "x$$p" = "x$$page_index" ]; then \ - page_found=1; \ - fi \ - done; \ - if [ "x$$page_found" = "x0" ]; then \ - echo $$path; \ - fi \ - done + @comm -13 <(echo '$(DOCSET_PAGE_PATHS)' | tr ' ' '\n') <(echo '$(DOCSET_INDEX_PATHS)' | tr ' ' '\n') .PHONY: clean clean: diff --git a/docs/docset/README.md b/docs/docset/README.md index 79a778eb8..9d962a91c 100644 --- a/docs/docset/README.md +++ b/docs/docset/README.md @@ -7,10 +7,11 @@ documentation browsers like [Dash](https://kapeli.com/dash), [Velocity](https:// The docset can be created with ```sh -make nlohmann_json.docset +make JSON_for_Modern_C++.docset ``` -The generated folder `nlohmann_json.docset` can then be opened in the documentation browser. +The generated folder `JSON_for_Modern_C++.docset` can then be opened in the documentation browser. `make all` builds a +`JSON_for_Modern_C++.tgz` archive instead, and `make install_docset_zeal` installs the docset for Zeal directly. A recent version is also part of the [Dash user contributions](https://github.com/Kapeli/Dash-User-Contributions/tree/master/docsets/JSON_for_Modern_C%2B%2B). diff --git a/docs/docset/docSet.sql b/docs/docset/docSet.sql index 3a97e405f..6ec90dcfc 100644 --- a/docs/docset/docSet.sql +++ b/docs/docset/docSet.sql @@ -210,6 +210,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('JSON_CATCH_USER', 'Macro', 'a INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DIAGNOSTICS', 'Macro', 'api/macros/json_diagnostics/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DIAGNOSTIC_POSITIONS', 'Macro', 'api/macros/json_diagnostic_positions/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DISABLE_ENUM_SERIALIZATION', 'Macro', 'api/macros/json_disable_enum_serialization/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DISABLE_TUPLE_REFERENCE_CONVERSION', 'Macro', 'api/macros/json_disable_tuple_reference_conversion/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_CPP_11', 'Macro', 'api/macros/json_has_cpp_11/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_CPP_14', 'Macro', 'api/macros/json_has_cpp_11/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_CPP_17', 'Macro', 'api/macros/json_has_cpp_11/index.html'); diff --git a/docs/mkdocs/docs/api/basic_json/basic_json.md b/docs/mkdocs/docs/api/basic_json/basic_json.md index 3a7a6eb8b..962f3e164 100644 --- a/docs/mkdocs/docs/api/basic_json/basic_json.md +++ b/docs/mkdocs/docs/api/basic_json/basic_json.md @@ -159,6 +159,8 @@ basic_json(basic_json&& other) noexcept; - `CompatibleType` is not `basic_json` (to avoid hijacking copy/move constructors), - `CompatibleType` is not a different `basic_json` type (i.e. with different template arguments) - `CompatibleType` is not a `basic_json` nested type (e.g., `json_pointer`, `iterator`, etc.) + - if [`JSON_DISABLE_TUPLE_REFERENCE_CONVERSION`](../macros/json_disable_tuple_reference_conversion.md) is defined + to `1`: `CompatibleType` is not a one-element `std::tuple` holding a reference to `basic_json` - `json_serializer` (with `U = uncvref_t`) has a `to_json(basic_json_t&, CompatibleType&&)` method diff --git a/docs/mkdocs/docs/api/basic_json/number_integer_t.md b/docs/mkdocs/docs/api/basic_json/number_integer_t.md index 9a2ffab7f..9b1d7ae74 100644 --- a/docs/mkdocs/docs/api/basic_json/number_integer_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_integer_t.md @@ -51,7 +51,7 @@ range will yield over/underflow when used in a constructor. During deserializati will automatically be stored as [`number_unsigned_t`](number_unsigned_t.md) or [`number_float_t`](number_float_t.md). [RFC 8259](https://tools.ietf.org/html/rfc8259) further states: -> Note that when such software is used, numbers that are integers and are in the range $[-2^{53}+1, 2^{53}-1]$ are +> Note that when such software is used, numbers that are integers and are in the range [-253+1, 253-1] are > interoperable in the sense that implementations will agree exactly on their numeric values. As this range is a subrange of the exactly supported range [INT64_MIN, INT64_MAX], this class's integer type is diff --git a/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md b/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md index 674f7711d..f4799b2e9 100644 --- a/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md @@ -52,7 +52,7 @@ when used in a constructor. During deserialization, too large or small integer n as [`number_integer_t`](number_integer_t.md) or [`number_float_t`](number_float_t.md). [RFC 8259](https://tools.ietf.org/html/rfc8259) further states: -> Note that when such software is used, numbers that are integers and are in the range $[-2^{53}+1, 2^{53}-1]$ are +> Note that when such software is used, numbers that are integers and are in the range [-253+1, 253-1] are > interoperable in the sense that implementations will agree exactly on their numeric values. As this range is a subrange (when considered in conjunction with the `number_integer_t` type) of the exactly supported diff --git a/docs/mkdocs/docs/api/basic_json/parse_error.md b/docs/mkdocs/docs/api/basic_json/parse_error.md index 74e16f41a..c3f49ef13 100644 --- a/docs/mkdocs/docs/api/basic_json/parse_error.md +++ b/docs/mkdocs/docs/api/basic_json/parse_error.md @@ -54,7 +54,7 @@ classDiagram ## Notes -For an input with $n$ bytes, 1 is the index of the first character and $n+1$ is the index of the terminating null byte +For an input with n bytes, 1 is the index of the first character and n+1 is the index of the terminating null byte or the end of file. This also holds true when reading a byte vector for binary formats. ## Examples diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index 9d2636eb2..e917e0ae2 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -53,6 +53,7 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_BRACE_INIT_COPY_SEMANTICS**](json_brace_init_copy_semantics.md) - opt in to copy/move semantics for single-element brace initialization - [**JSON_DISABLE_ENUM_SERIALIZATION**](json_disable_enum_serialization.md) - switch off default serialization/deserialization functions for enums +- [**JSON_DISABLE_TUPLE_REFERENCE_CONVERSION**](json_disable_tuple_reference_conversion.md) - switch off conversion from a one-element tuple of a JSON reference - [**JSON_USE_IMPLICIT_CONVERSIONS**](json_use_implicit_conversions.md) - control implicit conversions ## Comparison behavior diff --git a/docs/mkdocs/docs/api/macros/json_disable_tuple_reference_conversion.md b/docs/mkdocs/docs/api/macros/json_disable_tuple_reference_conversion.md new file mode 100644 index 000000000..c759fd65c --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_disable_tuple_reference_conversion.md @@ -0,0 +1,114 @@ +# JSON_DISABLE_TUPLE_REFERENCE_CONVERSION + +```cpp +#define JSON_DISABLE_TUPLE_REFERENCE_CONVERSION /* value */ +``` + +When defined to `1`, a `basic_json` value can no longer be constructed from a one-element `std::tuple` whose element is +a reference to that `basic_json` type, such as `std::tuple`, `std::tuple`, or `std::tuple`. +These are the tuples created by `std::forward_as_tuple(j)`. + +## Default definition + +The default value is `0` (disabled — existing behavior is preserved). + +```cpp +#define JSON_DISABLE_TUPLE_REFERENCE_CONVERSION 0 +``` + +## Notes + +!!! note "Background" + + By default, `basic_json` can be constructed from any `std::tuple` whose elements can be converted to JSON; the result + is an array. This includes `std::tuple`, which becomes a one-element array. + + `std::tuple` only converts another tuple element by element if its element type cannot be constructed from the whole + source tuple. Because `json` *can* be constructed from `std::tuple`, `std::tuple` instead converts the whole + tuple into a single `json` value. This has two surprising effects: + + ```cpp + json j = true; + + // rejected by some standard libraries (e.g., libc++); with others, the + // reference binds to a temporary that is destroyed right away + std::tuple t1(std::forward_as_tuple(j)); + + // compiles, but std::get<0>(t2) is [true], not true + std::tuple t2(std::forward_as_tuple(j)); + ``` + + Enabling this macro removes the conversion, so both tuples are converted element by element: `std::get<0>(t1)` + refers to `j`, and `std::get<0>(t2)` is a copy of `j` (see [#2226](https://github.com/nlohmann/json/issues/2226)). + +!!! warning "Opt-in only" + + This macro must be defined **before** including ``. Defining it after the include has no effect. + +!!! note "Affected conversions" + + Only one-element tuples holding a reference to the **same** `basic_json` type are affected. Constructing a JSON value + from them no longer compiles: + + ```cpp + json j = true; + json a = std::forward_as_tuple(j); // error with the macro enabled + json b = json::array({j}); // use this instead: [true] + ``` + + Tuples holding a JSON value (`std::make_tuple(j)`), tuples with more than one element, and tuples holding references + to other types (including other `basic_json` specializations) are converted to arrays as before. + +!!! hint "CMake option" + + This behavior can also be controlled with the CMake option + [`JSON_DisableTupleReferenceConversion`](../../integration/cmake.md#json_disabletuplereferenceconversion) + (`OFF` by default) which defines `JSON_DISABLE_TUPLE_REFERENCE_CONVERSION` accordingly. + +## Examples + +??? example "Default behavior (macro not defined)" + + ```cpp + #include + + using json = nlohmann::json; + + int main() + { + json j = true; + + std::tuple t(std::forward_as_tuple(j)); + // std::get<0>(t) is [true] -- the whole tuple was converted + } + ``` + +??? example "Conversion disabled (macro defined to 1)" + + ```cpp + #define JSON_DISABLE_TUPLE_REFERENCE_CONVERSION 1 + #include + + using json = nlohmann::json; + + int main() + { + json j = true; + + std::tuple t(std::forward_as_tuple(j)); + // std::get<0>(t) is true -- a copy of j + + std::tuple r(std::forward_as_tuple(j)); + // std::get<0>(r) refers to j + } + ``` + +## See also + +- [**basic_json(CompatibleType&&)**](../basic_json/basic_json.md) - the affected constructor +- [:simple-cmake: JSON_DisableTupleReferenceConversion](../../integration/cmake.md#json_disabletuplereferenceconversion) - + CMake option to control the macro + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/operator_gtgt.md b/docs/mkdocs/docs/api/operator_gtgt.md index b68889af9..35b8c1016 100644 --- a/docs/mkdocs/docs/api/operator_gtgt.md +++ b/docs/mkdocs/docs/api/operator_gtgt.md @@ -18,6 +18,10 @@ Deserializes an input stream to a JSON value. the stream `i` +## Exception safety + +Strong guarantee: if an exception is thrown, there are no changes in `j`. + ## Exceptions - Throws [`parse_error.101`](../home/exceptions.md#jsonexceptionparse_error101) in case of an unexpected token, or if @@ -125,3 +129,5 @@ being read. the stream; planned to become the default in version 4.0.0. - Fixed a null pointer dereference for an `std::istream` without a stream buffer (now throws `parse_error.101`), and a crash (`std::terminate`) when `i` has `eofbit` in its exception mask, in version 3.13.0. +- Changed to the strong exception safety guarantee in version 3.13.0: `j` is no longer left with a partially parsed + value if parsing throws. diff --git a/docs/mkdocs/docs/examples/parse__iterator_pair.link b/docs/mkdocs/docs/examples/parse__iterator_pair.link deleted file mode 100644 index f464e54c8..000000000 --- a/docs/mkdocs/docs/examples/parse__iterator_pair.link +++ /dev/null @@ -1 +0,0 @@ -online \ No newline at end of file diff --git a/docs/mkdocs/docs/examples/parse__pointers.link b/docs/mkdocs/docs/examples/parse__pointers.link deleted file mode 100644 index 9a93ef1c5..000000000 --- a/docs/mkdocs/docs/examples/parse__pointers.link +++ /dev/null @@ -1 +0,0 @@ -online \ No newline at end of file diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index a0c84edaf..cd73b07e2 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -61,7 +61,7 @@ The library uses the following mapping from JSON values types to BJData types ac The following values can **not** be converted to a BJData value: - - strings with more than 18446744073709551615 bytes, i.e., $2^{64}-1$ bytes (theoretical) + - strings with more than 18446744073709551615 bytes, i.e., 264-1 bytes (theoretical) !!! info "Unused BJData markers" diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index e4a628c3e..a4479712f 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -83,6 +83,13 @@ When defined, default parse and serialize functions for enums are excluded and h See [full documentation of `JSON_DISABLE_ENUM_SERIALIZATION`](../api/macros/json_disable_enum_serialization.md). +## `JSON_DISABLE_TUPLE_REFERENCE_CONVERSION` + +When defined to `1`, a JSON value can no longer be created from a one-element `std::tuple` holding a reference to a JSON +value, such as the result of `std::forward_as_tuple(j)`. This lets `std::tuple` convert such tuples element-wise. + +See [full documentation of `JSON_DISABLE_TUPLE_REFERENCE_CONVERSION`](../api/macros/json_disable_tuple_reference_conversion.md). + ## `JSON_NO_AUTOMATIC_UDLS` When defined, `` does not include `` with the user-defined string literals diff --git a/docs/mkdocs/docs/features/types/index.md b/docs/mkdocs/docs/features/types/index.md index e6078b825..354acd5ac 100644 --- a/docs/mkdocs/docs/features/types/index.md +++ b/docs/mkdocs/docs/features/types/index.md @@ -273,7 +273,7 @@ When the default type is used, the maximal unsigned integer number that can be s [RFC 8259](https://tools.ietf.org/html/rfc8259) further states: -> Note that when such software is used, numbers that are integers and are in the range $[-2^{53}+1, 2^{53}-1]$ are interoperable in the sense that implementations will agree exactly on their numeric values. +> Note that when such software is used, numbers that are integers and are in the range [-253+1, 253-1] are interoperable in the sense that implementations will agree exactly on their numeric values. As this range is a subrange of the exactly supported range [`INT64_MIN`, `INT64_MAX`], this class's integer type is interoperable. diff --git a/docs/mkdocs/docs/features/types/number_handling.md b/docs/mkdocs/docs/features/types/number_handling.md index cf37b044a..ef43054e5 100644 --- a/docs/mkdocs/docs/features/types/number_handling.md +++ b/docs/mkdocs/docs/features/types/number_handling.md @@ -48,7 +48,7 @@ On number interoperability, the following remarks are made: for numeric magnitude and precision than is widely available. Note that when such software is used, numbers that are integers and - are in the range $[-2^{53}+1, 2^{53}-1]$ are interoperable in the + are in the range [-253+1, 253-1] are interoperable in the sense that implementations will agree exactly on their numeric values. @@ -95,9 +95,9 @@ This is the same behavior as the code `#!c double x = 3.141592653589793238462643 !!! success "Interoperability" - - The library is interoperable with respect to the specification, because its supported range $[-2^{63}, 2^{64}-1]$ is - larger than the described range $[-2^{53}+1, 2^{53}-1]$. - - All integers outside the range $[-2^{63}, 2^{64}-1]$, as well as floating-point numbers are stored as `double`. + - The library is interoperable with respect to the specification, because its supported range [-263, 264-1] is + larger than the described range [-253+1, 253-1]. + - All integers outside the range [-263, 264-1], as well as floating-point numbers are stored as `double`. This also concurs with the specification above. ### Zeros diff --git a/docs/mkdocs/docs/features/types/template_parameters.md b/docs/mkdocs/docs/features/types/template_parameters.md index 8ed23e156..5b528fd79 100644 --- a/docs/mkdocs/docs/features/types/template_parameters.md +++ b/docs/mkdocs/docs/features/types/template_parameters.md @@ -389,6 +389,7 @@ using array_t = ArrayType>; | Functionality | Additional requirement | |-----------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| | [`diff`](../../api/basic_json/diff.md), [`items`](../../api/basic_json/items.md), [`std::hash`](../../api/basic_json/std_hash.md) | conversion of a `#!cpp std::size_t` to `StringType`: either assignability from the result of `#!cpp std::to_string`, or an ADL overload `#!cpp void int_to_string(StringType&, std::size_t)` | +| [`operator/(std::size_t)`](../../api/json_pointer/operator_slash.md) | the same conversion of a `#!cpp std::size_t` to `StringType` as `diff`, `items`, and `std::hash` above | | [`std::hash`](../../api/basic_json/std_hash.md) | additionally a specialization of `#!cpp std::hash` | | [`to_bson`](../../api/basic_json/to_bson.md) | `find(value_type)` and `npos` | | [`parse`](../../api/basic_json/parse.md) from a `string_t` | the input adapters must accept it; otherwise pass a character range | diff --git a/docs/mkdocs/docs/integration/cmake.md b/docs/mkdocs/docs/integration/cmake.md index 71512cbd5..4160584d5 100644 --- a/docs/mkdocs/docs/integration/cmake.md +++ b/docs/mkdocs/docs/integration/cmake.md @@ -169,6 +169,12 @@ Enable position diagnostics by defining macro [`JSON_DIAGNOSTIC_POSITIONS`](../a Disable default `enum` serialization by defining the macro [`JSON_DISABLE_ENUM_SERIALIZATION`](../api/macros/json_disable_enum_serialization.md). This option is `OFF` by default. +### `JSON_DisableTupleReferenceConversion` + +Disable the conversion from a one-element `std::tuple` holding a reference to a JSON value by defining the macro +[`JSON_DISABLE_TUPLE_REFERENCE_CONVERSION`](../api/macros/json_disable_tuple_reference_conversion.md). This option is +`OFF` by default. + ### `JSON_FastTests` Skip expensive/slow test suites. This option is `OFF` by default. Depends on `JSON_BuildTests`. diff --git a/docs/mkdocs/docs/integration/package_managers.md b/docs/mkdocs/docs/integration/package_managers.md index 792a0fa5a..28f95e584 100644 --- a/docs/mkdocs/docs/integration/package_managers.md +++ b/docs/mkdocs/docs/integration/package_managers.md @@ -678,11 +678,11 @@ to install the [nlohmann-json](https://ports.macports.org/port/nlohmann-json/) p 1. Create the following files: ```cpp title="example.cpp" - --8<-- "integration/homebrew/example.cpp" + --8<-- "integration/macports/example.cpp" ``` ```cmake title="CMakeLists.txt" - --8<-- "integration/homebrew/CMakeLists.txt" + --8<-- "integration/macports/CMakeLists.txt" ``` 2. Install the package: diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 71f119db9..0e1a1da46 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -288,6 +288,7 @@ nav: - 'JSON_DIAGNOSTICS': api/macros/json_diagnostics.md - 'JSON_DIAGNOSTIC_POSITIONS': api/macros/json_diagnostic_positions.md - 'JSON_DISABLE_ENUM_SERIALIZATION': api/macros/json_disable_enum_serialization.md + - 'JSON_DISABLE_TUPLE_REFERENCE_CONVERSION': api/macros/json_disable_tuple_reference_conversion.md - 'JSON_HAS_CPP_11, JSON_HAS_CPP_14, JSON_HAS_CPP_17, JSON_HAS_CPP_20': api/macros/json_has_cpp_11.md - 'JSON_HAS_EXPERIMENTAL_FILESYSTEM, JSON_HAS_FILESYSTEM': api/macros/json_has_filesystem.md - 'JSON_HAS_RANGES': api/macros/json_has_ranges.md @@ -351,7 +352,6 @@ markdown_extensions: - toc: permalink: true - md_in_html - - pymdownx.arithmatex - pymdownx.betterem: smart_enable: all - pymdownx.caret @@ -443,6 +443,3 @@ plugins: extra_css: - css/custom.css - -extra_javascript: - - https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.0/MathJax.js?config=TeX-MML-AM_CHTML diff --git a/docs/mkdocs/scripts/check_structure.py b/docs/mkdocs/scripts/check_structure.py index c8d637d06..78ad0d9fa 100755 --- a/docs/mkdocs/scripts/check_structure.py +++ b/docs/mkdocs/scripts/check_structure.py @@ -79,9 +79,9 @@ def check_structure() -> None: report("whitespace/line_length", f"{file}:{lineno+1} ({current_section})", f"line is too long ({len(line)} vs. 160 chars)") # sections in `` comments are treated as present - if line.startswith("") + nolint_match = re.match(r"", line) + if nolint_match: + current_section = nolint_match.group(1) existing_sections.append(current_section) # check if sections are correct @@ -97,7 +97,7 @@ def check_structure() -> None: if len(unexpected): report("style/numbering", f"{file}:{lineno} ({current_section})", f'unexpected overloads: {", ".join([f"({x})" for x in unexpected])}') - current_section = line.strip("## ") + current_section = line[3:] existing_sections.append(current_section) if current_section in expected_sections: @@ -141,7 +141,7 @@ def check_structure() -> None: # check that non-example admonitions have titles untitled_admonition = re.match(r"^(\?\?\?|!!!) ([^ ]+)$", line) if untitled_admonition and untitled_admonition.group(2) != "example": - report("style/admonition_title", f"{file}:{lineno} ({current_section})", f'"{untitled_admonition.group(2)}" admonitions should have a title') + report("style/admonition_title", f"{file}:{lineno+1} ({current_section})", f'"{untitled_admonition.group(2)}" admonitions should have a title') previous_line = line diff --git a/include/nlohmann/detail/conversions/to_json.hpp b/include/nlohmann/detail/conversions/to_json.hpp index fb48e4f82..92574c8ce 100644 --- a/include/nlohmann/detail/conversions/to_json.hpp +++ b/include/nlohmann/detail/conversions/to_json.hpp @@ -471,11 +471,13 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = { std::get(t)... }; } -#if JSON_BRACE_INIT_COPY_SEMANTICS -// JSON_BRACE_INIT_COPY_SEMANTICS makes a one-element braced list copy its -// element instead of wrapping it, which would serialize std::tuple{5} as 5 -// rather than [5]. Build what the default deduction builds instead: an object -// if the element is a [string, value] pair, a one-element array otherwise. +// A one-element braced list does not reliably wrap its element: with +// JSON_BRACE_INIT_COPY_SEMANTICS it copies it, which would serialize +// std::tuple{5} as 5 rather than [5], and some compilers (e.g., Apple clang +// 15 and 16) copy an element that is itself a basic_json even without it, so +// std::tuple{true} became true rather than [true]. Build what the default +// deduction builds instead: an object if the element is a [string, value] pair, +// a one-element array otherwise. template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence<0> /*unused*/) { @@ -493,7 +495,6 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = BasicJsonType::array({std::move(element)}); } } -#endif template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& /*unused*/, index_sequence<> /*unused*/) diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index 2f8adbc58..413ccb845 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -8,7 +8,6 @@ #pragma once -#include // generate_n #include // array #include // ldexp #include // size_t @@ -131,7 +130,8 @@ class binary_reader ~binary_reader() = default; /*! - @param[in] format the binary format to parse + @brief parse in the format the constructor was given + @param[in] sax_ a SAX event processor @param[in] strict whether to expect the input to be consumed completed @param[in] tag_handler how to treat CBOR tags @@ -139,9 +139,8 @@ class binary_reader @return whether parsing was successful: the input was read without errors, and no SAX event returned false */ - JSON_HEDLEY_NON_NULL(3) - bool sax_parse(const input_format_t format, - json_sax_t* sax_, + JSON_HEDLEY_NON_NULL(2) + bool sax_parse(json_sax_t* sax_, const bool strict = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { @@ -155,14 +154,14 @@ class binary_reader ndarray_open = 0; bool result = false; - switch (format) + switch (input_format) { case input_format_t::bson: result = parse_bson_internal(); break; case input_format_t::cbor: - result = parse_cbor_internal(true, tag_handler); + result = parse_cbor_internal(tag_handler); break; case input_format_t::msgpack: @@ -290,6 +289,22 @@ class binary_reader return enter_container(/*is_object*/true, len, type_marker); } + /*! + @brief close the innermost open array or object + + Pops the container opened by the matching @ref enter_container call and + emits the SAX end event. Every format-specific driver otherwise repeated + the same pop-then-dispatch sequence at its own close site. + + @return whether the SAX parser accepted the end event + */ + bool leave_container() + { + const bool is_object = container_stack.back().is_object; + container_stack.pop_back(); + return is_object ? sax->end_object() : sax->end_array(); + } + ////////// // BSON // ////////// @@ -548,8 +563,8 @@ class binary_reader if (element_type == 0) // end of the innermost document { // a copy, not a reference: it must stay valid across the - // pop_back() below, which destroys the container_stack - // element it would otherwise alias + // pop_back() inside leave_container() below, which destroys + // the container_stack element it would otherwise alias const container_frame top = container_stack.back(); if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(top.start_position, top.declared_size))) @@ -557,8 +572,7 @@ class binary_reader return false; } - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -1057,29 +1071,13 @@ class binary_reader return enter_array(conditional_static_cast(static_cast(current) & 0x1Fu)); case 0x98: // array (one-byte uint8_t for n follows) - { - std::uint8_t len{}; - return get_number(input_format_t::cbor, len) && enter_array(static_cast(len)); - } - case 0x99: // array (two-byte uint16_t for n follow) - { - std::uint16_t len{}; - return get_number(input_format_t::cbor, len) && enter_array(static_cast(len)); - } - case 0x9A: // array (four-byte uint32_t for n follow) - { - std::uint32_t len{}; - std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size); - } - case 0x9B: // array (eight-byte uint64_t for n follow) { std::uint64_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size); + return get_cbor_argument(len) && get_cbor_container_size(len, size, "array") && enter_array(size); } case 0x9F: // array (indefinite length) @@ -1113,35 +1111,19 @@ class binary_reader return enter_object(conditional_static_cast(static_cast(current) & 0x1Fu)); case 0xB8: // map (one-byte uint8_t for n follows) - { - std::uint8_t len{}; - return get_number(input_format_t::cbor, len) && enter_object(static_cast(len)); - } - case 0xB9: // map (two-byte uint16_t for n follow) - { - std::uint16_t len{}; - return get_number(input_format_t::cbor, len) && enter_object(static_cast(len)); - } - case 0xBA: // map (four-byte uint32_t for n follow) - { - std::uint32_t len{}; - std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size); - } - case 0xBB: // map (eight-byte uint64_t for n follow) { std::uint64_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size); + return get_cbor_argument(len) && get_cbor_container_size(len, size, "map") && enter_object(size); } case 0xBF: // map (indefinite length) return enter_object(detail::unknown_size()); - case 0xC0: // tagged item + case 0xC0: // tagged item (tag value 0-23, in the head itself) case 0xC1: case 0xC2: case 0xC3: @@ -1165,6 +1147,27 @@ class binary_reader case 0xD5: case 0xD6: case 0xD7: + { + if (tag_handler == cbor_tag_handler_t::error) + { + auto last_token = get_token_string(); + if (!report_repairable_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr))) + { + return false; + } + // when recovering, the tag is ignored, as RFC 8949, + // Section 6.1 suggests for converting to JSON + } + + // ignore and store: the tag value is already in the head, so + // there is nothing left to read here; the tagged value that + // follows is read by the loop in parse_cbor_internal() rather + // than by recursing here + tag_pending = true; + return true; + } + case 0xD8: // tagged item (1 byte follows) case 0xD9: // tagged item (2 bytes follow) case 0xDA: // tagged item (4 bytes follow) @@ -1187,47 +1190,11 @@ class binary_reader case cbor_tag_handler_t::ignore: { - // ignore binary subtype - switch (current) + // ignore the tag's binary subtype argument + std::uint64_t subtype_to_ignore{}; + if (!get_cbor_argument(subtype_to_ignore)) { - case 0xD8: - { - std::uint8_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - case 0xD9: - { - std::uint16_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - case 0xDA: - { - std::uint32_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - case 0xDB: - { - std::uint64_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - default: - break; + return false; } // the tagged value follows; it is read by the loop in // parse_cbor_internal() rather than by recursing here @@ -1237,57 +1204,15 @@ class binary_reader case cbor_tag_handler_t::store: { - binary_t b; // use binary subtype and store in a binary container - switch (current) + std::uint64_t subtype{}; + if (!get_cbor_argument(subtype)) { - case 0xD8: - { - std::uint8_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - case 0xD9: - { - std::uint16_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - case 0xDA: - { - std::uint32_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - case 0xDB: - { - std::uint64_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - default: - { - // as above, the tagged value is read by the caller - tag_pending = true; - return true; - } + return false; } + binary_t b; + b.set_subtype(detail::conditional_static_cast(subtype)); + get(); // a byte string (the heads accepted by get_cbor_binary) keeps the tag as subtype if ((current >= 0x40 && current <= 0x5B) || current == 0x5F) @@ -1318,52 +1243,7 @@ class binary_reader return sax->null(); case 0xF9: // Half-Precision Float (two-byte IEEE 754) - { - const auto byte1_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number"))) - { - return false; - } - const auto byte2_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number"))) - { - return false; - } - - const auto byte1 = static_cast(byte1_raw); - const auto byte2 = static_cast(byte2_raw); - - // Code from RFC 8949, Appendix D, Figure 3: - // As half-precision floating-point numbers were only added - // to IEEE 754 in 2008, today's programming platforms often - // still only have limited support for them. It is very - // easy to include at least decoding support for them even - // without such support. An example of a small decoder for - // half-precision floating-point numbers in the C language - // is shown in Fig. 3. - const auto half = static_cast((byte1 << 8u) + byte2); - const double val = [&half] - { - const int exp = (half >> 10u) & 0x1Fu; - const unsigned int mant = half & 0x3FFu; - JSON_ASSERT(exp <= 31); - JSON_ASSERT(mant <= 1023); - switch (exp) - { - case 0: - return std::ldexp(mant, -24); - case 31: - return (mant == 0) - ? std::numeric_limits::infinity() - : std::numeric_limits::quiet_NaN(); - default: - return std::ldexp(mant + 1024, exp - 25); - } - }(); - return sax->number_float((half & 0x8000u) != 0 - ? static_cast(-val) - : static_cast(val), ""); - } + return get_half_float(input_format_t::cbor, false); case 0xFA: // Single-Precision Float (four-byte IEEE 754) { @@ -1759,6 +1639,73 @@ class binary_reader } } + /*! + @brief read a CBOR argument (additional information 24-27) of the width + @ref current announces + + The lower 5 bits of @a current (0x18-0x1B) select a 1/2/4/8-byte + big-endian unsigned integer that follows the head byte; this is shared by + every major type that uses this encoding (unsigned/negative integers, + strings, arrays, maps, tags). Reading always goes through @ref get_number, + so EOF is reported the same way as before this helper existed. + + @param[out] value the decoded argument + @return whether reading succeeded + */ + bool get_cbor_argument(std::uint64_t& value) + { + switch (current & 0x1F) + { + case 0x18: // 1 byte + { + std::uint8_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + case 0x19: // 2 bytes + { + std::uint16_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + case 0x1A: // 4 bytes + { + std::uint32_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + case 0x1B: // 8 bytes + { + std::uint64_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + return false; // LCOV_EXCL_LINE + } + } + /*! @brief narrow a definite CBOR array/map length to std::size_t @@ -1791,19 +1738,15 @@ class binary_reader enclosing container after each element, so that the nesting depth of the input costs heap rather than native stack (see #5104). - @param[in] get_char whether a new character should be retrieved from the - input (true) or whether the last read character - @a current should be considered instead @param[in] tag_handler how CBOR tags should be treated @return whether reading the value succeeded */ - bool parse_cbor_internal(const bool get_char, - const cbor_tag_handler_t tag_handler) + bool parse_cbor_internal(const cbor_tag_handler_t tag_handler) { // whether the next value starts at a fresh byte or at the one already // read into `current` - bool fetch = get_char; + bool fetch = true; // the key currently being read; hoisted out of the loop so that its // capacity is reused across elements and across nesting levels @@ -1846,8 +1789,7 @@ class binary_reader if (at_end) { - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -2035,9 +1977,6 @@ class binary_reader // MsgPack // ///////////// - /*! - @return whether a valid MessagePack value was passed to the SAX parser - */ /*! @brief read one MessagePack value @@ -2741,8 +2680,7 @@ class binary_reader if (container_stack.back().remaining == 0) { - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -2936,20 +2874,16 @@ class binary_reader //////////// /*! - @param[in] get_char whether a new character should be retrieved from the - input (true, default) or whether the last read - character should be considered instead - @return whether a valid UBJSON value was passed to the SAX parser */ - bool parse_ubjson_internal(const bool get_char = true) + bool parse_ubjson_internal() { // the key currently being read; hoisted out of the loop so that its // capacity is reused across elements and across nesting levels string_t key; // the type marker of the value to read next - char_int_type prefix = get_char ? get_ignore_noop() : current; + char_int_type prefix = get_ignore_noop(); while (true) { @@ -3024,8 +2958,7 @@ class binary_reader break; } - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -3233,6 +3166,42 @@ class binary_reader return true; } + /*! + @brief read a UBJSON/BJData optimized-container count of a signed marker + type ('i', 'I', 'l', 'L') and narrow it to std::size_t + + Every signed count marker rejects a negative value the same way (error + 113); the value_in_range_of check additionally needed for 'L' is only + ever live when @a SignedType is std::int64_t on a target where + std::size_t is narrower (e.g. 32-bit), since 'i'/'I'/'l' can never exceed + std::size_t there. + + @tparam SignedType std::int8_t, std::int16_t, std::int32_t or std::int64_t + @param[out] result the count narrowed to std::size_t + @return whether reading and validating succeeded + */ + template + bool get_ubjson_signed_count(std::size_t& result) + { + SignedType number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) + { + return false; + } + if (JSON_HEDLEY_UNLIKELY(number < 0)) + { + return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read, + exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); + } + if (JSON_HEDLEY_UNLIKELY(!value_in_range_of(number))) + { + return report_error(chars_read, get_token_string(), out_of_range::create(408, + exception_message(input_format, "integer value overflow", "size"), nullptr)); + } + result = static_cast(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char + return true; + } + /*! @param[out] result determined size @param[in,out] is_ndarray for input, `true` means already inside an ndarray vector @@ -3265,73 +3234,16 @@ class binary_reader } case 'i': - { - std::int8_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - result = static_cast(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char - return true; - } + return get_ubjson_signed_count(result); case 'I': - { - std::int16_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - result = static_cast(number); - return true; - } + return get_ubjson_signed_count(result); case 'l': - { - std::int32_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - result = static_cast(number); - return true; - } + return get_ubjson_signed_count(result); case 'L': - { - std::int64_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - if (!value_in_range_of(number)) - { - return report_error(chars_read, get_token_string(), out_of_range::create(408, - exception_message(input_format, "integer value overflow", "size"), nullptr)); - } - result = static_cast(number); - return true; - } + return get_ubjson_signed_count(result); case 'u': { @@ -3423,16 +3335,23 @@ class binary_reader result = 1; for (auto i : dim) { - // Pre-multiplication overflow check: if i > 0 and result > SIZE_MAX/i, then result*i would overflow. - // This check must happen before multiplication since overflow detection after the fact is unreliable - // as modular arithmetic can produce any value, not just 0 or SIZE_MAX. - if (JSON_HEDLEY_UNLIKELY(i > 0 && result > (std::numeric_limits::max)() / i)) + // Pre-multiplication overflow check: since the loop above + // already rejected any zero dimension, i is always > 0 + // here, so result > SIZE_MAX/i means result*i would + // overflow. This check must happen before multiplication + // since overflow detection after the fact is unreliable, + // as modular arithmetic can produce any value, not just 0 + // or SIZE_MAX. + if (JSON_HEDLEY_UNLIKELY(result > (std::numeric_limits::max)() / i)) { return report_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr)); } result *= i; - // Additional post-multiplication check to catch any edge cases the pre-check might miss - if (result == 0 || result == npos) + // the pre-check above already rules out result becoming 0 + // by overflow; the only value it cannot rule out is an + // exact match with npos, the sentinel reserved for an + // unknown-size container (see get_ubjson_size_type()) + if (result == npos) { return report_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr)); } @@ -3494,7 +3413,7 @@ class binary_reader { result.second = get(); // must not ignore 'N', because 'N' maybe the type if (input_format == input_format_t::bjdata - && JSON_HEDLEY_UNLIKELY(std::binary_search(bjd_optimized_type_markers.begin(), bjd_optimized_type_markers.end(), result.second))) + && JSON_HEDLEY_UNLIKELY(is_bjd_excluded_optimized_type(result.second))) { auto last_token = get_token_string(); return report_error(chars_read, last_token, parse_error::create(112, chars_read, @@ -3638,50 +3557,7 @@ class binary_reader { break; } - const auto byte1_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number"))) - { - return false; - } - const auto byte2_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number"))) - { - return false; - } - - const auto byte1 = static_cast(byte1_raw); - const auto byte2 = static_cast(byte2_raw); - - // Code from RFC 8949, Appendix D, Figure 3: - // As half-precision floating-point numbers were only added - // to IEEE 754 in 2008, today's programming platforms often - // still only have limited support for them. It is very - // easy to include at least decoding support for them even - // without such support. An example of a small decoder for - // half-precision floating-point numbers in the C language - // is shown in Fig. 3. - const auto half = static_cast((byte2 << 8u) + byte1); - const double val = [&half] - { - const int exp = (half >> 10u) & 0x1Fu; - const unsigned int mant = half & 0x3FFu; - JSON_ASSERT(exp <= 31); - JSON_ASSERT(mant <= 1023); - switch (exp) - { - case 0: - return std::ldexp(mant, -24); - case 31: - return (mant == 0) - ? std::numeric_limits::infinity() - : std::numeric_limits::quiet_NaN(); - default: - return std::ldexp(mant + 1024, exp - 25); - } - }(); - return sax->number_float((half & 0x8000u) != 0 - ? static_cast(-val) - : static_cast(val), ""); + return get_half_float(input_format, true); } case 'd': @@ -3762,19 +3638,16 @@ class binary_reader if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0) { size_and_type.second &= ~(static_cast(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker - auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t) - { - return p.first < t; - }); + const char* type_name = bjd_type_name(size_and_type.second); string_t key = "_ArrayType_"; - if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second)) + if (JSON_HEDLEY_UNLIKELY(type_name == nullptr)) { auto last_token = get_token_string(); return report_error(chars_read, last_token, parse_error::create(112, chars_read, exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr)); } - string_t type = it->second; // sax->string() takes a reference + string_t type = type_name; // sax->string() takes a reference if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type))) { return false; @@ -4143,8 +4016,7 @@ class binary_reader if (at_end) { - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -4808,6 +4680,68 @@ class binary_reader return true; } + /*! + @brief read and decode an IEEE 754 half-precision (16-bit) float + + Used by CBOR (big endian) and BJData (little endian); the two formats + only differ in the byte order of the two bytes that make up the half. + + @param[in] format the current format (for diagnostics) + @param[in] little_endian whether the two bytes are little endian (BJData) + or big endian (CBOR) + + @return whether reading and decoding succeeded + */ + bool get_half_float(const input_format_t format, const bool little_endian) + { + const auto byte1_raw = get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number"))) + { + return false; + } + const auto byte2_raw = get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number"))) + { + return false; + } + + const auto byte1 = static_cast(byte1_raw); + const auto byte2 = static_cast(byte2_raw); + + // Code from RFC 8949, Appendix D, Figure 3: + // As half-precision floating-point numbers were only added + // to IEEE 754 in 2008, today's programming platforms often + // still only have limited support for them. It is very + // easy to include at least decoding support for them even + // without such support. An example of a small decoder for + // half-precision floating-point numbers in the C language + // is shown in Fig. 3. + const auto half = little_endian + ? static_cast((byte2 << 8u) + byte1) + : static_cast((byte1 << 8u) + byte2); + const double val = [&half] + { + const int exp = (half >> 10u) & 0x1Fu; + const unsigned int mant = half & 0x3FFu; + JSON_ASSERT(exp <= 31); + JSON_ASSERT(mant <= 1023); + switch (exp) + { + case 0: + return std::ldexp(mant, -24); + case 31: + return (mant == 0) + ? std::numeric_limits::infinity() + : std::numeric_limits::quiet_NaN(); + default: + return std::ldexp(mant + 1024, exp - 25); + } + }(); + return sax->number_float((half & 0x8000u) != 0 + ? static_cast(-val) + : static_cast(val), ""); + } + /*! @brief create a string by reading characters from the input @@ -5353,38 +5287,61 @@ class binary_reader /// open: none, its object, or its object and an array inside it std::uint8_t ndarray_open = 0; - // excluded markers in bjdata optimized type -#define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \ - make_array('F', 'H', 'N', 'S', 'T', 'Z', '[', '{') - -#define JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ \ - make_array( \ - bjd_type{'B', "byte"}, \ - bjd_type{'C', "char"}, \ - bjd_type{'D', "double"}, \ - bjd_type{'I', "int16"}, \ - bjd_type{'L', "int64"}, \ - bjd_type{'M', "uint64"}, \ - bjd_type{'U', "uint8"}, \ - bjd_type{'d', "single"}, \ - bjd_type{'i', "int8"}, \ - bjd_type{'l', "int32"}, \ - bjd_type{'m', "uint32"}, \ - bjd_type{'u', "uint16"}) - JSON_PRIVATE_UNLESS_TESTED: - // lookup tables - // NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes) - const decltype(JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_) bjd_optimized_type_markers = - JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_; + /*! + @brief whether @a marker is excluded from BJData's optimized ND-array types + @return whether @a marker is one of 'F', 'H', 'N', 'S', 'T', 'Z', '[', '{' - using bjd_type = std::pair; - // NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes) - const decltype(JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_) bjd_types_map = - JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_; + Mirrors binary_writer's @ref binary_writer::is_bjdata_excluded_type_marker + "is_bjdata_excluded_type_marker()`, which encodes the same list the other + way; keep the two in sync. + */ + static constexpr bool is_bjd_excluded_optimized_type(const char_int_type marker) noexcept + { + return marker == '[' || marker == '{' || marker == 'S' || marker == 'H' + || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; + } -#undef JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ -#undef JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ + /*! + @brief look up the ND-array element type name for a BJData dtype marker + @return the type name ("uint8", "int8", ...), or nullptr if @a marker does + not name a known dtype + + A C++11 `constexpr` function cannot contain a `switch`, so this is a + plain (non-constexpr) switch instead. + */ + static const char* bjd_type_name(const char_int_type marker) + { + switch (marker) + { + case 'B': + return "byte"; + case 'C': + return "char"; + case 'D': + return "double"; + case 'I': + return "int16"; + case 'L': + return "int64"; + case 'M': + return "uint64"; + case 'U': + return "uint8"; + case 'd': + return "single"; + case 'i': + return "int8"; + case 'l': + return "int32"; + case 'm': + return "uint32"; + case 'u': + return "uint16"; + default: + return nullptr; + } + } }; #ifndef JSON_HAS_CPP_17 diff --git a/include/nlohmann/detail/json_pointer.hpp b/include/nlohmann/detail/json_pointer.hpp index 0b9f9651a..78074988e 100644 --- a/include/nlohmann/detail/json_pointer.hpp +++ b/include/nlohmann/detail/json_pointer.hpp @@ -26,6 +26,7 @@ #include #include #include +#include #include NLOHMANN_JSON_NAMESPACE_BEGIN @@ -116,7 +117,7 @@ class json_pointer /// @sa https://json.nlohmann.me/api/json_pointer/operator_slasheq/ json_pointer& operator/=(std::size_t array_idx) { - return *this /= std::to_string(array_idx); + return *this /= detail::to_string(array_idx); } /// @brief create a new JSON pointer by appending the right JSON pointer at the end of the left JSON pointer @@ -752,7 +753,7 @@ class json_pointer // would throw out_of_range.404 -- contains() must not throw (see #5395) return false; } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !("0" <= reference_token && reference_token <= "9"))) + if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !('0' <= reference_token[0] && reference_token[0] <= '9'))) { // invalid char return false; @@ -780,7 +781,7 @@ class json_pointer // not throw (see #5395), so such a reference token is treated as "not found" errno = 0; // strtoull() does not reset errno on success char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.c_str(), &p_end, 10); // NOLINT(runtime/int) + const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) { diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index 0d524c710..7c5ae3089 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -359,6 +359,7 @@ void templated_json_throw(ExceptionType exception) // Macros to simplify conversion from/to types +// NLOHMANN_JSON_EXPAND to NLOHMANN_JSON_DOUBLE_PASTE63 are generated by tools/macro_builder (see its README.md) #define NLOHMANN_JSON_EXPAND( x ) x #define NLOHMANN_JSON_GET_MACRO(_1, _2, _3, _4, _5, _6, _7, _8, _9, _10, _11, _12, _13, _14, _15, _16, _17, _18, _19, _20, _21, _22, _23, _24, _25, _26, _27, _28, _29, _30, _31, _32, _33, _34, _35, _36, _37, _38, _39, _40, _41, _42, _43, _44, _45, _46, _47, _48, _49, _50, _51, _52, _53, _54, _55, _56, _57, _58, _59, _60, _61, _62, _63, _64, NAME,...) NAME #define NLOHMANN_JSON_PASTE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \ @@ -621,6 +622,10 @@ void templated_json_throw(ExceptionType exception) // arguments, so dispatching on Type,BaseType,member... directly would run out // one slot early and cap the derived-type macros at 62 members instead of the // 63 that NLOHMANN_JSON_PASTE supports. +// +// The slot table below (down to the closing NLOHMANN_JSON_TYPE_BODY_SENTINEL)) +// is generated by tools/macro_builder (see its README.md; run with the +// "type_body" argument). #define NLOHMANN_JSON_TYPE_BODY(Prefix, ...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \ Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ @@ -915,3 +920,7 @@ void templated_json_throw(ExceptionType exception) #ifndef JSON_DISABLE_ENUM_SERIALIZATION #define JSON_DISABLE_ENUM_SERIALIZATION 0 #endif + +#ifndef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION + #define JSON_DISABLE_TUPLE_REFERENCE_CONVERSION 0 +#endif diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index d2675d3f1..d6aa831d7 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -25,6 +25,7 @@ #undef JSON_INLINE_VARIABLE #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION +#undef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH diff --git a/include/nlohmann/detail/meta/cpp_future.hpp b/include/nlohmann/detail/meta/cpp_future.hpp index e109feacb..3e012b341 100644 --- a/include/nlohmann/detail/meta/cpp_future.hpp +++ b/include/nlohmann/detail/meta/cpp_future.hpp @@ -9,7 +9,6 @@ #pragma once -#include // array #include // size_t #include // conditional, enable_if, false_type, integral_constant, is_constructible, is_integral, is_same, remove_cv, remove_reference, true_type #include // index_sequence, make_index_sequence, index_sequence_for @@ -161,11 +160,5 @@ struct static_const constexpr T static_const::value; #endif -template -constexpr std::array make_array(Args&& ... args) -{ - return std::array {{static_cast(std::forward(args))...}}; -} - } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/meta/type_traits.hpp b/include/nlohmann/detail/meta/type_traits.hpp index 36573bf6f..96b70a774 100644 --- a/include/nlohmann/detail/meta/type_traits.hpp +++ b/include/nlohmann/detail/meta/type_traits.hpp @@ -636,6 +636,18 @@ template struct is_compatible_type : is_compatible_type_impl {}; +// a one-element std::tuple holding a reference to BasicJsonType, as created by +// std::forward_as_tuple(j); see JSON_DISABLE_TUPLE_REFERENCE_CONVERSION +template +struct is_basic_json_reference_tuple : std::false_type {}; + +template +struct is_basic_json_reference_tuple> +{ + static constexpr bool value = + std::is_reference::value && std::is_same, BasicJsonType>::value; +}; + template struct is_compatible_binary_type { diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 1ebd3c8ab..f20379436 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -10,7 +10,6 @@ #include // reverse #include // array -#include // map #include // isnan, isinf #include // uint8_t, uint16_t, uint32_t, uint64_t #include // memcpy @@ -115,7 +114,7 @@ class binary_writer /*! @param[in] j JSON value to serialize - @pre j.type() == value_t::object + @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) { @@ -204,7 +203,7 @@ class binary_writer } else { - write_compact_float(j.m_data.m_value.number_float, detail::input_format_t::cbor); + write_compact_float(j.m_data.m_value.number_float, to_char_type(0xFA), to_char_type(0xFB)); } break; } @@ -238,6 +237,13 @@ class binary_writer { if (j.m_data.m_value.binary->has_subtype()) { + // The subtype is always written as a tag with a 0xD8..0xDB + // head, never in the one-byte form 0xC0..0xD7 that CBOR + // allows for tags 0..23 (so this is not write_cbor_head). + // binary_reader with cbor_tag_handler_t::store only turns + // 0xD8..0xDB into a subtype and ignores the one-byte tags, + // so the shorter form would lose subtypes 0..23 on a round + // trip. if (j.m_data.m_value.binary->subtype() <= (std::numeric_limits::max)()) { write_number(static_cast(0xd8)); @@ -309,6 +315,43 @@ class binary_writer return static_cast(length); } + /*! + @brief write a non-negative integer using the MessagePack fixint/uint ladder + @param[in] n the value to write, already known to be non-negative + */ + void write_msgpack_unsigned(const std::uint64_t n) + { + if (n < 128) + { + // positive fixnum + write_number(static_cast(n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // uint 8 + oa.write_character(to_char_type(0xCC)); + write_number(static_cast(n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // uint 16 + oa.write_character(to_char_type(0xCD)); + write_number(static_cast(n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // uint 32 + oa.write_character(to_char_type(0xCE)); + write_number(static_cast(n)); + } + else + { + // uint 64 + oa.write_character(to_char_type(0xCF)); + write_number(n); + } + } + /*! @param[in] j JSON value to serialize */ @@ -335,38 +378,8 @@ class binary_writer if (j.m_data.m_value.number_integer >= 0) { // MessagePack does not differentiate between positive - // signed integers and unsigned integers. Therefore, we used - // the code from the value_t::number_unsigned case here. - const auto value_as_unsigned = static_cast(j.m_data.m_value.number_integer); - if (value_as_unsigned < 128) - { - // positive fixnum - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (value_as_unsigned <= (std::numeric_limits::max)()) - { - // uint 8 - oa.write_character(to_char_type(0xCC)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (value_as_unsigned <= (std::numeric_limits::max)()) - { - // uint 16 - oa.write_character(to_char_type(0xCD)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (value_as_unsigned <= (std::numeric_limits::max)()) - { - // uint 32 - oa.write_character(to_char_type(0xCE)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else - { - // uint 64 - oa.write_character(to_char_type(0xCF)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } + // signed integers and unsigned integers. + write_msgpack_unsigned(static_cast(j.m_data.m_value.number_integer)); } else { @@ -408,41 +421,13 @@ class binary_writer case value_t::number_unsigned: { - if (j.m_data.m_value.number_unsigned < 128) - { - // positive fixnum - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - // uint 8 - oa.write_character(to_char_type(0xCC)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - // uint 16 - oa.write_character(to_char_type(0xCD)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - // uint 32 - oa.write_character(to_char_type(0xCE)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else - { - // uint 64 - oa.write_character(to_char_type(0xCF)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } + write_msgpack_unsigned(static_cast(j.m_data.m_value.number_unsigned)); break; } case value_t::number_float: { - write_compact_float(j.m_data.m_value.number_float, detail::input_format_t::msgpack); + write_compact_float(j.m_data.m_value.number_float, to_char_type(0xCA), to_char_type(0xCB)); break; } @@ -1383,6 +1368,14 @@ class binary_writer oa.write_character(to_char_type(0x00)); if (parents.empty()) { + // calc_bson_sizes() and write_bson_document() are two + // hand-synchronized passes over the same structure, linked + // only by nested_sizes' visiting order; this checks that the + // write pass consumed exactly the sizes the size pass + // produced, so a future change that desyncs them (skips or + // rejects an entry in only one pass) is caught immediately + // instead of silently writing wrong length prefixes. + JSON_ASSERT(next_size == nested_sizes.size()); return; } current = std::move(parents.back()); @@ -1434,52 +1427,6 @@ class binary_writer } } - static constexpr CharType get_cbor_float_prefix(float /*unused*/) - { - return to_char_type(0xFA); // Single-Precision Float - } - - static constexpr CharType get_cbor_float_prefix(double /*unused*/) - { - return to_char_type(0xFB); // Double-Precision Float - } - - ///////////// - // MsgPack // - ///////////// - - static constexpr CharType get_msgpack_float_prefix(float /*unused*/) - { - return to_char_type(0xCA); // float 32 - } - - static constexpr CharType get_msgpack_float_prefix(double /*unused*/) - { - return to_char_type(0xCB); // float 64 - } - - /// @return the BON8 type marker for binary32 (float) or binary64 (double) - template - static constexpr CharType get_bon8_float_prefix() - { - return to_char_type(std::is_same::value ? 0x8E : 0x8F); - } - - /// @return the type marker for a FloatType value in @a format (CBOR, MessagePack, or BON8) - template - static CharType get_compact_float_prefix(const detail::input_format_t format) - { - if (format == detail::input_format_t::cbor) - { - return get_cbor_float_prefix(FloatType{}); - } - if (format == detail::input_format_t::bon8) - { - return get_bon8_float_prefix(); - } - return get_msgpack_float_prefix(FloatType{}); - } - //////////// // UBJSON // //////////// @@ -1493,207 +1440,127 @@ class binary_writer { if (add_prefix) { - oa.write_character(get_ubjson_float_prefix(n)); + oa.write_character(get_ubjson_float_prefix()); } write_number(n, use_bjdata); } - // UBJSON: write number (unsigned integer) + // UBJSON: write number (integer) template::value, int>::type = 0> + std::is_integral::value, int>::type = 0> void write_number_with_ubjson_prefix(const NumberType n, const bool add_prefix, const bool use_bjdata) { - if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('i')); // int8 - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('U')); // uint8 - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('I')); // int16 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('u')); // uint16 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('l')); // int32 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('m')); // uint32 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('L')); // int64 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata) - { - if (add_prefix) - { - oa.write_character(to_char_type('M')); // uint64 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else - { - if (add_prefix) - { - oa.write_character(to_char_type('H')); // high-precision number - } - - const auto number = BasicJsonType(n).dump(); - write_number_with_ubjson_prefix(number.size(), true, use_bjdata); - for (std::size_t i = 0; i < number.size(); ++i) - { - oa.write_character(to_char_type(static_cast(number[i]))); - } - } - } - - // UBJSON: write number (signed integer) - template < typename NumberType, typename std::enable_if < - std::is_signed::value&& - !std::is_floating_point::value, int >::type = 0 > - void write_number_with_ubjson_prefix(const NumberType n, - const bool add_prefix, - const bool use_bjdata) - { - if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('i')); // int8 - } - write_number(static_cast(n), use_bjdata); - } - else if (static_cast((std::numeric_limits::min)()) <= n && n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('U')); // uint8 - } - write_number(static_cast(n), use_bjdata); - } - else if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('I')); // int16 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && (static_cast((std::numeric_limits::min)()) <= n && n <= static_cast((std::numeric_limits::max)()))) - { - if (add_prefix) - { - oa.write_character(to_char_type('u')); // uint16 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('l')); // int32 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && (static_cast((std::numeric_limits::min)()) <= n && n <= static_cast((std::numeric_limits::max)()))) - { - if (add_prefix) - { - oa.write_character(to_char_type('m')); // uint32 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else - { - // every value of an integer type of at most 64 bits fits into an - // int64; only a wider type needs a range check - write_ubjson_int64_or_high_precision(n, add_prefix, use_bjdata, - std::integral_constant < bool, std::numeric_limits::digits <= std::numeric_limits::digits > {}); - } - } - - template - void write_ubjson_int64_or_high_precision(const NumberType n, const bool add_prefix, const bool use_bjdata, std::true_type /*fits_int64*/) - { + const CharType prefix = ubjson_integer_prefix(n, use_bjdata); if (add_prefix) { - oa.write_character(to_char_type('L')); // int64 + oa.write_character(prefix); } - write_number(static_cast(n), use_bjdata); + write_ubjson_integer_payload(prefix, n, use_bjdata); } + /*! + @brief determine the UBJSON/BJData type marker of an integer + + This is the only place that picks the marker of an integer: both + write_number_with_ubjson_prefix() and ubjson_prefix() use it. An optimized + container announces the marker of its first value after `$` and then + writes every value without a marker, so the two must never disagree. + + @param[in] n the integer + @param[in] use_bjdata whether the BJData-only markers `u`, `m`, and `M` + may be used + + @return the first marker of `i`, `U`, `I`, `u` (BJData), `l`, `m` (BJData), + `L`, `M` (BJData, unsigned types only), and `H` (high-precision + number) whose range contains @a n + */ template - void write_ubjson_int64_or_high_precision(const NumberType n, const bool add_prefix, const bool use_bjdata, std::false_type /*fits_int64*/) + static CharType ubjson_integer_prefix(const NumberType n, const bool use_bjdata) noexcept { - if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) + if (value_in_range_of(n)) { - write_ubjson_int64_or_high_precision(n, add_prefix, use_bjdata, std::true_type {}); - return; + return 'i'; } - - if (add_prefix) + if (value_in_range_of(n)) { - oa.write_character(to_char_type('H')); // high-precision number + return 'U'; } - - const auto number = BasicJsonType(n).dump(); - write_number_with_ubjson_prefix(number.size(), true, use_bjdata); - for (std::size_t i = 0; i < number.size(); ++i) + if (value_in_range_of(n)) { - oa.write_character(to_char_type(static_cast(number[i]))); + return 'I'; } + if (use_bjdata && value_in_range_of(n)) + { + return 'u'; + } + if (value_in_range_of(n)) + { + return 'l'; + } + if (use_bjdata && value_in_range_of(n)) + { + return 'm'; + } + if (value_in_range_of(n)) + { + return 'L'; + } + if (use_bjdata && std::is_unsigned::value) + { + return 'M'; + } + // anything else is treated as a high-precision number + return 'H'; } + /*! + @brief write the value of an integer for the marker chosen by + ubjson_integer_prefix() + */ template - static constexpr CharType ubjson_int64_or_high_precision_prefix(const NumberType /*n*/, std::true_type /*fits_int64*/) noexcept + void write_ubjson_integer_payload(const CharType prefix, const NumberType n, const bool use_bjdata) { - return 'L'; - } - - template - static CharType ubjson_int64_or_high_precision_prefix(const NumberType n, std::false_type /*fits_int64*/) noexcept - { - // anything outside of the range of an int64 is treated as a - // high-precision number - return ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) ? 'L' : 'H'; + switch (prefix) + { + case 'i': + write_number(static_cast(n), use_bjdata); + break; + case 'U': + write_number(static_cast(n), use_bjdata); + break; + case 'I': + write_number(static_cast(n), use_bjdata); + break; + case 'u': + write_number(static_cast(n), use_bjdata); + break; + case 'l': + write_number(static_cast(n), use_bjdata); + break; + case 'm': + write_number(static_cast(n), use_bjdata); + break; + case 'L': + write_number(static_cast(n), use_bjdata); + break; + case 'M': + write_number(static_cast(n), use_bjdata); + break; + default: + { + // high-precision number: the decimal digits as a string + JSON_ASSERT(prefix == 'H'); + const auto number = BasicJsonType(n).dump(); + write_number_with_ubjson_prefix(number.size(), true, use_bjdata); + for (std::size_t i = 0; i < number.size(); ++i) + { + oa.write_character(to_char_type(static_cast(number[i]))); + } + break; + } + } } /*! @@ -1710,77 +1577,13 @@ class binary_writer return j.m_data.m_value.boolean ? 'T' : 'F'; case value_t::number_integer: - { - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'i'; - } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'U'; - } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'I'; - } - if (use_bjdata && ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)())) - { - return 'u'; - } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'l'; - } - if (use_bjdata && ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)())) - { - return 'm'; - } - // every value of an integer type of at most 64 bits fits into - // an int64; only a wider type needs a range check - return ubjson_int64_or_high_precision_prefix(j.m_data.m_value.number_integer, - std::integral_constant < bool, std::numeric_limits::digits <= std::numeric_limits::digits > {}); - } + return ubjson_integer_prefix(j.m_data.m_value.number_integer, use_bjdata); case value_t::number_unsigned: - { - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'i'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'U'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'I'; - } - if (use_bjdata && j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'u'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'l'; - } - if (use_bjdata && j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'm'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'L'; - } - if (use_bjdata) - { - return 'M'; - } - // anything else is treated as a high-precision number - return 'H'; - } + return ubjson_integer_prefix(j.m_data.m_value.number_unsigned, use_bjdata); case value_t::number_float: - return get_ubjson_float_prefix(j.m_data.m_value.number_float); + return get_ubjson_float_prefix(); case value_t::string: return 'S'; @@ -1805,7 +1608,7 @@ class binary_writer Containers, strings, high-precision numbers, booleans and null cannot be declared as the single type of an optimized container in BJData; such a container is written unoptimized. The reader rejects them with the same - list (binary_reader::bjd_optimized_type_markers). + list (binary_reader::is_bjd_excluded_optimized_type()). */ static constexpr bool is_bjdata_excluded_type_marker(const CharType marker) noexcept { @@ -1813,14 +1616,16 @@ class binary_writer || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; } - static constexpr CharType get_ubjson_float_prefix(float /*unused*/) + /// @return the UBJSON/BJData type marker for a float or double value + /// + /// number_float_t must be float or double; a static_assert (rather than + /// an ambiguous overload) reports an unsupported number_float_t clearly. + template + static constexpr CharType get_ubjson_float_prefix() { - return 'd'; // float 32 - } - - static constexpr CharType get_ubjson_float_prefix(double /*unused*/) - { - return 'D'; // float 64 + static_assert(std::is_same::value || std::is_same::value, + "number_float_t must be float or double for the UBJSON/BJData writer"); + return std::is_same::value ? 'd' : 'D'; // float 32 / float 64 } /*! @@ -1837,35 +1642,184 @@ class binary_writer : value_in_range_of(el.template get()); } + /*! + @brief look up the BJData ND-array dtype marker for an `_ArrayType_` name + @return the one-character marker, or '\0' if @a name does not name a known dtype + + A C++11 `constexpr` function cannot contain a `switch`, so this is a plain + comparison chain instead; it is only reached once per ND-array candidate + object. Keep in sync with binary_reader's `bjd_type_name()`, which maps + the other way. + */ + static CharType bjdata_ndarray_type_marker(const string_t& name) + { + if (name == "uint8") + { + return 'U'; + } + if (name == "int8") + { + return 'i'; + } + if (name == "uint16") + { + return 'u'; + } + if (name == "int16") + { + return 'I'; + } + if (name == "uint32") + { + return 'm'; + } + if (name == "int32") + { + return 'l'; + } + if (name == "uint64") + { + return 'M'; + } + if (name == "int64") + { + return 'L'; + } + if (name == "single") + { + return 'd'; + } + if (name == "double") + { + return 'D'; + } + if (name == "char") + { + return 'C'; + } + if (name == "byte") + { + return 'B'; + } + return '\0'; + } + + /*! + @brief validate (dry_run) or write one BJData ND-array element of integer dtype @a T + @return whether @a el's value is in range of @a T; always true when @a dry_run is false + */ + template + bool write_bjdata_ndarray_element(const BasicJsonType& el, const bool dry_run) + { + if (dry_run) + { + return bjdata_ndarray_value_in_range(el); + } + using storage_type = typename std::conditional::value, std::uint64_t, std::int64_t>::type; + write_number(static_cast(el.template get()), true); + return true; + } + + /*! + @brief validate (dry_run) or write one BJData ND-array element of dtype 'd' (single precision) + @return whether @a el's value fits a float without overflow; always true when @a dry_run is false + */ + bool write_bjdata_ndarray_float_element(const BasicJsonType& el, const bool dry_run) + { + const auto dval = el.template get(); + if (dry_run) + { + return !std::isfinite(dval) || + (dval >= static_cast(std::numeric_limits::lowest()) && + dval <= static_cast((std::numeric_limits::max)())); + } + write_number(static_cast(dval), true); + return true; + } + + /*! + @brief validate or write every element of a BJData ND-array's `_ArrayData_` + @param[in] array_data the `_ArrayData_` array + @param[in] dtype the ND-array dtype marker, as returned by bjdata_ndarray_type_marker() + @param[in] dry_run true to only range-check each element, false to write it + @return whether every element is in range for @a dtype (always true when @a dry_run is false) + */ + bool write_bjdata_ndarray_elements(const BasicJsonType& array_data, const CharType dtype, const bool dry_run) + { + for (const auto& el : array_data) + { + bool ok = true; + switch (dtype) + { + case 'U': + case 'C': + case 'B': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'i': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'u': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'I': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'm': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'l': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'M': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'L': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'd': + ok = write_bjdata_ndarray_float_element(el, dry_run); + break; + case 'D': + default: + // 'D' (double) already spans the full range of number_float_t + if (!dry_run) + { + write_number(el.template get(), true); + } + break; + } + if (!ok) + { + return false; + } + } + return true; + } + /*! @return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid */ bool write_bjdata_ndarray(const typename BasicJsonType::object_t& value, const bool use_count, const bool use_type, const bjdata_version_t bjdata_version) { - std::map bjdtype = {{"uint8", 'U'}, {"int8", 'i'}, {"uint16", 'u'}, {"int16", 'I'}, - {"uint32", 'm'}, {"int32", 'l'}, {"uint64", 'M'}, {"int64", 'L'}, {"single", 'd'}, {"double", 'D'}, - {"char", 'C'}, {"byte", 'B'} - }; - - string_t key = "_ArrayType_"; + const auto& array_type = value.at("_ArrayType_"); // the type name is looked up as a string below; a non-string // annotation (e.g. a number, null, or an array) cannot name a known // dtype, so it is treated the same as an unrecognized type name and // falls back to a plain object encoding instead of throwing // type_error.302 out of get() - if (!value.at(key).is_string()) + if (!array_type.is_string()) { return true; } // use get() instead of static_cast to avoid an // ambiguous conversion under explicit instantiation on C++17 (see #4825) - auto it = bjdtype.find(value.at(key).template get()); - if (it == bjdtype.end()) + const CharType dtype = bjdata_ndarray_type_marker(array_type.template get()); + if (dtype == '\0') { return true; } - CharType dtype = it->second; // the 'B' (byte) marker is only defined from BJData Draft 3 onward; // emitting it under an earlier draft would produce a stream that an @@ -1877,12 +1831,12 @@ class binary_writer return true; } - key = "_ArraySize_"; + const auto& array_size = value.at("_ArraySize_"); // the dimensions are written verbatim as the header length below, so a // value that is not an array cannot produce a valid one: null emits 'Z' // and an object emits '{', neither of which a reader accepts after '#'. // Such an object is not a valid ndarray and falls back to a plain object. - if (!value.at(key).is_array()) + if (!array_size.is_array()) { return true; } @@ -1892,7 +1846,7 @@ class binary_writer // dimension, or a 1xN row vector is read back as a plain array, which // would silently drop the annotation, so such an object falls back to // a plain object encoding instead - const auto& dims = value.at(key); + const auto& dims = array_size; if (dims.size() < 2 || (dims.size() == 2 && dims.at(0).is_number_integer() && dims.at(0).template get() == 1)) { return true; @@ -1939,8 +1893,8 @@ class binary_writer // to be an array: size() is 0 for null and 1 for any other scalar, and // iterating an object visits its values, so any of these could match // the dimensions by accident and be encoded as an unrelated ND-array - key = "_ArrayData_"; - if (!value.at(key).is_array() || value.at(key).size() != len) + const auto& array_data = value.at("_ArrayData_"); + if (!array_data.is_array() || array_data.size() != len) { return true; } @@ -1955,7 +1909,7 @@ class binary_writer // API stores int literals as signed), so both are accepted here and the // writes below go through get<>, which reads the member that is active. const bool ndarray_is_float = (dtype == 'd' || dtype == 'D'); - for (const auto& el : value.at(key)) + for (const auto& el : array_data) { if (ndarray_is_float ? !el.is_number_float() : !el.is_number_integer()) { @@ -1968,134 +1922,19 @@ class binary_writer // wrap (integers) or overflow to infinity (the "single" precision // float) instead of being reported, so such an object falls back to // a plain object encoding as well - for (const auto& el : value.at(key)) + if (!write_bjdata_ndarray_elements(array_data, dtype, true)) { - bool in_range = true; - switch (dtype) - { - case 'U': - case 'C': - case 'B': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'i': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'u': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'I': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'm': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'l': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'M': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'L': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'd': - { - const auto dval = el.template get(); - in_range = !std::isfinite(dval) || - (dval >= static_cast(std::numeric_limits::lowest()) && - dval <= static_cast((std::numeric_limits::max)())); - break; - } - default: - // 'D' (double) already spans the full range of number_float_t - break; - } - if (!in_range) - { - return true; - } + return true; } - oa.write_character('['); - oa.write_character('$'); + oa.write_character(to_char_type('[')); + oa.write_character(to_char_type('$')); oa.write_character(dtype); - oa.write_character('#'); + oa.write_character(to_char_type('#')); - key = "_ArraySize_"; - write_ubjson(value.at(key), use_count, use_type, true, true, bjdata_version); + write_ubjson(array_size, use_count, use_type, true, true, bjdata_version); - key = "_ArrayData_"; - if (dtype == 'U' || dtype == 'C' || dtype == 'B') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'i') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'u') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'I') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'm') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'l') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'M') - { - for (const auto& el : value.at(key)) - { - write_number(el.template get(), true); - } - } - else if (dtype == 'L') - { - for (const auto& el : value.at(key)) - { - write_number(el.template get(), true); - } - } - else if (dtype == 'd') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'D') - { - for (const auto& el : value.at(key)) - { - write_number(el.template get(), true); - } - } + write_bjdata_ndarray_elements(array_data, dtype, false); return false; } @@ -2408,7 +2247,7 @@ class binary_writer } else { - write_compact_float(n, detail::input_format_t::bon8); + write_compact_float(n, to_char_type(0x8E), to_char_type(0x8F)); } #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP @@ -2419,19 +2258,6 @@ class binary_writer // Utility functions // /////////////////////// - /* - @brief write a number to output input - @param[in] n number of type @a NumberType - @param[in] OutputIsLittleEndian Set to true if output data is - required to be little endian - @tparam NumberType the type of the number - - @note This function needs to respect the system's endianness, because bytes - in CBOR, MessagePack, and UBJSON are stored in network order (big - endian) and therefore need reordering on little endian systems. - On the other hand, BSON and BJData use little endian and should reorder - on big endian systems. - */ // single-instruction byte swaps (compilers lower these to bswap/rev/movbe); // used to emit big-endian numbers without a per-byte std::reverse loop static std::uint16_t byte_swap(std::uint16_t x) noexcept @@ -2513,6 +2339,19 @@ class binary_writer std::reverse(a.begin(), a.end()); } + /*! + @brief write a number to the output + @param[in] n number of type @a NumberType + @param[in] OutputIsLittleEndian Set to true if output data is + required to be little endian + @tparam NumberType the type of the number + + @note This function needs to respect the system's endianness, because bytes + in CBOR, MessagePack, UBJSON, and BON8 are stored in network order + (big endian) and therefore need reordering on little endian systems. + On the other hand, BSON and BJData use little endian and should + reorder on big endian systems. + */ template void write_number(const NumberType n, const bool OutputIsLittleEndian = false) { @@ -2530,8 +2369,17 @@ class binary_writer oa.write_characters(vec.data(), sizeof(NumberType)); } - void write_compact_float(const number_float_t n, detail::input_format_t format) + /// @brief write @a n using @a float32_marker if it round-trips through + /// float, otherwise using @a float64_marker + /// + /// @a float32_marker and @a float64_marker are the format-specific type + /// markers (CBOR: 0xFA/0xFB, MessagePack: 0xCA/0xCB, BON8: 0x8E/0x8F); + /// each caller already knows them at compile time, so the format itself + /// no longer needs to be passed in. + void write_compact_float(const number_float_t n, const CharType float32_marker, const CharType float64_marker) { + static_assert(std::is_same::value || std::is_same::value, + "number_float_t must be float or double for the CBOR/MessagePack/BON8 writer"); #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_PUSH JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") @@ -2549,12 +2397,12 @@ class binary_writer static_cast(n) <= static_cast((std::numeric_limits::max)()) && static_cast(static_cast(n)) == static_cast(n)))) { - oa.write_character(get_compact_float_prefix(format)); + oa.write_character(float32_marker); write_number(static_cast(n)); } else { - oa.write_character(get_compact_float_prefix(format)); + oa.write_character(float64_marker); write_number(n); } #ifdef __GNUC__ @@ -2563,7 +2411,7 @@ class binary_writer } public: - // The following to_char_type functions are implement the conversion + // The following to_char_type functions implement the conversion // between uint8_t and CharType. In case CharType is not unsigned, // such a conversion is required to allow values greater than 128. // See for a discussion. diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index b7a22378e..5cb00b690 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -906,11 +906,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /*! @brief how many levels the operation going on in this thread has descended into - Copying a value and comparing two values share this count. The library never - nests one inside the other - copying a value does not compare one, and - comparing two values does not copy them - and where user code nests them - anyway, sharing the count only ends a descent sooner than it had to, which - costs a little speed and is never wrong. + Copying a value, converting one from another specialization, and comparing + two values share this count. The library never nests one of them inside + another - none of them does either of the other two on the way - and where + user code nests them anyway, sharing the count only ends a descent sooner + than it had to, which costs a little speed and is never wrong. A byte is enough: the count never exceeds the limit by more than the single level that notices the limit has been reached. @@ -1267,6 +1267,222 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec copy_iteratively(src); } + /*! + @brief convert the value @a val of another specialization into this null + value; @a val must be neither an object nor an array + + Converting such a value never descends, so both ways of converting an + object or an array (@ref convert_structured) leave their elements of this + kind to the converting constructor, which leaves them to this. + */ + template + void convert_leaf(const BasicJsonType& val) + { + using other_boolean_t = typename BasicJsonType::boolean_t; + using other_number_float_t = typename BasicJsonType::number_float_t; + using other_number_integer_t = typename BasicJsonType::number_integer_t; + using other_number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using other_string_t = typename BasicJsonType::string_t; + using other_binary_t = typename BasicJsonType::binary_t; + + switch (val.type()) + { + case value_t::boolean: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::number_float: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::number_integer: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::number_unsigned: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::string: + JSONSerializer::to_json(*this, val.template get_ref()); + break; + case value_t::binary: + JSONSerializer::to_json(*this, val.template get_ref()); + break; + case value_t::null: + // m_data.m_type is already value_t::null + break; + case value_t::discarded: + m_data.m_type = value_t::discarded; + break; + case value_t::object: // LCOV_EXCL_LINE + case value_t::array: // LCOV_EXCL_LINE + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + } + } + + /// scratch space for the converted elements of the arrays that + /// @ref convert_iteratively has yet to create + using convert_scratch_t = std::vector>; + + /*! + @brief create the object or array @a val converted into this null value + + Its converted elements are the last `val.size()` entries of @a elements (an + array) or of @a members (an object); they are moved into the container in + one go and then removed. + */ + template + void convert_level(const BasicJsonType& val, convert_scratch_t& elements, copy_scratch_t& members) + { + if (val.is_object()) + { + const auto first = members.end() - static_cast(val.size()); + m_data.m_value.object = create(std::make_move_iterator(first), + std::make_move_iterator(members.end())); + // only now that the object exists may this stop being a null value + m_data.m_type = value_t::object; + members.erase(first, members.end()); + } + else + { + const auto first = elements.end() - static_cast(val.size()); + m_data.m_value.array = create(std::make_move_iterator(first), + std::make_move_iterator(elements.end())); + // only now that the array exists may this stop being a null value + m_data.m_type = value_t::array; + elements.erase(first, elements.end()); + } + + set_parents(); + } + + /*! + @brief convert the object or array @a val of another specialization into + this null value without recursing + + The containers whose conversion has begun are kept on an explicit stack + rather than on the call stack. Unlike @ref copy_iteratively, this builds + every container from the bottom up: all its elements are converted first, + and the container is then created from them in one go, the way the range + constructor that converts the levels above the bound does. The two object + types need not enumerate their members in the same order, so the members + could not be paired up by position anyway, and building from a range keeps + what the range constructor does with keys that become equal on conversion. + + Every value is complete before it is handed on, and a container gets its + type only once it exists, so whatever throws, every value left behind can + be destroyed. + */ + template + void convert_iteratively(const BasicJsonType& val) + { + using other_const_iterator = typename BasicJsonType::const_iterator; + + // the containers whose conversion has begun, innermost last, each with + // its element to convert next + std::vector> pending; + + // the converted elements of the pending arrays and the converted + // members of the pending objects, those of the innermost one last + convert_scratch_t elements; + copy_scratch_t members; + + pending.emplace_back(&val, val.cbegin()); + + for (;;) + { + const BasicJsonType& container = *pending.back().first; + // a copy, as descending below can reallocate pending; the + // iterator kept in pending is only advanced through pending.back() + const other_const_iterator next = pending.back().second; + + if (next != container.cend()) + { + if (next->is_structured()) + { + // convert its elements first; next stays where it is until + // the converted container is handed back to this one + pending.emplace_back(&*next, next->cbegin()); + continue; + } + + // the converting constructor does not descend into this value + if (container.is_object()) + { + members.emplace_back(next.key(), *next); + } + else + { + elements.emplace_back(*next); + } + ++pending.back().second; + continue; + } + + // all elements of the container are converted: create it + pending.pop_back(); + + if (pending.empty()) + { + convert_level(container, elements, members); + return; + } + + basic_json converted; + converted.convert_level(container, elements, members); +#if JSON_DIAGNOSTIC_POSITIONS + converted.start_position = container.start_pos(); + converted.end_position = container.end_pos(); +#endif + + // hand it to the container it is an element of + if (pending.back().first->is_object()) + { + members.emplace_back(pending.back().second.key(), std::move(converted)); + } + else + { + elements.push_back(std::move(converted)); + } + ++pending.back().second; + } + } + + /*! + @brief convert the object or array @a val of another specialization into + this null value + + Converting a container converts its elements, so a value nested deeply + enough used to exhaust the call stack. The descent is bounded here as in + @ref copy_structured: the first `detail::recursion_depth_limit()` levels + are converted by the containers' range constructors, just as they always were, + and anything below that is converted without the call stack by + @ref convert_iteratively. + + @sa https://github.com/nlohmann/json/issues/5650 + */ + template + void convert_structured(const BasicJsonType& val) + { + const nesting_depth_guard guard; + + if (JSON_HEDLEY_LIKELY(guard.okay())) + { + // every element comes back to the converting constructor + if (val.is_object()) + { + using other_object_t = typename BasicJsonType::object_t; + JSONSerializer::to_json(*this, val.template get_ref()); + } + else + { + using other_array_t = typename BasicJsonType::array_t; + JSONSerializer::to_json(*this, val.template get_ref()); + } + return; + } + + convert_iteratively(val); + } + /// the result of comparing two values, including values that cannot be /// ordered at all, such as a discarded value or a NaN @@ -1596,7 +1812,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec template < typename CompatibleType, typename U = detail::uncvref_t, detail::enable_if_t < - !detail::is_basic_json::value && detail::is_compatible_type::value, int > = 0 > + !detail::is_basic_json::value && detail::is_compatible_type::value +#if JSON_DISABLE_TUPLE_REFERENCE_CONVERSION + // see https://github.com/nlohmann/json/issues/2226 + && !detail::is_basic_json_reference_tuple::value +#endif + , int > = 0 > basic_json(CompatibleType && val) noexcept(noexcept( // NOLINT(bugprone-forwarding-reference-overload,bugprone-exception-escape) JSONSerializer::to_json(std::declval(), std::forward(val)))) @@ -1617,49 +1838,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec end_position(val.end_pos()) #endif { - using other_boolean_t = typename BasicJsonType::boolean_t; - using other_number_float_t = typename BasicJsonType::number_float_t; - using other_number_integer_t = typename BasicJsonType::number_integer_t; - using other_number_unsigned_t = typename BasicJsonType::number_unsigned_t; - using other_string_t = typename BasicJsonType::string_t; - using other_object_t = typename BasicJsonType::object_t; - using other_array_t = typename BasicJsonType::array_t; - using other_binary_t = typename BasicJsonType::binary_t; - - switch (val.type()) + if (val.is_structured()) { - case value_t::boolean: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::number_float: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::number_integer: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::number_unsigned: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::string: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::object: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::array: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::binary: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::null: - *this = nullptr; - break; - case value_t::discarded: - m_data.m_type = value_t::discarded; - break; - default: // LCOV_EXCL_LINE - JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + convert_structured(val); + } + else + { + convert_leaf(val); } JSON_ASSERT(m_data.m_type == val.type()); @@ -5034,7 +5219,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = detail::input_adapter(std::forward(i)); return format == input_format_t::json ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) - : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } /// @brief generate SAX events (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5051,7 +5236,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = detail::input_adapter(std::move(first), std::move(last)); return format == input_format_t::json ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) - : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } /// @brief generate SAX events @@ -5073,7 +5258,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } #ifndef JSON_NO_IO /// @brief deserialize from stream @@ -5092,7 +5277,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/operator_gtgt/ friend std::istream& operator>>(std::istream& i, basic_json& j) { - parser(detail::input_adapter(i)).parse(false, j); + // parse into a temporary so that j is left unchanged if parsing fails + basic_json result; + parser(detail::input_adapter(i)).parse(false, result); + j = std::move(result); return i; } #endif // JSON_NO_IO @@ -5368,7 +5556,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5388,7 +5576,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5417,7 +5605,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5435,7 +5623,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5454,7 +5642,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5481,7 +5669,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5499,7 +5687,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5518,7 +5706,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5545,7 +5733,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5563,7 +5751,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5582,7 +5770,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5600,7 +5788,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(input_format_t::bon8, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5619,7 +5807,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(input_format_t::bon8, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5637,7 +5825,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5656,7 +5844,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5683,7 +5871,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 429d4a04f..1da7e8808 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -2772,6 +2772,7 @@ void templated_json_throw(ExceptionType exception) // Macros to simplify conversion from/to types +// NLOHMANN_JSON_EXPAND to NLOHMANN_JSON_DOUBLE_PASTE63 are generated by tools/macro_builder (see its README.md) #define NLOHMANN_JSON_EXPAND( x ) x #define NLOHMANN_JSON_GET_MACRO(_1, _2, _3, _4, _5, _6, _7, _8, _9, _10, _11, _12, _13, _14, _15, _16, _17, _18, _19, _20, _21, _22, _23, _24, _25, _26, _27, _28, _29, _30, _31, _32, _33, _34, _35, _36, _37, _38, _39, _40, _41, _42, _43, _44, _45, _46, _47, _48, _49, _50, _51, _52, _53, _54, _55, _56, _57, _58, _59, _60, _61, _62, _63, _64, NAME,...) NAME #define NLOHMANN_JSON_PASTE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \ @@ -3034,6 +3035,10 @@ void templated_json_throw(ExceptionType exception) // arguments, so dispatching on Type,BaseType,member... directly would run out // one slot early and cap the derived-type macros at 62 members instead of the // 63 that NLOHMANN_JSON_PASTE supports. +// +// The slot table below (down to the closing NLOHMANN_JSON_TYPE_BODY_SENTINEL)) +// is generated by tools/macro_builder (see its README.md; run with the +// "type_body" argument). #define NLOHMANN_JSON_TYPE_BODY(Prefix, ...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \ Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ @@ -3329,6 +3334,10 @@ void templated_json_throw(ExceptionType exception) #define JSON_DISABLE_ENUM_SERIALIZATION 0 #endif +#ifndef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION + #define JSON_DISABLE_TUPLE_REFERENCE_CONVERSION 0 +#endif + #if JSON_HAS_THREE_WAY_COMPARISON #include // partial_ordering #endif @@ -3660,7 +3669,6 @@ NLOHMANN_JSON_NAMESPACE_END -#include // array #include // size_t #include // conditional, enable_if, false_type, integral_constant, is_constructible, is_integral, is_same, remove_cv, remove_reference, true_type #include // index_sequence, make_index_sequence, index_sequence_for @@ -3813,12 +3821,6 @@ struct static_const constexpr T static_const::value; #endif -template -constexpr std::array make_array(Args&& ... args) -{ - return std::array {{static_cast(std::forward(args))...}}; -} - } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -4644,6 +4646,18 @@ template struct is_compatible_type : is_compatible_type_impl {}; +// a one-element std::tuple holding a reference to BasicJsonType, as created by +// std::forward_as_tuple(j); see JSON_DISABLE_TUPLE_REFERENCE_CONVERSION +template +struct is_basic_json_reference_tuple : std::false_type {}; + +template +struct is_basic_json_reference_tuple> +{ + static constexpr bool value = + std::is_reference::value && std::is_same, BasicJsonType>::value; +}; + template struct is_compatible_binary_type { @@ -7076,11 +7090,13 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = { std::get(t)... }; } -#if JSON_BRACE_INIT_COPY_SEMANTICS -// JSON_BRACE_INIT_COPY_SEMANTICS makes a one-element braced list copy its -// element instead of wrapping it, which would serialize std::tuple{5} as 5 -// rather than [5]. Build what the default deduction builds instead: an object -// if the element is a [string, value] pair, a one-element array otherwise. +// A one-element braced list does not reliably wrap its element: with +// JSON_BRACE_INIT_COPY_SEMANTICS it copies it, which would serialize +// std::tuple{5} as 5 rather than [5], and some compilers (e.g., Apple clang +// 15 and 16) copy an element that is itself a basic_json even without it, so +// std::tuple{true} became true rather than [true]. Build what the default +// deduction builds instead: an object if the element is a [string, value] pair, +// a one-element array otherwise. template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence<0> /*unused*/) { @@ -7098,7 +7114,6 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = BasicJsonType::array({std::move(element)}); } } -#endif template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& /*unused*/, index_sequence<> /*unused*/) @@ -7601,7 +7616,6 @@ NLOHMANN_JSON_NAMESPACE_END -#include // generate_n #include // array #include // ldexp #include // size_t @@ -14335,7 +14349,8 @@ class binary_reader ~binary_reader() = default; /*! - @param[in] format the binary format to parse + @brief parse in the format the constructor was given + @param[in] sax_ a SAX event processor @param[in] strict whether to expect the input to be consumed completed @param[in] tag_handler how to treat CBOR tags @@ -14343,9 +14358,8 @@ class binary_reader @return whether parsing was successful: the input was read without errors, and no SAX event returned false */ - JSON_HEDLEY_NON_NULL(3) - bool sax_parse(const input_format_t format, - json_sax_t* sax_, + JSON_HEDLEY_NON_NULL(2) + bool sax_parse(json_sax_t* sax_, const bool strict = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { @@ -14359,14 +14373,14 @@ class binary_reader ndarray_open = 0; bool result = false; - switch (format) + switch (input_format) { case input_format_t::bson: result = parse_bson_internal(); break; case input_format_t::cbor: - result = parse_cbor_internal(true, tag_handler); + result = parse_cbor_internal(tag_handler); break; case input_format_t::msgpack: @@ -14494,6 +14508,22 @@ class binary_reader return enter_container(/*is_object*/true, len, type_marker); } + /*! + @brief close the innermost open array or object + + Pops the container opened by the matching @ref enter_container call and + emits the SAX end event. Every format-specific driver otherwise repeated + the same pop-then-dispatch sequence at its own close site. + + @return whether the SAX parser accepted the end event + */ + bool leave_container() + { + const bool is_object = container_stack.back().is_object; + container_stack.pop_back(); + return is_object ? sax->end_object() : sax->end_array(); + } + ////////// // BSON // ////////// @@ -14752,8 +14782,8 @@ class binary_reader if (element_type == 0) // end of the innermost document { // a copy, not a reference: it must stay valid across the - // pop_back() below, which destroys the container_stack - // element it would otherwise alias + // pop_back() inside leave_container() below, which destroys + // the container_stack element it would otherwise alias const container_frame top = container_stack.back(); if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(top.start_position, top.declared_size))) @@ -14761,8 +14791,7 @@ class binary_reader return false; } - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -15261,29 +15290,13 @@ class binary_reader return enter_array(conditional_static_cast(static_cast(current) & 0x1Fu)); case 0x98: // array (one-byte uint8_t for n follows) - { - std::uint8_t len{}; - return get_number(input_format_t::cbor, len) && enter_array(static_cast(len)); - } - case 0x99: // array (two-byte uint16_t for n follow) - { - std::uint16_t len{}; - return get_number(input_format_t::cbor, len) && enter_array(static_cast(len)); - } - case 0x9A: // array (four-byte uint32_t for n follow) - { - std::uint32_t len{}; - std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size); - } - case 0x9B: // array (eight-byte uint64_t for n follow) { std::uint64_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size); + return get_cbor_argument(len) && get_cbor_container_size(len, size, "array") && enter_array(size); } case 0x9F: // array (indefinite length) @@ -15317,35 +15330,19 @@ class binary_reader return enter_object(conditional_static_cast(static_cast(current) & 0x1Fu)); case 0xB8: // map (one-byte uint8_t for n follows) - { - std::uint8_t len{}; - return get_number(input_format_t::cbor, len) && enter_object(static_cast(len)); - } - case 0xB9: // map (two-byte uint16_t for n follow) - { - std::uint16_t len{}; - return get_number(input_format_t::cbor, len) && enter_object(static_cast(len)); - } - case 0xBA: // map (four-byte uint32_t for n follow) - { - std::uint32_t len{}; - std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size); - } - case 0xBB: // map (eight-byte uint64_t for n follow) { std::uint64_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size); + return get_cbor_argument(len) && get_cbor_container_size(len, size, "map") && enter_object(size); } case 0xBF: // map (indefinite length) return enter_object(detail::unknown_size()); - case 0xC0: // tagged item + case 0xC0: // tagged item (tag value 0-23, in the head itself) case 0xC1: case 0xC2: case 0xC3: @@ -15369,6 +15366,27 @@ class binary_reader case 0xD5: case 0xD6: case 0xD7: + { + if (tag_handler == cbor_tag_handler_t::error) + { + auto last_token = get_token_string(); + if (!report_repairable_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr))) + { + return false; + } + // when recovering, the tag is ignored, as RFC 8949, + // Section 6.1 suggests for converting to JSON + } + + // ignore and store: the tag value is already in the head, so + // there is nothing left to read here; the tagged value that + // follows is read by the loop in parse_cbor_internal() rather + // than by recursing here + tag_pending = true; + return true; + } + case 0xD8: // tagged item (1 byte follows) case 0xD9: // tagged item (2 bytes follow) case 0xDA: // tagged item (4 bytes follow) @@ -15391,47 +15409,11 @@ class binary_reader case cbor_tag_handler_t::ignore: { - // ignore binary subtype - switch (current) + // ignore the tag's binary subtype argument + std::uint64_t subtype_to_ignore{}; + if (!get_cbor_argument(subtype_to_ignore)) { - case 0xD8: - { - std::uint8_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - case 0xD9: - { - std::uint16_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - case 0xDA: - { - std::uint32_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - case 0xDB: - { - std::uint64_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - default: - break; + return false; } // the tagged value follows; it is read by the loop in // parse_cbor_internal() rather than by recursing here @@ -15441,57 +15423,15 @@ class binary_reader case cbor_tag_handler_t::store: { - binary_t b; // use binary subtype and store in a binary container - switch (current) + std::uint64_t subtype{}; + if (!get_cbor_argument(subtype)) { - case 0xD8: - { - std::uint8_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - case 0xD9: - { - std::uint16_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - case 0xDA: - { - std::uint32_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - case 0xDB: - { - std::uint64_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - default: - { - // as above, the tagged value is read by the caller - tag_pending = true; - return true; - } + return false; } + binary_t b; + b.set_subtype(detail::conditional_static_cast(subtype)); + get(); // a byte string (the heads accepted by get_cbor_binary) keeps the tag as subtype if ((current >= 0x40 && current <= 0x5B) || current == 0x5F) @@ -15522,52 +15462,7 @@ class binary_reader return sax->null(); case 0xF9: // Half-Precision Float (two-byte IEEE 754) - { - const auto byte1_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number"))) - { - return false; - } - const auto byte2_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number"))) - { - return false; - } - - const auto byte1 = static_cast(byte1_raw); - const auto byte2 = static_cast(byte2_raw); - - // Code from RFC 8949, Appendix D, Figure 3: - // As half-precision floating-point numbers were only added - // to IEEE 754 in 2008, today's programming platforms often - // still only have limited support for them. It is very - // easy to include at least decoding support for them even - // without such support. An example of a small decoder for - // half-precision floating-point numbers in the C language - // is shown in Fig. 3. - const auto half = static_cast((byte1 << 8u) + byte2); - const double val = [&half] - { - const int exp = (half >> 10u) & 0x1Fu; - const unsigned int mant = half & 0x3FFu; - JSON_ASSERT(exp <= 31); - JSON_ASSERT(mant <= 1023); - switch (exp) - { - case 0: - return std::ldexp(mant, -24); - case 31: - return (mant == 0) - ? std::numeric_limits::infinity() - : std::numeric_limits::quiet_NaN(); - default: - return std::ldexp(mant + 1024, exp - 25); - } - }(); - return sax->number_float((half & 0x8000u) != 0 - ? static_cast(-val) - : static_cast(val), ""); - } + return get_half_float(input_format_t::cbor, false); case 0xFA: // Single-Precision Float (four-byte IEEE 754) { @@ -15963,6 +15858,73 @@ class binary_reader } } + /*! + @brief read a CBOR argument (additional information 24-27) of the width + @ref current announces + + The lower 5 bits of @a current (0x18-0x1B) select a 1/2/4/8-byte + big-endian unsigned integer that follows the head byte; this is shared by + every major type that uses this encoding (unsigned/negative integers, + strings, arrays, maps, tags). Reading always goes through @ref get_number, + so EOF is reported the same way as before this helper existed. + + @param[out] value the decoded argument + @return whether reading succeeded + */ + bool get_cbor_argument(std::uint64_t& value) + { + switch (current & 0x1F) + { + case 0x18: // 1 byte + { + std::uint8_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + case 0x19: // 2 bytes + { + std::uint16_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + case 0x1A: // 4 bytes + { + std::uint32_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + case 0x1B: // 8 bytes + { + std::uint64_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + return false; // LCOV_EXCL_LINE + } + } + /*! @brief narrow a definite CBOR array/map length to std::size_t @@ -15995,19 +15957,15 @@ class binary_reader enclosing container after each element, so that the nesting depth of the input costs heap rather than native stack (see #5104). - @param[in] get_char whether a new character should be retrieved from the - input (true) or whether the last read character - @a current should be considered instead @param[in] tag_handler how CBOR tags should be treated @return whether reading the value succeeded */ - bool parse_cbor_internal(const bool get_char, - const cbor_tag_handler_t tag_handler) + bool parse_cbor_internal(const cbor_tag_handler_t tag_handler) { // whether the next value starts at a fresh byte or at the one already // read into `current` - bool fetch = get_char; + bool fetch = true; // the key currently being read; hoisted out of the loop so that its // capacity is reused across elements and across nesting levels @@ -16050,8 +16008,7 @@ class binary_reader if (at_end) { - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -16239,9 +16196,6 @@ class binary_reader // MsgPack // ///////////// - /*! - @return whether a valid MessagePack value was passed to the SAX parser - */ /*! @brief read one MessagePack value @@ -16945,8 +16899,7 @@ class binary_reader if (container_stack.back().remaining == 0) { - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -17140,20 +17093,16 @@ class binary_reader //////////// /*! - @param[in] get_char whether a new character should be retrieved from the - input (true, default) or whether the last read - character should be considered instead - @return whether a valid UBJSON value was passed to the SAX parser */ - bool parse_ubjson_internal(const bool get_char = true) + bool parse_ubjson_internal() { // the key currently being read; hoisted out of the loop so that its // capacity is reused across elements and across nesting levels string_t key; // the type marker of the value to read next - char_int_type prefix = get_char ? get_ignore_noop() : current; + char_int_type prefix = get_ignore_noop(); while (true) { @@ -17228,8 +17177,7 @@ class binary_reader break; } - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -17437,6 +17385,42 @@ class binary_reader return true; } + /*! + @brief read a UBJSON/BJData optimized-container count of a signed marker + type ('i', 'I', 'l', 'L') and narrow it to std::size_t + + Every signed count marker rejects a negative value the same way (error + 113); the value_in_range_of check additionally needed for 'L' is only + ever live when @a SignedType is std::int64_t on a target where + std::size_t is narrower (e.g. 32-bit), since 'i'/'I'/'l' can never exceed + std::size_t there. + + @tparam SignedType std::int8_t, std::int16_t, std::int32_t or std::int64_t + @param[out] result the count narrowed to std::size_t + @return whether reading and validating succeeded + */ + template + bool get_ubjson_signed_count(std::size_t& result) + { + SignedType number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) + { + return false; + } + if (JSON_HEDLEY_UNLIKELY(number < 0)) + { + return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read, + exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); + } + if (JSON_HEDLEY_UNLIKELY(!value_in_range_of(number))) + { + return report_error(chars_read, get_token_string(), out_of_range::create(408, + exception_message(input_format, "integer value overflow", "size"), nullptr)); + } + result = static_cast(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char + return true; + } + /*! @param[out] result determined size @param[in,out] is_ndarray for input, `true` means already inside an ndarray vector @@ -17469,73 +17453,16 @@ class binary_reader } case 'i': - { - std::int8_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - result = static_cast(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char - return true; - } + return get_ubjson_signed_count(result); case 'I': - { - std::int16_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - result = static_cast(number); - return true; - } + return get_ubjson_signed_count(result); case 'l': - { - std::int32_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - result = static_cast(number); - return true; - } + return get_ubjson_signed_count(result); case 'L': - { - std::int64_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return report_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - if (!value_in_range_of(number)) - { - return report_error(chars_read, get_token_string(), out_of_range::create(408, - exception_message(input_format, "integer value overflow", "size"), nullptr)); - } - result = static_cast(number); - return true; - } + return get_ubjson_signed_count(result); case 'u': { @@ -17627,16 +17554,23 @@ class binary_reader result = 1; for (auto i : dim) { - // Pre-multiplication overflow check: if i > 0 and result > SIZE_MAX/i, then result*i would overflow. - // This check must happen before multiplication since overflow detection after the fact is unreliable - // as modular arithmetic can produce any value, not just 0 or SIZE_MAX. - if (JSON_HEDLEY_UNLIKELY(i > 0 && result > (std::numeric_limits::max)() / i)) + // Pre-multiplication overflow check: since the loop above + // already rejected any zero dimension, i is always > 0 + // here, so result > SIZE_MAX/i means result*i would + // overflow. This check must happen before multiplication + // since overflow detection after the fact is unreliable, + // as modular arithmetic can produce any value, not just 0 + // or SIZE_MAX. + if (JSON_HEDLEY_UNLIKELY(result > (std::numeric_limits::max)() / i)) { return report_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr)); } result *= i; - // Additional post-multiplication check to catch any edge cases the pre-check might miss - if (result == 0 || result == npos) + // the pre-check above already rules out result becoming 0 + // by overflow; the only value it cannot rule out is an + // exact match with npos, the sentinel reserved for an + // unknown-size container (see get_ubjson_size_type()) + if (result == npos) { return report_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr)); } @@ -17698,7 +17632,7 @@ class binary_reader { result.second = get(); // must not ignore 'N', because 'N' maybe the type if (input_format == input_format_t::bjdata - && JSON_HEDLEY_UNLIKELY(std::binary_search(bjd_optimized_type_markers.begin(), bjd_optimized_type_markers.end(), result.second))) + && JSON_HEDLEY_UNLIKELY(is_bjd_excluded_optimized_type(result.second))) { auto last_token = get_token_string(); return report_error(chars_read, last_token, parse_error::create(112, chars_read, @@ -17842,50 +17776,7 @@ class binary_reader { break; } - const auto byte1_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number"))) - { - return false; - } - const auto byte2_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number"))) - { - return false; - } - - const auto byte1 = static_cast(byte1_raw); - const auto byte2 = static_cast(byte2_raw); - - // Code from RFC 8949, Appendix D, Figure 3: - // As half-precision floating-point numbers were only added - // to IEEE 754 in 2008, today's programming platforms often - // still only have limited support for them. It is very - // easy to include at least decoding support for them even - // without such support. An example of a small decoder for - // half-precision floating-point numbers in the C language - // is shown in Fig. 3. - const auto half = static_cast((byte2 << 8u) + byte1); - const double val = [&half] - { - const int exp = (half >> 10u) & 0x1Fu; - const unsigned int mant = half & 0x3FFu; - JSON_ASSERT(exp <= 31); - JSON_ASSERT(mant <= 1023); - switch (exp) - { - case 0: - return std::ldexp(mant, -24); - case 31: - return (mant == 0) - ? std::numeric_limits::infinity() - : std::numeric_limits::quiet_NaN(); - default: - return std::ldexp(mant + 1024, exp - 25); - } - }(); - return sax->number_float((half & 0x8000u) != 0 - ? static_cast(-val) - : static_cast(val), ""); + return get_half_float(input_format, true); } case 'd': @@ -17966,19 +17857,16 @@ class binary_reader if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0) { size_and_type.second &= ~(static_cast(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker - auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t) - { - return p.first < t; - }); + const char* type_name = bjd_type_name(size_and_type.second); string_t key = "_ArrayType_"; - if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second)) + if (JSON_HEDLEY_UNLIKELY(type_name == nullptr)) { auto last_token = get_token_string(); return report_error(chars_read, last_token, parse_error::create(112, chars_read, exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr)); } - string_t type = it->second; // sax->string() takes a reference + string_t type = type_name; // sax->string() takes a reference if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type))) { return false; @@ -18347,8 +18235,7 @@ class binary_reader if (at_end) { - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -19012,6 +18899,68 @@ class binary_reader return true; } + /*! + @brief read and decode an IEEE 754 half-precision (16-bit) float + + Used by CBOR (big endian) and BJData (little endian); the two formats + only differ in the byte order of the two bytes that make up the half. + + @param[in] format the current format (for diagnostics) + @param[in] little_endian whether the two bytes are little endian (BJData) + or big endian (CBOR) + + @return whether reading and decoding succeeded + */ + bool get_half_float(const input_format_t format, const bool little_endian) + { + const auto byte1_raw = get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number"))) + { + return false; + } + const auto byte2_raw = get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number"))) + { + return false; + } + + const auto byte1 = static_cast(byte1_raw); + const auto byte2 = static_cast(byte2_raw); + + // Code from RFC 8949, Appendix D, Figure 3: + // As half-precision floating-point numbers were only added + // to IEEE 754 in 2008, today's programming platforms often + // still only have limited support for them. It is very + // easy to include at least decoding support for them even + // without such support. An example of a small decoder for + // half-precision floating-point numbers in the C language + // is shown in Fig. 3. + const auto half = little_endian + ? static_cast((byte2 << 8u) + byte1) + : static_cast((byte1 << 8u) + byte2); + const double val = [&half] + { + const int exp = (half >> 10u) & 0x1Fu; + const unsigned int mant = half & 0x3FFu; + JSON_ASSERT(exp <= 31); + JSON_ASSERT(mant <= 1023); + switch (exp) + { + case 0: + return std::ldexp(mant, -24); + case 31: + return (mant == 0) + ? std::numeric_limits::infinity() + : std::numeric_limits::quiet_NaN(); + default: + return std::ldexp(mant + 1024, exp - 25); + } + }(); + return sax->number_float((half & 0x8000u) != 0 + ? static_cast(-val) + : static_cast(val), ""); + } + /*! @brief create a string by reading characters from the input @@ -19557,38 +19506,61 @@ class binary_reader /// open: none, its object, or its object and an array inside it std::uint8_t ndarray_open = 0; - // excluded markers in bjdata optimized type -#define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \ - make_array('F', 'H', 'N', 'S', 'T', 'Z', '[', '{') - -#define JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ \ - make_array( \ - bjd_type{'B', "byte"}, \ - bjd_type{'C', "char"}, \ - bjd_type{'D', "double"}, \ - bjd_type{'I', "int16"}, \ - bjd_type{'L', "int64"}, \ - bjd_type{'M', "uint64"}, \ - bjd_type{'U', "uint8"}, \ - bjd_type{'d', "single"}, \ - bjd_type{'i', "int8"}, \ - bjd_type{'l', "int32"}, \ - bjd_type{'m', "uint32"}, \ - bjd_type{'u', "uint16"}) - JSON_PRIVATE_UNLESS_TESTED: - // lookup tables - // NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes) - const decltype(JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_) bjd_optimized_type_markers = - JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_; + /*! + @brief whether @a marker is excluded from BJData's optimized ND-array types + @return whether @a marker is one of 'F', 'H', 'N', 'S', 'T', 'Z', '[', '{' - using bjd_type = std::pair; - // NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes) - const decltype(JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_) bjd_types_map = - JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_; + Mirrors binary_writer's @ref binary_writer::is_bjdata_excluded_type_marker + "is_bjdata_excluded_type_marker()`, which encodes the same list the other + way; keep the two in sync. + */ + static constexpr bool is_bjd_excluded_optimized_type(const char_int_type marker) noexcept + { + return marker == '[' || marker == '{' || marker == 'S' || marker == 'H' + || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; + } -#undef JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ -#undef JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ + /*! + @brief look up the ND-array element type name for a BJData dtype marker + @return the type name ("uint8", "int8", ...), or nullptr if @a marker does + not name a known dtype + + A C++11 `constexpr` function cannot contain a `switch`, so this is a + plain (non-constexpr) switch instead. + */ + static const char* bjd_type_name(const char_int_type marker) + { + switch (marker) + { + case 'B': + return "byte"; + case 'C': + return "char"; + case 'D': + return "double"; + case 'I': + return "int16"; + case 'L': + return "int64"; + case 'M': + return "uint64"; + case 'U': + return "uint8"; + case 'd': + return "single"; + case 'i': + return "int8"; + case 'l': + return "int32"; + case 'm': + return "uint32"; + case 'u': + return "uint16"; + default: + return nullptr; + } + } }; #ifndef JSON_HAS_CPP_17 @@ -21961,6 +21933,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include @@ -22052,7 +22026,7 @@ class json_pointer /// @sa https://json.nlohmann.me/api/json_pointer/operator_slasheq/ json_pointer& operator/=(std::size_t array_idx) { - return *this /= std::to_string(array_idx); + return *this /= detail::to_string(array_idx); } /// @brief create a new JSON pointer by appending the right JSON pointer at the end of the left JSON pointer @@ -22688,7 +22662,7 @@ class json_pointer // would throw out_of_range.404 -- contains() must not throw (see #5395) return false; } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !("0" <= reference_token && reference_token <= "9"))) + if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !('0' <= reference_token[0] && reference_token[0] <= '9'))) { // invalid char return false; @@ -22716,7 +22690,7 @@ class json_pointer // not throw (see #5395), so such a reference token is treated as "not found" errno = 0; // strtoull() does not reset errno on success char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.c_str(), &p_end, 10); // NOLINT(runtime/int) + const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) { @@ -23215,7 +23189,6 @@ NLOHMANN_JSON_NAMESPACE_END #include // reverse #include // array -#include // map #include // isnan, isinf #include // uint8_t, uint16_t, uint32_t, uint64_t #include // memcpy @@ -23545,7 +23518,7 @@ class binary_writer /*! @param[in] j JSON value to serialize - @pre j.type() == value_t::object + @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) { @@ -23634,7 +23607,7 @@ class binary_writer } else { - write_compact_float(j.m_data.m_value.number_float, detail::input_format_t::cbor); + write_compact_float(j.m_data.m_value.number_float, to_char_type(0xFA), to_char_type(0xFB)); } break; } @@ -23668,6 +23641,13 @@ class binary_writer { if (j.m_data.m_value.binary->has_subtype()) { + // The subtype is always written as a tag with a 0xD8..0xDB + // head, never in the one-byte form 0xC0..0xD7 that CBOR + // allows for tags 0..23 (so this is not write_cbor_head). + // binary_reader with cbor_tag_handler_t::store only turns + // 0xD8..0xDB into a subtype and ignores the one-byte tags, + // so the shorter form would lose subtypes 0..23 on a round + // trip. if (j.m_data.m_value.binary->subtype() <= (std::numeric_limits::max)()) { write_number(static_cast(0xd8)); @@ -23739,6 +23719,43 @@ class binary_writer return static_cast(length); } + /*! + @brief write a non-negative integer using the MessagePack fixint/uint ladder + @param[in] n the value to write, already known to be non-negative + */ + void write_msgpack_unsigned(const std::uint64_t n) + { + if (n < 128) + { + // positive fixnum + write_number(static_cast(n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // uint 8 + oa.write_character(to_char_type(0xCC)); + write_number(static_cast(n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // uint 16 + oa.write_character(to_char_type(0xCD)); + write_number(static_cast(n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // uint 32 + oa.write_character(to_char_type(0xCE)); + write_number(static_cast(n)); + } + else + { + // uint 64 + oa.write_character(to_char_type(0xCF)); + write_number(n); + } + } + /*! @param[in] j JSON value to serialize */ @@ -23765,38 +23782,8 @@ class binary_writer if (j.m_data.m_value.number_integer >= 0) { // MessagePack does not differentiate between positive - // signed integers and unsigned integers. Therefore, we used - // the code from the value_t::number_unsigned case here. - const auto value_as_unsigned = static_cast(j.m_data.m_value.number_integer); - if (value_as_unsigned < 128) - { - // positive fixnum - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (value_as_unsigned <= (std::numeric_limits::max)()) - { - // uint 8 - oa.write_character(to_char_type(0xCC)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (value_as_unsigned <= (std::numeric_limits::max)()) - { - // uint 16 - oa.write_character(to_char_type(0xCD)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (value_as_unsigned <= (std::numeric_limits::max)()) - { - // uint 32 - oa.write_character(to_char_type(0xCE)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else - { - // uint 64 - oa.write_character(to_char_type(0xCF)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } + // signed integers and unsigned integers. + write_msgpack_unsigned(static_cast(j.m_data.m_value.number_integer)); } else { @@ -23838,41 +23825,13 @@ class binary_writer case value_t::number_unsigned: { - if (j.m_data.m_value.number_unsigned < 128) - { - // positive fixnum - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - // uint 8 - oa.write_character(to_char_type(0xCC)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - // uint 16 - oa.write_character(to_char_type(0xCD)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - // uint 32 - oa.write_character(to_char_type(0xCE)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else - { - // uint 64 - oa.write_character(to_char_type(0xCF)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } + write_msgpack_unsigned(static_cast(j.m_data.m_value.number_unsigned)); break; } case value_t::number_float: { - write_compact_float(j.m_data.m_value.number_float, detail::input_format_t::msgpack); + write_compact_float(j.m_data.m_value.number_float, to_char_type(0xCA), to_char_type(0xCB)); break; } @@ -24813,6 +24772,14 @@ class binary_writer oa.write_character(to_char_type(0x00)); if (parents.empty()) { + // calc_bson_sizes() and write_bson_document() are two + // hand-synchronized passes over the same structure, linked + // only by nested_sizes' visiting order; this checks that the + // write pass consumed exactly the sizes the size pass + // produced, so a future change that desyncs them (skips or + // rejects an entry in only one pass) is caught immediately + // instead of silently writing wrong length prefixes. + JSON_ASSERT(next_size == nested_sizes.size()); return; } current = std::move(parents.back()); @@ -24864,52 +24831,6 @@ class binary_writer } } - static constexpr CharType get_cbor_float_prefix(float /*unused*/) - { - return to_char_type(0xFA); // Single-Precision Float - } - - static constexpr CharType get_cbor_float_prefix(double /*unused*/) - { - return to_char_type(0xFB); // Double-Precision Float - } - - ///////////// - // MsgPack // - ///////////// - - static constexpr CharType get_msgpack_float_prefix(float /*unused*/) - { - return to_char_type(0xCA); // float 32 - } - - static constexpr CharType get_msgpack_float_prefix(double /*unused*/) - { - return to_char_type(0xCB); // float 64 - } - - /// @return the BON8 type marker for binary32 (float) or binary64 (double) - template - static constexpr CharType get_bon8_float_prefix() - { - return to_char_type(std::is_same::value ? 0x8E : 0x8F); - } - - /// @return the type marker for a FloatType value in @a format (CBOR, MessagePack, or BON8) - template - static CharType get_compact_float_prefix(const detail::input_format_t format) - { - if (format == detail::input_format_t::cbor) - { - return get_cbor_float_prefix(FloatType{}); - } - if (format == detail::input_format_t::bon8) - { - return get_bon8_float_prefix(); - } - return get_msgpack_float_prefix(FloatType{}); - } - //////////// // UBJSON // //////////// @@ -24923,207 +24844,127 @@ class binary_writer { if (add_prefix) { - oa.write_character(get_ubjson_float_prefix(n)); + oa.write_character(get_ubjson_float_prefix()); } write_number(n, use_bjdata); } - // UBJSON: write number (unsigned integer) + // UBJSON: write number (integer) template::value, int>::type = 0> + std::is_integral::value, int>::type = 0> void write_number_with_ubjson_prefix(const NumberType n, const bool add_prefix, const bool use_bjdata) { - if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('i')); // int8 - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('U')); // uint8 - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('I')); // int16 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('u')); // uint16 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('l')); // int32 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('m')); // uint32 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('L')); // int64 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata) - { - if (add_prefix) - { - oa.write_character(to_char_type('M')); // uint64 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else - { - if (add_prefix) - { - oa.write_character(to_char_type('H')); // high-precision number - } - - const auto number = BasicJsonType(n).dump(); - write_number_with_ubjson_prefix(number.size(), true, use_bjdata); - for (std::size_t i = 0; i < number.size(); ++i) - { - oa.write_character(to_char_type(static_cast(number[i]))); - } - } - } - - // UBJSON: write number (signed integer) - template < typename NumberType, typename std::enable_if < - std::is_signed::value&& - !std::is_floating_point::value, int >::type = 0 > - void write_number_with_ubjson_prefix(const NumberType n, - const bool add_prefix, - const bool use_bjdata) - { - if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('i')); // int8 - } - write_number(static_cast(n), use_bjdata); - } - else if (static_cast((std::numeric_limits::min)()) <= n && n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('U')); // uint8 - } - write_number(static_cast(n), use_bjdata); - } - else if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('I')); // int16 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && (static_cast((std::numeric_limits::min)()) <= n && n <= static_cast((std::numeric_limits::max)()))) - { - if (add_prefix) - { - oa.write_character(to_char_type('u')); // uint16 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('l')); // int32 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && (static_cast((std::numeric_limits::min)()) <= n && n <= static_cast((std::numeric_limits::max)()))) - { - if (add_prefix) - { - oa.write_character(to_char_type('m')); // uint32 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else - { - // every value of an integer type of at most 64 bits fits into an - // int64; only a wider type needs a range check - write_ubjson_int64_or_high_precision(n, add_prefix, use_bjdata, - std::integral_constant < bool, std::numeric_limits::digits <= std::numeric_limits::digits > {}); - } - } - - template - void write_ubjson_int64_or_high_precision(const NumberType n, const bool add_prefix, const bool use_bjdata, std::true_type /*fits_int64*/) - { + const CharType prefix = ubjson_integer_prefix(n, use_bjdata); if (add_prefix) { - oa.write_character(to_char_type('L')); // int64 + oa.write_character(prefix); } - write_number(static_cast(n), use_bjdata); + write_ubjson_integer_payload(prefix, n, use_bjdata); } + /*! + @brief determine the UBJSON/BJData type marker of an integer + + This is the only place that picks the marker of an integer: both + write_number_with_ubjson_prefix() and ubjson_prefix() use it. An optimized + container announces the marker of its first value after `$` and then + writes every value without a marker, so the two must never disagree. + + @param[in] n the integer + @param[in] use_bjdata whether the BJData-only markers `u`, `m`, and `M` + may be used + + @return the first marker of `i`, `U`, `I`, `u` (BJData), `l`, `m` (BJData), + `L`, `M` (BJData, unsigned types only), and `H` (high-precision + number) whose range contains @a n + */ template - void write_ubjson_int64_or_high_precision(const NumberType n, const bool add_prefix, const bool use_bjdata, std::false_type /*fits_int64*/) + static CharType ubjson_integer_prefix(const NumberType n, const bool use_bjdata) noexcept { - if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) + if (value_in_range_of(n)) { - write_ubjson_int64_or_high_precision(n, add_prefix, use_bjdata, std::true_type {}); - return; + return 'i'; } - - if (add_prefix) + if (value_in_range_of(n)) { - oa.write_character(to_char_type('H')); // high-precision number + return 'U'; } - - const auto number = BasicJsonType(n).dump(); - write_number_with_ubjson_prefix(number.size(), true, use_bjdata); - for (std::size_t i = 0; i < number.size(); ++i) + if (value_in_range_of(n)) { - oa.write_character(to_char_type(static_cast(number[i]))); + return 'I'; } + if (use_bjdata && value_in_range_of(n)) + { + return 'u'; + } + if (value_in_range_of(n)) + { + return 'l'; + } + if (use_bjdata && value_in_range_of(n)) + { + return 'm'; + } + if (value_in_range_of(n)) + { + return 'L'; + } + if (use_bjdata && std::is_unsigned::value) + { + return 'M'; + } + // anything else is treated as a high-precision number + return 'H'; } + /*! + @brief write the value of an integer for the marker chosen by + ubjson_integer_prefix() + */ template - static constexpr CharType ubjson_int64_or_high_precision_prefix(const NumberType /*n*/, std::true_type /*fits_int64*/) noexcept + void write_ubjson_integer_payload(const CharType prefix, const NumberType n, const bool use_bjdata) { - return 'L'; - } - - template - static CharType ubjson_int64_or_high_precision_prefix(const NumberType n, std::false_type /*fits_int64*/) noexcept - { - // anything outside of the range of an int64 is treated as a - // high-precision number - return ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) ? 'L' : 'H'; + switch (prefix) + { + case 'i': + write_number(static_cast(n), use_bjdata); + break; + case 'U': + write_number(static_cast(n), use_bjdata); + break; + case 'I': + write_number(static_cast(n), use_bjdata); + break; + case 'u': + write_number(static_cast(n), use_bjdata); + break; + case 'l': + write_number(static_cast(n), use_bjdata); + break; + case 'm': + write_number(static_cast(n), use_bjdata); + break; + case 'L': + write_number(static_cast(n), use_bjdata); + break; + case 'M': + write_number(static_cast(n), use_bjdata); + break; + default: + { + // high-precision number: the decimal digits as a string + JSON_ASSERT(prefix == 'H'); + const auto number = BasicJsonType(n).dump(); + write_number_with_ubjson_prefix(number.size(), true, use_bjdata); + for (std::size_t i = 0; i < number.size(); ++i) + { + oa.write_character(to_char_type(static_cast(number[i]))); + } + break; + } + } } /*! @@ -25140,77 +24981,13 @@ class binary_writer return j.m_data.m_value.boolean ? 'T' : 'F'; case value_t::number_integer: - { - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'i'; - } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'U'; - } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'I'; - } - if (use_bjdata && ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)())) - { - return 'u'; - } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'l'; - } - if (use_bjdata && ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)())) - { - return 'm'; - } - // every value of an integer type of at most 64 bits fits into - // an int64; only a wider type needs a range check - return ubjson_int64_or_high_precision_prefix(j.m_data.m_value.number_integer, - std::integral_constant < bool, std::numeric_limits::digits <= std::numeric_limits::digits > {}); - } + return ubjson_integer_prefix(j.m_data.m_value.number_integer, use_bjdata); case value_t::number_unsigned: - { - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'i'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'U'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'I'; - } - if (use_bjdata && j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'u'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'l'; - } - if (use_bjdata && j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'm'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'L'; - } - if (use_bjdata) - { - return 'M'; - } - // anything else is treated as a high-precision number - return 'H'; - } + return ubjson_integer_prefix(j.m_data.m_value.number_unsigned, use_bjdata); case value_t::number_float: - return get_ubjson_float_prefix(j.m_data.m_value.number_float); + return get_ubjson_float_prefix(); case value_t::string: return 'S'; @@ -25235,7 +25012,7 @@ class binary_writer Containers, strings, high-precision numbers, booleans and null cannot be declared as the single type of an optimized container in BJData; such a container is written unoptimized. The reader rejects them with the same - list (binary_reader::bjd_optimized_type_markers). + list (binary_reader::is_bjd_excluded_optimized_type()). */ static constexpr bool is_bjdata_excluded_type_marker(const CharType marker) noexcept { @@ -25243,14 +25020,16 @@ class binary_writer || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; } - static constexpr CharType get_ubjson_float_prefix(float /*unused*/) + /// @return the UBJSON/BJData type marker for a float or double value + /// + /// number_float_t must be float or double; a static_assert (rather than + /// an ambiguous overload) reports an unsupported number_float_t clearly. + template + static constexpr CharType get_ubjson_float_prefix() { - return 'd'; // float 32 - } - - static constexpr CharType get_ubjson_float_prefix(double /*unused*/) - { - return 'D'; // float 64 + static_assert(std::is_same::value || std::is_same::value, + "number_float_t must be float or double for the UBJSON/BJData writer"); + return std::is_same::value ? 'd' : 'D'; // float 32 / float 64 } /*! @@ -25267,35 +25046,184 @@ class binary_writer : value_in_range_of(el.template get()); } + /*! + @brief look up the BJData ND-array dtype marker for an `_ArrayType_` name + @return the one-character marker, or '\0' if @a name does not name a known dtype + + A C++11 `constexpr` function cannot contain a `switch`, so this is a plain + comparison chain instead; it is only reached once per ND-array candidate + object. Keep in sync with binary_reader's `bjd_type_name()`, which maps + the other way. + */ + static CharType bjdata_ndarray_type_marker(const string_t& name) + { + if (name == "uint8") + { + return 'U'; + } + if (name == "int8") + { + return 'i'; + } + if (name == "uint16") + { + return 'u'; + } + if (name == "int16") + { + return 'I'; + } + if (name == "uint32") + { + return 'm'; + } + if (name == "int32") + { + return 'l'; + } + if (name == "uint64") + { + return 'M'; + } + if (name == "int64") + { + return 'L'; + } + if (name == "single") + { + return 'd'; + } + if (name == "double") + { + return 'D'; + } + if (name == "char") + { + return 'C'; + } + if (name == "byte") + { + return 'B'; + } + return '\0'; + } + + /*! + @brief validate (dry_run) or write one BJData ND-array element of integer dtype @a T + @return whether @a el's value is in range of @a T; always true when @a dry_run is false + */ + template + bool write_bjdata_ndarray_element(const BasicJsonType& el, const bool dry_run) + { + if (dry_run) + { + return bjdata_ndarray_value_in_range(el); + } + using storage_type = typename std::conditional::value, std::uint64_t, std::int64_t>::type; + write_number(static_cast(el.template get()), true); + return true; + } + + /*! + @brief validate (dry_run) or write one BJData ND-array element of dtype 'd' (single precision) + @return whether @a el's value fits a float without overflow; always true when @a dry_run is false + */ + bool write_bjdata_ndarray_float_element(const BasicJsonType& el, const bool dry_run) + { + const auto dval = el.template get(); + if (dry_run) + { + return !std::isfinite(dval) || + (dval >= static_cast(std::numeric_limits::lowest()) && + dval <= static_cast((std::numeric_limits::max)())); + } + write_number(static_cast(dval), true); + return true; + } + + /*! + @brief validate or write every element of a BJData ND-array's `_ArrayData_` + @param[in] array_data the `_ArrayData_` array + @param[in] dtype the ND-array dtype marker, as returned by bjdata_ndarray_type_marker() + @param[in] dry_run true to only range-check each element, false to write it + @return whether every element is in range for @a dtype (always true when @a dry_run is false) + */ + bool write_bjdata_ndarray_elements(const BasicJsonType& array_data, const CharType dtype, const bool dry_run) + { + for (const auto& el : array_data) + { + bool ok = true; + switch (dtype) + { + case 'U': + case 'C': + case 'B': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'i': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'u': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'I': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'm': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'l': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'M': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'L': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'd': + ok = write_bjdata_ndarray_float_element(el, dry_run); + break; + case 'D': + default: + // 'D' (double) already spans the full range of number_float_t + if (!dry_run) + { + write_number(el.template get(), true); + } + break; + } + if (!ok) + { + return false; + } + } + return true; + } + /*! @return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid */ bool write_bjdata_ndarray(const typename BasicJsonType::object_t& value, const bool use_count, const bool use_type, const bjdata_version_t bjdata_version) { - std::map bjdtype = {{"uint8", 'U'}, {"int8", 'i'}, {"uint16", 'u'}, {"int16", 'I'}, - {"uint32", 'm'}, {"int32", 'l'}, {"uint64", 'M'}, {"int64", 'L'}, {"single", 'd'}, {"double", 'D'}, - {"char", 'C'}, {"byte", 'B'} - }; - - string_t key = "_ArrayType_"; + const auto& array_type = value.at("_ArrayType_"); // the type name is looked up as a string below; a non-string // annotation (e.g. a number, null, or an array) cannot name a known // dtype, so it is treated the same as an unrecognized type name and // falls back to a plain object encoding instead of throwing // type_error.302 out of get() - if (!value.at(key).is_string()) + if (!array_type.is_string()) { return true; } // use get() instead of static_cast to avoid an // ambiguous conversion under explicit instantiation on C++17 (see #4825) - auto it = bjdtype.find(value.at(key).template get()); - if (it == bjdtype.end()) + const CharType dtype = bjdata_ndarray_type_marker(array_type.template get()); + if (dtype == '\0') { return true; } - CharType dtype = it->second; // the 'B' (byte) marker is only defined from BJData Draft 3 onward; // emitting it under an earlier draft would produce a stream that an @@ -25307,12 +25235,12 @@ class binary_writer return true; } - key = "_ArraySize_"; + const auto& array_size = value.at("_ArraySize_"); // the dimensions are written verbatim as the header length below, so a // value that is not an array cannot produce a valid one: null emits 'Z' // and an object emits '{', neither of which a reader accepts after '#'. // Such an object is not a valid ndarray and falls back to a plain object. - if (!value.at(key).is_array()) + if (!array_size.is_array()) { return true; } @@ -25322,7 +25250,7 @@ class binary_writer // dimension, or a 1xN row vector is read back as a plain array, which // would silently drop the annotation, so such an object falls back to // a plain object encoding instead - const auto& dims = value.at(key); + const auto& dims = array_size; if (dims.size() < 2 || (dims.size() == 2 && dims.at(0).is_number_integer() && dims.at(0).template get() == 1)) { return true; @@ -25369,8 +25297,8 @@ class binary_writer // to be an array: size() is 0 for null and 1 for any other scalar, and // iterating an object visits its values, so any of these could match // the dimensions by accident and be encoded as an unrelated ND-array - key = "_ArrayData_"; - if (!value.at(key).is_array() || value.at(key).size() != len) + const auto& array_data = value.at("_ArrayData_"); + if (!array_data.is_array() || array_data.size() != len) { return true; } @@ -25385,7 +25313,7 @@ class binary_writer // API stores int literals as signed), so both are accepted here and the // writes below go through get<>, which reads the member that is active. const bool ndarray_is_float = (dtype == 'd' || dtype == 'D'); - for (const auto& el : value.at(key)) + for (const auto& el : array_data) { if (ndarray_is_float ? !el.is_number_float() : !el.is_number_integer()) { @@ -25398,134 +25326,19 @@ class binary_writer // wrap (integers) or overflow to infinity (the "single" precision // float) instead of being reported, so such an object falls back to // a plain object encoding as well - for (const auto& el : value.at(key)) + if (!write_bjdata_ndarray_elements(array_data, dtype, true)) { - bool in_range = true; - switch (dtype) - { - case 'U': - case 'C': - case 'B': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'i': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'u': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'I': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'm': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'l': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'M': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'L': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'd': - { - const auto dval = el.template get(); - in_range = !std::isfinite(dval) || - (dval >= static_cast(std::numeric_limits::lowest()) && - dval <= static_cast((std::numeric_limits::max)())); - break; - } - default: - // 'D' (double) already spans the full range of number_float_t - break; - } - if (!in_range) - { - return true; - } + return true; } - oa.write_character('['); - oa.write_character('$'); + oa.write_character(to_char_type('[')); + oa.write_character(to_char_type('$')); oa.write_character(dtype); - oa.write_character('#'); + oa.write_character(to_char_type('#')); - key = "_ArraySize_"; - write_ubjson(value.at(key), use_count, use_type, true, true, bjdata_version); + write_ubjson(array_size, use_count, use_type, true, true, bjdata_version); - key = "_ArrayData_"; - if (dtype == 'U' || dtype == 'C' || dtype == 'B') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'i') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'u') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'I') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'm') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'l') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'M') - { - for (const auto& el : value.at(key)) - { - write_number(el.template get(), true); - } - } - else if (dtype == 'L') - { - for (const auto& el : value.at(key)) - { - write_number(el.template get(), true); - } - } - else if (dtype == 'd') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'D') - { - for (const auto& el : value.at(key)) - { - write_number(el.template get(), true); - } - } + write_bjdata_ndarray_elements(array_data, dtype, false); return false; } @@ -25838,7 +25651,7 @@ class binary_writer } else { - write_compact_float(n, detail::input_format_t::bon8); + write_compact_float(n, to_char_type(0x8E), to_char_type(0x8F)); } #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP @@ -25849,19 +25662,6 @@ class binary_writer // Utility functions // /////////////////////// - /* - @brief write a number to output input - @param[in] n number of type @a NumberType - @param[in] OutputIsLittleEndian Set to true if output data is - required to be little endian - @tparam NumberType the type of the number - - @note This function needs to respect the system's endianness, because bytes - in CBOR, MessagePack, and UBJSON are stored in network order (big - endian) and therefore need reordering on little endian systems. - On the other hand, BSON and BJData use little endian and should reorder - on big endian systems. - */ // single-instruction byte swaps (compilers lower these to bswap/rev/movbe); // used to emit big-endian numbers without a per-byte std::reverse loop static std::uint16_t byte_swap(std::uint16_t x) noexcept @@ -25943,6 +25743,19 @@ class binary_writer std::reverse(a.begin(), a.end()); } + /*! + @brief write a number to the output + @param[in] n number of type @a NumberType + @param[in] OutputIsLittleEndian Set to true if output data is + required to be little endian + @tparam NumberType the type of the number + + @note This function needs to respect the system's endianness, because bytes + in CBOR, MessagePack, UBJSON, and BON8 are stored in network order + (big endian) and therefore need reordering on little endian systems. + On the other hand, BSON and BJData use little endian and should + reorder on big endian systems. + */ template void write_number(const NumberType n, const bool OutputIsLittleEndian = false) { @@ -25960,8 +25773,17 @@ class binary_writer oa.write_characters(vec.data(), sizeof(NumberType)); } - void write_compact_float(const number_float_t n, detail::input_format_t format) + /// @brief write @a n using @a float32_marker if it round-trips through + /// float, otherwise using @a float64_marker + /// + /// @a float32_marker and @a float64_marker are the format-specific type + /// markers (CBOR: 0xFA/0xFB, MessagePack: 0xCA/0xCB, BON8: 0x8E/0x8F); + /// each caller already knows them at compile time, so the format itself + /// no longer needs to be passed in. + void write_compact_float(const number_float_t n, const CharType float32_marker, const CharType float64_marker) { + static_assert(std::is_same::value || std::is_same::value, + "number_float_t must be float or double for the CBOR/MessagePack/BON8 writer"); #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_PUSH JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") @@ -25979,12 +25801,12 @@ class binary_writer static_cast(n) <= static_cast((std::numeric_limits::max)()) && static_cast(static_cast(n)) == static_cast(n)))) { - oa.write_character(get_compact_float_prefix(format)); + oa.write_character(float32_marker); write_number(static_cast(n)); } else { - oa.write_character(get_compact_float_prefix(format)); + oa.write_character(float64_marker); write_number(n); } #ifdef __GNUC__ @@ -25993,7 +25815,7 @@ class binary_writer } public: - // The following to_char_type functions are implement the conversion + // The following to_char_type functions implement the conversion // between uint8_t and CharType. In case CharType is not unsigned, // such a conversion is required to allow values greater than 128. // See for a discussion. @@ -30156,11 +29978,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /*! @brief how many levels the operation going on in this thread has descended into - Copying a value and comparing two values share this count. The library never - nests one inside the other - copying a value does not compare one, and - comparing two values does not copy them - and where user code nests them - anyway, sharing the count only ends a descent sooner than it had to, which - costs a little speed and is never wrong. + Copying a value, converting one from another specialization, and comparing + two values share this count. The library never nests one of them inside + another - none of them does either of the other two on the way - and where + user code nests them anyway, sharing the count only ends a descent sooner + than it had to, which costs a little speed and is never wrong. A byte is enough: the count never exceeds the limit by more than the single level that notices the limit has been reached. @@ -30517,6 +30339,222 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec copy_iteratively(src); } + /*! + @brief convert the value @a val of another specialization into this null + value; @a val must be neither an object nor an array + + Converting such a value never descends, so both ways of converting an + object or an array (@ref convert_structured) leave their elements of this + kind to the converting constructor, which leaves them to this. + */ + template + void convert_leaf(const BasicJsonType& val) + { + using other_boolean_t = typename BasicJsonType::boolean_t; + using other_number_float_t = typename BasicJsonType::number_float_t; + using other_number_integer_t = typename BasicJsonType::number_integer_t; + using other_number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using other_string_t = typename BasicJsonType::string_t; + using other_binary_t = typename BasicJsonType::binary_t; + + switch (val.type()) + { + case value_t::boolean: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::number_float: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::number_integer: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::number_unsigned: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::string: + JSONSerializer::to_json(*this, val.template get_ref()); + break; + case value_t::binary: + JSONSerializer::to_json(*this, val.template get_ref()); + break; + case value_t::null: + // m_data.m_type is already value_t::null + break; + case value_t::discarded: + m_data.m_type = value_t::discarded; + break; + case value_t::object: // LCOV_EXCL_LINE + case value_t::array: // LCOV_EXCL_LINE + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + } + } + + /// scratch space for the converted elements of the arrays that + /// @ref convert_iteratively has yet to create + using convert_scratch_t = std::vector>; + + /*! + @brief create the object or array @a val converted into this null value + + Its converted elements are the last `val.size()` entries of @a elements (an + array) or of @a members (an object); they are moved into the container in + one go and then removed. + */ + template + void convert_level(const BasicJsonType& val, convert_scratch_t& elements, copy_scratch_t& members) + { + if (val.is_object()) + { + const auto first = members.end() - static_cast(val.size()); + m_data.m_value.object = create(std::make_move_iterator(first), + std::make_move_iterator(members.end())); + // only now that the object exists may this stop being a null value + m_data.m_type = value_t::object; + members.erase(first, members.end()); + } + else + { + const auto first = elements.end() - static_cast(val.size()); + m_data.m_value.array = create(std::make_move_iterator(first), + std::make_move_iterator(elements.end())); + // only now that the array exists may this stop being a null value + m_data.m_type = value_t::array; + elements.erase(first, elements.end()); + } + + set_parents(); + } + + /*! + @brief convert the object or array @a val of another specialization into + this null value without recursing + + The containers whose conversion has begun are kept on an explicit stack + rather than on the call stack. Unlike @ref copy_iteratively, this builds + every container from the bottom up: all its elements are converted first, + and the container is then created from them in one go, the way the range + constructor that converts the levels above the bound does. The two object + types need not enumerate their members in the same order, so the members + could not be paired up by position anyway, and building from a range keeps + what the range constructor does with keys that become equal on conversion. + + Every value is complete before it is handed on, and a container gets its + type only once it exists, so whatever throws, every value left behind can + be destroyed. + */ + template + void convert_iteratively(const BasicJsonType& val) + { + using other_const_iterator = typename BasicJsonType::const_iterator; + + // the containers whose conversion has begun, innermost last, each with + // its element to convert next + std::vector> pending; + + // the converted elements of the pending arrays and the converted + // members of the pending objects, those of the innermost one last + convert_scratch_t elements; + copy_scratch_t members; + + pending.emplace_back(&val, val.cbegin()); + + for (;;) + { + const BasicJsonType& container = *pending.back().first; + // a copy, as descending below can reallocate pending; the + // iterator kept in pending is only advanced through pending.back() + const other_const_iterator next = pending.back().second; + + if (next != container.cend()) + { + if (next->is_structured()) + { + // convert its elements first; next stays where it is until + // the converted container is handed back to this one + pending.emplace_back(&*next, next->cbegin()); + continue; + } + + // the converting constructor does not descend into this value + if (container.is_object()) + { + members.emplace_back(next.key(), *next); + } + else + { + elements.emplace_back(*next); + } + ++pending.back().second; + continue; + } + + // all elements of the container are converted: create it + pending.pop_back(); + + if (pending.empty()) + { + convert_level(container, elements, members); + return; + } + + basic_json converted; + converted.convert_level(container, elements, members); +#if JSON_DIAGNOSTIC_POSITIONS + converted.start_position = container.start_pos(); + converted.end_position = container.end_pos(); +#endif + + // hand it to the container it is an element of + if (pending.back().first->is_object()) + { + members.emplace_back(pending.back().second.key(), std::move(converted)); + } + else + { + elements.push_back(std::move(converted)); + } + ++pending.back().second; + } + } + + /*! + @brief convert the object or array @a val of another specialization into + this null value + + Converting a container converts its elements, so a value nested deeply + enough used to exhaust the call stack. The descent is bounded here as in + @ref copy_structured: the first `detail::recursion_depth_limit()` levels + are converted by the containers' range constructors, just as they always were, + and anything below that is converted without the call stack by + @ref convert_iteratively. + + @sa https://github.com/nlohmann/json/issues/5650 + */ + template + void convert_structured(const BasicJsonType& val) + { + const nesting_depth_guard guard; + + if (JSON_HEDLEY_LIKELY(guard.okay())) + { + // every element comes back to the converting constructor + if (val.is_object()) + { + using other_object_t = typename BasicJsonType::object_t; + JSONSerializer::to_json(*this, val.template get_ref()); + } + else + { + using other_array_t = typename BasicJsonType::array_t; + JSONSerializer::to_json(*this, val.template get_ref()); + } + return; + } + + convert_iteratively(val); + } + /// the result of comparing two values, including values that cannot be /// ordered at all, such as a discarded value or a NaN @@ -30846,7 +30884,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec template < typename CompatibleType, typename U = detail::uncvref_t, detail::enable_if_t < - !detail::is_basic_json::value && detail::is_compatible_type::value, int > = 0 > + !detail::is_basic_json::value && detail::is_compatible_type::value +#if JSON_DISABLE_TUPLE_REFERENCE_CONVERSION + // see https://github.com/nlohmann/json/issues/2226 + && !detail::is_basic_json_reference_tuple::value +#endif + , int > = 0 > basic_json(CompatibleType && val) noexcept(noexcept( // NOLINT(bugprone-forwarding-reference-overload,bugprone-exception-escape) JSONSerializer::to_json(std::declval(), std::forward(val)))) @@ -30867,49 +30910,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec end_position(val.end_pos()) #endif { - using other_boolean_t = typename BasicJsonType::boolean_t; - using other_number_float_t = typename BasicJsonType::number_float_t; - using other_number_integer_t = typename BasicJsonType::number_integer_t; - using other_number_unsigned_t = typename BasicJsonType::number_unsigned_t; - using other_string_t = typename BasicJsonType::string_t; - using other_object_t = typename BasicJsonType::object_t; - using other_array_t = typename BasicJsonType::array_t; - using other_binary_t = typename BasicJsonType::binary_t; - - switch (val.type()) + if (val.is_structured()) { - case value_t::boolean: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::number_float: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::number_integer: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::number_unsigned: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::string: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::object: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::array: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::binary: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::null: - *this = nullptr; - break; - case value_t::discarded: - m_data.m_type = value_t::discarded; - break; - default: // LCOV_EXCL_LINE - JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + convert_structured(val); + } + else + { + convert_leaf(val); } JSON_ASSERT(m_data.m_type == val.type()); @@ -34284,7 +34291,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = detail::input_adapter(std::forward(i)); return format == input_format_t::json ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) - : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } /// @brief generate SAX events (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -34301,7 +34308,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = detail::input_adapter(std::move(first), std::move(last)); return format == input_format_t::json ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) - : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } /// @brief generate SAX events @@ -34323,7 +34330,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } #ifndef JSON_NO_IO /// @brief deserialize from stream @@ -34342,7 +34349,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/operator_gtgt/ friend std::istream& operator>>(std::istream& i, basic_json& j) { - parser(detail::input_adapter(i)).parse(false, j); + // parse into a temporary so that j is left unchanged if parsing fails + basic_json result; + parser(detail::input_adapter(i)).parse(false, result); + j = std::move(result); return i; } #endif // JSON_NO_IO @@ -34618,7 +34628,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34638,7 +34648,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34667,7 +34677,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34685,7 +34695,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34704,7 +34714,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34731,7 +34741,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34749,7 +34759,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34768,7 +34778,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34795,7 +34805,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34813,7 +34823,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34832,7 +34842,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34850,7 +34860,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(input_format_t::bon8, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34869,7 +34879,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(input_format_t::bon8, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34887,7 +34897,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34906,7 +34916,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -34933,7 +34943,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -35892,6 +35902,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_INLINE_VARIABLE #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION +#undef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH diff --git a/tests/src/unit-allocator.cpp b/tests/src/unit-allocator.cpp index ce498558d..151c268d7 100644 --- a/tests/src/unit-allocator.cpp +++ b/tests/src/unit-allocator.cpp @@ -479,6 +479,77 @@ TEST_CASE("deep copy uses the provided allocator") CHECK(copy == j); } +namespace +{ +// the number of constructions countdown_allocator lets happen, including the +// one that fails; 0 means none ever fails +std::size_t constructions_until_failure = 0; + +template +struct countdown_allocator : std::allocator +{ + using std::allocator::allocator; + + template + void construct(U* p, Args&& ... args) + { + if (constructions_until_failure != 0 && --constructions_until_failure == 0) + { + throw std::bad_alloc(); + } + + ::new (static_cast(p)) U(std::forward(args)...); + } + + template + struct rebind + { + using other = countdown_allocator; + }; +}; +} // namespace + +TEST_CASE("converting a deeply nested value from another specialization fails cleanly (#5650)") +{ + using countdown_json = nlohmann::basic_json; + + // deeper than the 128 levels the converting constructor descends into, so + // that failures land on both sides of the bound - or, built with + // JSON_NO_THREAD_LOCAL, all in the iterative conversion + json j = {1, "two", {{"three", 3}}}; + for (std::size_t i = 0; i < 150; ++i) + { + j = json{{"a", json::array({j, "sibling"})}}; + } + + // Fail every construction in turn. Each failure has to reach the caller, + // and everything built until then has to be destroyed cleanly. + std::size_t failures = 0; + for (std::size_t n = 1;; ++n) + { + constructions_until_failure = n; + try + { + const countdown_json converted = j; + constructions_until_failure = 0; + CHECK(converted.dump() == j.dump()); + break; + } + catch (const std::bad_alloc&) + { + ++failures; + } + } + CHECK(failures > 0); +} + namespace { template diff --git a/tests/src/unit-alt-string.cpp b/tests/src/unit-alt-string.cpp index 421d5730d..2d58860b7 100644 --- a/tests/src/unit-alt-string.cpp +++ b/tests/src/unit-alt-string.cpp @@ -352,6 +352,41 @@ TEST_CASE("alternative string type") CHECK(j2.flatten().unflatten() == j2); } + SECTION("contains(json_pointer)") + { + // contains(json_pointer) must compile and work with a string_t that has + // no c_str() and no comparison with const char* (see #5666) + auto j = alt_json::parse(R"({"foo": ["bar", "baz"]})"); + + // present: object key and array indices + CHECK(j.contains(alt_json::json_pointer("/foo"))); + CHECK(j.contains(alt_json::json_pointer("/foo/0"))); + CHECK(j.contains(alt_json::json_pointer("/foo/1"))); + + // missing: absent object key and out-of-range array index + CHECK_FALSE(j.contains(alt_json::json_pointer("/bar"))); + CHECK_FALSE(j.contains(alt_json::json_pointer("/foo/2"))); + + // "-" always fails the range check + CHECK_FALSE(j.contains(alt_json::json_pointer("/foo/-"))); + + // an array index must not have a leading zero + CHECK_FALSE(j.contains(alt_json::json_pointer("/foo/01"))); + + // a reference token that is not a number + CHECK_FALSE(j.contains(alt_json::json_pointer("/foo/bar"))); + } + + SECTION("operator/(std::size_t)") + { + // json_pointer::operator/=(std::size_t) must compile without string_t + // being constructible from std::string (see #5666) + auto j = alt_json::parse(R"({"foo": ["bar", "baz"]})"); + + CHECK(j.at(alt_json::json_pointer("/foo") / std::size_t(0)) == j["foo"][0]); + CHECK(j.at(alt_json::json_pointer("/foo") / std::size_t(1)) == j["foo"][1]); + } + SECTION("patch") { alt_json const patch1 = alt_json::parse(R"([{ "op": "add", "path": "/a/b", "value": [ "foo", "bar" ] }])"); diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index a58507c15..c56d09e75 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -210,15 +210,42 @@ TEST_CASE_TEMPLATE_INVOKE(value_in_range_of_test, \ TEST_CASE("BJData") { - SECTION("binary_reader BJData LUT arrays are sorted") + SECTION("binary_reader BJData lookup tables") { std::vector const data; auto ia = nlohmann::detail::input_adapter(data); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) nlohmann::detail::binary_reader const br{std::move(ia), json::input_format_t::bjdata}; - CHECK(std::is_sorted(br.bjd_optimized_type_markers.begin(), br.bjd_optimized_type_markers.end())); - CHECK(std::is_sorted(br.bjd_types_map.begin(), br.bjd_types_map.end())); + // the excluded optimized-type markers must match binary_writer's + // is_bjdata_excluded_type_marker(), which encodes the same 8 markers + for (const char marker : + {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z' + }) + { + CHECK(br.is_bjd_excluded_optimized_type(marker)); + } + for (const char marker : + {'U', 'i', 'u', 'I', 'm', 'l', 'M', 'L', 'd', 'D', 'C', 'B', 'x' + }) + { + CHECK(!br.is_bjd_excluded_optimized_type(marker)); + } + + // every dtype marker must round-trip to its ND-array type name + const std::vector> types + { + {'B', "byte"}, {'C', "char"}, {'D', "double"}, {'I', "int16"}, + {'L', "int64"}, {'M', "uint64"}, {'U', "uint8"}, {'d', "single"}, + {'i', "int8"}, {'l', "int32"}, {'m', "uint32"}, {'u', "uint16"} + }; + for (const auto& type : types) + { + const char* name = br.bjd_type_name(type.first); + REQUIRE(name != nullptr); + CHECK(std::string(name) == type.second); + } + CHECK(br.bjd_type_name('x') == nullptr); } SECTION("individual values") diff --git a/tests/src/unit-diagnostic-positions.cpp b/tests/src/unit-diagnostic-positions.cpp index 5326094c7..f5c6bc648 100644 --- a/tests/src/unit-diagnostic-positions.cpp +++ b/tests/src/unit-diagnostic-positions.cpp @@ -141,6 +141,58 @@ TEST_CASE("Better diagnostics with positions") check_objects(300); } + SECTION("converting keeps the positions of nested values (#5650)") + { + // Values nested deeper than the converting constructor's descent bound + // are converted without the call stack, on a path that has to carry the + // positions of every value over itself. Objects and arrays take turns, + // and the innermost value is null, which used to lose its positions. + const auto check_conversion = [](std::size_t depth) + { + CAPTURE(depth) + + std::string text; + std::string closing; + for (std::size_t i = 0; i < depth; ++i) + { + text += (i % 2 == 0) ? "[12, " : R"({"b":1, "a":)"; + closing += (i % 2 == 0) ? ']' : '}'; + } + text += "null"; + text.append(closing.rbegin(), closing.rend()); + + const json original = json::parse(text); + const nlohmann::ordered_json converted = original; + + const json* o = &original; + const nlohmann::ordered_json* c = &converted; + for (std::size_t level = 0; level <= depth; ++level) + { + CAPTURE(level) + REQUIRE(c->start_pos() == o->start_pos()); + REQUIRE(c->end_pos() == o->end_pos()); + + if (level < depth) + { + // the number beside the value nested next + const json& o_number = o->is_object() ? o->at("b") : o->at(0); + const nlohmann::ordered_json& c_number = c->is_object() ? c->at("b") : c->at(0); + REQUIRE(c_number.start_pos() == o_number.start_pos()); + REQUIRE(c_number.end_pos() == o_number.end_pos()); + + o = o->is_object() ? &o->at("a") : &o->at(1); + c = c->is_object() ? &c->at("a") : &c->at(1); + } + } + }; + + check_conversion(1); + check_conversion(127); + check_conversion(128); + check_conversion(129); + check_conversion(300); + } + SECTION("JSON patch add to primitive parent (#4292)") { // the JSON Patch "add" target /foo/bar/baz has a string parent diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index 46f41f252..ee7360eb1 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -19,6 +19,7 @@ using nlohmann::json; #include #include +#include TEST_CASE("Better diagnostics") { @@ -340,6 +341,36 @@ TEST_CASE("Regression tests for extended diagnostics") } } + SECTION("Regression test for issue #5650 - converting keeps the parents of nested values") + { + // A value nested deeper than the converting constructor's descent bound + // is converted without the call stack. Every container that path creates + // has to have the parents of its children set, or the JSON Pointer in the + // diagnostic is cut short. Objects and arrays take turns. + const std::size_t pairs = 150; + + json j = "not a number"; + std::string pointer; + for (std::size_t i = 0; i < pairs; ++i) + { + j = json{{"a", json::array({j})}}; + pointer += "/a/0"; + } + + const nlohmann::ordered_json converted = j; + + const nlohmann::ordered_json* inner = &converted; + for (std::size_t i = 0; i < pairs; ++i) + { + inner = &inner->at("a").at(0); + } + + std::string const expected = "[json.exception.type_error.302] (" + pointer + ") type must be number, but is string"; + int i = 0; + CHECK_THROWS_WITH_AS(i = inner->get(), expected.c_str(), nlohmann::ordered_json::type_error); + CHECK(i == 0); + } + SECTION("Regression test for issue #5668 - wrong path for std::map/unordered_map with non-string keys") { // a map with non-string keys is read from an array of [key, value] arrays; @@ -492,6 +523,21 @@ TEST_CASE("Regression tests for extended diagnostics") CHECK(copy == j); } } + + SECTION("Regression test for issue #5652 - operator>> leaves a partial value in its target on a parse error") + { + json j = "old value"; + std::istringstream is("[1, x"); + CHECK_THROWS_WITH_AS(is >> j, "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: '1, x'", json::parse_error); + + // j must be left unchanged, as json::parse() guarantees for its result + CHECK(j == "old value"); + + // copying j must not trigger assert_invariant(): a failed parse must + // not leave array/object elements without a parent pointer + json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + } } TEST_CASE("Better diagnostics past the descent bound of update() and merge_patch()") diff --git a/tests/src/unit-disable-tuple-reference-conversion.cpp b/tests/src/unit-disable-tuple-reference-conversion.cpp new file mode 100644 index 000000000..6d978620b --- /dev/null +++ b/tests/src/unit-disable-tuple-reference-conversion.cpp @@ -0,0 +1,93 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// This file tests the opt-in JSON_DISABLE_TUPLE_REFERENCE_CONVERSION, so it +// defines the macro itself rather than relying on a -D flag, and runs in every +// build. +#ifdef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION + #undef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION +#endif + +#define JSON_DISABLE_TUPLE_REFERENCE_CONVERSION 1 + +#include +using nlohmann::json; +using nlohmann::ordered_json; + +#include +#include +#include +#include + +// clang before 4 and GCC before 5 cannot create a std::tuple of basic_json +// references at all, with or without JSON_DISABLE_TUPLE_REFERENCE_CONVERSION: +// the tuple constructors make them instantiate basic_json's conversion operator +// for libstdc++'s internal tuple bases, which fails hard +#if (defined(__clang__) && __clang_major__ < 4) || (!defined(__clang__) && defined(__GNUC__) && __GNUC__ < 5) + #define SKIP_TESTS_FOR_JSON_REFERENCE_TUPLES +#endif + +TEST_CASE("JSON_DISABLE_TUPLE_REFERENCE_CONVERSION") +{ + SECTION("json is not constructible from a one-element tuple of a json reference") + { + CHECK_FALSE(std::is_constructible>::value); + CHECK_FALSE(std::is_constructible>::value); + CHECK_FALSE(std::is_constructible < json, std::tuple < json && >>::value); + CHECK_FALSE(std::is_constructible&>::value); + CHECK_FALSE(std::is_constructible>::value); + } + +#ifndef SKIP_TESTS_FOR_JSON_REFERENCE_TUPLES + SECTION("issue #2226 - tuple from tuple keeps the reference") + { + json j = true; + const std::tuple tup(std::forward_as_tuple(j)); + CHECK(&std::get<0>(tup) == &j); + } + + SECTION("tuple from tuple copies the element") + { + const json j = {{"key", "value"}}; + const std::tuple t1(std::forward_as_tuple(j)); + CHECK(std::get<0>(t1) == j); + + json j2 = "text"; + const std::tuple t2(std::forward_as_tuple(std::move(j2))); + CHECK(std::get<0>(t2) == "text"); + } +#endif + + SECTION("other tuple conversions are not affected") + { + const json j = true; + + // one-element tuple holding a json value + CHECK(json(std::make_tuple(j)) == json::array({true})); + + // tuples with more than one element, even when holding references + int i = 1; +#ifndef SKIP_TESTS_FOR_JSON_REFERENCE_TUPLES + CHECK(json(std::forward_as_tuple(i, j)) == json::array({1, true})); + CHECK(json(std::forward_as_tuple(j, j)) == json::array({true, true})); +#endif + + // one-element tuples holding references to other types + std::string s = "text"; + CHECK(json(std::forward_as_tuple(s)) == json::array({"text"})); + CHECK(json(std::forward_as_tuple(i)) == json::array({1})); + +#ifndef SKIP_TESTS_FOR_JSON_REFERENCE_TUPLES + // a reference to a different basic_json specialization + ordered_json oj = true; + CHECK(json(std::forward_as_tuple(oj)) == json::array({true})); +#endif + } +} diff --git a/tests/src/unit-large_json.cpp b/tests/src/unit-large_json.cpp index 8204ed9b6..97f665848 100644 --- a/tests/src/unit-large_json.cpp +++ b/tests/src/unit-large_json.cpp @@ -13,6 +13,7 @@ using nlohmann::json; #include #include +#include TEST_CASE("tests on very large JSONs") { @@ -53,6 +54,24 @@ const json* innermost_value(const json& j, std::size_t& depth) return current; } +// The text of a value nested depth levels deep around the number 0. Level i is +// an array if pattern[i % pattern.size()] is '[', and otherwise an object with +// the single member "a", which every object type enumerates in the same order. +std::string nested_text(std::size_t depth, const std::string& pattern) +{ + std::string text; + std::string closing; + for (std::size_t i = 0; i < depth; ++i) + { + const bool array = pattern[i % pattern.size()] == '['; + text += array ? "[" : "{\"a\":"; + closing += array ? ']' : '}'; + } + text += '0'; + text.append(closing.rbegin(), closing.rend()); + return text; +} + } // namespace TEST_CASE("tests on deeply nested JSONs") @@ -224,5 +243,114 @@ TEST_CASE("tests on deeply nested JSONs") CHECK(*innermost_value(j, unused) == 0); } } + + SECTION("issue #5650 - stack overflow converting between specializations") + { + const std::vector patterns = {"[", "{", "[{"}; + + SECTION("json to ordered_json") + { + for (const auto& pattern : patterns) + { + CAPTURE(pattern); + const std::string text = nested_text(depth, pattern); + const json j = json::parse(text); + + const nlohmann::ordered_json converted = j; + CHECK(converted.dump() == text); + } + } + + SECTION("ordered_json to json") + { + for (const auto& pattern : patterns) + { + CAPTURE(pattern); + const std::string text = nested_text(depth, pattern); + const nlohmann::ordered_json o = nlohmann::ordered_json::parse(text); + + const json converted = o; + CHECK(converted.dump() == text); + } + } + + SECTION("get()") + { + for (const auto& pattern : patterns) + { + CAPTURE(pattern); + const std::string text = nested_text(depth, pattern); + const json j = json::parse(text); + + CHECK(j.get().dump() == text); + } + } + + SECTION("depths around the bound of the recursive descent") + { + for (std::size_t d = 1; d <= 300; ++d) + { + CAPTURE(d); + for (const auto& pattern : patterns) + { + CAPTURE(pattern); + const std::string text = nested_text(d, pattern); + const json j = json::parse(text); + + const nlohmann::ordered_json converted = j; + CHECK(converted.dump() == text); + const json back = converted; + CHECK(back.dump() == text); + } + } + } + + SECTION("values below the bound are converted as values above it") + { + // Bury a value below the bound, where it is converted without the + // call stack, and compare it with the same value converted on its + // own by the containers' range constructors. Its objects have + // members that the two object types enumerate in different orders. + const auto bury = [](nlohmann::ordered_json value) + { + for (std::size_t i = 0; i < 200; ++i) + { + value = nlohmann::ordered_json::array({std::move(value)}); + } + return value; + }; + const auto dig = [](const json & value) + { + const json* current = &value; + for (std::size_t i = 0; i < 200; ++i) + { + current = ¤t->at(0); + } + return current; + }; + + nlohmann::ordered_json value = nlohmann::ordered_json::object(); + value["z"] = {1, -2, 3U, 4.5, true, nullptr, "six", nlohmann::ordered_json::binary({7, 8}, 9), + nlohmann::ordered_json::binary({10}), nlohmann::ordered_json::array(), nlohmann::ordered_json::object() + }; + value["y"] = {{"x", {{"w", 1}, {"v", 2}}}, {"u", {3, {{"t", 4}, {"s", 5}}}}}; + value["r"] = nlohmann::ordered_json::array({nlohmann::ordered_json(nlohmann::ordered_json::value_t::discarded)}); + + const json converted_above = value; + const json buried = bury(value); + const json& converted_below = *dig(buried); + + CHECK(converted_below.dump() == converted_above.dump()); + CHECK(converted_below.at("z").at(7).get_binary().subtype() == 9); + CHECK_FALSE(converted_below.at("z").at(8).get_binary().has_subtype()); + CHECK(converted_below.at("r").at(0).is_discarded()); + + // a discarded value is never equal to anything, so compare the rest + value.erase("r"); + const json without_discarded_above = value; + const json without_discarded_buried = bury(value); + CHECK(*dig(without_discarded_buried) == without_discarded_above); + } + } } diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index c49b00c7c..cd9eeac55 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -18,6 +18,19 @@ // for some reason including this after the json header leads to linker errors with VS 2017... #include +// skip tests if JSON_DISABLE_TUPLE_REFERENCE_CONVERSION=1 (#2226) +#if defined(JSON_DISABLE_TUPLE_REFERENCE_CONVERSION) && (JSON_DISABLE_TUPLE_REFERENCE_CONVERSION == 1) + #define SKIP_TESTS_FOR_TUPLE_REFERENCE_CONVERSION +#endif + +// clang before 4 and GCC before 5 cannot create a std::tuple of basic_json +// references at all, with or without JSON_DISABLE_TUPLE_REFERENCE_CONVERSION: +// the tuple constructors make them instantiate basic_json's conversion operator +// for libstdc++'s internal tuple bases, which fails hard +#if (defined(__clang__) && __clang_major__ < 4) || (!defined(__clang__) && defined(__GNUC__) && __GNUC__ < 5) + #define SKIP_TESTS_FOR_JSON_REFERENCE_TUPLES +#endif + #define JSON_TESTS_PRIVATE #include using json = nlohmann::json; @@ -28,6 +41,7 @@ using ordered_json = nlohmann::ordered_json; #include #include +#include #include #include @@ -542,6 +556,20 @@ TEST_CASE("regression tests 2") ))); } +#ifndef SKIP_TESTS_FOR_TUPLE_REFERENCE_CONVERSION + SECTION("issue #2226 - std::tuple dangling reference - implicit conversion") + { + // by default, a one-element tuple holding a json reference converts to + // a one-element array; JSON_DISABLE_TUPLE_REFERENCE_CONVERSION removes + // this conversion (see unit-disable-tuple-reference-conversion.cpp) + const json j = true; + CHECK(std::is_constructible>::value); +#ifndef SKIP_TESTS_FOR_JSON_REFERENCE_TUPLES + CHECK(json(std::forward_as_tuple(j)) == json::array({true})); +#endif + } +#endif + SECTION("PR #2181 - regression bug with lvalue") { // see https://github.com/nlohmann/json/pull/2181#issuecomment-653326060 diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index c315fac94..80c771d42 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -3033,3 +3033,224 @@ TEST_CASE("UBJSON optimized array of unsigned integers beyond int64") CHECK(json::to_ubjson(j, true, true) == expected); CHECK(json::from_ubjson(expected) == j); } + +namespace +{ +// the bytes that follow the marker of an integer: the value in the width of +// the marker (big endian for UBJSON, little endian for BJData), or, for a +// high-precision number, the length and the decimal digits +std::vector integer_payload(const char marker, const json& value, const bool little_endian) +{ + std::size_t width = 0; + switch (marker) + { + case 'i': + case 'U': + width = 1; + break; + case 'I': + case 'u': + width = 2; + break; + case 'l': + case 'm': + width = 4; + break; + case 'L': + case 'M': + width = 8; + break; + default: + { + const std::string digits = value.dump(); + std::vector result = {'i', static_cast(digits.size())}; + for (const char c : digits) + { + result.push_back(static_cast(c)); + } + return result; + } + } + + const std::uint64_t bits = value.is_number_unsigned() + ? value.get() + : static_cast(value.get()); + std::vector result(width); + for (std::size_t i = 0; i < width; ++i) + { + result[little_endian ? i : width - 1 - i] = static_cast(bits >> (8 * i)); + } + return result; +} + +json i64(const std::int64_t v) +{ + return v; +} + +json u64(const std::uint64_t v) +{ + return v; +} +} // namespace + +TEST_CASE("UBJSON and BJData integer markers at every range edge") +{ + // An optimized container announces the marker of its values after `$` and + // then writes every value without a marker, so the marker the writer + // announces and the width it writes must match for every value. This + // checks both for the values around each edge of the integer types, as + // scalars and as the values of optimized arrays and objects. + struct integer_case + { + json value; + char ubjson; // expected UBJSON marker + char bjdata; // expected BJData marker + }; + + const std::int64_t int64_min = (std::numeric_limits::min)(); + const std::int64_t int64_max = (std::numeric_limits::max)(); + const std::uint64_t uint64_max = (std::numeric_limits::max)(); + + const std::vector cases = + { + // int8 + {i64(-129), 'I', 'I'}, + {i64(-128), 'i', 'i'}, + {i64(-127), 'i', 'i'}, + {i64(-1), 'i', 'i'}, + {i64(0), 'i', 'i'}, + {u64(0), 'i', 'i'}, + {i64(126), 'i', 'i'}, + {i64(127), 'i', 'i'}, + {u64(127), 'i', 'i'}, + {i64(128), 'U', 'U'}, + {u64(128), 'U', 'U'}, + // uint8 + {i64(254), 'U', 'U'}, + {i64(255), 'U', 'U'}, + {u64(255), 'U', 'U'}, + {i64(256), 'I', 'I'}, + {u64(256), 'I', 'I'}, + // int16 + {i64(-32769), 'l', 'l'}, + {i64(-32768), 'I', 'I'}, + {i64(-32767), 'I', 'I'}, + {i64(32766), 'I', 'I'}, + {i64(32767), 'I', 'I'}, + {u64(32767), 'I', 'I'}, + {i64(32768), 'l', 'u'}, + {u64(32768), 'l', 'u'}, + // uint16 (BJData only) + {i64(65534), 'l', 'u'}, + {i64(65535), 'l', 'u'}, + {u64(65535), 'l', 'u'}, + {i64(65536), 'l', 'l'}, + {u64(65536), 'l', 'l'}, + // int32 + {i64(-2147483649LL), 'L', 'L'}, + {i64(-2147483648LL), 'l', 'l'}, + {i64(-2147483647LL), 'l', 'l'}, + {i64(2147483646LL), 'l', 'l'}, + {i64(2147483647LL), 'l', 'l'}, + {u64(2147483647ULL), 'l', 'l'}, + {i64(2147483648LL), 'L', 'm'}, + {u64(2147483648ULL), 'L', 'm'}, + // uint32 (BJData only) + {i64(4294967294LL), 'L', 'm'}, + {i64(4294967295LL), 'L', 'm'}, + {u64(4294967295ULL), 'L', 'm'}, + {i64(4294967296LL), 'L', 'L'}, + {u64(4294967296ULL), 'L', 'L'}, + // int64 + {i64(int64_min), 'L', 'L'}, + {i64(int64_min + 1), 'L', 'L'}, + {i64(int64_max - 1), 'L', 'L'}, + {i64(int64_max), 'L', 'L'}, + {u64(static_cast(int64_max)), 'L', 'L'}, + // uint64 (BJData only; UBJSON writes a high-precision number) + {u64(static_cast(int64_max) + 1), 'H', 'M'}, + {u64(uint64_max - 1), 'H', 'M'}, + {u64(uint64_max), 'H', 'M'}, + }; + + for (const auto& c : cases) + { + for (const bool bjdata : + { + false, true + }) + { + const char marker = bjdata ? c.bjdata : c.ubjson; + const std::vector payload = integer_payload(marker, c.value, bjdata); + const auto to_binary = [bjdata](const json & j, const bool use_size, const bool use_type) + { + return bjdata ? json::to_bjdata(j, use_size, use_type) : json::to_ubjson(j, use_size, use_type); + }; + const auto from_binary = [bjdata](const std::vector& v) + { + return bjdata ? json::from_bjdata(v) : json::from_ubjson(v); + }; + INFO("value = " << c.value.dump() << (c.value.is_number_unsigned() ? " (unsigned)" : "") << ", format = " << (bjdata ? "BJData" : "UBJSON")); + + // scalar + std::vector expected = {static_cast(marker)}; + expected.insert(expected.end(), payload.begin(), payload.end()); + for (const bool use_size : + { + false, true + }) + { + CHECK(to_binary(c.value, use_size, false) == expected); + } + CHECK(from_binary(expected) == c.value); + + const json arr = {c.value, c.value, c.value}; + + // array without count or type: every value has its marker + expected = {'['}; + for (int i = 0; i < 3; ++i) + { + expected.push_back(static_cast(marker)); + expected.insert(expected.end(), payload.begin(), payload.end()); + } + expected.push_back(']'); + CHECK(to_binary(arr, false, false) == expected); + CHECK(from_binary(expected) == arr); + + // array with count: every value has its marker + expected = {'[', '#', 'i', 3}; + for (int i = 0; i < 3; ++i) + { + expected.push_back(static_cast(marker)); + expected.insert(expected.end(), payload.begin(), payload.end()); + } + CHECK(to_binary(arr, true, false) == expected); + CHECK(from_binary(expected) == arr); + + // array with type and count: the marker once, then the payloads + expected = {'[', '$', static_cast(marker), '#', 'i', 3}; + for (int i = 0; i < 3; ++i) + { + expected.insert(expected.end(), payload.begin(), payload.end()); + } + CHECK(to_binary(arr, true, true) == expected); + CHECK(from_binary(expected) == arr); + + // object with type and count: the marker once, then key and payload + const json obj = {{"a", c.value}, {"b", c.value}}; + expected = {'{', '$', static_cast(marker), '#', 'i', 2}; + for (const char key : + {'a', 'b' + }) + { + expected.push_back('i'); + expected.push_back(1); + expected.push_back(static_cast(key)); + expected.insert(expected.end(), payload.begin(), payload.end()); + } + CHECK(to_binary(obj, true, true) == expected); + CHECK(from_binary(expected) == obj); + } + } +} diff --git a/tools/amalgamate/README.md b/tools/amalgamate/README.md index 33ea9687b..658e57cbd 100644 --- a/tools/amalgamate/README.md +++ b/tools/amalgamate/README.md @@ -1,8 +1,8 @@ # amalgamate.py - Amalgamate C source and header files -Origin: https://bitbucket.org/erikedlund/amalgamate - -Mirror: https://github.com/edlund/amalgamate +Origin: https://github.com/edlund/amalgamate (formerly hosted at +https://bitbucket.org/erikedlund/amalgamate, which no longer exists; see +`CHANGES.md` for the upstream commit this copy is based on) `amalgamate.py` aims to make it easy to use SQLite-style C source and header amalgamation in projects. @@ -41,21 +41,22 @@ results. ## Installing amalgamate.py -Python v.2.7.0 or higher is required. +Python 3 is required. -`amalgamate.py` can be tested and installed using the following commands: - - ./test.sh && sudo -k cp ./amalgamate.py /usr/local/bin/ +In this repository, `amalgamate.py` is not installed separately; it is run in +place through `make amalgamate`, which calls it once for `json.hpp` and once +for `json_fwd.hpp` (see the root `Makefile`). ## Using amalgamate.py - amalgamate.py [-v] -c path/to/config.json -s path/to/source/dir \ - [-p path/to/prologue.(c|h)] + amalgamate.py -c path/to/config.json -s path/to/source/dir \ + [-p path/to/prologue.(c|h)] [--verbose=yes|no] * The `-c, --config` option should specify the path to a JSON config file which lists the source files, include paths and where to write the resulting - amalgamation. Have a look at `test/source.c.json` and `test/include.h.json` - to see two examples. + amalgamation. `config_json.json` and `config_json_fwd.json` in this + directory are the configs used for `json.hpp` and `json_fwd.hpp`; each + sets `target`, `sources` and `include_paths`. The optional `external` list names include paths that are kept as `#include` directives instead of being inlined, e.g. `["nlohmann/json.hpp"]` for a header @@ -68,3 +69,6 @@ Python v.2.7.0 or higher is required. * The `-p, --prologue` option should specify the path to a file which will be added to the beginning of the amalgamation. It is optional. + * The `-v, --verbose` option takes `yes` or `no` (for example + `--verbose=yes`, as used by the Makefile). It is optional. + diff --git a/tools/generate_natvis/README.md b/tools/generate_natvis/README.md index e11f29eec..3995758c2 100644 --- a/tools/generate_natvis/README.md +++ b/tools/generate_natvis/README.md @@ -2,8 +2,26 @@ Generate the Natvis debugger visualization file for all supported namespace combinations. +The ABI tag list and the library version are parsed from +`include/nlohmann/detail/abi_macros.hpp`, so this script must be re-run (via +`make natvis`) whenever an `NLOHMANN_JSON_ABI_TAG_*` macro is added to that +file or the library version is bumped — otherwise the committed +`nlohmann_json.natvis` drifts from the header it visualizes, and +`make check-amalgamation` fails. + ## Usage ```shell -./generate_natvis.py --version X.Y.Z output_directory/ +make natvis ``` + +or, equivalently: + +```shell +./generate_natvis.py [--version X.Y.Z] [repository_root/] +``` + +`--version` and the output/repository-root directory both default to values +derived from this script's own location, so they only need to be given +explicitly when generating a Natvis file for a different checkout or a +version other than the one in `abi_macros.hpp`. diff --git a/tools/generate_natvis/generate_natvis.py b/tools/generate_natvis/generate_natvis.py index 8562e8f1a..f4cd6fcd2 100755 --- a/tools/generate_natvis/generate_natvis.py +++ b/tools/generate_natvis/generate_natvis.py @@ -7,21 +7,75 @@ import os import re import sys +# Directory of the repository, assuming this script stays at +# tools/generate_natvis/generate_natvis.py. Used only as the default value +# for the "output" argument below. +REPO_ROOT = os.path.normpath(os.path.join(sys.path[0], '..', '..')) + + def semver(v): if not re.fullmatch(r'\d+\.\d+\.\d+', v): raise ValueError return v + +def abi_info(repo_root): + """Derive the ABI tag list (in NLOHMANN_JSON_ABI_TAGS_CONCAT order) and the + library version from /include/nlohmann/detail/abi_macros.hpp, + so this script cannot drift from the header it visualizes.""" + abi_macros_hpp = os.path.join(repo_root, 'include', 'nlohmann', 'detail', 'abi_macros.hpp') + with open(abi_macros_hpp) as f: + content = f.read() + + # find the NLOHMANN_JSON_ABI_TAGS_CONCAT(...) invocation that lists the + # NLOHMANN_JSON_ABI_TAG_* identifiers in order (not its own #define, which + # only names its formal parameters a, b, c, ...) + tag_idents = None + for args in re.findall(r'NLOHMANN_JSON_ABI_TAGS_CONCAT\(\s*(.*?)\)', content, re.S): + idents = re.findall(r'NLOHMANN_JSON_ABI_TAG_\w+', args) + if idents: + tag_idents = idents + break + if not tag_idents: + raise ValueError(f'could not find NLOHMANN_JSON_ABI_TAGS_CONCAT(...) in {abi_macros_hpp}') + + abi_tags = [] + for ident in tag_idents: + # each tag is #define'd to its suffix (e.g. _diag) when the matching + # JSON_* option is enabled, and to nothing in the #else branch; only + # the non-empty definition matches here + match = re.search(r'#define\s+' + re.escape(ident) + r'\s+(_\w+)\s*\n', content) + if not match: + raise ValueError(f'could not find a non-empty #define for {ident} in {abi_macros_hpp}') + abi_tags.append(match.group(1)) + + version = {} + for part in ('MAJOR', 'MINOR', 'PATCH'): + match = re.search(r'#define\s+NLOHMANN_JSON_VERSION_' + part + r'\s+(\d+)', content) + if not match: + raise ValueError(f'could not find NLOHMANN_JSON_VERSION_{part} in {abi_macros_hpp}') + version[part] = match.group(1) + + return abi_tags, '{MAJOR}.{MINOR}.{PATCH}'.format(**version) + + if __name__ == '__main__': parser = argparse.ArgumentParser() - parser.add_argument('--version', required=True, type=semver, help='Library version number') - parser.add_argument('output', help='Output directory for nlohmann_json.natvis') + parser.add_argument('--version', type=semver, + help='Library version number (default: parsed from ' + 'include/nlohmann/detail/abi_macros.hpp below "output")') + parser.add_argument('output', nargs='?', default=REPO_ROOT, + help='Repository root: where include/nlohmann/detail/abi_macros.hpp is ' + 'read from and where nlohmann_json.natvis is written ' + '(default: the repository root this script lives in)') args = parser.parse_args() + derived_tags, derived_version = abi_info(args.output) + namespaces = ['nlohmann'] abi_prefix = 'json_abi' - abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics', '_psp', '_snul'] - version = '_v' + args.version.replace('.', '_') + abi_tags = derived_tags + version = '_v' + (args.version or derived_version).replace('.', '_') inline_namespaces = [] # generate all combinations of inline namespace names diff --git a/tools/macro_builder/README.md b/tools/macro_builder/README.md new file mode 100644 index 000000000..f425f22c9 --- /dev/null +++ b/tools/macro_builder/README.md @@ -0,0 +1,51 @@ +# macro_builder + +Generates the argument-counting macros behind the `NLOHMANN_DEFINE_TYPE_*` and `NLOHMANN_DEFINE_DERIVED_TYPE_*` +macros in [`include/nlohmann/detail/macro_scope.hpp`](../../include/nlohmann/detail/macro_scope.hpp): + +- `NLOHMANN_JSON_EXPAND` +- `NLOHMANN_JSON_GET_MACRO`, which selects a macro by the number of its arguments (64 slots) +- `NLOHMANN_JSON_PASTE`, which calls a function-like macro for each member, and its helpers `NLOHMANN_JSON_PASTE2` + to `NLOHMANN_JSON_PASTE64` +- `NLOHMANN_JSON_DOUBLE_PASTE`, which the `*_WITH_NAMES` macros use to call a function-like macro for each + (JSON name, member) pair, and its helpers `NLOHMANN_JSON_DOUBLE_PASTE3` to `NLOHMANN_JSON_DOUBLE_PASTE63` +- the slot table of `NLOHMANN_JSON_TYPE_BODY`, which dispatches `NLOHMANN_DEFINE_TYPE_*(Type)` (no further + arguments) to the zero-member implementation and every other argument count to the one-or-more-member + implementation + +The number of slots (`max_args` in [`main.cpp`](main.cpp)) sets the member limit of these macros. +`NLOHMANN_JSON_PASTE` and `NLOHMANN_JSON_TYPE_BODY` take the function/prefix as their first argument, so 64 slots +allow 63 members; `NLOHMANN_JSON_DOUBLE_PASTE` additionally consumes its members two at a time (name, member), so +it only defines the odd helpers up to `NLOHMANN_JSON_DOUBLE_PASTE63`. + +## Usage + +From the project root: + +```shell +c++ -std=c++11 tools/macro_builder/main.cpp -o macro_builder +./macro_builder +./macro_builder type_body +``` + +1. Run `./macro_builder` (no arguments). In `include/nlohmann/detail/macro_scope.hpp`, replace the lines from + `#define NLOHMANN_JSON_EXPAND( x ) x` to the `#define NLOHMANN_JSON_DOUBLE_PASTE63(...)` line with the output, + without its trailing empty line. +2. Run `./macro_builder type_body`. Replace the lines from `#define NLOHMANN_JSON_TYPE_BODY(Prefix, ...)` to the + `NLOHMANN_JSON_TYPE_BODY_SENTINEL))` line with the output. +3. Run `make amalgamate`. It updates `single_include/nlohmann/json.hpp` and runs `make pretty`, which indents the + continuation lines that the tool writes unindented. + +With an unchanged `main.cpp`, these steps reproduce both blocks of `macro_scope.hpp` byte for byte. `make +macro_builder_check` (also run by CI, see `.github/workflows/check_amalgamation.yml`) automates this: it builds +`main.cpp`, regenerates both blocks, and fails on a diff against the checked-in header. + +## Maintained by hand + +The tool does not generate everything that depends on the number of slots. When changing `max_args`, also update: + +- the documented limit of 63 members in `docs/mkdocs/docs` and the tests at that limit in + `tests/src/unit-udt_macro.cpp` + +All three tables pass one macro name per slot to `NLOHMANN_JSON_GET_MACRO`, so they need exactly as many entries as +it has slots. diff --git a/tools/macro_builder/main.cpp b/tools/macro_builder/main.cpp index e676daacc..f035048f2 100644 --- a/tools/macro_builder/main.cpp +++ b/tools/macro_builder/main.cpp @@ -1,10 +1,14 @@ #include #include #include +#include using namespace std; -void build_code(int max_args) +// Builds NLOHMANN_JSON_EXPAND, NLOHMANN_JSON_GET_MACRO, and the +// NLOHMANN_JSON_PASTE / NLOHMANN_JSON_PASTE2..PASTE dispatch table +// and recursive definitions. +string build_paste_code(int max_args) { stringstream ss; ss << "#define NLOHMANN_JSON_EXPAND( x ) x" << endl; @@ -12,32 +16,93 @@ void build_code(int max_args) for (int i = 0 ; i < max_args ; i++) ss << "_" << i + 1 << ", "; ss << "NAME,...) NAME" << endl; - + ss << "#define NLOHMANN_JSON_PASTE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \\" << endl; for (int i = max_args ; i > 1 ; i--) ss << "NLOHMANN_JSON_PASTE" << i << ", \\" << endl; ss << "NLOHMANN_JSON_PASTE1)(__VA_ARGS__))" << endl; - + ss << "#define NLOHMANN_JSON_PASTE2(func, v1) func(v1)" << endl; for (int i = 3 ; i <= max_args ; i++) { - ss << "#define NLOHMANN_JSON_PASTE" << i << "(func, "; + ss << "#define NLOHMANN_JSON_PASTE" << i << "(func, "; for (int j = 1 ; j < i -1 ; j++) - ss << "v" << j << ", "; + ss << "v" << j << ", "; ss << "v" << i-1 << ") NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE" << i-1 << "(func, "; for (int j = 2 ; j < i-1 ; j++) ss << "v" << j << ", "; ss << "v" << i-1 << ")" << endl; } - - cout << ss.str() << endl; + + return ss.str(); } -int main(int argc, char** argv) +// Builds the NLOHMANN_JSON_DOUBLE_PASTE dispatch table and recursive +// definitions used by the *_WITH_NAMES macros. Its GET_MACRO dispatch reuses +// the same max_args slots as NLOHMANN_JSON_PASTE, but DOUBLE_PASTE consumes +// its arguments two at a time (name, member), so an even slot count falls +// back to the next lower odd NLOHMANN_JSON_DOUBLE_PASTE. +string build_double_paste_code(int max_args) +{ + stringstream ss; + ss << "#define NLOHMANN_JSON_DOUBLE_PASTE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \\" << endl; + for (int i = max_args ; i > 1 ; i--) + { + int k = (i % 2 == 1) ? i : i - 1; + ss << "NLOHMANN_JSON_DOUBLE_PASTE" << k << ", \\" << endl; + } + ss << "NLOHMANN_JSON_DOUBLE_PASTE1)(__VA_ARGS__))" << endl; + + ss << "#define NLOHMANN_JSON_DOUBLE_PASTE3(func, v1, v2) func(v1, v2)" << endl; + for (int k = 5 ; k <= max_args - 1 ; k += 2) + { + ss << "#define NLOHMANN_JSON_DOUBLE_PASTE" << k << "(func, "; + for (int j = 1 ; j < k - 1 ; j++) + ss << "v" << j << ", "; + ss << "v" << k - 1 << ") NLOHMANN_JSON_DOUBLE_PASTE3(func, v1, v2) NLOHMANN_JSON_DOUBLE_PASTE" << k - 2 << "(func, "; + for (int j = 3 ; j < k - 1 ; j++) + ss << "v" << j << ", "; + ss << "v" << k - 1 << ")" << endl; + } + + return ss.str(); +} + +// Builds the NLOHMANN_JSON_TYPE_BODY dispatch table: max_args - 1 slots +// selecting the *_MEMBERS implementation and a final slot selecting +// *_EMPTY, so NLOHMANN_DEFINE_TYPE_*(Type) with no further arguments still +// resolves (issue #4041). +string build_type_body_table(int max_args) +{ + stringstream ss; + ss << "#define NLOHMANN_JSON_TYPE_BODY(Prefix, ...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \\" << endl; + const int per_line = 8; + for (int i = 1 ; i <= max_args ; i++) + { + ss << (i == max_args ? "Prefix ## EMPTY" : "Prefix ## MEMBERS") << ", "; + if (i % per_line == 0) + ss << "\\" << endl; + } + ss << "NLOHMANN_JSON_TYPE_BODY_SENTINEL))" << endl; + + return ss.str(); +} + +int main(int argc, char** argv) { int max_args = 64; - build_code(max_args); - + + // With "type_body", print only the NLOHMANN_JSON_TYPE_BODY dispatch + // table (a separate insertion point in macro_scope.hpp); otherwise + // print the EXPAND/GET_MACRO/PASTE/DOUBLE_PASTE block that precedes it. + if (argc > 1 && string(argv[1]) == "type_body") + { + cout << build_type_body_table(max_args); + } + else + { + cout << build_paste_code(max_args) << build_double_paste_code(max_args); + } + return 0; } - diff --git a/tools/serve_header/serve_header.py b/tools/serve_header/serve_header.py index 1f29cb589..5b5bf03f4 100755 --- a/tools/serve_header/serve_header.py +++ b/tools/serve_header/serve_header.py @@ -5,6 +5,8 @@ import logging import os import re import shutil +import socket +import ssl import sys import subprocess @@ -34,7 +36,7 @@ JSON_VERSION_RE = re.compile(r'\s*#\s*define\s+NLOHMANN_JSON_VERSION_MAJOR\s+') class ExitHandler(logging.StreamHandler): def __init__(self, level): - """.""" + """Exit the process on log records at or above level.""" super().__init__() self.level = level @@ -54,7 +56,7 @@ def is_project_root(test_dir='.'): class DirectoryEventBucket: def __init__(self, callback, delay=1.2, threshold=0.8): - """.""" + """Batch directory events and pass their common path to callback.""" self.delay = delay self.threshold = timedelta(seconds=threshold) self.callback = callback @@ -99,7 +101,7 @@ class WorkTree: make_command = 'make' def __init__(self, root_dir, tree_dir): - """.""" + """Track the working tree at tree_dir and its amalgamated header.""" self.root_dir = root_dir self.tree_dir = tree_dir self.rel_dir = os.path.relpath(tree_dir, root_dir) @@ -114,11 +116,11 @@ class WorkTree: self.build_time = t.strftime(DATETIME_FORMAT) def __hash__(self): - """.""" + """Hash by working tree directory.""" return hash((self.tree_dir)) def __eq__(self, other): - """.""" + """Compare by working tree directory.""" if not isinstance(other, type(self)): return NotImplemented return self.tree_dir == other.tree_dir @@ -150,7 +152,7 @@ class WorkTree: class WorkTrees(FileSystemEventHandler): def __init__(self, root_dir): - """.""" + """Find the working trees below root_dir and watch it for changes.""" super().__init__() self.root_dir = root_dir self.trees = set([]) @@ -250,11 +252,11 @@ class WorkTrees(FileSystemEventHandler): self.observer.stop() self.observer.join() -class HeaderRequestHandler(SimpleHTTPRequestHandler): # lgtm[py/missing-call-to-init] +class HeaderRequestHandler(SimpleHTTPRequestHandler): cors_origins = DEFAULT_CORS_ORIGINS def __init__(self, request, client_address, server): - """.""" + """Handle a request for a header below the working trees' root directory.""" self.worktrees = server.worktrees self.worktree = None try: @@ -336,7 +338,7 @@ class HeaderRequestHandler(SimpleHTTPRequestHandler): # lgtm[py/missing-call-to- class DualStackServer(ThreadingHTTPServer): def __init__(self, addr, worktrees): - """.""" + """Serve the headers of worktrees on addr.""" self.worktrees = worktrees super().__init__(addr, HeaderRequestHandler) @@ -349,8 +351,6 @@ class DualStackServer(ThreadingHTTPServer): if __name__ == '__main__': import argparse - import ssl - import socket import yaml # exit code