From ed8ba0201fe9ee3c6dcc97c92c93bab38faec2a1 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 7 Oct 2026 22:20:05 +0200 Subject: [PATCH 01/10] Install CMake package config files with Meson, and add Meson options (#5587) * Install CMake package config files with Meson, and add Meson options Squashed onto develop from: - Install CMake package config files with Meson - meson: Indent code inside an if block - meson: set a minimum Meson version - meson: use `override_dependency()` to set dependencies - meson: use `install_subdir` for headers - meson: set the C++ standard to C++11 - meson: handle single header and multiheader the same way CMake does - meson: add support for the GlobalUDLs option - meson: add support for the ImplictConversions option - Add meson information to FILES.md - Complete the Meson options and match CMake's compile definitions - Add the JSON_* definitions to the CMake pkg-config file - Document the Meson options and check them in CI - Fix the Meson CMake target for includedir or datadir outside the prefix - Avoid //include in Meson-generated CMake target for prefix / - Add DisableTupleReferenceConversion to Meson and the CMake pkg-config file - Check in CI that Meson and pkg-config offer the CMake options - Remove accidentally committed Python bytecode Co-authored-by: Dylan Baker Signed-off-by: Dylan Baker Signed-off-by: Niels Lohmann * Add the StrictBinaryUTF8 and DeleteDeprecatedFunctions Meson options JSON_StrictBinaryUTF8 (#5741) and JSON_DeleteDeprecatedFunctions (#5755) arrived with develop but were missing from the Meson build and the pkg-config file, so check_build_options failed. Both Meson options default to false and add JSON_STRICT_BINARY_UTF8=1 and JSON_DELETE_DEPRECATED_FUNCTIONS=1 to the dependency, the pkg-config file, and the generated CMake target; CMake's pkg-config file now carries both definitions too. The ci_meson_install job sets and checks them in its non-default install. Signed-off-by: Niels Lohmann --------- Signed-off-by: Dylan Baker Signed-off-by: Niels Lohmann Co-authored-by: Dylan Baker --- .github/workflows/ubuntu.yml | 45 ++++++ CMakeLists.txt | 34 ++++- FILES.md | 26 +++- Makefile | 8 +- cmake/nlohmann_jsonTargets.cmake.in | 42 +++++ cmake/pkg-config.pc.in | 2 +- .../docs/integration/package_managers.md | 22 ++- meson.build | 114 ++++++++++++-- meson_options.txt | 66 ++++++++ tools/check_build_options/README.md | 21 +++ .../check_build_options.py | 144 ++++++++++++++++++ 11 files changed, 505 insertions(+), 19 deletions(-) create mode 100644 cmake/nlohmann_jsonTargets.cmake.in create mode 100644 meson_options.txt create mode 100644 tools/check_build_options/README.md create mode 100755 tools/check_build_options/check_build_options.py diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index 1fa2f9fc6..3c94bc1e4 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -31,6 +31,51 @@ jobs: - name: Build run: cmake --build build --target ci_test_gcc + ci_meson_install: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - name: Get latest CMake and ninja + uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2 + - name: Check that Meson and pkg-config offer the CMake options + run: make check_build_options + - name: Install Meson + run: pip install meson + - name: Install with Meson + run: | + meson setup build-meson --prefix=${{ github.workspace }}/install + meson install -C build-meson + - name: Use the installed package with find_package + run: | + cmake -S tests/cmake_import/project -B build-import -DCMAKE_PREFIX_PATH=${{ github.workspace }}/install + cmake --build build-import + - name: Install with Meson and non-default options + run: | + meson setup build-meson-options --prefix=${{ github.workspace }}/install-options -DMultipleHeaders=true -DDiagnostics=true -DGlobalUDLs=false -DDisableTupleReferenceConversion=true -DStrictBinaryUTF8=true -DDeleteDeprecatedFunctions=true + meson install -C build-meson-options + - name: Check that the options reach the installed files + run: | + test -d install-options/include/nlohmann/detail + cflags=$(PKG_CONFIG_PATH=${{ github.workspace }}/install-options/share/pkgconfig pkg-config --cflags nlohmann_json) + echo "$cflags" + echo "$cflags" | grep -q -- '-DJSON_DIAGNOSTICS=1' + echo "$cflags" | grep -q -- '-DJSON_USE_GLOBAL_UDLS=0' + echo "$cflags" | grep -q -- '-DJSON_DISABLE_TUPLE_REFERENCE_CONVERSION=1' + echo "$cflags" | grep -q -- '-DJSON_STRICT_BINARY_UTF8=1' + echo "$cflags" | grep -q -- '-DJSON_DELETE_DEPRECATED_FUNCTIONS=1' + grep -q 'JSON_USE_GLOBAL_UDLS=0;JSON_DISABLE_TUPLE_REFERENCE_CONVERSION=1;JSON_DIAGNOSTICS=1' install-options/share/cmake/nlohmann_json/nlohmann_jsonTargets.cmake + grep -q 'JSON_STRICT_BINARY_UTF8=1;JSON_DELETE_DEPRECATED_FUNCTIONS=1' install-options/share/cmake/nlohmann_json/nlohmann_jsonTargets.cmake + cmake -S tests/cmake_import/project -B build-import-options -DCMAKE_PREFIX_PATH=${{ github.workspace }}/install-options + cmake --build build-import-options + - name: Install with Meson and the include directory outside the prefix + run: | + meson setup build-meson-split --prefix=${{ github.workspace }}/install-split --includedir=${{ github.workspace }}/install-split-dev/include + meson install -C build-meson-split + cmake -S tests/cmake_import/project -B build-import-split -DCMAKE_PREFIX_PATH=${{ github.workspace }}/install-split + cmake --build build-import-split + ci_infer: runs-on: ubuntu-latest steps: diff --git a/CMakeLists.txt b/CMakeLists.txt index fed9e3171..19f5dca30 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -189,7 +189,39 @@ if (MSVC) ) endif() -# Install a pkg-config file, so other tools can find this. +# Install a pkg-config file, so other tools can find this. It carries the same +# compile definitions as the target above. +set(NLOHMANN_JSON_PKGCONFIG_CFLAGS "") +if (NOT JSON_GlobalUDLs) + string(APPEND NLOHMANN_JSON_PKGCONFIG_CFLAGS " -DJSON_USE_GLOBAL_UDLS=0") +endif() +if (NOT JSON_ImplicitConversions) + string(APPEND NLOHMANN_JSON_PKGCONFIG_CFLAGS " -DJSON_USE_IMPLICIT_CONVERSIONS=0") +endif() +if (JSON_DisableEnumSerialization) + string(APPEND NLOHMANN_JSON_PKGCONFIG_CFLAGS " -DJSON_DISABLE_ENUM_SERIALIZATION=1") +endif() +if (JSON_DisableTupleReferenceConversion) + string(APPEND NLOHMANN_JSON_PKGCONFIG_CFLAGS " -DJSON_DISABLE_TUPLE_REFERENCE_CONVERSION=1") +endif() +if (JSON_Diagnostics) + string(APPEND NLOHMANN_JSON_PKGCONFIG_CFLAGS " -DJSON_DIAGNOSTICS=1") +endif() +if (JSON_Diagnostic_Positions) + string(APPEND NLOHMANN_JSON_PKGCONFIG_CFLAGS " -DJSON_DIAGNOSTIC_POSITIONS=1") +endif() +if (JSON_LegacyDiscardedValueComparison) + string(APPEND NLOHMANN_JSON_PKGCONFIG_CFLAGS " -DJSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1") +endif() +if (JSON_StrictNulHandling) + string(APPEND NLOHMANN_JSON_PKGCONFIG_CFLAGS " -DJSON_STRICT_NUL_HANDLING=1") +endif() +if (JSON_StrictBinaryUTF8) + string(APPEND NLOHMANN_JSON_PKGCONFIG_CFLAGS " -DJSON_STRICT_BINARY_UTF8=1") +endif() +if (JSON_DeleteDeprecatedFunctions) + string(APPEND NLOHMANN_JSON_PKGCONFIG_CFLAGS " -DJSON_DELETE_DEPRECATED_FUNCTIONS=1") +endif() configure_file( "${CMAKE_CURRENT_SOURCE_DIR}/cmake/pkg-config.pc.in" "${CMAKE_CURRENT_BINARY_DIR}/${PROJECT_NAME}.pc" diff --git a/FILES.md b/FILES.md index 861fdf610..fbb57815a 100644 --- a/FILES.md +++ b/FILES.md @@ -249,9 +249,31 @@ make BUILD.bazel The "Check amalgamation" workflow fails if the file is out of date. -### `meson.build` +### `meson.build` and `meson_options.txt` -The build definition for the [Meson](https://mesonbuild.com) build system. +Meson build definitions suitable for use as a subproject ("wrap" in Meson terminology). + +Projects wishing to use the wrap can execute: +```sh +meson wrap install nlohmann_json +``` + +Which allows Meson to build from source when a system provided dependency isn't available. + +To build directly: +```sh +meson setup builddir +ninja -C builddir +``` + +`meson_options.txt` defines the options, which mirror the CMake options that change the library's target (for example, +`-DDiagnostics=true`). Meson requires this file next to `meson.build`, so it is also part of `include.zip`. `make check_build_options` +([`tools/check_build_options`](tools/check_build_options/README.md)) checks in CI that both files and the pkg-config files +stay in sync with the CMake options. + +When installing, `meson.build` installs the headers, a pkg-config file, and the CMake package config files, so that +`find_package(nlohmann_json)` works. As Meson cannot generate `nlohmann_jsonTargets.cmake` itself, it is created from +the template `cmake/nlohmann_jsonTargets.cmake.in`, which is only used by Meson. ### `Package.swift` diff --git a/Makefile b/Makefile index 11b9227a4..093fbb7b5 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: pretty clean ChangeLog.md release update_hedley update_hedley_undef BUILD.bazel natvis macro_builder_check +.PHONY: pretty clean ChangeLog.md release update_hedley update_hedley_undef BUILD.bazel natvis macro_builder_check check_build_options ########################################################################## # configuration @@ -127,6 +127,10 @@ macro_builder_check: diff "$$TMPDIR/paste.hpp" "$$TMPDIR/paste_actual.hpp" || (echo "===================================================================\n $(MACRO_SCOPE_HPP) (NLOHMANN_JSON_EXPAND..NLOHMANN_JSON_DOUBLE_PASTE63) is out of date!\n Regenerate it, see tools/macro_builder/README.md.\n===================================================================" ; exit 1); \ diff "$$TMPDIR/type_body.hpp" "$$TMPDIR/type_body_actual.hpp" || (echo "===================================================================\n $(MACRO_SCOPE_HPP) (NLOHMANN_JSON_TYPE_BODY) is out of date!\n Regenerate it, see tools/macro_builder/README.md.\n===================================================================" ; exit 1) +# check that the Meson build and the pkg-config files offer the options of the CMake target +check_build_options: + python3 tools/check_build_options/check_build_options.py . + # check if file single_include/nlohmann/json.hpp has been amalgamated from the nlohmann sources check-amalgamation: @mv $(AMALGAMATED_FILE) $(AMALGAMATED_FILE)~ @@ -184,7 +188,7 @@ json.tar.xz: # We use `-X` to make the resulting ZIP file reproducible, see # . include.zip: BUILD.bazel - zip -9 --recurse-paths -X include.zip $(SRCS) $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) BUILD.bazel MODULE.bazel meson.build LICENSE.MIT + zip -9 --recurse-paths -X include.zip $(SRCS) $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) BUILD.bazel MODULE.bazel meson.build meson_options.txt LICENSE.MIT # Create the files for a release and add signatures and hashes. release: include.zip json.tar.xz diff --git a/cmake/nlohmann_jsonTargets.cmake.in b/cmake/nlohmann_jsonTargets.cmake.in new file mode 100644 index 000000000..b2e57a526 --- /dev/null +++ b/cmake/nlohmann_jsonTargets.cmake.in @@ -0,0 +1,42 @@ +# Imported target for installations made with Meson (see meson.build). +# +# CMake installations generate this file with install(EXPORT ...). Meson cannot +# do that, but as the library is header-only, the target only needs an include +# directory, the C++ standard, and the compile definitions of the options that +# differ from their defaults. Paths are computed relative to this file so that +# the installation can be relocated (e.g., into a sysroot), unless includedir or +# datadir is outside the prefix. + +if(TARGET @PROJECT_NAME@::@NLOHMANN_JSON_TARGET_NAME@) + return() +endif() + +get_filename_component(_IMPORT_PREFIX "${CMAKE_CURRENT_LIST_DIR}/@NLOHMANN_JSON_CONFIG_TO_PREFIX@" ABSOLUTE) +# As in CMake's generated file: avoid "//include" for an installation to "/". +if(_IMPORT_PREFIX STREQUAL "/") + set(_IMPORT_PREFIX "") +endif() + +add_library(@PROJECT_NAME@::@NLOHMANN_JSON_TARGET_NAME@ INTERFACE IMPORTED) +set_target_properties(@PROJECT_NAME@::@NLOHMANN_JSON_TARGET_NAME@ PROPERTIES + INTERFACE_INCLUDE_DIRECTORIES "@NLOHMANN_JSON_INCLUDE_DIR@" +) +if(CMAKE_VERSION VERSION_LESS 3.8) + set_target_properties(@PROJECT_NAME@::@NLOHMANN_JSON_TARGET_NAME@ PROPERTIES + INTERFACE_COMPILE_FEATURES cxx_range_for + ) +else() + set_target_properties(@PROJECT_NAME@::@NLOHMANN_JSON_TARGET_NAME@ PROPERTIES + INTERFACE_COMPILE_FEATURES cxx_std_11 + ) +endif() + +set(_NLOHMANN_JSON_COMPILE_DEFINITIONS "@NLOHMANN_JSON_COMPILE_DEFINITIONS@") +if(_NLOHMANN_JSON_COMPILE_DEFINITIONS) + set_target_properties(@PROJECT_NAME@::@NLOHMANN_JSON_TARGET_NAME@ PROPERTIES + INTERFACE_COMPILE_DEFINITIONS "${_NLOHMANN_JSON_COMPILE_DEFINITIONS}" + ) +endif() + +unset(_NLOHMANN_JSON_COMPILE_DEFINITIONS) +unset(_IMPORT_PREFIX) diff --git a/cmake/pkg-config.pc.in b/cmake/pkg-config.pc.in index 21a91a3cf..70407f09b 100644 --- a/cmake/pkg-config.pc.in +++ b/cmake/pkg-config.pc.in @@ -4,4 +4,4 @@ includedir=${prefix}/@CMAKE_INSTALL_INCLUDEDIR@ Name: @PROJECT_NAME@ Description: JSON for Modern C++ Version: @PROJECT_VERSION@ -Cflags: -I${includedir} +Cflags: -I${includedir}@NLOHMANN_JSON_PKGCONFIG_CFLAGS@ diff --git a/docs/mkdocs/docs/integration/package_managers.md b/docs/mkdocs/docs/integration/package_managers.md index d5357b8ff..bfa994340 100644 --- a/docs/mkdocs/docs/integration/package_managers.md +++ b/docs/mkdocs/docs/integration/package_managers.md @@ -125,10 +125,24 @@ meson wrap install nlohmann_json Please see the Meson project for any issues regarding the packaging. The provided `meson.build` can also be used as an alternative to CMake for installing `nlohmann_json` system-wide in -which case a [pkg-config](pkg-config.md) file is installed. To use it, have your build system require the -`nlohmann_json` pkg-config dependency. In Meson, it is preferred to use the -[`dependency()`](https://mesonbuild.com/Reference-manual.html#dependency) object with a subproject fallback, rather than -using the subproject directly. +which case a [pkg-config](pkg-config.md) file and the CMake package config files are installed. To use it, have your build system require +the `nlohmann_json` pkg-config dependency, or use [`find_package(nlohmann_json)`](cmake.md#external) in CMake. In Meson, +it is preferred to use the [`dependency()`](https://mesonbuild.com/Reference-manual.html#dependency) object with a +subproject fallback, rather than using the subproject directly. + +The options that change the library's configuration are available in Meson as well, named like the +[CMake options](cmake.md#cmake-options) without the `JSON_` prefix: `MultipleHeaders`, `GlobalUDLs`, +`ImplicitConversions`, `DisableEnumSerialization`, `DisableTupleReferenceConversion`, `Diagnostics`, +`Diagnostic_Positions`, `LegacyDiscardedValueComparison`, `StrictNulHandling`, `StrictBinaryUTF8`, and +`DeleteDeprecatedFunctions`. They have the same defaults as in CMake, except that +`MultipleHeaders` is `false`. Set them with `-D` when setting up the build, or with the subproject name as prefix when +the library is used as a subproject: + +```shell +meson setup build -Dnlohmann_json:Diagnostics=true +``` + +The resulting compile definitions are part of the Meson dependency, the pkg-config file, and the CMake target. ??? example "Example: Wrap" diff --git a/meson.build b/meson.build index 2800957ec..19f734784 100644 --- a/meson.build +++ b/meson.build @@ -2,24 +2,120 @@ project('nlohmann_json', 'cpp', version : '3.12.0', license : 'MIT', + meson_version : '>= 0.64', + default_options: ['cpp_std=c++11'], ) +if get_option('MultipleHeaders') + incdir = 'include' +else + incdir = 'single_include' +endif + +# The same compile definitions as the CMake target (see target_compile_definitions +# in CMakeLists.txt): only an option that differs from its default adds one. +json_defines = [] +if not get_option('GlobalUDLs') + json_defines += 'JSON_USE_GLOBAL_UDLS=0' +endif +if not get_option('ImplicitConversions') + json_defines += 'JSON_USE_IMPLICIT_CONVERSIONS=0' +endif +if get_option('DisableEnumSerialization') + json_defines += 'JSON_DISABLE_ENUM_SERIALIZATION=1' +endif +if get_option('DisableTupleReferenceConversion') + json_defines += 'JSON_DISABLE_TUPLE_REFERENCE_CONVERSION=1' +endif +if get_option('Diagnostics') + json_defines += 'JSON_DIAGNOSTICS=1' +endif +if get_option('Diagnostic_Positions') + json_defines += 'JSON_DIAGNOSTIC_POSITIONS=1' +endif +if get_option('LegacyDiscardedValueComparison') + json_defines += 'JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1' +endif +if get_option('StrictNulHandling') + json_defines += 'JSON_STRICT_NUL_HANDLING=1' +endif +if get_option('StrictBinaryUTF8') + json_defines += 'JSON_STRICT_BINARY_UTF8=1' +endif +if get_option('DeleteDeprecatedFunctions') + json_defines += 'JSON_DELETE_DEPRECATED_FUNCTIONS=1' +endif + +cpp_args = [] +foreach define : json_defines + cpp_args += '-D' + define +endforeach + nlohmann_json_dep = declare_dependency( - include_directories: include_directories('single_include') + compile_args: cpp_args, + include_directories: include_directories(incdir) ) +meson.override_dependency('nlohmann_json', nlohmann_json_dep) +# The multi-header version under the name earlier versions of this file used nlohmann_json_multiple_headers = declare_dependency( + compile_args: cpp_args, include_directories: include_directories('include') ) if not meson.is_subproject() -install_headers('single_include/nlohmann/json.hpp', subdir: 'nlohmann') -install_headers('single_include/nlohmann/json_fwd.hpp', subdir: 'nlohmann') -install_headers('single_include/nlohmann/json_literals.hpp', subdir: 'nlohmann') + install_subdir( + incdir / 'nlohmann', + install_dir: get_option('includedir'), + install_tag: 'devel', + ) -pkgc = import('pkgconfig') -pkgc.generate(name: 'nlohmann_json', - version: meson.project_version(), - description: 'JSON for Modern C++' -) + pkgc = import('pkgconfig') + pkgc.generate(name: 'nlohmann_json', + version: meson.project_version(), + description: 'JSON for Modern C++', + extra_cflags: cpp_args, + install_dir: get_option('datadir') / 'pkgconfig', + ) + + # CMake package config files, so that find_package(nlohmann_json) works. The + # include directory is given relative to the config files, so that the + # installation can be relocated. This is not possible if includedir or datadir + # is an absolute path outside the prefix (e.g., with the separate outputs of + # Nix); then the absolute include directory is used, as CMake does. + fs = import('fs') + cmake_install_dir = get_option('datadir') / 'cmake' / meson.project_name() + cmake_to_prefix = [] + if fs.is_absolute(get_option('includedir')) or fs.is_absolute(cmake_install_dir) + cmake_include_dir = (get_option('prefix') / get_option('includedir')).replace('\\', '/') + else + foreach component : cmake_install_dir.split('/') + cmake_to_prefix += '..' + endforeach + cmake_include_dir = '${_IMPORT_PREFIX}/' + get_option('includedir') + endif + + cmake_conf = configuration_data() + cmake_conf.set('PROJECT_NAME', meson.project_name()) + cmake_conf.set('PROJECT_VERSION', meson.project_version()) + cmake_conf.set('PROJECT_VERSION_MAJOR', meson.project_version().split('.')[0]) + cmake_conf.set('NLOHMANN_JSON_TARGET_NAME', meson.project_name()) + cmake_conf.set('NLOHMANN_JSON_TARGETS_EXPORT_NAME', meson.project_name() + 'Targets') + cmake_conf.set('NLOHMANN_JSON_INCLUDE_DIR', cmake_include_dir) + cmake_conf.set('NLOHMANN_JSON_CONFIG_TO_PREFIX', '/'.join(cmake_to_prefix)) + cmake_conf.set('NLOHMANN_JSON_COMPILE_DEFINITIONS', ';'.join(json_defines)) + + foreach cmake_file : [ + ['cmake/config.cmake.in', 'nlohmann_jsonConfig.cmake'], + ['cmake/nlohmann_jsonConfigVersion.cmake.in', 'nlohmann_jsonConfigVersion.cmake'], + ['cmake/nlohmann_jsonTargets.cmake.in', 'nlohmann_jsonTargets.cmake'], + ] + configure_file( + input: cmake_file[0], + output: cmake_file[1], + configuration: cmake_conf, + format: 'cmake@', + install_dir: cmake_install_dir, + ) + endforeach endif diff --git a/meson_options.txt b/meson_options.txt new file mode 100644 index 000000000..d4bfb201f --- /dev/null +++ b/meson_options.txt @@ -0,0 +1,66 @@ +option( + 'MultipleHeaders', + type: 'boolean', + value: false, + description: 'Use non-amalgamated version of the library', +) +option( + 'GlobalUDLs', + type: 'boolean', + value: true, + description: 'Place user-defined string literals in the global namespace', +) +option( + 'ImplicitConversions', + type: 'boolean', + value: true, + description: 'Enable implicit conversions', +) +option( + 'DisableEnumSerialization', + type: 'boolean', + value: false, + description: 'Disable default integer enum serialization', +) +option( + 'DisableTupleReferenceConversion', + type: 'boolean', + value: false, + description: 'Disable conversion from a one-element tuple of a JSON reference', +) +option( + 'Diagnostics', + type: 'boolean', + value: false, + description: 'Use extended diagnostic messages', +) +option( + 'Diagnostic_Positions', + type: 'boolean', + value: false, + description: 'Enable diagnostic positions', +) +option( + 'LegacyDiscardedValueComparison', + type: 'boolean', + value: false, + description: 'Enable legacy discarded value comparison', +) +option( + 'StrictNulHandling', + type: 'boolean', + value: false, + description: 'Enable strict NUL-byte handling', +) +option( + 'StrictBinaryUTF8', + type: 'boolean', + value: false, + description: 'Enable UTF-8 checks in the CBOR, UBJSON, BJData, and BSON writers', +) +option( + 'DeleteDeprecatedFunctions', + type: 'boolean', + value: false, + description: 'Delete the deprecated functions instead of only deprecating them', +) diff --git a/tools/check_build_options/README.md b/tools/check_build_options/README.md new file mode 100644 index 000000000..1b24a55ff --- /dev/null +++ b/tools/check_build_options/README.md @@ -0,0 +1,21 @@ +# check_build_options + +Checks that the Meson build and the pkg-config files offer the same options as the CMake target, so that a new CMake +option is not forgotten in one of them. + +The compile definitions of the CMake target (`target_compile_definitions` in [`CMakeLists.txt`](../../CMakeLists.txt)) +are the reference. When you add an option there, also add it to + +- the pkg-config block in `CMakeLists.txt` (`NLOHMANN_JSON_PKGCONFIG_CFLAGS`), +- [`meson_options.txt`](../../meson_options.txt), named without the `JSON_` prefix and with the same default, +- [`meson.build`](../../meson.build) (`json_defines`), and +- the list of Meson options in + [`docs/mkdocs/docs/integration/package_managers.md`](../../docs/mkdocs/docs/integration/package_managers.md). + +Run the check with + +```shell +make check_build_options +``` + +It needs only Python 3 and runs in the `ci_meson_install` CI job. diff --git a/tools/check_build_options/check_build_options.py b/tools/check_build_options/check_build_options.py new file mode 100755 index 000000000..9b9debfdc --- /dev/null +++ b/tools/check_build_options/check_build_options.py @@ -0,0 +1,144 @@ +#!/usr/bin/env python3 +"""Check that the Meson build and the pkg-config files offer the CMake options. + +The compile definitions of the CMake target (target_compile_definitions in +CMakeLists.txt) are the reference. For every option used there, the script +checks that + +- the CMake pkg-config file adds the same definition under the same condition, +- meson_options.txt has a boolean option of the same name without the "JSON_" + prefix and with the same default, +- meson.build adds the same definition under the same condition, and +- the Meson section of the package manager documentation lists the option. + +Meson's MultipleHeaders option selects the include directory and adds no +definition; it is the only Meson option without a definition. +""" + +import argparse +import os +import re +import sys + +REPO_ROOT = os.path.normpath(os.path.join(sys.path[0], '..', '..')) +DOCS = os.path.join('docs', 'mkdocs', 'docs', 'integration', 'package_managers.md') + +# Meson options that add no compile definition, with their default +MESON_ONLY = {'MultipleHeaders': 'false'} + + +def read(root, path): + with open(os.path.join(root, path), encoding='utf-8') as f: + return f.read() + + +def cmake_target_definitions(cmake): + """Return {option: (definition, add_if_on)} from target_compile_definitions.""" + block = re.search(r'target_compile_definitions\(\s*\$\{NLOHMANN_JSON_TARGET_NAME\}\s*INTERFACE(.*?)\n\)', cmake, re.S) + if not block: + sys.exit('CMakeLists.txt: target_compile_definitions of the target not found') + result = {} + for line in block.group(1).split('\n'): + line = line.strip() + if not line: + continue + m = re.fullmatch(r'\$<\$>:(\w+=\w+)>', line) + if m: + result[m.group(1)] = (m.group(2), False) + continue + m = re.fullmatch(r'\$<\$:(\w+=\w+)>', line) + if m: + result[m.group(1)] = (m.group(2), True) + continue + sys.exit(f'CMakeLists.txt: unexpected line in target_compile_definitions: {line}') + return result + + +def cmake_defaults(cmake): + """Return {option: 'true'/'false'} for the option() calls of the form JSON_.""" + return {m.group(1): 'true' if m.group(2) == 'ON' else 'false' + for m in re.finditer(r'^option\(JSON_(\w+)\s+"[^"]*"\s+(ON|OFF)\)', cmake, re.M)} + + +def cmake_pkgconfig_definitions(cmake): + """Return {option: (definition, add_if_on)} from the pkg-config block.""" + return {m.group(2): (m.group(3), m.group(1) is None) + for m in re.finditer(r'if \((NOT )?JSON_(\w+)\)\s*\n\s*string\(APPEND NLOHMANN_JSON_PKGCONFIG_CFLAGS " -D(\w+=\w+)"\)', cmake)} + + +def meson_options(options): + """Return {option: default} for the boolean options in meson_options.txt.""" + result = {} + for block in re.findall(r'option\((.*?)\)', options, re.S): + name = re.search(r"'(\w+)'", block).group(1) + kind = re.search(r"type\s*:\s*'(\w+)'", block) + value = re.search(r'value\s*:\s*(\w+)', block) + result[name] = value.group(1) if kind and kind.group(1) == 'boolean' and value else None + return result + + +def meson_definitions(meson): + """Return {option: (definition, add_if_on)} from meson.build.""" + return {m.group(2): (m.group(3), m.group(1) is None) + for m in re.finditer(r"if (not )?get_option\('(\w+)'\)\s*\n\s*json_defines \+= '(\w+=\w+)'", meson)} + + +def describe(definition): + name, add_if_on = definition + return f'{name} if {"enabled" if add_if_on else "disabled"}' + + +def compare(errors, where, expected, actual): + for option, definition in expected.items(): + if option not in actual: + errors.append(f'{where}: no definition for option {option} (expected {describe(definition)})') + elif actual[option] != definition: + errors.append(f'{where}: option {option} adds {describe(actual[option])}, expected {describe(definition)}') + for option in actual.keys() - expected.keys(): + errors.append(f'{where}: definition for option {option}, which the CMake target does not have') + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.split('\n')[0]) + parser.add_argument('root', nargs='?', default=REPO_ROOT, help='repository root (default: %(default)s)') + root = parser.parse_args().root + + cmake = read(root, 'CMakeLists.txt') + reference = cmake_target_definitions(cmake) + defaults = cmake_defaults(cmake) + errors = [] + + compare(errors, 'CMakeLists.txt (pkg-config)', reference, cmake_pkgconfig_definitions(cmake)) + compare(errors, 'meson.build', reference, meson_definitions(read(root, 'meson.build'))) + + options = meson_options(read(root, 'meson_options.txt')) + for option in reference: + if option not in defaults: + errors.append(f'CMakeLists.txt: no option(JSON_{option} ... ON|OFF)') + elif option not in options: + errors.append(f'meson_options.txt: option {option} missing') + elif options[option] != defaults[option]: + errors.append(f'meson_options.txt: option {option} must be boolean with value {defaults[option]} as in CMake') + for option, default in MESON_ONLY.items(): + if options.get(option) != default: + errors.append(f'meson_options.txt: option {option} must be boolean with value {default}') + for option in options.keys() - reference.keys() - MESON_ONLY.keys(): + errors.append(f'meson_options.txt: option {option} has no counterpart in the CMake target') + + docs = read(root, DOCS) + for option in sorted(set(options) & (reference.keys() | MESON_ONLY.keys())): + if f'`{option}`' not in docs: + errors.append(f'{DOCS}: Meson option {option} not listed') + + for error in errors: + print(error, file=sys.stderr) + if errors: + print('The Meson build and the pkg-config files must offer the options of the CMake target; see ' + 'tools/check_build_options/README.md.', file=sys.stderr) + return 1 + print(f'OK: {len(reference)} options with compile definitions agree between CMake, pkg-config, and Meson.') + return 0 + + +if __name__ == '__main__': + sys.exit(main()) From a269794db7ce14879dc07b0ad510e878b593d4d2 Mon Sep 17 00:00:00 2001 From: Suyog Verma Date: Thu, 8 Oct 2026 12:19:32 +0530 Subject: [PATCH 02/10] Use MSVC intrinsics for full multiplication (#5782) * Use MSVC intrinsics for full multiplication Signed-off-by: Suyog Verma * Fix formatting in unit-class_lexer Signed-off-by: Suyog Verma * Address review feedback Signed-off-by: Suyog Verma --------- Signed-off-by: Suyog Verma --- include/nlohmann/detail/bit_ops.hpp | 13 ++++++++-- single_include/nlohmann/json.hpp | 13 ++++++++-- tests/src/unit-class_lexer.cpp | 37 ++++++++++++++++++++++++++--- 3 files changed, 56 insertions(+), 7 deletions(-) diff --git a/include/nlohmann/detail/bit_ops.hpp b/include/nlohmann/detail/bit_ops.hpp index 9655ff18b..9ccb82c66 100644 --- a/include/nlohmann/detail/bit_ops.hpp +++ b/include/nlohmann/detail/bit_ops.hpp @@ -9,12 +9,15 @@ #pragma once #include // uint64_t +#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + #include // __umulh, _umul128 +#endif #include // Portable bit-level helpers for the number and string scanners. They use -// compiler builtins where available and plain C++ otherwise, so they need no -// platform headers and work regardless of byte order. +// compiler builtins or platform-specific intrinsics where available and plain +// C++ otherwise, so they work regardless of byte order. NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -52,6 +55,12 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc __extension__ using uint128 = unsigned __int128; const uint128 r = static_cast(a) * b; return {static_cast(r), static_cast(r >> 64u)}; +#elif defined(_MSC_VER) && defined(_M_X64) + std::uint64_t high = 0; + const std::uint64_t low = _umul128(a, b, &high); + return {low, high}; +#elif defined(_MSC_VER) && defined(_M_ARM64) + return {a * b, __umulh(a, b)}; #else const std::uint64_t a_lo = a & 0xFFFFFFFFu; const std::uint64_t a_hi = a >> 32u; diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4f1ded530..72c95feea 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -8791,13 +8791,16 @@ NLOHMANN_JSON_NAMESPACE_END #include // uint64_t +#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + #include // __umulh, _umul128 +#endif // #include // Portable bit-level helpers for the number and string scanners. They use -// compiler builtins where available and plain C++ otherwise, so they need no -// platform headers and work regardless of byte order. +// compiler builtins or platform-specific intrinsics where available and plain +// C++ otherwise, so they work regardless of byte order. NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -8835,6 +8838,12 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc __extension__ using uint128 = unsigned __int128; const uint128 r = static_cast(a) * b; return {static_cast(r), static_cast(r >> 64u)}; +#elif defined(_MSC_VER) && defined(_M_X64) + std::uint64_t high = 0; + const std::uint64_t low = _umul128(a, b, &high); + return {low, high}; +#elif defined(_MSC_VER) && defined(_M_ARM64) + return {a * b, __umulh(a, b)}; #else const std::uint64_t a_lo = a & 0xFFFFFFFFu; const std::uint64_t a_hi = a >> 32u; diff --git a/tests/src/unit-class_lexer.cpp b/tests/src/unit-class_lexer.cpp index 1d20901d6..860af81d6 100644 --- a/tests/src/unit-class_lexer.cpp +++ b/tests/src/unit-class_lexer.cpp @@ -17,6 +17,7 @@ using nlohmann::json; #include // uint32_t, uint64_t #include // strtod #include // memcpy +#include // numeric_limits #include // stringstream #include // string #include // pair @@ -891,8 +892,39 @@ TEST_CASE("Eisel-Lemire float conversion") SECTION("128-bit products and leading zeros") { + const auto check_product = [](std::uint64_t a, std::uint64_t b) + { + const auto product = nlohmann::detail::full_multiplication(a, b); + CHECK(big_from(product.high, product.low) == big_mul(big_from(0, a), big_from(0, b))); + }; + + const std::uint64_t max = (std::numeric_limits::max)(); + const std::array, 13> edge_cases = + { + { + {0, 0}, + {0, 1}, + {1, 1}, + {1, max}, + {0xFFFFFFFFu, 0x100000000u}, + {0x100000000u, 0x100000000u}, + {0x100000001u, 0x100000001u}, + {max, max}, + {max, 2}, + {0xFFFFFFFF00000000u, 0x100000001u}, + {0x100000001u, 0xFFFFFFFF00000000u}, + {max, 1}, + {2, max}, + } + }; + + for (const auto& test : edge_cases) + { + check_product(test.first, test.second); + } + // whichever implementation the compiler gets (with or without a - // 128-bit integer type or a builtin) + // 128-bit integer type or a builtin / intrinsic) std::uint64_t state = 42; for (int i = 0; i < 10000; ++i) { @@ -901,8 +933,7 @@ TEST_CASE("Eisel-Lemire float conversion") state ^= state << 17u; const std::uint64_t a = state; const std::uint64_t b = (state * 0x9E3779B97F4A7C15u) >> (i % 64); - const auto product = nlohmann::detail::full_multiplication(a, b); - CHECK(big_from(product.high, product.low) == big_mul(big_from(0, a), big_from(0, b))); + check_product(a, b); const int k = i % 64; const std::uint64_t x = (std::uint64_t{1} << k) | (a & ((std::uint64_t{1} << k) - 1)); From 3d7f554927d597b8f45927803eaadb0c267e1998 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 8 Oct 2026 16:11:16 +0200 Subject: [PATCH 03/10] Use the with_*_t aliases in tests, examples, and docs (#5787) * Use the with_*_t aliases in tests, examples, and docs Replace spelled-out basic_json<...> instantiations that only change one or two template parameters with nlohmann::json::with_*_t (or ordered_json::with_*_t when the object type is ordered_map). Types that change all three number types chain with_integers_t and with_float_t. The raw basic_json<...> spelling stays where the template parameter list itself is the subject: the alias tests in unit-udt.cpp, explicit instantiations, and the ordered_json/compile-time docs. Signed-off-by: Niels Lohmann * Fix unit-large_json for clang and JSON_DIAGNOSTICS Two test problems from #5781 broke CI on develop: CAPTURE(depth); trips clang's -Wextra-semi-stmt, and the type_error.321 messages did not account for the diagnostics path prefix. Signed-off-by: Niels Lohmann * Use static_cast in unit-hash for clang-tidy #5772 added functional casts that clang-tidy reports as C-style casts (google-readability-casting). Also append a char instead of a one-character string in unit-large_json. Signed-off-by: Niels Lohmann * Declare the expected message prefix const in unit-large_json Without JSON_DIAGNOSTICS the prefix was never modified, which clang-tidy reports (misc-const-correctness). Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .../macros/json_use_implicit_conversions.md | 2 +- .../docs/examples/custom_array_type.cpp | 3 +- .../docs/examples/custom_binary_type.cpp | 8 +-- .../docs/examples/custom_object_type.cpp | 3 +- .../docs/examples/custom_string_type.cpp | 4 +- .../docs/features/types/number_handling.md | 5 +- .../features/types/template_parameters.md | 8 ++- .../docs/integration/migration_guide.md | 2 +- tests/src/unit-allocator.cpp | 54 +++---------------- tests/src/unit-binary_formats.cpp | 2 +- tests/src/unit-comparison.cpp | 6 +-- tests/src/unit-custom-array-type.cpp | 4 +- tests/src/unit-custom-object-type.cpp | 6 +-- tests/src/unit-hash.cpp | 6 +-- tests/src/unit-large_json.cpp | 24 ++++++--- tests/src/unit-locale-cpp.cpp | 6 +-- tests/src/unit-msgpack.cpp | 4 +- tests/src/unit-regression1.cpp | 14 +++-- tests/src/unit-regression2.cpp | 21 ++------ tests/src/unit-regression3.cpp | 15 +----- tests/src/unit-serialization.cpp | 3 +- 21 files changed, 68 insertions(+), 132 deletions(-) diff --git a/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md b/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md index 675a8acee..2e7781378 100644 --- a/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md +++ b/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md @@ -67,7 +67,7 @@ By default, implicit conversions are enabled. `JSON_USE_IMPLICIT_CONVERSIONS` is defined to `0`: ```cpp - using wjson = nlohmann::basic_json; + using wjson = nlohmann::json::with_string_t; void load(const nlohmann::json& j); diff --git a/docs/mkdocs/docs/examples/custom_array_type.cpp b/docs/mkdocs/docs/examples/custom_array_type.cpp index 63651cadf..9765c51e8 100644 --- a/docs/mkdocs/docs/examples/custom_array_type.cpp +++ b/docs/mkdocs/docs/examples/custom_array_type.cpp @@ -1,11 +1,10 @@ #include -#include #include #include "custom_array_type.hpp" -using custom_json = nlohmann::basic_json; +using custom_json = nlohmann::json::with_array_t; int main() { diff --git a/docs/mkdocs/docs/examples/custom_binary_type.cpp b/docs/mkdocs/docs/examples/custom_binary_type.cpp index cea34ea38..0a835d475 100644 --- a/docs/mkdocs/docs/examples/custom_binary_type.cpp +++ b/docs/mkdocs/docs/examples/custom_binary_type.cpp @@ -1,16 +1,10 @@ -#include #include -#include -#include -#include #include #include "custom_binary_type.hpp" -using custom_json = nlohmann::basic_json; +using custom_json = nlohmann::json::with_binary_t; int main() { diff --git a/docs/mkdocs/docs/examples/custom_object_type.cpp b/docs/mkdocs/docs/examples/custom_object_type.cpp index d4f99ccf0..9d4b43daf 100644 --- a/docs/mkdocs/docs/examples/custom_object_type.cpp +++ b/docs/mkdocs/docs/examples/custom_object_type.cpp @@ -1,12 +1,11 @@ #include #include -#include #include #include "custom_object_type.hpp" -using custom_json = nlohmann::basic_json; +using custom_json = nlohmann::json::with_object_t; int main() { diff --git a/docs/mkdocs/docs/examples/custom_string_type.cpp b/docs/mkdocs/docs/examples/custom_string_type.cpp index 63b798fc6..2e71c9378 100644 --- a/docs/mkdocs/docs/examples/custom_string_type.cpp +++ b/docs/mkdocs/docs/examples/custom_string_type.cpp @@ -1,12 +1,10 @@ #include -#include -#include #include #include "custom_string_type.hpp" -using custom_json = nlohmann::basic_json; +using custom_json = nlohmann::json::with_string_t; int main() { diff --git a/docs/mkdocs/docs/features/types/number_handling.md b/docs/mkdocs/docs/features/types/number_handling.md index bae315160..13b6dfd43 100644 --- a/docs/mkdocs/docs/features/types/number_handling.md +++ b/docs/mkdocs/docs/features/types/number_handling.md @@ -351,9 +351,8 @@ The number types can be changed with template parameters. A `basic_json` type that uses `#!c long double` as floating-point type. - ```cpp hl_lines="2" - using json_ld = nlohmann::basic_json; + ```cpp hl_lines="1" + using json_ld = nlohmann::json::with_float_t; ``` Note values should then be parsed with `json_ld::parse` rather than `json::parse` as the latter would parse diff --git a/docs/mkdocs/docs/features/types/template_parameters.md b/docs/mkdocs/docs/features/types/template_parameters.md index a83b7a56a..f20102878 100644 --- a/docs/mkdocs/docs/features/types/template_parameters.md +++ b/docs/mkdocs/docs/features/types/template_parameters.md @@ -8,6 +8,10 @@ these requirements so they do not have to be discovered by trial and error. Each that are known to work for that parameter and the ones that do not, checked against Boost 1.83, Abseil 20250127.0, Folly, EASTL 3.21, `ankerl::unordered_dense`, `phmap`, `gtl`, `robin_hood`, `tsl::ordered_map`, and Qt 6. +To change a single template parameter and keep the others, use the member alias templates +[`with_*_t`](../../api/basic_json/with_t.md); for instance, `nlohmann::json::with_float_t` is `json` with +`#!cpp long double` as [`number_float_t`](../../api/basic_json/number_float_t.md). + ## How to read this page Requirements are split into two groups: @@ -143,7 +147,7 @@ struct unordered_map_object using base_t::base_t; }; -using unordered_json = nlohmann::basic_json; +using unordered_json = nlohmann::json::with_object_t; ``` Whether `#!cpp std::unordered_map` can be instantiated at all depends on the standard library: `object_t` is formed @@ -176,7 +180,7 @@ struct flat_hash_object using base_t::base_t; }; -using flat_hash_json = nlohmann::basic_json; +using flat_hash_json = nlohmann::json::with_object_t; ``` `absl::node_hash_map` keeps references to the mapped values valid across insertions; `absl::flat_hash_map` does not, diff --git a/docs/mkdocs/docs/integration/migration_guide.md b/docs/mkdocs/docs/integration/migration_guide.md index 17b2eb360..bfbf850e0 100644 --- a/docs/mkdocs/docs/integration/migration_guide.md +++ b/docs/mkdocs/docs/integration/migration_guide.md @@ -116,7 +116,7 @@ function to use instead. === "Deprecated" ```cpp - using my_json = nlohmann::basic_json; + using my_json = nlohmann::json::with_string_t; nlohmann::json_pointer ptr("/foo/bar/1"); ``` diff --git a/tests/src/unit-allocator.cpp b/tests/src/unit-allocator.cpp index 07eb21c4e..4062870d5 100644 --- a/tests/src/unit-allocator.cpp +++ b/tests/src/unit-allocator.cpp @@ -370,14 +370,7 @@ TEST_CASE("copy of a deeply nested value survives a failing allocation (#5640)") #if !(defined(_ITERATOR_DEBUG_LEVEL) && _ITERATOR_DEBUG_LEVEL > 0) SECTION("std::map-backed object_t") { - using bad_alloc_json = nlohmann::basic_json; + using bad_alloc_json = nlohmann::json::with_allocator_t; check_deep_copy_survives_failing_allocation(false); check_deep_copy_survives_failing_allocation(true); @@ -385,14 +378,7 @@ TEST_CASE("copy of a deeply nested value survives a failing allocation (#5640)") SECTION("ordered_map-backed object_t") { - using bad_alloc_ordered_json = nlohmann::basic_json; + using bad_alloc_ordered_json = nlohmann::ordered_json::with_allocator_t; check_deep_copy_survives_failing_allocation(false); check_deep_copy_survives_failing_allocation(true); @@ -450,14 +436,7 @@ struct scratch_counting_allocator : std::allocator TEST_CASE("deep copy uses the provided allocator") { - using counting_json = nlohmann::basic_json; + using counting_json = nlohmann::json::with_allocator_t; // deeper than the 128 levels the copy constructor descends into, so the // innermost objects are copied by the iterative deep copy @@ -516,14 +495,7 @@ TEST_CASE("converting a deeply nested value from another specialization fails cl // the allocator in noexcept constructors, so a failing construction crashes // the program there instead of throwing std::bad_alloc. Nothing to check. #if !(defined(_MSC_VER) && _MSC_VER < 1910 && defined(_ITERATOR_DEBUG_LEVEL) && _ITERATOR_DEBUG_LEVEL > 0) - using countdown_json = nlohmann::basic_json; + using countdown_json = nlohmann::json::with_allocator_t; // deeper than the 128 levels the converting constructor descends into, so // that failures land on both sides of the bound - or, built with @@ -631,14 +603,7 @@ TEST_CASE("destructor performs no allocation, only deallocation") // Since that stack could itself throw bad_alloc from inside the // noexcept destructor (#5135), destroy() no longer allocates anything: // it only ever frees what is already there. - using counting_json = nlohmann::basic_json; + using counting_json = nlohmann::json::with_allocator_t; SECTION("array") { @@ -683,14 +648,7 @@ TEST_CASE("destructor performs no allocation, only deallocation") TEST_CASE("a failed allocation leaves the value unchanged") { // create JSON type using the throwing allocator - using my_json = nlohmann::basic_json; + using my_json = nlohmann::json::with_allocator_t; // Each of these creates a string, array, object, or binary value. The // value must be created before the type is changed: otherwise, a failed diff --git a/tests/src/unit-binary_formats.cpp b/tests/src/unit-binary_formats.cpp index 3e5edf765..03dbc94da 100644 --- a/tests/src/unit-binary_formats.cpp +++ b/tests/src/unit-binary_formats.cpp @@ -237,7 +237,7 @@ namespace // the binary formats as function pointers for "Binary formats with narrow number types"; // named functions rather than lambdas, because clang 3.5 cannot convert a lambda // to a function pointer in the braced initializer of the format table -using narrow_json = nlohmann::basic_json; +using narrow_json = nlohmann::json::with_integers_t::with_float_t; using bytes = std::vector; bytes encode_cbor(const json& j) diff --git a/tests/src/unit-comparison.cpp b/tests/src/unit-comparison.cpp index b9257cb17..1e409bfb0 100644 --- a/tests/src/unit-comparison.cpp +++ b/tests/src/unit-comparison.cpp @@ -824,7 +824,7 @@ struct unordered_object_t : std::map, Allocator> return !(lhs == rhs); } }; -using unordered_json = nlohmann::basic_json; +using unordered_json = nlohmann::json::with_object_t; // the entries "0" to "9", enumerated in ascending or in descending order unordered_json make_unordered_object(const bool descending) @@ -875,7 +875,7 @@ struct key_case_less template using key_case_map = std::map; -using key_case_json = nlohmann::basic_json; +using key_case_json = nlohmann::json::with_object_t; // the innermost value of a chain of single-element arrays template @@ -905,7 +905,7 @@ struct case_insensitive_less template using case_insensitive_map = std::map; -using ci_json = nlohmann::basic_json; +using ci_json = nlohmann::json::with_object_t; } // namespace TEST_CASE("equality of objects whose entries have no fixed order") diff --git a/tests/src/unit-custom-array-type.cpp b/tests/src/unit-custom-array-type.cpp index 7a374b612..6f015d234 100644 --- a/tests/src/unit-custom-array-type.cpp +++ b/tests/src/unit-custom-array-type.cpp @@ -22,7 +22,7 @@ namespace // std::deque has no capacity() member function, which the library only needs // to detect a reallocation for JSON_DIAGNOSTICS -using deque_json = nlohmann::basic_json; +using deque_json = nlohmann::json::with_array_t; // a std::vector whose at() is hidden: the library performs its own bounds // check and must not fall back to the container's checked accessor @@ -39,7 +39,7 @@ class vector_without_at : public std::vector void at() = delete; }; -using no_at_json = nlohmann::basic_json; +using no_at_json = nlohmann::json::with_array_t; } // namespace diff --git a/tests/src/unit-custom-object-type.cpp b/tests/src/unit-custom-object-type.cpp index 0eaaf2d3c..fdd063292 100644 --- a/tests/src/unit-custom-object-type.cpp +++ b/tests/src/unit-custom-object-type.cpp @@ -179,7 +179,7 @@ class no_key_compare_map } }; -using no_key_compare_json = nlohmann::basic_json; +using no_key_compare_json = nlohmann::json::with_object_t; // An ObjectType whose erase(iterator) returns void rather than the following // iterator, as for instance Abseil's hash maps do @@ -196,7 +196,7 @@ struct void_erase_map : std::map } }; -using void_erase_json = nlohmann::basic_json; +using void_erase_json = nlohmann::json::with_object_t; // wraps an iterator, but only offers the LegacyForwardIterator operations, // like the iterators of std::unordered_map and other hash maps @@ -388,7 +388,7 @@ class forward_only_map } }; -using forward_only_json = nlohmann::basic_json; +using forward_only_json = nlohmann::json::with_object_t; } // namespace diff --git a/tests/src/unit-hash.cpp b/tests/src/unit-hash.cpp index e70b39b5a..117bc0151 100644 --- a/tests/src/unit-hash.cpp +++ b/tests/src/unit-hash.cpp @@ -157,13 +157,13 @@ TEST_CASE("hash") // the ends of the integer ranges, which equal floats exactly const auto int_min = (std::numeric_limits::min)(); const auto int_max = (std::numeric_limits::max)(); - const auto two_63 = json::number_unsigned_t(1) << 63U; + const auto two_63 = static_cast(1) << 63U; CHECK(json(int_min) == json(-9223372036854775808.0)); CHECK(std::hash {}(json(int_min)) == std::hash {}(json(-9223372036854775808.0))); CHECK(json(two_63) == json(9223372036854775808.0)); CHECK(std::hash {}(json(two_63)) == std::hash {}(json(9223372036854775808.0))); - CHECK(json(json::number_unsigned_t(int_max)) == json(int_max)); - CHECK(std::hash {}(json(json::number_unsigned_t(int_max))) == std::hash {}(json(int_max))); + CHECK(json(static_cast(int_max)) == json(int_max)); + CHECK(std::hash {}(json(static_cast(int_max))) == std::hash {}(json(int_max))); } TEST_CASE("hash") diff --git a/tests/src/unit-large_json.cpp b/tests/src/unit-large_json.cpp index 6ce5626d8..cd615f4c3 100644 --- a/tests/src/unit-large_json.cpp +++ b/tests/src/unit-large_json.cpp @@ -426,7 +426,7 @@ TEST_CASE("issue #5392 - binary writers on deeply nested values") { for (std::size_t depth = 120; depth <= 140; ++depth) { - CAPTURE(depth); + CAPTURE(depth) const json array = nested_array(depth, json(7)); CHECK(json::from_cbor(json::to_cbor(array)) == array); @@ -464,7 +464,7 @@ TEST_CASE("issue #5392 - binary writers on deeply nested values") nlohmann::detail::recursion_depth_limit() + 1, nlohmann::detail::recursion_depth_limit() + 2 }) { - CAPTURE(depth); + CAPTURE(depth) const json array = nested_array(depth, json(0)); std::vector expected_cbor(depth, 0x81); @@ -505,10 +505,22 @@ TEST_CASE("issue #5392 - binary writers on deeply nested values") const json discarded_leaf(json::value_t::discarded); const json deep_discarded = nested_array(depth, discarded_leaf); - CHECK_THROWS_WITH_AS(json::to_cbor(deep_discarded), "[json.exception.type_error.321] cannot serialize discarded value to CBOR", json::type_error); - CHECK_THROWS_WITH_AS(json::to_msgpack(deep_discarded), "[json.exception.type_error.321] cannot serialize discarded value to MessagePack", json::type_error); - CHECK_THROWS_WITH_AS(json::to_ubjson(deep_discarded), "[json.exception.type_error.321] cannot serialize discarded value to UBJSON", json::type_error); - CHECK_THROWS_WITH_AS(json::to_bjdata(deep_discarded), "[json.exception.type_error.321] cannot serialize discarded value to BJData", json::type_error); + // with diagnostics, the message names the path to the discarded leaf +#if JSON_DIAGNOSTICS + std::string path; + for (std::size_t i = 0; i < depth; ++i) + { + path += "/0"; + } + const std::string prefix = "[json.exception.type_error.321] (" + path + ") "; +#else + const std::string prefix = "[json.exception.type_error.321] "; +#endif + + CHECK_THROWS_WITH_AS(json::to_cbor(deep_discarded), (prefix + "cannot serialize discarded value to CBOR").c_str(), json::type_error); + CHECK_THROWS_WITH_AS(json::to_msgpack(deep_discarded), (prefix + "cannot serialize discarded value to MessagePack").c_str(), json::type_error); + CHECK_THROWS_WITH_AS(json::to_ubjson(deep_discarded), (prefix + "cannot serialize discarded value to UBJSON").c_str(), json::type_error); + CHECK_THROWS_WITH_AS(json::to_bjdata(deep_discarded), (prefix + "cannot serialize discarded value to BJData").c_str(), json::type_error); } SECTION("does not overflow the C++ stack") diff --git a/tests/src/unit-locale-cpp.cpp b/tests/src/unit-locale-cpp.cpp index 626296825..4865e5687 100644 --- a/tests/src/unit-locale-cpp.cpp +++ b/tests/src/unit-locale-cpp.cpp @@ -172,7 +172,7 @@ TEST_CASE("locale-dependent test (LC_NUMERIC=de_DE)") // a floating-point type that is not a float or a double is written // with snprintf, whose locale-specific decimal point and thousands // separator are undone afterwards - using long_double_json = nlohmann::basic_json; + using long_double_json = nlohmann::json::with_float_t; CHECK(long_double_json(12345.5L).dump() == "12345.5"); CHECK(long_double_json(1.0L).dump() == "1.0"); CHECK(long_double_json(-0.25L).dump() == "-0.25"); @@ -272,7 +272,7 @@ TEST_CASE("locale changes between lexer construction and number conversion (#519 } text += "]"; - using long_double_json = nlohmann::basic_json; + using long_double_json = nlohmann::json::with_float_t; // reference values, parsed without a locale switch REQUIRE(std::setlocale(LC_NUMERIC, "C") != nullptr); @@ -432,7 +432,7 @@ TEST_CASE("locale changes during a single dump() (#5709 item 3)") // long double on 64-bit Arm, where it is IEEE-754 double) takes the // locale-independent to_chars() path instead, and this test is a no-op // there. - using long_double_json = nlohmann::basic_json; + using long_double_json = nlohmann::json::with_float_t; using ld_limits = std::numeric_limits; const bool is_ieee_single_or_double = (ld_limits::is_iec559 && ld_limits::digits == 24 && ld_limits::max_exponent == 128) || diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index b705f94b7..d29769738 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -2432,8 +2432,8 @@ TEST_CASE("MessagePack numbers use the active union member (see #5644)") // used to read the union member that was not the active one, writing // wrong bytes for some values; std::int64_t/std::uint64_t (the default // types, where both members have the same width) were not affected - using int32_json = nlohmann::basic_json; - using int16_json = nlohmann::basic_json; + using int32_json = nlohmann::json::with_integers_t; + using int16_json = nlohmann::json::with_integers_t; SECTION("number_integer_t = std::int32_t") { diff --git a/tests/src/unit-regression1.cpp b/tests/src/unit-regression1.cpp index 8f0d4c747..084eb79fd 100644 --- a/tests/src/unit-regression1.cpp +++ b/tests/src/unit-regression1.cpp @@ -39,7 +39,7 @@ using nlohmann::json; template using my_workaround_fifo_map = nlohmann::fifo_map, A>; -using my_json = nlohmann::basic_json; +using my_json = nlohmann::json::with_object_t; ///////////////////////////////////////////////////////////////////// // for #977 @@ -86,8 +86,7 @@ struct foo_serializer < T, typename std::enable_if < !std::is_same::valu }; } // namespace ns -using foo_json = nlohmann::basic_json>; +using foo_json = nlohmann::json::with_json_serializer_t; ///////////////////////////////////////////////////////////////////// // for #805 @@ -254,7 +253,7 @@ TEST_CASE("regression tests 1") { // create JSON class with nonstandard integer number type using custom_json = - nlohmann::basic_json; + nlohmann::json::with_integers_t::with_float_t; custom_json j; j["int_1"] = 1; CHECK(j["int_1"] == 1); @@ -470,18 +469,17 @@ TEST_CASE("regression tests 1") // create JSON class with nonstandard float number type // float - nlohmann::basic_json const j_float = + nlohmann::json::with_integers_t::with_float_t const j_float = 1.23e25f; CHECK(j_float.get() == 1.23e25f); // double - nlohmann::basic_json const j_double = + nlohmann::json const j_double = 1.23e35; CHECK(j_double.get() == 1.23e35); // long double - nlohmann::basic_json - const j_long_double = 1.23e45L; + nlohmann::json::with_float_t const j_long_double = 1.23e45L; CHECK(j_long_double.get() == 1.23e45L); } diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index d084e6432..b532e75ae 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -64,18 +64,7 @@ using ordered_json = nlohmann::ordered_json; ///////////////////////////////////////////////////////////////////// // for #4804 ///////////////////////////////////////////////////////////////////// - using json_4804 = nlohmann::basic_json, // BinaryType - void // CustomBaseClass - >; + using json_4804 = nlohmann::json::with_binary_t>; #endif #ifdef JSON_HAS_CPP_20 @@ -107,7 +96,7 @@ DOCTEST_CLANG_SUPPRESS_WARNING("-Wexit-time-destructors") // for #1021 ///////////////////////////////////////////////////////////////////// -using float_json = nlohmann::basic_json; +using float_json = nlohmann::json::with_float_t; #if (defined(__cpp_exceptions) || defined(__EXCEPTIONS) || defined(_CPPUNWIND)) && !defined(JSON_NOEXCEPTION) namespace @@ -155,10 +144,8 @@ struct failing_allocator : std::allocator }; }; -using failing_json = nlohmann::basic_json; -using failing_ordered_json = nlohmann::basic_json; +using failing_json = nlohmann::json::with_allocator_t; +using failing_ordered_json = nlohmann::ordered_json::with_allocator_t; // builds `depth` levels of nesting around a scalar, iteratively (never // recursing: each wrap only moves the previous, already-built value, which diff --git a/tests/src/unit-regression3.cpp b/tests/src/unit-regression3.cpp index 69a95fb9b..f8d9740fa 100644 --- a/tests/src/unit-regression3.cpp +++ b/tests/src/unit-regression3.cpp @@ -62,18 +62,7 @@ using ordered_json = nlohmann::ordered_json; ///////////////////////////////////////////////////////////////////// // for #4804 ///////////////////////////////////////////////////////////////////// - using json_4804 = nlohmann::basic_json, // BinaryType - void // CustomBaseClass - >; + using json_4804 = nlohmann::json::with_binary_t>; #endif #ifdef JSON_HAS_CPP_20 @@ -930,7 +919,7 @@ TEST_CASE("regression test #5476 - array type without reserve()") { // the capacity reserved for definite-length arrays must not require the // array type to have a reserve() member function - using deque_json = nlohmann::basic_json; + using deque_json = nlohmann::json::with_array_t; SECTION("std::deque") { diff --git a/tests/src/unit-serialization.cpp b/tests/src/unit-serialization.cpp index 1aa2111e7..c56c4f5c1 100644 --- a/tests/src/unit-serialization.cpp +++ b/tests/src/unit-serialization.cpp @@ -367,8 +367,7 @@ TEST_CASE("dump for basic_json with long double number_float_t") // serializer::dump_float(x, std::false_type). That branch must use the // "%.*Lg" format specifier; using "%.*g" with a long double argument is // undefined behavior and corrupts the output. - using long_double_json = nlohmann::basic_json; + using long_double_json = nlohmann::json::with_float_t; SECTION("round-trip dump/parse") { From d8dfc0d0f9cf1097dd2768e3a8770bbddcc0aa08 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 8 Oct 2026 16:21:59 +0200 Subject: [PATCH 04/10] Fix unused-result warnings in the contains and parse_error examples (#5788) contains(json_pointer) is marked JSON_HEDLEY_WARN_UNUSED_RESULT since #5477, and parse() has been for longer. The contains example ignored the result in two try blocks waiting for a parse_error that contains() never throws, so they printed nothing; print the result for those pointers instead. Signed-off-by: Niels Lohmann --- .../docs/examples/contains__json_pointer.cpp | 26 ++++--------------- .../examples/contains__json_pointer.output | 2 ++ docs/mkdocs/docs/examples/parse_error.cpp | 2 +- 3 files changed, 8 insertions(+), 22 deletions(-) diff --git a/docs/mkdocs/docs/examples/contains__json_pointer.cpp b/docs/mkdocs/docs/examples/contains__json_pointer.cpp index 14d8514b4..985ad9ec5 100644 --- a/docs/mkdocs/docs/examples/contains__json_pointer.cpp +++ b/docs/mkdocs/docs/examples/contains__json_pointer.cpp @@ -19,25 +19,9 @@ int main() << j.contains("/array/1"_json_pointer) << '\n' << j.contains("/array/-"_json_pointer) << '\n' << j.contains("/array/4"_json_pointer) << '\n' - << j.contains("/baz"_json_pointer) << std::endl; - - try - { - // try to use an array index with leading '0' - j.contains("/array/01"_json_pointer); - } - catch (const json::parse_error& e) - { - std::cout << e.what() << '\n'; - } - - try - { - // try to use an array index that is not a number - j.contains("/array/one"_json_pointer); - } - catch (const json::parse_error& e) - { - std::cout << e.what() << '\n'; - } + << j.contains("/baz"_json_pointer) << '\n' + // an array index with a leading '0' is not found + << j.contains("/array/01"_json_pointer) << '\n' + // an array index that is not a number is not found + << j.contains("/array/one"_json_pointer) << std::endl; } diff --git a/docs/mkdocs/docs/examples/contains__json_pointer.output b/docs/mkdocs/docs/examples/contains__json_pointer.output index dd1eb38c1..b989a7c30 100644 --- a/docs/mkdocs/docs/examples/contains__json_pointer.output +++ b/docs/mkdocs/docs/examples/contains__json_pointer.output @@ -5,3 +5,5 @@ true false false false +false +false diff --git a/docs/mkdocs/docs/examples/parse_error.cpp b/docs/mkdocs/docs/examples/parse_error.cpp index ce15ebe22..2f73ac65f 100644 --- a/docs/mkdocs/docs/examples/parse_error.cpp +++ b/docs/mkdocs/docs/examples/parse_error.cpp @@ -8,7 +8,7 @@ int main() try { // parsing input with a syntax error - json::parse("[1,2,3,]"); + json j = json::parse("[1,2,3,]"); } catch (const json::parse_error& e) { From d33068da730aa1ffa3f1acb332e36653681efa2c Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 8 Oct 2026 17:44:34 +0200 Subject: [PATCH 05/10] Address review comments on the API stability docs and a test comment (#5784) * Address review comments on #5775 and #5779 Allow new defaulted parameters and new default arguments in the API stability rules, mention the macro opt-in, and drop the redundant recompile advice. Describe test-diagnostics-optimized as the regression test for the fixed #5742. Signed-off-by: Niels Lohmann * Document what counts as a breaking change in the API stability rules Spell out the 3.x compatibility rules in the roadmap: new defaulted parameters, new default arguments, noexcept/constexpr, template parameters, parse and dump results, accepted input, key iteration order, iterator invalidation, implicit conversions, to_json/from_json lookup, json_sax, value_t enumerators, and documented macros, CMake options and headers. Also list std::hash values as not part of the public API, and link the macro overview from the section. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/community/roadmap.md | 32 +++++++++++++++++++-------- tests/CMakeLists.txt | 3 ++- 2 files changed, 25 insertions(+), 10 deletions(-) diff --git a/docs/mkdocs/docs/community/roadmap.md b/docs/mkdocs/docs/community/roadmap.md index 27afeb4c5..1b1f287a9 100644 --- a/docs/mkdocs/docs/community/roadmap.md +++ b/docs/mkdocs/docs/community/roadmap.md @@ -36,13 +36,26 @@ work items are tracked in the [GitHub milestones](https://github.com/nlohmann/js ## API stability Releases follow [semantic versioning](https://semver.org): a minor or patch release of version 3.x does not break code -that uses the public API. In particular, a 3.x release does not: +that uses the public API, unless that code opts in to a change with a macro as described [below](#version-40). In +particular, a 3.x release does not: -- change the signature of a function (its parameter types, return type, number of parameters, or the const-ness of a - member function); -- remove or rename a function or class; +- make breaking changes to the signature of a function: the types or order of its existing parameters, its return type, + its `noexcept` or `constexpr` specifier, or the const-ness of a member function. New parameters may be added if they + have a default value; +- remove or rename a function or class, or change the template parameters of a public class template; - change which exceptions a function throws, or the [exception ids](../home/exceptions.md); -- change access specifiers or default arguments. +- change access specifiers, or change or remove existing default arguments. New default arguments may be added; +- change the JSON type that a valid input parses to, or the text that `dump()` produces for a valid value; +- accept input that was rejected before, or reject input that was accepted before; +- change the order in which the keys of an object are iterated. The default type sorts keys, and + [`ordered_json`](../api/ordered_json.md) keeps insertion order; +- change when iterators, pointers, or references are invalidated, or the state of a moved-from `basic_json`; +- add or remove implicit conversions from `basic_json`; +- change how `to_json` and `from_json` functions are found, or the behavior of + [`adl_serializer`](../api/adl_serializer/index.md); +- add pure virtual functions to the [`json_sax`](../api/json_sax/index.md) interface; +- remove, rename, renumber, or add enumerators of `value_t`; +- remove or rename a documented macro, CMake option, CMake target, or header, or change what a documented macro does. Exceptions to these rules, for instance when fixing a bug requires changing the exception a function throws, are documented in the [release notes](../home/releases.md). @@ -51,13 +64,14 @@ The following are **not** part of the public API and may change in any release, - The text of exception messages returned by `what()`. Use the [exception id](../home/exceptions.md) to tell errors apart. -- The ABI, including `sizeof(basic_json)` and the memory layout of its values. Recompile your code when you upgrade the - library. The [versioned inline namespace](../features/namespace.md) turns mixing versions into a link error. +- The ABI, including `sizeof(basic_json)` and the memory layout of its values. The + [versioned inline namespace](../features/namespace.md) turns mixing versions into a link error. +- The hash values returned by `std::hash` for `basic_json`. Numbers that compare equal still hash equally. - Everything in namespace `nlohmann::detail`, and macros and type traits that are not documented in the [API reference](../api/basic_json/index.md). -Changes that would break the public API are only added behind a macro whose default keeps the 3.x behavior, see -[Version 4.0](#version-40). +Breaking changes are only added behind a macro whose default keeps the 3.x behavior. See [Version 4.0](#version-40) and +the [macro overview](../features/macros.md). ## Version 4.0 diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index e092616b7..9318838bb 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -138,7 +138,8 @@ json_test_set_test_options(test-disabled_exceptions # only the #972 regression test needs thirdparty/fifo_map on its include path json_test_set_test_options(test-regression1 LINK_LIBRARIES fifo_map_include) -# GCC's false -Warray-bounds error with JSON_DIAGNOSTICS only shows up when optimizing (#5742). +# Regression test for GCC's false -Warray-bounds error with JSON_DIAGNOSTICS (#5742, fixed in #5585). It only +# showed up when optimizing, so build this test with -O3 and the warning as an error. # -O3 makes the optimizer-driven warnings of the ci_test_gcc flag set (-Winline, # -Wsuggest-attribute=...) fire on the library's inline functions; they are not # what this test checks, so turn them off for it. From 69a0c1b82ca0d5c7e3e82524ce09abd407b4a0d6 Mon Sep 17 00:00:00 2001 From: Alex Prabhat Bara Date: Fri, 9 Oct 2026 16:52:07 +0530 Subject: [PATCH 06/10] Avoid allocating temporary basic_json for cbor and msgpack object keys (#5328) * avoid allocating temporary basic_json for CBOR and MessagePack object keys Signed-off-by: alexprabhat99 * add size() to the custom object key test type UBJSON and BJData access object keys through size() and c_str() directly, so the key type now provides both and the comment says why. Signed-off-by: alexprabhat99 * address review: drop key size()/c_str(), test keys below the depth limit Nothing in the library calls size() or c_str() on an object key, so the test key type only keeps data(), which JSON_DIAGNOSTICS needs. The CBOR and MessagePack custom key tests now also nest objects deeper than detail::recursion_depth_limit(), so keys written by write_cbor_iterative and write_msgpack_iterative are covered as well. Signed-off-by: alexprabhat99 --------- Signed-off-by: alexprabhat99 --- docs/mkdocs/docs/api/basic_json/object_t.md | 3 +- .../nlohmann/detail/output/binary_writer.hpp | 180 ++++++++++-------- single_include/nlohmann/json.hpp | 180 ++++++++++-------- tests/src/custom_object_key_type.hpp | 75 ++++++++ tests/src/unit-cbor.cpp | 53 ++++++ tests/src/unit-msgpack.cpp | 53 ++++++ 6 files changed, 393 insertions(+), 151 deletions(-) create mode 100644 tests/src/custom_object_key_type.hpp diff --git a/docs/mkdocs/docs/api/basic_json/object_t.md b/docs/mkdocs/docs/api/basic_json/object_t.md index 204e5ab28..5e7cab9e9 100644 --- a/docs/mkdocs/docs/api/basic_json/object_t.md +++ b/docs/mkdocs/docs/api/basic_json/object_t.md @@ -26,7 +26,8 @@ To store objects in C++, a type is defined by the template parameters described `StringType` : the type of the keys or names (e.g., `std::string`). The comparison function `std::less` is used to - order elements inside the container. + order elements inside the container. `object_t::key_type` must be implicitly convertible to `string_t` (required by the + binary formats). `AllocatorType` : the allocator to use for objects (e.g., `std::allocator`) diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 1cb64f08d..4af3ff9fb 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -244,16 +244,7 @@ class binary_writer case value_t::string: { - string_t storage; - const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); - - // step 1: write control byte and the string length - write_cbor_head(0x60, value.size()); - - // step 2: write the string - oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + write_cbor_string(*j.m_data.m_value.string, j); break; } @@ -316,23 +307,20 @@ class binary_writer case value_t::object: { + static_assert( + std::is_convertible < + typename BasicJsonType::object_t::key_type, + string_t >::value, + "object_t::key_type must be implicitly convertible to string_t"); + // step 1: write control byte and the object size write_cbor_head(0xA0, j.m_data.m_value.object->size()); for (const auto& el : *j.m_data.m_value.object) { - // el.first is checked here, against the object as - // diagnostics context, because write_cbor(el.first) - // converts it to a temporary basic_json that would be - // used as the context instead; for error_handler_t::keep - // and ::replace/::ignore the recursive write_cbor(el.first) - // call below handles the key like any other string, so no - // separate check is needed here for those - if (error_handler == error_handler_t::strict) - { - check_utf8(el.first, j); - } - write_cbor(el.first); + // el.first is written directly (not via a temporary + // basic_json), with the object as diagnostics context + write_cbor_string(el.first, j); write_cbor(el.second, depth + 1); } break; @@ -491,39 +479,7 @@ class binary_writer case value_t::string: { - string_t storage; - const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); - - // step 1: write control byte and the string length - const auto N = to_msgpack_length(value.size(), j); - if (N <= 31) - { - // fixstr - write_number(static_cast(0xA0 | N)); - } - else if (N <= (std::numeric_limits::max)()) - { - // str 8 - oa.write_character(to_char_type(0xD9)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - // str 16 - oa.write_character(to_char_type(0xDA)); - write_number(static_cast(N)); - } - else - { - // str 32 - oa.write_character(to_char_type(0xDB)); - write_number(static_cast(N)); - } - - // step 2: write the string - oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + write_msgpack_string(*j.m_data.m_value.string, j); break; } @@ -629,19 +585,20 @@ class binary_writer case value_t::object: { + static_assert( + std::is_convertible < + typename BasicJsonType::object_t::key_type, + string_t >::value, + "object_t::key_type must be implicitly convertible to string_t"); + // step 1: write control byte and the object size write_msgpack_object_prefix(j.m_data.m_value.object->size(), j); for (const auto& el : *j.m_data.m_value.object) { - // as in write_cbor, el.first is checked here against the - // object as diagnostics context; the recursive call below - // handles keep/replace/ignore like any other string - if (error_handler == error_handler_t::strict) - { - check_utf8(el.first, j); - } - write_msgpack(el.first); + // as in write_cbor, el.first is written directly with the + // object as diagnostics context + write_msgpack_string(el.first, j); write_msgpack(el.second, depth + 1); } break; @@ -1025,13 +982,9 @@ class binary_writer continue; } - // el.first is checked here, against the object as diagnostics - // context, like the matching check in write_cbor's object case - if (error_handler == error_handler_t::strict) - { - check_utf8(current.object_it->first, *current.value); - } - write_cbor(current.object_it->first); + // the key is written directly (not via a temporary basic_json), + // with the object as diagnostics context, as in write_cbor + write_cbor_string(current.object_it->first, *current.value); const BasicJsonType* child = &(current.object_it->second); ++stack.back().object_it; write_cbor_value_or_push(*child, stack); @@ -1106,11 +1059,9 @@ class binary_writer continue; } - if (error_handler == error_handler_t::strict) - { - check_utf8(current.object_it->first, *current.value); - } - write_msgpack(current.object_it->first); + // as in write_cbor_iterative, the key is written directly with + // the object as diagnostics context + write_msgpack_string(current.object_it->first, *current.value); const BasicJsonType* child = &(current.object_it->second); ++stack.back().object_it; write_msgpack_value_or_push(*child, stack); @@ -1988,6 +1939,85 @@ class binary_writer } } + /*! + @brief write a CBOR text string + + @a value is checked or sanitized according to @ref error_handler, with + @a context (the string value itself, or the object a key belongs to) used + as diagnostics context; this avoids converting object keys to a temporary + basic_json just to write them + + @note When object_t::key_type is not string_t, @a value is a temporary + string_t converted from the key, which lives only until the end of + the caller's statement. The reference returned by + @ref sanitize_utf8_for_write may refer to it, so it must not escape + this function. + */ + void write_cbor_string(const string_t& value, const BasicJsonType& context) + { + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, context, storage); + + // step 1: write control byte and the string length + write_cbor_head(0x60, sanitized.size()); + + // step 2: write the string + oa.write_characters( + reinterpret_cast(sanitized.data()), + sanitized.size()); + } + + ///////////// + // MsgPack // + ///////////// + + /*! + @brief write a MessagePack str + + @a value is checked or sanitized according to @ref error_handler, with + @a context used as diagnostics context, as in @ref write_cbor_string + + @note As in @ref write_cbor_string, @a value may be a temporary string_t + converted from a key, so the reference returned by + @ref sanitize_utf8_for_write must not escape this function. + */ + void write_msgpack_string(const string_t& value, const BasicJsonType& context) + { + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, context, storage); + + // step 1: write control byte and the string length + const auto N = to_msgpack_length(sanitized.size(), context); + if (N <= 31) + { + // fixstr + write_number(static_cast(0xA0 | N)); + } + else if (N <= (std::numeric_limits::max)()) + { + // str 8 + oa.write_character(to_char_type(0xD9)); + write_number(static_cast(N)); + } + else if (N <= (std::numeric_limits::max)()) + { + // str 16 + oa.write_character(to_char_type(0xDA)); + write_number(static_cast(N)); + } + else + { + // str 32 + oa.write_character(to_char_type(0xDB)); + write_number(static_cast(N)); + } + + // step 2: write the string + oa.write_characters( + reinterpret_cast(sanitized.data()), + sanitized.size()); + } + //////////// // UBJSON // //////////// diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 72c95feea..428a6abbf 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -21749,16 +21749,7 @@ class binary_writer case value_t::string: { - string_t storage; - const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); - - // step 1: write control byte and the string length - write_cbor_head(0x60, value.size()); - - // step 2: write the string - oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + write_cbor_string(*j.m_data.m_value.string, j); break; } @@ -21821,23 +21812,20 @@ class binary_writer case value_t::object: { + static_assert( + std::is_convertible < + typename BasicJsonType::object_t::key_type, + string_t >::value, + "object_t::key_type must be implicitly convertible to string_t"); + // step 1: write control byte and the object size write_cbor_head(0xA0, j.m_data.m_value.object->size()); for (const auto& el : *j.m_data.m_value.object) { - // el.first is checked here, against the object as - // diagnostics context, because write_cbor(el.first) - // converts it to a temporary basic_json that would be - // used as the context instead; for error_handler_t::keep - // and ::replace/::ignore the recursive write_cbor(el.first) - // call below handles the key like any other string, so no - // separate check is needed here for those - if (error_handler == error_handler_t::strict) - { - check_utf8(el.first, j); - } - write_cbor(el.first); + // el.first is written directly (not via a temporary + // basic_json), with the object as diagnostics context + write_cbor_string(el.first, j); write_cbor(el.second, depth + 1); } break; @@ -21996,39 +21984,7 @@ class binary_writer case value_t::string: { - string_t storage; - const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); - - // step 1: write control byte and the string length - const auto N = to_msgpack_length(value.size(), j); - if (N <= 31) - { - // fixstr - write_number(static_cast(0xA0 | N)); - } - else if (N <= (std::numeric_limits::max)()) - { - // str 8 - oa.write_character(to_char_type(0xD9)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - // str 16 - oa.write_character(to_char_type(0xDA)); - write_number(static_cast(N)); - } - else - { - // str 32 - oa.write_character(to_char_type(0xDB)); - write_number(static_cast(N)); - } - - // step 2: write the string - oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + write_msgpack_string(*j.m_data.m_value.string, j); break; } @@ -22134,19 +22090,20 @@ class binary_writer case value_t::object: { + static_assert( + std::is_convertible < + typename BasicJsonType::object_t::key_type, + string_t >::value, + "object_t::key_type must be implicitly convertible to string_t"); + // step 1: write control byte and the object size write_msgpack_object_prefix(j.m_data.m_value.object->size(), j); for (const auto& el : *j.m_data.m_value.object) { - // as in write_cbor, el.first is checked here against the - // object as diagnostics context; the recursive call below - // handles keep/replace/ignore like any other string - if (error_handler == error_handler_t::strict) - { - check_utf8(el.first, j); - } - write_msgpack(el.first); + // as in write_cbor, el.first is written directly with the + // object as diagnostics context + write_msgpack_string(el.first, j); write_msgpack(el.second, depth + 1); } break; @@ -22530,13 +22487,9 @@ class binary_writer continue; } - // el.first is checked here, against the object as diagnostics - // context, like the matching check in write_cbor's object case - if (error_handler == error_handler_t::strict) - { - check_utf8(current.object_it->first, *current.value); - } - write_cbor(current.object_it->first); + // the key is written directly (not via a temporary basic_json), + // with the object as diagnostics context, as in write_cbor + write_cbor_string(current.object_it->first, *current.value); const BasicJsonType* child = &(current.object_it->second); ++stack.back().object_it; write_cbor_value_or_push(*child, stack); @@ -22611,11 +22564,9 @@ class binary_writer continue; } - if (error_handler == error_handler_t::strict) - { - check_utf8(current.object_it->first, *current.value); - } - write_msgpack(current.object_it->first); + // as in write_cbor_iterative, the key is written directly with + // the object as diagnostics context + write_msgpack_string(current.object_it->first, *current.value); const BasicJsonType* child = &(current.object_it->second); ++stack.back().object_it; write_msgpack_value_or_push(*child, stack); @@ -23493,6 +23444,85 @@ class binary_writer } } + /*! + @brief write a CBOR text string + + @a value is checked or sanitized according to @ref error_handler, with + @a context (the string value itself, or the object a key belongs to) used + as diagnostics context; this avoids converting object keys to a temporary + basic_json just to write them + + @note When object_t::key_type is not string_t, @a value is a temporary + string_t converted from the key, which lives only until the end of + the caller's statement. The reference returned by + @ref sanitize_utf8_for_write may refer to it, so it must not escape + this function. + */ + void write_cbor_string(const string_t& value, const BasicJsonType& context) + { + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, context, storage); + + // step 1: write control byte and the string length + write_cbor_head(0x60, sanitized.size()); + + // step 2: write the string + oa.write_characters( + reinterpret_cast(sanitized.data()), + sanitized.size()); + } + + ///////////// + // MsgPack // + ///////////// + + /*! + @brief write a MessagePack str + + @a value is checked or sanitized according to @ref error_handler, with + @a context used as diagnostics context, as in @ref write_cbor_string + + @note As in @ref write_cbor_string, @a value may be a temporary string_t + converted from a key, so the reference returned by + @ref sanitize_utf8_for_write must not escape this function. + */ + void write_msgpack_string(const string_t& value, const BasicJsonType& context) + { + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, context, storage); + + // step 1: write control byte and the string length + const auto N = to_msgpack_length(sanitized.size(), context); + if (N <= 31) + { + // fixstr + write_number(static_cast(0xA0 | N)); + } + else if (N <= (std::numeric_limits::max)()) + { + // str 8 + oa.write_character(to_char_type(0xD9)); + write_number(static_cast(N)); + } + else if (N <= (std::numeric_limits::max)()) + { + // str 16 + oa.write_character(to_char_type(0xDA)); + write_number(static_cast(N)); + } + else + { + // str 32 + oa.write_character(to_char_type(0xDB)); + write_number(static_cast(N)); + } + + // step 2: write the string + oa.write_characters( + reinterpret_cast(sanitized.data()), + sanitized.size()); + } + //////////// // UBJSON // //////////// diff --git a/tests/src/custom_object_key_type.hpp b/tests/src/custom_object_key_type.hpp new file mode 100644 index 000000000..fa23736d4 --- /dev/null +++ b/tests/src/custom_object_key_type.hpp @@ -0,0 +1,75 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include +#include +#include +#include +#include + +namespace custom_object_key_test +{ +class key +{ + public: + key() = default; + + key(const char* value) + : m_value(value) + {} + + key(std::string value) + : m_value(std::move(value)) + {} + + operator std::string() const + { + return m_value; + } + + // Required by JSON_DIAGNOSTICS, which reads object keys through data() + // when building the path of an exception. + const char* data() const noexcept + { + return m_value.data(); + } + + friend bool operator<(const key& lhs, const key& rhs) + { + return lhs.m_value < rhs.m_value; + } + + private: + std::string m_value; +}; + +template +class object + : public std::map < + key, + Value, + std::less, // NOLINT(modernize-use-transparent-functors) + typename std::allocator_traits::template rebind_alloc < + std::pair>> +{ + private: + using allocator_type = + typename std::allocator_traits::template rebind_alloc < + std::pair>; + + using base_type = + std::map, allocator_type>; // NOLINT(modernize-use-transparent-functors) + + public: + using base_type::base_type; +}; + +using json = nlohmann::basic_json; +} // namespace custom_object_key_test diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index 9ee371495..9b57d0664 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -28,6 +28,7 @@ using nlohmann::json; #include "make_test_data_available.hpp" #include "round_trip_corpus.hpp" #include "test_utils.hpp" +#include "custom_object_key_type.hpp" #include "sax_countdown.hpp" using utils::SaxCountdown; @@ -3357,3 +3358,55 @@ TEST_CASE("CBOR large strings and binaries (chunked reader)") } } } + +TEST_CASE("CBOR supports custom object key types") +{ + using custom_json = custom_object_key_test::json; + using custom_key = custom_object_key_test::key; + + custom_json::object_t object; + object.emplace(custom_key{"short"}, 1); + object.emplace( + custom_key{"a key longer than twenty-three characters"}, + 2); + + const custom_json value(std::move(object)); + const auto encoded = custom_json::to_cbor(value); + + CHECK(nlohmann::json::from_cbor(encoded) == nlohmann::json + { + {"short", 1}, + {"a key longer than twenty-three characters", 2} + }); +} + +TEST_CASE("CBOR supports custom object key types nested deeper than the recursion depth limit") +{ + // below detail::recursion_depth_limit(), keys are written by + // write_cbor_iterative instead of write_cbor + using custom_json = custom_object_key_test::json; + using custom_key = custom_object_key_test::key; + + const std::size_t depth = nlohmann::detail::recursion_depth_limit() + 10; + + custom_json value = 1; + nlohmann::json expected = 1; + for (std::size_t i = 0; i < depth; ++i) + { + // alternate short keys with ones long enough to need a length byte + const std::string name = (i % 2 == 0) ? "k" + std::to_string(i) + : "a key longer than thirty-one characters " + std::to_string(i); + + custom_json::object_t object; + object.emplace(custom_key{name}, std::move(value)); + value = custom_json(std::move(object)); + + nlohmann::json::object_t expected_object; + expected_object.emplace(name, std::move(expected)); + expected = nlohmann::json(std::move(expected_object)); + } + + const auto encoded = custom_json::to_cbor(value); + CHECK(encoded == nlohmann::json::to_cbor(expected)); + CHECK(nlohmann::json::from_cbor(encoded) == expected); +} diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index d29769738..0e07c8ea4 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -31,6 +31,7 @@ using nlohmann::json; #include "make_test_data_available.hpp" #include "round_trip_corpus.hpp" #include "test_utils.hpp" +#include "custom_object_key_type.hpp" #include "sax_countdown.hpp" using utils::SaxCountdown; @@ -2522,3 +2523,55 @@ TEST_CASE("MessagePack large strings and binaries (chunked reader)") } } } + +TEST_CASE("MessagePack supports custom object key types") +{ + using custom_json = custom_object_key_test::json; + using custom_key = custom_object_key_test::key; + + custom_json::object_t object; + object.emplace(custom_key{"short"}, 1); + object.emplace( + custom_key{"a key longer than thirty-one characters"}, + 2); + + const custom_json value(std::move(object)); + const auto encoded = custom_json::to_msgpack(value); + + CHECK(nlohmann::json::from_msgpack(encoded) == nlohmann::json + { + {"short", 1}, + {"a key longer than thirty-one characters", 2} + }); +} + +TEST_CASE("MessagePack supports custom object key types nested deeper than the recursion depth limit") +{ + // below detail::recursion_depth_limit(), keys are written by + // write_msgpack_iterative instead of write_msgpack + using custom_json = custom_object_key_test::json; + using custom_key = custom_object_key_test::key; + + const std::size_t depth = nlohmann::detail::recursion_depth_limit() + 10; + + custom_json value = 1; + nlohmann::json expected = 1; + for (std::size_t i = 0; i < depth; ++i) + { + // alternate short keys with ones long enough to need a length byte + const std::string name = (i % 2 == 0) ? "k" + std::to_string(i) + : "a key longer than thirty-one characters " + std::to_string(i); + + custom_json::object_t object; + object.emplace(custom_key{name}, std::move(value)); + value = custom_json(std::move(object)); + + nlohmann::json::object_t expected_object; + expected_object.emplace(name, std::move(expected)); + expected = nlohmann::json(std::move(expected_object)); + } + + const auto encoded = custom_json::to_msgpack(value); + CHECK(encoded == nlohmann::json::to_msgpack(expected)); + CHECK(nlohmann::json::from_msgpack(encoded) == expected); +} From d540750f21899cb063b5ca1b6933cc1e49bdc2fc Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 16:56:14 +0200 Subject: [PATCH 07/10] Use the with_*_t aliases in the remaining tests and docs (#5790) #5787 missed the instantiations spelled as `basic_json <` (astyle's formatting) or ending in the CustomBaseClass argument. Convert them, including the binary_t.md example pointed out in review. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/binary_t.md | 13 +----- docs/mkdocs/docs/examples/as_base_class.cpp | 14 +----- tests/src/unit-bson.cpp | 8 +--- tests/src/unit-custom-base-class.cpp | 29 +----------- tests/src/unit-custom-binary-type.cpp | 8 +--- tests/src/unit-msgpack.cpp | 52 +++------------------ tests/src/unit-ordered_json2.cpp | 11 +---- 7 files changed, 15 insertions(+), 120 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/binary_t.md b/docs/mkdocs/docs/api/basic_json/binary_t.md index 46b0f7ae2..da13c22a2 100644 --- a/docs/mkdocs/docs/api/basic_json/binary_t.md +++ b/docs/mkdocs/docs/api/basic_json/binary_t.md @@ -64,18 +64,7 @@ values of that type directly to a `basic_json` instance, and they will automatic rather than arrays: ```cpp -using custom_json = nlohmann::basic_json< - nlohmann::ordered_map, // ObjectType - std::vector, // ArrayType - std::string, // StringType - bool, // BooleanType - std::int64_t, // NumberIntegerType - std::uint64_t, // NumberUnsignedType - double, // NumberFloatType - std::allocator, // AllocatorType - nlohmann::adl_serializer, - std::vector // Custom BinaryType ->; +using custom_json = nlohmann::ordered_json::with_binary_t>; std::vector data{std::byte{1}, std::byte{2}, std::byte{3}}; custom_json j = data; // Creates a binary value, not an array diff --git a/docs/mkdocs/docs/examples/as_base_class.cpp b/docs/mkdocs/docs/examples/as_base_class.cpp index 48357b045..b5b62d78f 100644 --- a/docs/mkdocs/docs/examples/as_base_class.cpp +++ b/docs/mkdocs/docs/examples/as_base_class.cpp @@ -15,19 +15,7 @@ class base_class_with_hidden_members } }; -using json = nlohmann::basic_json < - std::map, - std::vector, - std::string, - bool, - std::int64_t, - std::uint64_t, - double, - std::allocator, - nlohmann::adl_serializer, - std::vector, - base_class_with_hidden_members - >; +using json = nlohmann::json::with_base_class_t; int main() { diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index ea5ca3130..03c901cee 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -46,9 +46,7 @@ class huge_binary_t : public std::vector } }; -using huge_binary_json = nlohmann::basic_json < - std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, huge_binary_t, void >; +using huge_binary_json = nlohmann::json::with_binary_t; // a string type that can be made to report a size beyond INT32_MAX without // allocating that much memory, so BSON length overflow can be tested for @@ -96,9 +94,7 @@ class huge_string_t : public std::string bool pretend_huge = false; }; -using huge_string_json = nlohmann::basic_json < - std::map, std::vector, huge_string_t, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, std::vector, void >; +using huge_string_json = nlohmann::json::with_string_t; } // namespace TEST_CASE("BSON") diff --git a/tests/src/unit-custom-base-class.cpp b/tests/src/unit-custom-base-class.cpp index 9718d1cda..673c7ddbb 100644 --- a/tests/src/unit-custom-base-class.cpp +++ b/tests/src/unit-custom-base-class.cpp @@ -400,20 +400,7 @@ class base_class_with_hidden_members std::size_t m_size = 42; }; -using json_with_hidden_base_members = - nlohmann::basic_json < - std::map, - std::vector, - std::string, - bool, - std::int64_t, - std::uint64_t, - double, - std::allocator, - nlohmann::adl_serializer, - std::vector, - base_class_with_hidden_members - >; +using json_with_hidden_base_members = nlohmann::json::with_base_class_t; TEST_CASE("JSON Node as_base_class") { @@ -459,19 +446,7 @@ struct const_member_base const int id = 7; // NOLINT(misc-non-private-member-variables-in-classes) }; -using json_with_const_base = nlohmann::basic_json < - std::map, - std::vector, - std::string, - bool, - std::int64_t, - std::uint64_t, - double, - std::allocator, - nlohmann::adl_serializer, - std::vector, - const_member_base - >; +using json_with_const_base = nlohmann::json::with_base_class_t; // build an array nested @a depth levels deep, with the innermost value 1; // every level is constructed (never assigned), since const_member_base does diff --git a/tests/src/unit-custom-binary-type.cpp b/tests/src/unit-custom-binary-type.cpp index d885376cb..6b504bd51 100644 --- a/tests/src/unit-custom-binary-type.cpp +++ b/tests/src/unit-custom-binary-type.cpp @@ -26,15 +26,11 @@ namespace // a BinaryType whose value type is signed: the elements must still be // processed as the numbers 0..255 -using char_binary_json = nlohmann::basic_json < - std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, std::vector, void >; +using char_binary_json = nlohmann::json::with_binary_t>; #ifdef JSON_HAS_CPP_17 // a BinaryType whose value type is not an integer type at all - using byte_binary_json = nlohmann::basic_json < - std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, std::vector, void >; + using byte_binary_json = nlohmann::json::with_binary_t>; #endif } // namespace diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index 0e07c8ea4..1e1005ad7 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -2202,10 +2202,7 @@ struct huge_array : std::vector } }; -using huge_array_json = nlohmann::basic_json < - std::map, huge_array, std::string, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, - std::vector, void >; +using huge_array_json = nlohmann::json::with_array_t; TEST_CASE("MessagePack Size above uint32 for array") { @@ -2250,18 +2247,7 @@ template, - void >; +using huge_object_json = nlohmann::json::with_object_t; TEST_CASE("MessagePack Size above uint32 for object") { @@ -2296,18 +2282,7 @@ struct huge_string : std::string } }; -using huge_string_json = nlohmann::basic_json < - std::map, - std::vector, - huge_string, - bool, - std::int64_t, - std::uint64_t, - double, - std::allocator, - nlohmann::adl_serializer, - std::vector, - void >; +using huge_string_json = nlohmann::json::with_string_t; TEST_CASE("MessagePack Size above uint32 for string") { @@ -2330,18 +2305,7 @@ struct huge_binary : std::vector } }; -using huge_binary_json = nlohmann::basic_json < - std::map, - std::vector, - std::string, - bool, - std::int64_t, - std::uint64_t, - double, - std::allocator, - nlohmann::adl_serializer, - huge_binary, - void >; +using huge_binary_json = nlohmann::json::with_binary_t; TEST_CASE("MessagePack Size above uint32 for binary") { @@ -2391,14 +2355,10 @@ class beyond_uint32_string_t : public std::string } }; -using beyond_uint32_string_json = nlohmann::basic_json < - std::map, std::vector, beyond_uint32_string_t, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, std::vector, void >; +using beyond_uint32_string_json = nlohmann::json::with_string_t; #endif -using beyond_uint32_binary_json = nlohmann::basic_json < - std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, beyond_uint32_binary_t, void >; +using beyond_uint32_binary_json = nlohmann::json::with_binary_t; } // namespace TEST_CASE("MessagePack lengths beyond UINT32_MAX cannot be serialized") diff --git a/tests/src/unit-ordered_json2.cpp b/tests/src/unit-ordered_json2.cpp index 653820628..4f4e18844 100644 --- a/tests/src/unit-ordered_json2.cpp +++ b/tests/src/unit-ordered_json2.cpp @@ -217,16 +217,7 @@ void int_to_string(alt_string& target, std::size_t value) target = std::to_string(value).c_str(); } -using alt_json = nlohmann::basic_json < - std::map, - std::vector, - alt_string, - bool, - std::int64_t, - std::uint64_t, - double, - std::allocator, - nlohmann::adl_serializer >; +using alt_json = nlohmann::json::with_string_t; bool operator<(const char* op1, const alt_string& op2) noexcept { From 374dfe4f0f3f259b1ccaab879a4ab706dbf97e4c Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 17:05:36 +0200 Subject: [PATCH 08/10] Keep converted object keys alive while writing UBJSON and BJData (#5791) * Keep converted object keys alive while writing UBJSON and BJData Since #5746, write_ubjson and write_ubjson_iterative pass each object key to sanitize_utf8_for_write and keep the returned reference. When object_t::key_type is not string_t but converts to it, the argument is a temporary that is destroyed at the end of the statement, and the function returns a reference to it in every case but a sanitized copy, so the key bytes are read from a dead object (AddressSanitizer: stack-use-after-scope). Default json and ordered_json are unaffected. Bind the key to a named object_key_string_t first: a reference when key_type is string_t, so no copy is added there, and a converted copy otherwise. A deleted overload of sanitize_utf8_for_write for anything other than string_t turns a recurrence into a compile error. Signed-off-by: Niels Lohmann * Suppress -Wunused-member-function for the converting_key test type converting_key::data() is only called when JSON_DIAGNOSTICS is enabled. Signed-off-by: Niels Lohmann * Fix clang-tidy findings in the UBJSON/BJData converted-key fix Suppress hicpp/modernize-use-equals-delete on the deleted sanitize_utf8_for_write overload: it guards a private helper and must stay private. Replace the C-style array in the new test with std::array. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .../nlohmann/detail/output/binary_writer.hpp | 23 ++- single_include/nlohmann/json.hpp | 23 ++- tests/src/unit-binary_utf8_error_handler.cpp | 157 ++++++++++++++++++ 3 files changed, 199 insertions(+), 4 deletions(-) diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 4af3ff9fb..212d2658f 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -85,6 +85,12 @@ template::value, + const string_t&, string_t >::type; using binary_t = typename BasicJsonType::binary_t; using number_float_t = typename BasicJsonType::number_float_t; @@ -776,8 +782,10 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { + // a converted key must outlive the reference returned by sanitize_utf8_for_write + const object_key_string_t key_string = el.first; string_t storage; - const string_t& key = sanitize_utf8_for_write(el.first, j, storage); + const string_t& key = sanitize_utf8_for_write(key_string, j, storage); write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(key.data()), @@ -1317,8 +1325,10 @@ class binary_writer continue; } + // a converted key must outlive the reference returned by sanitize_utf8_for_write + const object_key_string_t key_string = current.object_it->first; string_t storage; - const string_t& key = sanitize_utf8_for_write(current.object_it->first, j, storage); + const string_t& key = sanitize_utf8_for_write(key_string, j, storage); write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(key.data()), @@ -2760,6 +2770,11 @@ class binary_writer itself in every case but a sanitized `replace`/`ignore` one, so @a storage must outlive the returned reference only then. + @a s must be an lvalue that outlives the returned reference. An object key + whose `key_type` is not @ref string_t must therefore first be converted + into a named string_t (see @ref object_key_string_t); the deleted overload + below enforces this at compile time. + @param[in] s the string (value or object key) to write @param[in] context the value @a s belongs to (for diagnostics) @param[out] storage backing storage for a sanitized copy @@ -2789,6 +2804,10 @@ class binary_writer } } + /// deleted: anything but a string_t would bind a temporary that dies before the returned reference is used + template < typename T, enable_if_t < !std::is_same::value, int > = 0 > + const string_t& sanitize_utf8_for_write(const T& /*s*/, const BasicJsonType& /*context*/, string_t& /*storage*/) const = delete; // NOLINT(hicpp-use-equals-delete,modernize-use-equals-delete): a private helper's guard, not part of the interface + /*! @brief write an integer in the shortest encoding diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 428a6abbf..2e6838bbc 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -21590,6 +21590,12 @@ template::value, + const string_t&, string_t >::type; using binary_t = typename BasicJsonType::binary_t; using number_float_t = typename BasicJsonType::number_float_t; @@ -22281,8 +22287,10 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { + // a converted key must outlive the reference returned by sanitize_utf8_for_write + const object_key_string_t key_string = el.first; string_t storage; - const string_t& key = sanitize_utf8_for_write(el.first, j, storage); + const string_t& key = sanitize_utf8_for_write(key_string, j, storage); write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(key.data()), @@ -22822,8 +22830,10 @@ class binary_writer continue; } + // a converted key must outlive the reference returned by sanitize_utf8_for_write + const object_key_string_t key_string = current.object_it->first; string_t storage; - const string_t& key = sanitize_utf8_for_write(current.object_it->first, j, storage); + const string_t& key = sanitize_utf8_for_write(key_string, j, storage); write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(key.data()), @@ -24265,6 +24275,11 @@ class binary_writer itself in every case but a sanitized `replace`/`ignore` one, so @a storage must outlive the returned reference only then. + @a s must be an lvalue that outlives the returned reference. An object key + whose `key_type` is not @ref string_t must therefore first be converted + into a named string_t (see @ref object_key_string_t); the deleted overload + below enforces this at compile time. + @param[in] s the string (value or object key) to write @param[in] context the value @a s belongs to (for diagnostics) @param[out] storage backing storage for a sanitized copy @@ -24294,6 +24309,10 @@ class binary_writer } } + /// deleted: anything but a string_t would bind a temporary that dies before the returned reference is used + template < typename T, enable_if_t < !std::is_same::value, int > = 0 > + const string_t& sanitize_utf8_for_write(const T& /*s*/, const BasicJsonType& /*context*/, string_t& /*storage*/) const = delete; // NOLINT(hicpp-use-equals-delete,modernize-use-equals-delete): a private helper's guard, not part of the interface + /*! @brief write an integer in the shortest encoding diff --git a/tests/src/unit-binary_utf8_error_handler.cpp b/tests/src/unit-binary_utf8_error_handler.cpp index c3f2435fd..114c92885 100644 --- a/tests/src/unit-binary_utf8_error_handler.cpp +++ b/tests/src/unit-binary_utf8_error_handler.cpp @@ -12,7 +12,11 @@ #include using nlohmann::json; +#include +#include +#include #include +#include #include namespace @@ -55,6 +59,53 @@ std::string dump_and_parse(const std::string& raw, eh error_handler) return json::parse(json(raw).dump(-1, ' ', false, error_handler)).get(); } +// an object key type that is not string_t, but converts implicitly to it; +// data() is only used when JSON_DIAGNOSTICS is enabled +DOCTEST_CLANG_SUPPRESS_WARNING_PUSH +DOCTEST_CLANG_SUPPRESS_WARNING("-Wunused-member-function") +class converting_key +{ + public: + converting_key(const char* s) : m_value(s) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions) + converting_key(std::string s) : m_value(std::move(s)) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions) + + // the conversion yields a temporary string_t + operator std::string() const // NOLINT(google-explicit-constructor,hicpp-explicit-conversions) + { + return m_value; + } + + // read by the exception messages when JSON_DIAGNOSTICS is enabled + const char* data() const noexcept + { + return m_value.data(); + } + + friend bool operator<(const converting_key& lhs, const converting_key& rhs) + { + return lhs.m_value < rhs.m_value; + } + + private: + std::string m_value; +}; +DOCTEST_CLANG_SUPPRESS_WARNING_POP + +// ObjectType using converting_key; the Key template argument is ignored +template +class converting_key_object : public std::map, // NOLINT(modernize-use-transparent-functors) + typename std::allocator_traits::template rebind_alloc>> +{ + using base_type = std::map, // NOLINT(modernize-use-transparent-functors) + typename std::allocator_traits::template rebind_alloc>>; + + public: + using base_type::base_type; + using base_type::operator=; +}; + +using converting_key_json = nlohmann::basic_json; + } // namespace TEST_CASE("UTF-8 error_handler for the binary readers and writers") @@ -370,3 +421,109 @@ TEST_CASE("UTF-8 error_handler for the binary readers and writers") CHECK(json::from_bson(bson_bytes)["k"].get() == ill_formed_cases()[0].bytes); } } + +// The UBJSON and BJData writers bind the (possibly sanitized) key to a const +// string_t&. If key_type is not string_t but converts to it, the converted +// temporary must outlive that reference; this was a use-after-scope found by +// AddressSanitizer. Keys exceed the small string optimization on purpose. +TEST_CASE("UBJSON and BJData writers with an object_t whose key_type is not string_t") +{ + const std::string long_prefix(70, 'k'); + + SECTION("well-formed keys, every error_handler") + { + const std::string key1 = long_prefix + "-first"; + const std::string key2 = long_prefix + "-second"; + + converting_key_json::object_t o; + o.emplace(converting_key(key1), 1); + o.emplace(converting_key(key2), "value"); + const converting_key_json v(std::move(o)); + + json expected; + expected[key1] = 1; + expected[key2] = "value"; + + const std::array, 3> combos = {{{false, false}, {true, false}, {true, true}}}; + for (const auto h : all_handlers()) + { + CAPTURE(static_cast(h)) + for (const auto& combo : combos) + { + const bool use_count = combo.first; + const bool use_type = combo.second; + CAPTURE(use_count) + CAPTURE(use_type) + + CHECK(json::from_ubjson(converting_key_json::to_ubjson(v, use_count, use_type, h)) == expected); + CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, use_count, use_type, json::bjdata_version_t::draft2, h)) == expected); + CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, use_count, use_type, json::bjdata_version_t::draft3, h)) == expected); + } + } + } + + SECTION("ill-formed keys") + { + for (const auto& c : ill_formed_cases()) + { + CAPTURE(c.name) + const std::string key = long_prefix + c.bytes; + + converting_key_json::object_t o; + o.emplace(converting_key(key), 1); + const converting_key_json v(std::move(o)); + + CHECK_THROWS_AS(converting_key_json::to_ubjson(v, false, false, eh::strict), converting_key_json::type_error&); + CHECK_THROWS_AS(converting_key_json::to_bjdata(v, false, false, json::bjdata_version_t::draft2, eh::strict), converting_key_json::type_error&); + + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)) + const std::string expected = dump_and_parse(key, h); + + CHECK(json::from_ubjson(converting_key_json::to_ubjson(v, false, false, h)).begin().key() == expected); + CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, false, false, json::bjdata_version_t::draft2, h)).begin().key() == expected); + } + + CHECK(json::from_ubjson(converting_key_json::to_ubjson(v, false, false, eh::keep)).begin().key() == key); + CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, false, false, json::bjdata_version_t::draft2, eh::keep)).begin().key() == key); + } + } + + SECTION("nested deeper than the recursion limit") + { + // wrap the previous value, innermost first + converting_key_json v = 42; + json expected = 42; + for (int i = 199; i >= 0; --i) + { + const std::string key = "level-" + std::to_string(i) + "-" + std::string(64, 'x'); + + converting_key_json::object_t o; + o.emplace(converting_key(key), std::move(v)); + v = converting_key_json(std::move(o)); + + json e; + e[key] = std::move(expected); + expected = std::move(e); + } + + for (const auto h : all_handlers()) + { + CAPTURE(static_cast(h)) + for (const bool use_count : + { + false, true + }) + { + CAPTURE(use_count) + + CHECK(json::from_ubjson(converting_key_json::to_ubjson(v, use_count, false, h)) == expected); + CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, use_count, false, json::bjdata_version_t::draft2, h)) == expected); + } + } + } +} From 44e8597701aaed2e7ff23312abeee497d203950b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 17:06:41 +0200 Subject: [PATCH 09/10] Flatten deeply nested values without recursing per nesting level (#5792) * Flatten deeply nested values without recursing per nesting level json_pointer::flatten() called itself once per nesting level, so flatten() on a value nested deeply enough exhausted the call stack. #5547 and #5548 fixed merge_patch() and diff() from #5393, but flatten() was left out. flatten() now walks the value with an explicit stack and keeps the path in one buffer that grows and shrinks with it. It has a single code path and no depth limit: the old version built a new path string per child, so the iterative one is no slower on shallow values and much faster on deep ones. The output, including the order of an ordered_json result, is unchanged. Signed-off-by: Niels Lohmann * Construct flatten frames in place Give the frame a constructor so both call sites can use emplace_back, as suggested in the review; index starts at 0 for every frame. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/detail/json_pointer.hpp | 163 ++++++++++++++++------- single_include/nlohmann/json.hpp | 163 ++++++++++++++++------- tests/src/unit-json_pointer.cpp | 78 +++++++++++ 3 files changed, 308 insertions(+), 96 deletions(-) diff --git a/include/nlohmann/detail/json_pointer.hpp b/include/nlohmann/detail/json_pointer.hpp index 92f52aedc..0a1c0f43f 100644 --- a/include/nlohmann/detail/json_pointer.hpp +++ b/include/nlohmann/detail/json_pointer.hpp @@ -878,64 +878,131 @@ class json_pointer @param[in,out] result the result object to insert values to @note Empty objects or arrays are flattened to `null`. + + The value is walked with an explicit stack rather than the call stack, so + arbitrarily deeply nested values can be flattened. + + @sa https://github.com/nlohmann/json/issues/5393 */ template static void flatten(const string_t& reference_string, const BasicJsonType& value, BasicJsonType& result) { - switch (value.type()) + using object_const_iterator = typename BasicJsonType::object_t::const_iterator; + + // an array or object being walked: the container, the array index or + // object iterator of the next child, and the length of the path of the + // container itself + struct frame { - case detail::value_t::array: - { - if (value.m_data.m_value.array->empty()) - { - // flatten empty array as null - result[reference_string] = nullptr; - } - else - { - // iterate array and use index as a reference string - for (std::size_t i = 0; i < value.m_data.m_value.array->size(); ++i) - { - flatten(detail::concat(reference_string, '/', std::to_string(i)), - value.m_data.m_value.array->operator[](i), result); - } - } - break; - } + frame(const BasicJsonType* container_, object_const_iterator member_, const std::size_t path_length_) noexcept + : container(container_), member(std::move(member_)), path_length(path_length_) + {} - case detail::value_t::object: - { - if (value.m_data.m_value.object->empty()) - { - // flatten empty object as null - result[reference_string] = nullptr; - } - else - { - // iterate object and use keys as reference string - for (const auto& element : *value.m_data.m_value.object) - { - flatten(detail::concat(reference_string, '/', detail::escape(element.first)), element.second, result); - } - } - break; - } + const BasicJsonType* container; + std::size_t index = 0; + object_const_iterator member; + std::size_t path_length; + }; - case detail::value_t::null: - case detail::value_t::string: - case detail::value_t::boolean: - case detail::value_t::number_integer: - case detail::value_t::number_unsigned: - case detail::value_t::number_float: - case detail::value_t::binary: - case detail::value_t::discarded: - default: + // The containers being flattened are kept on an explicit stack, and + // every child is flattened completely before the next one, so the + // entries come out in the same order as with a recursive walk. The + // path of the value being flattened is kept in one buffer that grows + // and shrinks with the stack, rather than in a new string per level. + std::vector stack; + string_t path = reference_string; + + // flatten `v`, whose path is `path`: primitives and empty containers + // are added to the result right away; other containers get a frame + const auto enter = [&stack, &path, &result](const BasicJsonType & v) + { + switch (v.type()) { - // add a primitive value with its reference string - result[reference_string] = value; - break; + case detail::value_t::array: + { + if (v.m_data.m_value.array->empty()) + { + // flatten empty array as null + result[path] = nullptr; + } + else + { + stack.emplace_back(&v, object_const_iterator(), path.size()); + } + return; + } + + case detail::value_t::object: + { + if (v.m_data.m_value.object->empty()) + { + // flatten empty object as null + result[path] = nullptr; + } + else + { + stack.emplace_back(&v, v.m_data.m_value.object->begin(), path.size()); + } + return; + } + + case detail::value_t::null: + case detail::value_t::string: + case detail::value_t::boolean: + case detail::value_t::number_integer: + case detail::value_t::number_unsigned: + case detail::value_t::number_float: + case detail::value_t::binary: + case detail::value_t::discarded: + default: + { + // add a primitive value with its reference string + result[path] = v; + return; + } + } + }; + + enter(value); + while (!stack.empty()) + { + // the frame is changed through stack.back(): enter() may push a + // frame, which would invalidate a reference to it + const BasicJsonType* const container = stack.back().container; + + // drop the path of the previous child + path.resize(stack.back().path_length); + + if (container->is_array()) + { + const auto& array = *container->m_data.m_value.array; + const std::size_t i = stack.back().index; + if (i == array.size()) + { + stack.pop_back(); + continue; + } + + // iterate array and use index as a reference string + ++stack.back().index; + detail::concat_into(path, '/', detail::to_string(i)); + enter(array[i]); + } + else + { + const object_const_iterator it = stack.back().member; + if (it == container->m_data.m_value.object->end()) + { + stack.pop_back(); + continue; + } + + // iterate object and use keys as reference string + ++stack.back().member; + detail::concat_into(path, '/', detail::escape(it->first)); + enter(it->second); } } } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 2e6838bbc..f0b1077f5 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -20796,64 +20796,131 @@ class json_pointer @param[in,out] result the result object to insert values to @note Empty objects or arrays are flattened to `null`. + + The value is walked with an explicit stack rather than the call stack, so + arbitrarily deeply nested values can be flattened. + + @sa https://github.com/nlohmann/json/issues/5393 */ template static void flatten(const string_t& reference_string, const BasicJsonType& value, BasicJsonType& result) { - switch (value.type()) + using object_const_iterator = typename BasicJsonType::object_t::const_iterator; + + // an array or object being walked: the container, the array index or + // object iterator of the next child, and the length of the path of the + // container itself + struct frame { - case detail::value_t::array: - { - if (value.m_data.m_value.array->empty()) - { - // flatten empty array as null - result[reference_string] = nullptr; - } - else - { - // iterate array and use index as a reference string - for (std::size_t i = 0; i < value.m_data.m_value.array->size(); ++i) - { - flatten(detail::concat(reference_string, '/', std::to_string(i)), - value.m_data.m_value.array->operator[](i), result); - } - } - break; - } + frame(const BasicJsonType* container_, object_const_iterator member_, const std::size_t path_length_) noexcept + : container(container_), member(std::move(member_)), path_length(path_length_) + {} - case detail::value_t::object: - { - if (value.m_data.m_value.object->empty()) - { - // flatten empty object as null - result[reference_string] = nullptr; - } - else - { - // iterate object and use keys as reference string - for (const auto& element : *value.m_data.m_value.object) - { - flatten(detail::concat(reference_string, '/', detail::escape(element.first)), element.second, result); - } - } - break; - } + const BasicJsonType* container; + std::size_t index = 0; + object_const_iterator member; + std::size_t path_length; + }; - case detail::value_t::null: - case detail::value_t::string: - case detail::value_t::boolean: - case detail::value_t::number_integer: - case detail::value_t::number_unsigned: - case detail::value_t::number_float: - case detail::value_t::binary: - case detail::value_t::discarded: - default: + // The containers being flattened are kept on an explicit stack, and + // every child is flattened completely before the next one, so the + // entries come out in the same order as with a recursive walk. The + // path of the value being flattened is kept in one buffer that grows + // and shrinks with the stack, rather than in a new string per level. + std::vector stack; + string_t path = reference_string; + + // flatten `v`, whose path is `path`: primitives and empty containers + // are added to the result right away; other containers get a frame + const auto enter = [&stack, &path, &result](const BasicJsonType & v) + { + switch (v.type()) { - // add a primitive value with its reference string - result[reference_string] = value; - break; + case detail::value_t::array: + { + if (v.m_data.m_value.array->empty()) + { + // flatten empty array as null + result[path] = nullptr; + } + else + { + stack.emplace_back(&v, object_const_iterator(), path.size()); + } + return; + } + + case detail::value_t::object: + { + if (v.m_data.m_value.object->empty()) + { + // flatten empty object as null + result[path] = nullptr; + } + else + { + stack.emplace_back(&v, v.m_data.m_value.object->begin(), path.size()); + } + return; + } + + case detail::value_t::null: + case detail::value_t::string: + case detail::value_t::boolean: + case detail::value_t::number_integer: + case detail::value_t::number_unsigned: + case detail::value_t::number_float: + case detail::value_t::binary: + case detail::value_t::discarded: + default: + { + // add a primitive value with its reference string + result[path] = v; + return; + } + } + }; + + enter(value); + while (!stack.empty()) + { + // the frame is changed through stack.back(): enter() may push a + // frame, which would invalidate a reference to it + const BasicJsonType* const container = stack.back().container; + + // drop the path of the previous child + path.resize(stack.back().path_length); + + if (container->is_array()) + { + const auto& array = *container->m_data.m_value.array; + const std::size_t i = stack.back().index; + if (i == array.size()) + { + stack.pop_back(); + continue; + } + + // iterate array and use index as a reference string + ++stack.back().index; + detail::concat_into(path, '/', detail::to_string(i)); + enter(array[i]); + } + else + { + const object_const_iterator it = stack.back().member; + if (it == container->m_data.m_value.object->end()) + { + stack.pop_back(); + continue; + } + + // iterate object and use keys as reference string + ++stack.back().member; + detail::concat_into(path, '/', detail::escape(it->first)); + enter(it->second); } } } diff --git a/tests/src/unit-json_pointer.cpp b/tests/src/unit-json_pointer.cpp index b4ed2c9cc..20f2331e3 100644 --- a/tests/src/unit-json_pointer.cpp +++ b/tests/src/unit-json_pointer.cpp @@ -939,3 +939,81 @@ TEST_CASE("unescaping keeps a '~' that does not start an escape sequence") nlohmann::detail::unescape(s); CHECK(s == "~/~"); } + +TEST_CASE("flatten of structured values") +{ + SECTION("values nested too deeply for the call stack (#5393)") + { + // flatten() used to recurse once per nesting level + const std::size_t depth = 100000; + for (const bool objects : + { + false, true + }) + { + CAPTURE(objects) + std::string text; + std::string path; + for (std::size_t i = 0; i < depth; ++i) + { + text += objects ? "{\"a\":" : "["; + path += objects ? "/a" : "/0"; + } + text += "0"; + text += std::string(depth, objects ? '}' : ']'); + const auto value = json::parse(text); + + const auto flat = value.flatten(); + REQUIRE(flat.size() == 1); + REQUIRE(flat.begin().key().size() == path.size()); + CHECK(flat.begin().key() == path); + CHECK(flat.begin().value() == 0); + + // unflatten() is not iterative: it takes time and memory + // quadratic in the depth, so it is only roundtripped for a + // moderate depth + std::string small_text; + for (std::size_t i = 0; i < 500; ++i) + { + small_text += objects ? "{\"a\":" : "["; + } + small_text += "0"; + small_text += std::string(500, objects ? '}' : ']'); + const auto small_value = json::parse(small_text); + CHECK(small_value.flatten().unflatten() == small_value); + } + } + + SECTION("objects and arrays interleaved") + { + const json value = + { + {"a", {1, {{"b", json::array()}, {"c", json::object()}}, json::array({{{"x~/", {true, nullptr}}}})}}, + {"a/b", {{"~", 1}}}, + {"z", "s"} + }; + + const json expected = + { + {"/a/0", 1}, + {"/a/1/b", nullptr}, + {"/a/1/c", nullptr}, + {"/a/2/0/x~0~1/0", true}, + {"/a/2/0/x~0~1/1", nullptr}, + {"/a~1b/~0", 1}, + {"/z", "s"} + }; + + CHECK(value.flatten() == expected); + } + + SECTION("order of the entries of an ordered_json") + { + const auto value = nlohmann::ordered_json::parse( + R"({"z":"s","a/b":{"~":1,"k":[]},"a":[1,{"c":{},"b":[]},[{"x~/":[true,null],"w":2}]]})"); + + const auto flat = value.flatten(); + CHECK(flat.dump() == + R"({"/z":"s","/a~1b/~0":1,"/a~1b/k":null,"/a/0":1,"/a/1/c":null,"/a/1/b":null,"/a/2/0/x~0~1/0":true,"/a/2/0/x~0~1/1":null,"/a/2/0/w":2})"); + } +} From 2913e9643383b8078b9f6d5572401081047ea28e Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 9 Oct 2026 17:57:31 +0200 Subject: [PATCH 10/10] Unflatten in time and memory linear in the pointer depth (#5793) * Unflatten in time and memory linear in the pointer depth #5443 made unflatten() decide between arrays and objects independently of the iteration order by collecting the pointer prefixes that have a reference token 0 below them in a std::set>. Every such prefix was stored as a copy of all its reference tokens, and get_and_create() compared whole prefix vectors at every step, so unflattening a pointer of depth d took time and memory quadratic in d: a 10,000-level array pointer took 18 s and 1.3 GB, a 100,000-level one did not finish. The prefixes are now numbered nodes of a tree, so each is stored once and get_and_create() follows the tree token by token. The result is unchanged, including its independence of the iteration order. Signed-off-by: Niels Lohmann * Initialize prefix_tree members to satisfy -Weffc++ Signed-off-by: Niels Lohmann * Move prefix_tree setup and child insertion into member functions The constructor now creates the root node, add_child() inserts a reference token below a prefix and returns the child's number, and find_child() looks one up for get_and_create(). Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/detail/json_pointer.hpp | 83 ++++++++++++++++++------ single_include/nlohmann/json.hpp | 83 ++++++++++++++++++------ tests/src/unit-json_pointer.cpp | 57 ++++++++++++---- 3 files changed, 173 insertions(+), 50 deletions(-) diff --git a/include/nlohmann/detail/json_pointer.hpp b/include/nlohmann/detail/json_pointer.hpp index 0a1c0f43f..2ec4e3b58 100644 --- a/include/nlohmann/detail/json_pointer.hpp +++ b/include/nlohmann/detail/json_pointer.hpp @@ -16,8 +16,8 @@ #include // ostream #endif // JSON_NO_IO #include // max +#include // map #include // accumulate -#include // set #include // string #include // move #include // vector @@ -359,33 +359,82 @@ class json_pointer private: /*! - @brief the reference token sequences that denote arrays + @brief the pointer prefixes of a flattened object, and which of them denote arrays @ref unflatten collects the pointer prefixes that have a reference token 0 among their children; @ref get_and_create creates arrays exactly below those prefixes and objects everywhere else. Deciding this up front keeps the result independent of the order in which the flattened object is iterated, which is unspecified for some object types. + + The prefixes form a tree and are numbered, so each of them is stored only + once (as a node) rather than as a copy of all of its reference tokens. */ - using array_parents_t = std::set>; + struct prefix_tree + { + // children[id] maps a reference token to the number of the prefix + // extended by that token; number 0 is the empty prefix + std::vector> children; + // is_array[id] is true iff some flattened key has the reference token + // 0 directly below the prefix with number id + std::vector is_array; + + // start with the empty prefix only + prefix_tree() + : children(1) + , is_array(1, false) + {} + + // return the number of the prefix with number id extended by + // reference_token, adding it if it is new + std::size_t add_child(std::size_t id, string_t&& reference_token) + { + if (reference_token == "0") + { + is_array[id] = true; + } + + // read the number before the emplace_back below, which may + // reallocate children and invalidate the iterator + const std::size_t next = children.size(); + const auto inserted = children[id].emplace(std::move(reference_token), next); + const std::size_t child = inserted.first->second; + if (inserted.second) + { + children.emplace_back(); + is_array.push_back(false); + } + return child; + } + + // return the number of the prefix with number id extended by + // reference_token, which must have been added before + std::size_t find_child(std::size_t id, const string_t& reference_token) const + { + const auto it = children[id].find(reference_token); + JSON_ASSERT(it != children[id].end()); + return it->second; + } + }; /*! @brief create and return a reference to the pointed to value - Complexity: Linear in the number of reference tokens. + Complexity: Linear in the number of reference tokens (times the logarithm + of the number of siblings for the prefix lookup). @throw parse_error.106 if an array index begins with '0' @throw parse_error.109 if array index is not a number @throw type_error.313 if value cannot be unflattened */ template - BasicJsonType& get_and_create(BasicJsonType& j, const array_parents_t& array_parents) const + BasicJsonType& get_and_create(BasicJsonType& j, const prefix_tree& tree) const { auto* result = &j; - // the reference tokens that have been consumed so far; used to look up - // whether the value to be created below is an array or an object - std::vector prefix; + // the number of the prefix consumed so far; used to look up whether + // the value to be created below is an array or an object + std::size_t id = 0; // in case no reference tokens exist, return a reference to the JSON value // j which will be overwritten by a primitive value @@ -395,7 +444,7 @@ class json_pointer { case detail::value_t::null: { - if (array_parents.find(prefix) != array_parents.end()) + if (tree.is_array[id]) { // some reference token below this position is 0, so the // value is an array @@ -440,7 +489,7 @@ class json_pointer JSON_THROW(detail::type_error::create(313, "invalid value to unflatten", &j)); } - prefix.push_back(reference_token); + id = tree.find_child(id, reference_token); } return *result; @@ -1030,19 +1079,15 @@ class json_pointer // collect the pointer prefixes that have a reference token 0 among // their children; the values below them are arrays, all others are - // objects (see array_parents_t) - array_parents_t array_parents; + // objects (see prefix_tree) + prefix_tree tree; for (const auto& element : *value.m_data.m_value.object) { json_pointer ptr(element.first); - std::vector prefix; + std::size_t id = 0; for (auto& reference_token : ptr.reference_tokens) { - if (reference_token == "0") - { - array_parents.insert(prefix); - } - prefix.push_back(std::move(reference_token)); + id = tree.add_child(id, std::move(reference_token)); } } @@ -1058,7 +1103,7 @@ class json_pointer // that if the JSON pointer is "" (i.e., points to the whole value), // function get_and_create returns a reference to the result itself. // An assignment will then create a primitive value. - json_pointer(element.first).get_and_create(result, array_parents) = element.second; + json_pointer(element.first).get_and_create(result, tree) = element.second; } return result; diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index f0b1077f5..9de8ff2b2 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -19928,8 +19928,8 @@ NLOHMANN_JSON_NAMESPACE_END #include // ostream #endif // JSON_NO_IO #include // max +#include // map #include // accumulate -#include // set #include // string #include // move #include // vector @@ -20277,33 +20277,82 @@ class json_pointer private: /*! - @brief the reference token sequences that denote arrays + @brief the pointer prefixes of a flattened object, and which of them denote arrays @ref unflatten collects the pointer prefixes that have a reference token 0 among their children; @ref get_and_create creates arrays exactly below those prefixes and objects everywhere else. Deciding this up front keeps the result independent of the order in which the flattened object is iterated, which is unspecified for some object types. + + The prefixes form a tree and are numbered, so each of them is stored only + once (as a node) rather than as a copy of all of its reference tokens. */ - using array_parents_t = std::set>; + struct prefix_tree + { + // children[id] maps a reference token to the number of the prefix + // extended by that token; number 0 is the empty prefix + std::vector> children; + // is_array[id] is true iff some flattened key has the reference token + // 0 directly below the prefix with number id + std::vector is_array; + + // start with the empty prefix only + prefix_tree() + : children(1) + , is_array(1, false) + {} + + // return the number of the prefix with number id extended by + // reference_token, adding it if it is new + std::size_t add_child(std::size_t id, string_t&& reference_token) + { + if (reference_token == "0") + { + is_array[id] = true; + } + + // read the number before the emplace_back below, which may + // reallocate children and invalidate the iterator + const std::size_t next = children.size(); + const auto inserted = children[id].emplace(std::move(reference_token), next); + const std::size_t child = inserted.first->second; + if (inserted.second) + { + children.emplace_back(); + is_array.push_back(false); + } + return child; + } + + // return the number of the prefix with number id extended by + // reference_token, which must have been added before + std::size_t find_child(std::size_t id, const string_t& reference_token) const + { + const auto it = children[id].find(reference_token); + JSON_ASSERT(it != children[id].end()); + return it->second; + } + }; /*! @brief create and return a reference to the pointed to value - Complexity: Linear in the number of reference tokens. + Complexity: Linear in the number of reference tokens (times the logarithm + of the number of siblings for the prefix lookup). @throw parse_error.106 if an array index begins with '0' @throw parse_error.109 if array index is not a number @throw type_error.313 if value cannot be unflattened */ template - BasicJsonType& get_and_create(BasicJsonType& j, const array_parents_t& array_parents) const + BasicJsonType& get_and_create(BasicJsonType& j, const prefix_tree& tree) const { auto* result = &j; - // the reference tokens that have been consumed so far; used to look up - // whether the value to be created below is an array or an object - std::vector prefix; + // the number of the prefix consumed so far; used to look up whether + // the value to be created below is an array or an object + std::size_t id = 0; // in case no reference tokens exist, return a reference to the JSON value // j which will be overwritten by a primitive value @@ -20313,7 +20362,7 @@ class json_pointer { case detail::value_t::null: { - if (array_parents.find(prefix) != array_parents.end()) + if (tree.is_array[id]) { // some reference token below this position is 0, so the // value is an array @@ -20358,7 +20407,7 @@ class json_pointer JSON_THROW(detail::type_error::create(313, "invalid value to unflatten", &j)); } - prefix.push_back(reference_token); + id = tree.find_child(id, reference_token); } return *result; @@ -20948,19 +20997,15 @@ class json_pointer // collect the pointer prefixes that have a reference token 0 among // their children; the values below them are arrays, all others are - // objects (see array_parents_t) - array_parents_t array_parents; + // objects (see prefix_tree) + prefix_tree tree; for (const auto& element : *value.m_data.m_value.object) { json_pointer ptr(element.first); - std::vector prefix; + std::size_t id = 0; for (auto& reference_token : ptr.reference_tokens) { - if (reference_token == "0") - { - array_parents.insert(prefix); - } - prefix.push_back(std::move(reference_token)); + id = tree.add_child(id, std::move(reference_token)); } } @@ -20976,7 +21021,7 @@ class json_pointer // that if the JSON pointer is "" (i.e., points to the whole value), // function get_and_create returns a reference to the result itself. // An assignment will then create a primitive value. - json_pointer(element.first).get_and_create(result, array_parents) = element.second; + json_pointer(element.first).get_and_create(result, tree) = element.second; } return result; diff --git a/tests/src/unit-json_pointer.cpp b/tests/src/unit-json_pointer.cpp index 20f2331e3..1faf511f4 100644 --- a/tests/src/unit-json_pointer.cpp +++ b/tests/src/unit-json_pointer.cpp @@ -969,21 +969,54 @@ TEST_CASE("flatten of structured values") CHECK(flat.begin().key() == path); CHECK(flat.begin().value() == 0); - // unflatten() is not iterative: it takes time and memory - // quadratic in the depth, so it is only roundtripped for a - // moderate depth - std::string small_text; - for (std::size_t i = 0; i < 500; ++i) - { - small_text += objects ? "{\"a\":" : "["; - } - small_text += "0"; - small_text += std::string(500, objects ? '}' : ']'); - const auto small_value = json::parse(small_text); - CHECK(small_value.flatten().unflatten() == small_value); + // unflatten() is linear in the depth, so the value roundtrips + CHECK(flat.unflatten() == value); } } + SECTION("unflatten of a deeply nested pointer") + { + const std::size_t depth = 100000; + for (const bool objects : + { + false, true + }) + { + CAPTURE(objects) + std::string path; + for (std::size_t i = 0; i < depth; ++i) + { + path += objects ? "/a" : "/0"; + } + + json flat = json::object(); + flat[path] = 1; + const json value = flat.unflatten(); + + // walk down iteratively + std::size_t levels = 0; + const json* current = &value; + while (objects ? current->is_object() : current->is_array()) + { + REQUIRE(current->size() == 1); + current = objects ? ¤t->at("a") : ¤t->at(0); + ++levels; + } + CHECK(levels == depth); + CHECK(*current == 1); + } + } + + SECTION("unflatten does not depend on the iteration order") + { + // the "0" key comes after its sibling in iteration order + const nlohmann::ordered_json flat_array = nlohmann::ordered_json::parse(R"({"/a/1": 2, "/a/0": 1})"); + CHECK(flat_array.unflatten() == nlohmann::ordered_json::parse(R"({"a": [1, 2]})")); + + const nlohmann::ordered_json flat_object = nlohmann::ordered_json::parse(R"({"/b/1": 2})"); + CHECK(flat_object.unflatten() == nlohmann::ordered_json::parse(R"({"b": {"1": 2}})")); + } + SECTION("objects and arrays interleaved") { const json value =