mirror of
https://github.com/nlohmann/json.git
synced 2026-09-05 15:57:58 +00:00
Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3678634bf2 | ||
|
|
dd26d3ff07 |
@@ -100,7 +100,7 @@ jobs:
|
||||
container: ubuntu:focal
|
||||
strategy:
|
||||
matrix:
|
||||
target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_simdutf]
|
||||
target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls]
|
||||
steps:
|
||||
- name: Install build-essential
|
||||
run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev
|
||||
|
||||
@@ -212,24 +212,6 @@ add_custom_target(ci_test_legacycomparison
|
||||
COMMENT "Compile and test with legacy discarded value comparison enabled"
|
||||
)
|
||||
|
||||
###############################################################################
|
||||
# Validate UTF-8 with simdutf.
|
||||
###############################################################################
|
||||
|
||||
add_custom_target(ci_test_simdutf
|
||||
COMMAND ${CMAKE_COMMAND}
|
||||
-DCMAKE_BUILD_TYPE=Debug -GNinja
|
||||
-DJSON_BuildTests=ON -DJSON_TestSimdutf=ON
|
||||
# simdutf needs C++17, so the library falls back to its scalar validator
|
||||
# below that: build the suite at C++11 to cover the fallback with the macro
|
||||
# defined, and at C++17 to run every test against simdutf itself
|
||||
"-DJSON_TestStandards=11\;17"
|
||||
-S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_simdutf
|
||||
COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_simdutf
|
||||
COMMAND cd ${PROJECT_BINARY_DIR}/build_simdutf && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure
|
||||
COMMENT "Compile and test with simdutf UTF-8 validation enabled"
|
||||
)
|
||||
|
||||
###############################################################################
|
||||
# Enable brace-init copy semantics.
|
||||
###############################################################################
|
||||
|
||||
@@ -24,7 +24,6 @@ header. See also the [macro overview page](../../features/macros.md).
|
||||
- [**JSON_NO_IO**](json_no_io.md) - switch off functions relying on certain C++ I/O headers
|
||||
- [**JSON_SKIP_UNSUPPORTED_COMPILER_CHECK**](json_skip_unsupported_compiler_check.md) - do not warn about unsupported compilers
|
||||
- [**JSON_USE_GLOBAL_UDLS**](json_use_global_udls.md) - place user-defined string literals (UDLs) into the global namespace
|
||||
- [**JSON_USE_SIMDUTF**](json_use_simdutf.md) - use the simdutf library to accelerate UTF-8 validation
|
||||
|
||||
## Library version
|
||||
|
||||
|
||||
@@ -1,71 +0,0 @@
|
||||
# JSON_USE_SIMDUTF
|
||||
|
||||
```cpp
|
||||
#define JSON_USE_SIMDUTF
|
||||
```
|
||||
|
||||
When defined, the parser validates the UTF-8 content of JSON strings that come from a **contiguous byte input**
|
||||
(`std::string`, `std::vector<char>`/`<std::uint8_t>`, string literals, `const char*` ranges, …) using the
|
||||
[simdutf](https://github.com/simdutf/simdutf) library instead of the built-in scalar validator. On text with many
|
||||
non-ASCII characters (e.g. CJK or emoji) this can validate several times faster.
|
||||
|
||||
This is an **opt-in external dependency**. The library itself remains header-only and its behavior is unchanged: the
|
||||
same input is accepted or rejected either way, and every parse error is reported at the same position with the same
|
||||
message (simdutf is only used to fast-path *valid* runs; anything it flags falls back to the scalar path so the exact
|
||||
diagnostic is preserved). Streaming inputs (files, `std::istream`, wide strings, user-defined adapters) always use the
|
||||
scalar path.
|
||||
|
||||
When `JSON_USE_SIMDUTF` is defined you must make the `simdutf.h` header available on the include path and link the
|
||||
simdutf library. When it is not defined, no simdutf header is included and there is no dependency.
|
||||
|
||||
!!! note "Requires C++17"
|
||||
|
||||
simdutf requires C++17 and its header rejects older standards with an `#!cpp #error`. The backend is therefore only
|
||||
compiled in from C++17 on. In C++11 and C++14 the macro has no effect and the scalar validator is used, which
|
||||
accepts and rejects exactly the same input -- only throughput differs. Setting the macro project-wide is therefore
|
||||
safe even when some translation units are built with an older standard.
|
||||
|
||||
!!! warning "Define consistently"
|
||||
|
||||
The macro selects between two definitions of the same inline validation function. It must therefore be defined
|
||||
identically for **every** translation unit that includes the library; mixing translation units that define it with
|
||||
ones that do not is an ODR violation. Prefer setting it as a compile definition on the target rather than with
|
||||
`#!cpp #define` in individual source files.
|
||||
|
||||
## Default definition
|
||||
|
||||
By default, `#!cpp JSON_USE_SIMDUTF` is not defined and the portable C++11 scalar validator is used.
|
||||
|
||||
```cpp
|
||||
#undef JSON_USE_SIMDUTF
|
||||
```
|
||||
|
||||
## Examples
|
||||
|
||||
??? example
|
||||
|
||||
The code below enables the simdutf backend for UTF-8 validation.
|
||||
|
||||
```cpp
|
||||
#define JSON_USE_SIMDUTF 1
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
...
|
||||
```
|
||||
|
||||
The project must also link against simdutf, e.g. with CMake:
|
||||
|
||||
```cmake
|
||||
target_compile_definitions(your_target PRIVATE JSON_USE_SIMDUTF)
|
||||
target_link_libraries(your_target PRIVATE simdutf::simdutf)
|
||||
```
|
||||
|
||||
!!! hint "Testing this configuration"
|
||||
|
||||
The unit tests can be built against the simdutf backend with the CMake option `JSON_TestSimdutf` (`OFF` by
|
||||
default), which fetches simdutf and defines `JSON_USE_SIMDUTF` for every test target. The `ci_test_simdutf` target
|
||||
runs the whole test suite in that configuration.
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.13.0.
|
||||
@@ -137,14 +137,6 @@ behavior is deprecated and switched off (`0`) by default.
|
||||
|
||||
See [full documentation of `JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md).
|
||||
|
||||
## `JSON_USE_SIMDUTF`
|
||||
|
||||
When defined, UTF-8 validation of JSON strings read from contiguous byte input is delegated to the
|
||||
[simdutf](https://github.com/simdutf/simdutf) library instead of the built-in scalar validator. This is an opt-in
|
||||
external dependency and is not defined by default.
|
||||
|
||||
See [full documentation of `JSON_USE_SIMDUTF`](../api/macros/json_use_simdutf.md).
|
||||
|
||||
## `NLOHMANN_DEFINE_TYPE_*(...)`, `NLOHMANN_DEFINE_DERIVED_TYPE_*(...)`
|
||||
|
||||
The library defines 12 macros to simplify the serialization/deserialization of types. See the page on
|
||||
|
||||
@@ -8,41 +8,72 @@ the result of an internet search. If you know further customers of the library,
|
||||
## Space Exploration
|
||||
|
||||
- [**Peregrine Lunar Lander Flight 01**](https://en.wikipedia.org/wiki/Peregrine_Mission_One) - The library was used for payload management in the **Peregrine Moon Lander**, developed by **Astrobotic Technology** and launched as part of NASA's **Commercial Lunar Payload Services (CLPS)** program. After six days in orbit, the spacecraft was intentionally redirected into Earth's atmosphere, where it burned up over the Pacific Ocean on **January 18, 2024**.
|
||||
- [**NASA Unsteady Pressure-Sensitive Paint Processing**](https://github.com/nasa/upsp-processing), NASA software for processing high-speed video recordings of wind tunnel tests on launch vehicle and aircraft models
|
||||
- [**Terma TEMU**](https://temu.terma.com/docs/public/temu-release-notes/latest/copying/json-for-modern-cpp.html), an emulator of spacecraft on-board computers used to develop and validate flight software for European space missions
|
||||
|
||||
## Automotive
|
||||
|
||||
- [**Alexa Auto SDK**](https://github.com/alexa/alexa-auto-sdk), a software development kit enabling the integration of Alexa into automotive systems
|
||||
- [**Apollo**](https://github.com/ApolloAuto/apollo), a framework for building autonomous driving systems
|
||||
- [**Automotive Grade Linux (AGL)**](https://download.automotivelinux.org/AGL/release/jellyfish/latest/qemux86-64/deploy/licenses/nlohmann-json/), a collaborative open-source platform for automotive software development
|
||||
- [**Autoware**](https://github.com/autowarefoundation/autoware_universe), an open-source software stack for autonomous driving built on ROS 2
|
||||
- [**Eclipse S-CORE**](https://github.com/eclipse-score/nlohmann_json), an open-source software platform for the software-defined vehicle backed by major automotive manufacturers and suppliers
|
||||
- [**Genesis Motor** (infotainment)](http://webmanual.genesis.com/ccIC/AVNT/JW/KOR/English/reference010.html), a luxury automotive brand
|
||||
- [**Hyundai** (infotainment)](https://www.hyundai.com/wsvc/ww/download.file.do?id=/content/hyundai/ww/data/opensource/data/GN7-2022/licenseCode/info), a global automotive brand
|
||||
- [**Kia** (infotainment)](http://webmanual.kia.com/PREM_GEN6/AVNT/RJPE/KOR/Korean/reference010.html), a global automotive brand
|
||||
- [**Mercedes-Benz Operating System (MB.OS)**](https://group.mercedes-benz.com/careers/about-us/mercedes-benz-operating-system/), a core component of the vehicle software ecosystem from Mercedes-Benz
|
||||
- [**NVIDIA DRIVE OS**](https://developer.nvidia.com/docs/drive/drive-os/6.0.5/public/driveworks-nvcgf/dwx_open_source_attribution.html), the operating system and DriveWorks SDK powering NVIDIA's platform for autonomous vehicles
|
||||
- [**Rivian** (infotainment)](https://assets.ctfassets.net/2md5qhoeajym/3cwyo4eoufk4yingUwusFt/ded2c47da620fdfc99c88c7156d2c1d8/In-Vehicle_OSS_Attribution_2024__11-24_.pdf), an electric vehicle manufacturer
|
||||
- [**Suzuki** (infotainment)](https://www.globalsuzuki.com/motorcycle/ipc/oss/oss_48KA_00.pdf), a global automotive and motorcycle manufacturer
|
||||
|
||||
## Gaming and Entertainment
|
||||
|
||||
- [**Anno 117: Pax Romana**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a city-building strategy game set in the Roman Empire
|
||||
- [**Assassin's Creed: Mirage**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a stealth-action game set in the Middle East, focusing on the journey of a young assassin with classic parkour and stealth mechanics
|
||||
- [**Battlefield 6**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a military first-person shooter known for its large-scale multiplayer battles
|
||||
- [**Battlefield: REDSEC**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a free-to-play battle royale experience set in the Battlefield universe
|
||||
- [**BioMenace: Remastered**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a remaster of the classic side-scrolling platform shooter
|
||||
- [**Chasm: The Rift**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a first-person shooter blending horror and adventure, where players navigate dark realms and battle monsters
|
||||
- [**College Football 25**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a college football simulation game featuring gameplay that mimics real-life college teams and competitions
|
||||
- [**College Football 26**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a college football simulation game featuring licensed teams and stadiums
|
||||
- [**College Football 27**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), the latest installment of the college football simulation series
|
||||
- [**Concepts**](https://concepts.app/en/licenses), a digital sketching app designed for creative professionals, offering flexible drawing tools for illustration, design, and brainstorming
|
||||
- [**Depthkit**](https://www.depthkit.tv/third-party-licenses), a tool for creating and capturing volumetric video, enabling immersive 3D experiences and interactive content
|
||||
- [**Dune: Awakening**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), an open-world survival MMO set on the desert planet Arrakis
|
||||
- [**EA Sports FC 25**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), an association football simulation with club, career, and online modes
|
||||
- [**EA Sports FC 26**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), the latest installment of the association football simulation series
|
||||
- [**EA Sports UFC 6**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a mixed martial arts fighting simulation
|
||||
- [**FiveM**](https://github.com/citizenfx/fivem), a modification framework for Grand Theft Auto V that powers custom multiplayer servers
|
||||
- [**FLUX:: Immersive**](https://doc.flux.audio/syrah/Credits.html), a suite of professional audio processing and immersive mixing plugins used in music and post-production
|
||||
- [**IMG.LY**](https://img.ly/acknowledgements), a platform offering creative tools and SDKs for integrating advanced image and video editing in applications
|
||||
- [**immersivetech**](https://immersitech.io/open-source-third-party-software/), a technology company focused on immersive experiences, providing tools and solutions for virtual and augmented reality applications
|
||||
- [**Kodi**](https://github.com/xbmc/xbmc/blob/master/xbmc/utils/JSONVariantWriter.cpp), a home theater and media center application
|
||||
- [**LOOT**](https://loot.readthedocs.io/_/downloads/en/0.13.0/pdf/), a tool for optimizing the load order of game plugins, commonly used in The Elder Scrolls and Fallout series
|
||||
- [**LunaTranslator**](https://github.com/HIllya51/LunaTranslator/blob/main/src/NativeImpl/LunaSubprocess/aspatch.cpp), a real-time translation tool for visual novels
|
||||
- [**MaaAssistantArknights**](https://github.com/MaaAssistantArknights/MaaAssistantArknights/blob/dev-v2/src/MaaCore/Vision/Roguelike/BlackFlow/BlackFlowMapAnalyzer.cpp), an automation assistant for the mobile game Arknights
|
||||
- [**Madden NFL 25**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a sports simulation game capturing the excitement of American football with realistic gameplay and team management features
|
||||
- [**Madden NFL 26**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), an American football simulation with franchise and team management modes
|
||||
- [**Madden NFL 27**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), the latest installment of the American football simulation series
|
||||
- [**Marne**](https://marne.io/licenses), an unofficial private server platform for hosting custom Battlefield 1 game experiences
|
||||
- [**Minecraft**](https://www.minecraft.net/zh-hant/attribution), a popular sandbox video game
|
||||
- [**Mumble**](https://github.com/mumble-voip/mumble), a low-latency, open-source voice chat application widely used by gaming communities
|
||||
- [**NHL 22**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a hockey simulation game offering realistic gameplay, team management, and various modes to enhance the hockey experience
|
||||
- [**OBS Studio**](https://github.com/obsproject/obs-studio), a free and open-source suite for video recording and live streaming
|
||||
- [**OpenRCT2**](https://github.com/OpenRCT2/OpenRCT2/blob/develop/src/openrct2/core/JsonFwd.hpp), an open source re-implementation of RollerCoaster Tycoon 2
|
||||
- [**Pixelpart**](https://pixelpart.net/documentation/book/third-party.html), a 2D animation and video compositing software that allows users to create animated graphics and visual effects with a focus on simplicity and ease of use
|
||||
- [**Razer Cortex**](https://mysupport.razer.com/app/answers/detail/a_id/14146/~/open-source-software-for-razer-software), a gaming performance optimizer and system booster designed to enhance the gaming experience
|
||||
- [**Red Dead Redemption II**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), an open-world action-adventure game following an outlaw's story in the late 1800s, emphasizing deep storytelling and immersive gameplay
|
||||
- [**RetroArch**](https://github.com/libretro/RetroArch), a frontend for emulators, game engines, and media players built on the libretro API
|
||||
- [**shadPS4**](https://github.com/shadps4-emu/shadPS4/blob/main/src/core/user_manager.h), a PlayStation 4 emulator for Windows, Linux and macOS
|
||||
- [**skate.**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a free-to-play skateboarding game set in an open world
|
||||
- [**Snapchat**](https://www.snap.com/terms/license-android), a multimedia messaging and augmented reality app for communication and entertainment
|
||||
- [**Steel Century Groove**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), an action game released in 2026
|
||||
- [**Sunshine**](https://github.com/LizardByte/Sunshine/blob/master/src/confighttp.h), a self-hosted game streaming host compatible with Moonlight clients
|
||||
- [**Tactics Ogre: Reborn**](https://www.square-enix-games.com/en_US/documents/tactics-ogre-reborn-pc-installer-software-and-associated-plug-ins-disclosure), a tactical role-playing game featuring strategic battles and deep storytelling elements
|
||||
- [**Throne and Liberty**](https://www.amazon.com/gp/help/customer/display.html?nodeId=T7fLNw5oAevCMtJFPj&pop-up=1), an MMORPG that offers an expansive fantasy world with dynamic gameplay and immersive storytelling
|
||||
- [**Unity Vivox**](https://docs.unity3d.com/Packages/com.unity.services.vivox@15.1/license/Third%20Party%20Notices.html), a communication service that enables voice and text chat functionality in multiplayer games developed with Unity
|
||||
- [**xemu**](https://github.com/xemu-project/xemu), an emulator of the original Xbox console
|
||||
- [**Zool: Redimensioned**](https://www.mobygames.com/person/1195889/niels-lohmann/credits/), a modern reimagining of the classic platformer featuring fast-paced gameplay and vibrant environments
|
||||
- [**immersivetech**](https://immersitech.io/open-source-third-party-software/), a technology company focused on immersive experiences, providing tools and solutions for virtual and augmented reality applications
|
||||
|
||||
## Consumer Electronics
|
||||
|
||||
@@ -50,109 +81,195 @@ the result of an internet search. If you know further customers of the library,
|
||||
- [**Canon CanoScan LIDE**](https://carolburo.com/wp-content/uploads/2024/06/LiDE400_OnlineManual_Win_FR_V02.pdf), a series of flatbed scanners offering high-resolution image scanning for home and office use
|
||||
- [**Canon PIXMA Printers**](https://www.mediaexpert.pl/products/files/73/7338196/Instrukcja-obslugi-CANON-Pixma-TS7450i.pdf), a line of all-in-one inkjet printers known for high-quality printing and wireless connectivity
|
||||
- [**Cisco Webex Desk Camera**](https://www.cisco.com/c/dam/en_us/about/doing_business/open_source/docs/CiscoWebexDeskCamera-23-1622100417.pdf), a video camera designed for professional-quality video conferencing and remote collaboration
|
||||
- [**DJI Edge SDK**](https://github.com/dji-sdk/Edge-SDK-V2-Demo), the reference applications for DJI's Edge SDK, used to build edge computing services on DJI drone docks
|
||||
- [**Elgato Stream Deck**](https://github.com/elgatosf/streamdeck-obs-plugin2), a family of programmable control surfaces for content creators and their plugin ecosystem
|
||||
- [**Instagrid**](https://instagrid.co/intellectual-property/foss), a manufacturer of portable, high-performance battery systems for professional mobile power supply
|
||||
- [**iRobot**](https://iot-content.irobot.com/iw/sfsites/c/cms/delivery/media/MCKRLTPDJSSJBNJKDA5SG5UVVIIQ), a manufacturer of autonomous home robots including the Roomba vacuum cleaner range
|
||||
- [**Logitech Logi Bolt**](https://opensource.logitech.com/wiki/Logi_BoltApp/), the management application for Logitech's secure wireless connectivity technology
|
||||
- [**Novitus**](https://novitus.pl/licencjepensource), a manufacturer of fiscal cash registers and point-of-sale devices
|
||||
- [**Philips Hue Personal Wireless Lighting**](http://2ak5ape.257.cz/), a smart lighting system for customizable and wireless home illumination
|
||||
- [**Ray-Ban Meta Smart glasses**](https://www.meta.com/de/en/legal/smart-glasses/third-party-notices-android/03/), a pair of smart glasses designed for capturing photos and videos with integrated connectivity and social features
|
||||
- [**Razer Synapse**](https://mysupport.razer.com/app/answers/detail/a_id/14146/~/open-source-software-for-razer-software), a unified configuration software enabling hardware customization for Razer devices
|
||||
- [**Sharp Professional Displays**](https://jp.sharp/restricted/business/lcd-display/cms/images/source_pnla862/PN-LA652_752_862_LicenseInformation.pdf), a range of large-format interactive displays for business and education
|
||||
- [**Siemens SINEMA Remote Connect**](https://cache.industry.siemens.com/dl/files/790/109793790/att_1054961/v2/OSS_SINEMA-RC_86.pdf), a remote connectivity solution for monitoring and managing industrial networks and devices securely
|
||||
- [**Skydio**](https://pages.skydio.com/rs/784-TUF-591/images/Open%20Source%20Software%20Notice%20v0.2.html), a manufacturer of autonomous drones for inspection, public safety, and defense applications
|
||||
- [**Sony PlayStation 4**](https://doc.dl.playstation.net/doc/ps4-oss/index.html), a gaming console developed by Sony that offers a wide range of games and multimedia entertainment features
|
||||
- [**Sony Spatial Reality Display**](https://www.sony.co.jp/en/Products/Developer-Spatial-Reality-display/download/dcc-tools/blender-plugin/SpatiaRealityDisplayPluginforPreviewBL_Manual.pdf), a glasses-free stereoscopic 3D display and its plugins for Blender, 3ds Max, and ZBrush
|
||||
- [**Sony Virtual Webcam Driver for Remote Camera**](https://helpguide.sony.net/rc/vwd/v1/zh-cn/print.pdf), a software driver that enables the use of Sony cameras as virtual webcams for video conferencing and streaming
|
||||
- [**Yamaha Clavinova**](https://usa.yamaha.com/files/download/other_assets/1/2298171/CLP-800_oss_license.pdf), a series of digital pianos combining acoustic piano feel with digital sound technology
|
||||
|
||||
## Operating Systems
|
||||
## Operating Systems and Platforms
|
||||
|
||||
- [**Apple iOS and macOS**](https://www.apple.com/macos), a family of operating systems developed by Apple, including iOS for mobile devices and macOS for desktop computers
|
||||
- [**Chromium**](https://chromium.googlesource.com/chromium/src/+/main/third_party/nlohmann_json/), the open-source browser project that Google Chrome, Microsoft Edge, and many other browsers are built on, where the library is used as data container for on-device model execution
|
||||
- [**Google Fuchsia**](https://fuchsia.googlesource.com/third_party/json/), an open-source operating system developed by Google, designed to be secure, updatable, and adaptable across various devices
|
||||
- [**LG webOS**](https://github.com/webosose/com.webos.service.camera), a Linux-based operating system used in LG smart TVs, signage, and embedded devices
|
||||
- [**Microsoft Azure Linux**](https://github.com/microsoft/azurelinux), a Linux distribution developed by Microsoft for Azure infrastructure and edge workloads
|
||||
- [**OpenHarmony**](https://github.com/openharmony/third_party_json), an open-source operating system for smart devices and the foundation of HarmonyOS
|
||||
- [**SerenityOS**](https://github.com/SerenityOS/serenity), an open-source operating system that aims to provide a simple and beautiful user experience with a focus on simplicity and elegance
|
||||
- [**Windows Subsystem for Linux**](https://github.com/microsoft/WSL), a compatibility layer that runs Linux environments natively on Windows
|
||||
- [**Yocto**](http://ftp.emacinc.com/openembedded-sw/kirkstone-icop-5.15-kirkstone-6.0/archive-2024-10/pn8m-090t-ppc/licenses/nlohmann-json/), a Linux-based build system for creating custom operating systems and software distributions, tailored for embedded devices and IoT applications
|
||||
|
||||
## Development Tools and IDEs
|
||||
|
||||
- [**Accentize SpectralBalance**](https://www.accentize.com/products/SpectralBalanceManual.pdf), an adaptive speech analysis tool designed to enhance audio quality by optimizing frequency balance in recordings
|
||||
- [**Airbus Ghidralligator**](https://www.cyber.airbus.com/en/newsroom/stories/2025-06-ghidralligator), a Ghidra-based emulator from Airbus CyberSecurity used to fuzz and analyse embedded firmware
|
||||
- [**Apache brpc**](https://github.com/apache/brpc/blob/master/src/butil/iobuf.h), an industrial-grade remote procedure call framework for C++
|
||||
- [**Arm Compiler for Linux**](https://documentation-service.arm.com/static/66558e9d876c8d213b7843e4), a software development toolchain for compiling and optimizing applications on Arm-based Linux systems
|
||||
- [**BBEdit**](https://s3.amazonaws.com/BBSW-download/BBEdit_15.1.2_User_Manual.pdf), a professional text and code editor for macOS
|
||||
- [**CoderPad**](https://coderpad.io), a collaborative coding platform that enables real-time code interviews and assessments for developers; the library is included in every CoderPad instance and can be accessed with a simple `#include "json.hpp"`
|
||||
- [**Codon**](https://github.com/exaloop/codon/blob/develop/jupyter/jupyter.h), an ahead-of-time compiler for a Python-like language
|
||||
- [**Compiler Explorer**](https://godbolt.org), a web-based tool that allows users to write, compile, and visualize the assembly output of code in various programming languages; the library is readily available and accessible with the directive `#include <nlohmann/json.hpp>`.
|
||||
- [**GitHub CodeQL**](https://github.com/github/codeql), a code analysis tool used for identifying security vulnerabilities and bugs in software through semantic queries
|
||||
- [**Flutter**](https://github.com/flutter/flutter/blob/master/engine/src/flutter/impeller/compiler/reflector.cc), a UI toolkit for building natively compiled applications for mobile, web, and desktop from a single codebase
|
||||
- [**Fraunhofer VVenC**](https://github.com/fraunhoferhhi/vvenc), a fast and efficient encoder for the Versatile Video Coding (H.266/VVC) standard
|
||||
- [**GitHub CodeQL**](https://github.com/github/codeql/blob/main/shared/cpp/Diagnostics.h), a code analysis tool used for identifying security vulnerabilities and bugs in software through semantic queries
|
||||
- [**GoPro ngfx**](https://github.com/gopro/ngfx), a low-level graphics abstraction and profiling framework developed by GoPro
|
||||
- [**gRPC**](https://github.com/grpc/grpc/blob/master/tools/artifact_gen/utils.h), a high-performance universal remote procedure call framework
|
||||
- [**Hex-Rays**](https://docs.hex-rays.com/user-guide/user-interface/licenses), a reverse engineering toolset for analyzing and decompiling binaries, primarily used for security research and vulnerability analysis
|
||||
- [**ImHex**](https://github.com/WerWolv/ImHex), a hex editor designed for reverse engineering, providing advanced features for data analysis and manipulation
|
||||
- [**Intel GITS**](https://github.com/intel/gits), a tool for capturing and replaying graphics API calls for debugging and performance analysis
|
||||
- [**Intel GPA Framework**](https://intel.github.io/gpasdk-doc/src/licenses.html), a suite of cross-platform tools for capturing, analyzing, and optimizing graphics applications across different APIs
|
||||
- [**Intopix**](https://www.intopix.com/software-licensing), a provider of advanced image processing and compression solutions used in software development and AV workflows
|
||||
- [**Java SE**](https://www.oracle.com/a/tech/docs/jdk8-lium.pdf), the core Java platform that provides the libraries and runtime needed to build and run general-purpose Java applications
|
||||
- [**MKVToolNix**](https://mkvtoolnix.download/doc/README.md), a set of tools for creating, editing, and inspecting MKV (Matroska) multimedia container files
|
||||
- [**Meta Yoga**](https://github.com/facebook/yoga), a layout engine that facilitates flexible and efficient user interface design across multiple platforms
|
||||
- [**NVIDIA Nsight Compute**](https://docs.nvidia.com/nsight-compute/2022.2/pdf/CopyrightAndLicenses.pdf), a performance analysis tool for CUDA applications that provides detailed insights into GPU performance metrics
|
||||
- [**MKVToolNix**](https://mkvtoolnix.download/doc/README.md), a set of tools for creating, editing, and inspecting MKV (Matroska) multimedia container files
|
||||
- [**MRTech IFF SDK**](https://mr-technologies.com/pub/iff-sdk-manual-2-0-1/iff-sdk-manual-2-0-1.pdf), an image processing SDK for machine vision applications with GPU-accelerated pipelines
|
||||
- [**Nix**](https://github.com/NixOS/nix/blob/master/src/nix/build.cc), a purely functional package manager
|
||||
- [**Notepad++**](https://github.com/notepad-plus-plus/notepad-plus-plus), a free source code editor that supports various programming languages
|
||||
- [**NVIDIA Nsight Compute**](https://docs.nvidia.com/nsight-compute/2022.2/pdf/CopyrightAndLicenses.pdf), a performance analysis tool for CUDA applications that provides detailed insights into GPU performance metrics
|
||||
- [**openFrameworks**](https://github.com/openframeworks/openFrameworks/blob/master/libs/openFrameworks/utils/ofJson.h), a community-developed C++ toolkit for creative coding
|
||||
- [**OpenRGB**](https://gitlab.com/CalcProgrammer1/OpenRGB), an open source RGB lighting control that doesn't depend on manufacturer software
|
||||
- [**OpenTelemetry C++**](https://github.com/open-telemetry/opentelemetry-cpp), a library for collecting and exporting observability data in C++, enabling developers to implement distributed tracing and metrics in their application
|
||||
- [**Oracle GraalVM**](https://docs.oracle.com/en/graalvm/jdk/21/docs/licensing-information/), a high-performance JDK distribution with ahead-of-time compilation and polyglot runtime support
|
||||
- [**Philips amp-cucumber-cpp-runner**](https://github.com/philips-software/amp-cucumber-cpp-runner), a behaviour-driven development test runner for embedded C++ software developed at Philips
|
||||
- [**Qt Creator**](https://doc.qt.io/qtcreator/qtcreator-attribution-json-nlohmann.html), an IDE for developing applications using the Qt application framework
|
||||
- [**Qt for MCUs**](https://doc.qt.io/QtForMCUs/quickultralite-attribution-nlohmann-json.html), a graphics framework for building fluid user interfaces on microcontrollers
|
||||
- [**React Native**](https://github.com/react/react-native/blob/main/packages/react-native/ReactCxxPlatform/react/devsupport/PackagerConnection.cpp), a framework for building native mobile applications using React
|
||||
- [**Scanbot SDK**](https://docs.scanbot.io/barcode-scanner-sdk/web/third-party-libraries/), a software development kit (SDK) that provides tools for integrating advanced document scanning and barcode scanning capabilities into applications
|
||||
- [**STMicroelectronics TouchGFX**](https://www.st.com/resource/en/additional_license_terms/additional-license-terms-x-cube-touchgfx.html), a graphical user interface framework shipped with STM32 microcontrollers for building embedded HMIs
|
||||
- [**swagger-codegen**](https://github.com/swagger-api/swagger-codegen/blob/master/samples/server/petstore/pistache-server/model/Pet.h), a template-driven engine that generates API clients and server stubs from an OpenAPI specification
|
||||
- [**Swoole**](https://github.com/swoole/swoole-src/blob/master/ext-src/swoole_admin_server.cc), a coroutine-based concurrency engine for PHP
|
||||
- [**Tracy Profiler**](https://github.com/wolfpld/tracy/blob/master/profiler/src/profiler/TracyLlm.hpp), a real-time frame profiler for games and other applications
|
||||
- [**WasmEdge**](https://github.com/WasmEdge/WasmEdge/blob/master/plugins/wasi_nn/GGML/tts/tts_core.cpp), a lightweight WebAssembly runtime for edge and cloud workloads
|
||||
- [**x64dbg**](https://github.com/x64dbg/x64dbg/blob/development/src/cross/remote_table/TableRpcData.h), an open source user mode debugger for Windows, aimed at reverse engineering and malware analysis
|
||||
|
||||
## Machine Learning and AI
|
||||
|
||||
- [**Alibaba MNN**](https://github.com/alibaba/MNN), a lightweight deep learning inference engine for mobile and embedded devices
|
||||
- [**AMD Gaia**](https://github.com/amd/gaia), an open-source framework for running generative AI applications locally on AMD hardware
|
||||
- [**AMD Vitis AI (VAIP)**](https://github.com/amd/vaip), the execution provider stack that runs AI models on AMD Ryzen AI and adaptive computing devices
|
||||
- [**Apple Core ML Tools**](https://github.com/apple/coremltools), a set of tools for converting and configuring machine learning models for deployment in Apple's Core ML framework
|
||||
- [**Avular Mobile Robotics**](https://www.avular.com/licenses/nlohmann-json-3.9.1.txt), a platform for developing and deploying mobile robotics solutions
|
||||
- [**FunASR**](https://github.com/modelscope/FunASR/blob/main/runtime/http/bin/asr_sessions.h), a speech recognition toolkit for training and deploying end-to-end models
|
||||
- [**Google gemma.cpp**](https://github.com/google/gemma.cpp), a lightweight C++ inference engine designed for running AI models from the Gemma family
|
||||
- [**Google Magenta The Infinite Crate**](https://github.com/magenta/the-infinite-crate), an open-source generative AI plugin for digital audio workstations from Google's Magenta research team
|
||||
- [**GPT4All**](https://github.com/nomic-ai/gpt4all/blob/main/gpt4all-chat/src/tool.h), a desktop application for running local large language models on consumer hardware
|
||||
- [**Huawei MindSpore**](https://github.com/mindspore-ai/mindspore/blob/master/Third_Party_Open_Source_Software_Notice), a deep learning framework for training and inference across device, edge, and cloud
|
||||
- [**KTransformers**](https://github.com/kvcache-ai/ktransformers/blob/main/archive/csrc/balance_serve/sched/model_config.h), a framework for heterogeneous large language model inference
|
||||
- [**llama.cpp**](https://github.com/ggerganov/llama.cpp), a C++ library designed for efficient inference of large language models (LLMs), enabling streamlined integration into applications
|
||||
- [**LocalAI**](https://github.com/mudler/LocalAI/blob/master/backend/cpp/ds4/dsml_renderer.cpp), a self-hosted inference engine that exposes local models through an OpenAI-compatible API
|
||||
- [**MLX**](https://github.com/ml-explore/mlx), an array framework for machine learning on Apple Silicon
|
||||
- [**Mozilla llamafile**](https://github.com/Mozilla-Ocho/llamafile), a tool designed for distributing and executing large language models (LLMs) efficiently using a single file format
|
||||
- [**NVIDIA ACE**](https://docs.nvidia.com/ace/latest/index.html), a suite of real-time AI solutions designed for the development of interactive avatars and digital human applications, enabling scalable and sophisticated user interactions
|
||||
- [**NVIDIA Instant NGP**](https://github.com/NVlabs/instant-ngp/blob/master/src/nerf_loader.cu), an implementation of instant neural graphics primitives for rapid scene reconstruction
|
||||
- [**NVIDIA TensorRT**](https://github.com/NVIDIA/TensorRT), an SDK for high-performance deep learning inference, including its TensorRT-LLM extension for large language models
|
||||
- [**NVIDIA TensorRT-LLM**](https://github.com/NVIDIA/TensorRT-LLM/blob/main/cpp/tensorrt_llm/common/safetensors.cpp), a toolkit for optimizing and serving large language model inference on GPUs
|
||||
- [**ONNX Runtime**](https://github.com/microsoft/onnxruntime), a cross-platform inference and training accelerator for machine learning models
|
||||
- [**OpenVINO**](https://github.com/openvinotoolkit/openvino), Intel's toolkit for optimizing and deploying deep learning inference across CPUs, GPUs, and NPUs
|
||||
- [**PaddleOCR**](https://github.com/PaddlePaddle/PaddleOCR/blob/main/deploy/cpp_infer/src/modules/text_detection/result.cc), an optical character recognition toolkit that turns documents and images into structured data
|
||||
- [**PaddlePaddle**](https://github.com/PaddlePaddle/Paddle/blob/develop/paddle/ap/src/axpr/anf_expr.cc), a deep learning framework for distributed training and inference
|
||||
- [**Peer**](https://support.peer.inc/hc/en-us/articles/17261335054235-Licenses), a platform offering personalized AI assistants for interactive learning and creative collaboration
|
||||
- [**PyTorch**](https://github.com/pytorch/pytorch), a machine learning framework for building and training neural networks, widely used in research and production
|
||||
- [**Qualcomm AI Engine Direct**](https://github.com/qualcomm/qai-appbuilder), a toolchain for building and running generative AI applications on Snapdragon devices
|
||||
- [**sherpa-onnx**](https://github.com/k2-fsa/sherpa-onnx/blob/master/sherpa-onnx/csrc/sentence-piece-tokenizer.cc), a speech toolkit for on-device recognition, synthesis and speaker diarization
|
||||
- [**stable-diffusion.cpp**](https://github.com/leejet/stable-diffusion.cpp), a C++ implementation of the Stable Diffusion image generation model
|
||||
- [**TanvasTouch**](https://tanvas.co/tanvastouch-sdk-third-party-acknowledgments), a software development kit (SDK) that enables developers to create tactile experiences on touchscreens, allowing users to feel textures and physical sensations in a digital environment
|
||||
- [**TensorFlow**](https://github.com/tensorflow/tensorflow), a machine learning framework that facilitates the development and training of models, supporting data serialization and efficient data exchange between components
|
||||
- [**whisper.cpp**](https://github.com/ggml-org/whisper.cpp), a C++ implementation of OpenAI's Whisper automatic speech recognition model
|
||||
|
||||
## Scientific Research and Analysis
|
||||
|
||||
- [**BLACK**](https://www.black-sat.org/en/stable/installation/linux.html), a bounded linear temporal logic (LTL) satisfiability checker
|
||||
- [**CERN ALICE O2**](https://github.com/AliceO2Group/AliceO2), the online-offline computing framework of the ALICE heavy-ion experiment at the Large Hadron Collider
|
||||
- [**CERN Atlas Athena**](https://gitlab.cern.ch/atlas/athena/-/blob/main/Control/PerformanceMonitoring/PerfMonComps/src/PerfMonMTSvc.h), a software framework used in the ATLAS experiment at the Large Hadron Collider (LHC) for performance monitoring
|
||||
- [**CERN CMSSW**](https://github.com/cms-sw/cmssw), the offline software framework of the CMS experiment at the Large Hadron Collider
|
||||
- [**CERN Gaudi**](https://gitlab.cern.ch/gaudi/Gaudi), the event-processing framework used by the LHCb and ATLAS experiments at the Large Hadron Collider
|
||||
- [**ICU**](https://github.com/unicode-org/icu), the International Components for Unicode, a mature library for software globalization and multilingual support
|
||||
- [**KAMERA**](https://github.com/Kitware/kamera), a platform for synchronized data collection and real-time deep learning to map marine species like polar bears and seals, aiding Arctic ecosystem research
|
||||
- [**KiCad**](https://gitlab.com/kicad/code/kicad/-/tree/master/thirdparty/nlohmann_json), a free and open-source software suite for electronic design automation
|
||||
- [**LLNL ROSE**](https://github.com/llnl/rose), a compiler infrastructure from Lawrence Livermore National Laboratory for building source-to-source program analysis and transformation tools
|
||||
- [**Maple**](https://www.maplesoft.com/support/help/Maple/view.aspx?path=copyright), a symbolic and numeric computing environment for advanced mathematical modeling and analysis
|
||||
- [**MeVisLab**](https://mevislabdownloads.mevis.de/docs/current/MeVis/ThirdParty/Documentation/Publish/ThirdPartyReference/index.html), a software framework for medical image processing and visualization.
|
||||
- [**MITK**](https://github.com/MITK/MITK), the Medical Imaging Interaction Toolkit, a framework for developing interactive medical image processing software
|
||||
- [**OpenPMD API**](https://openpmd-api.readthedocs.io/en/0.8.0-alpha/backends/json.html), a versatile programming interface for accessing and managing scientific data, designed to facilitate the efficient storage, retrieval, and sharing of simulation data across various applications and platforms
|
||||
- [**ORNL DataFed**](https://github.com/ORNL/DataFed), a federated scientific data management system developed at Oak Ridge National Laboratory
|
||||
- [**ParaView**](https://github.com/Kitware/ParaView), an open-source tool for large-scale data visualization and analysis across various scientific domains
|
||||
- [**QGIS**](https://gitlab.b-data.ch/qgis/qgis/-/blob/backport-57658-to-release-3_34/external/nlohmann/json.hpp), a free and open-source geographic information system (GIS) application that allows users to create, edit, visualize, and analyze geospatial data across a variety of formats
|
||||
- [**VTK**](https://github.com/Kitware/VTK), a software library for 3D computer graphics, image processing, and visualization
|
||||
- [**Sandia InterSpec**](https://github.com/sandialabs/InterSpec), spectral radiation analysis software from Sandia National Laboratories for identifying radioactive isotopes
|
||||
- [**VolView**](https://github.com/Kitware/VolView), a lightweight application for interactive visualization and analysis of 3D medical imaging data.
|
||||
- [**VTK**](https://github.com/Kitware/VTK), a software library for 3D computer graphics, image processing, and visualization
|
||||
|
||||
## Business and Productivity Software
|
||||
|
||||
- [**ArcGIS PRO**](https://www.esri.com/content/dam/esrisites/en-us/media/legal/open-source-acknowledgements/arcgis-pro-2-8-attribution-report.html), a desktop geographic information system (GIS) application developed by Esri for mapping and spatial analysis
|
||||
- [**Autodesk Desktop**](https://damassets.autodesk.net/content/dam/autodesk/www/Company/legal-notices-trademarks/autodesk-desktop-platform-components/internal-autodesk-components-web-page-2023.pdf), a software platform developed by Autodesk for creating and managing desktop applications and services
|
||||
- [**Check Point**](https://www.checkpoint.com/about-us/copyright-and-trademarks/), a cybersecurity company specializing in threat prevention and network security solutions, offering a range of products designed to protect enterprises from cyber threats and ensure data integrity
|
||||
- [**EasyEffects**](https://github.com/wwmm/easyeffects/blob/master/src/presets_manager.hpp), an audio effects processor for PipeWire offering limiting, compression and equalization
|
||||
- [**espanso**](https://github.com/espanso/espanso/blob/dev/espanso-ui/src/win32/native.cpp), a cross-platform text expander
|
||||
- [**Karabiner-Elements**](https://github.com/pqrs-org/Karabiner-Elements/blob/main/src/share/app_icon.hpp), a keyboard customizer for macOS
|
||||
- [**MacType**](https://github.com/snowie2000/mactype/blob/directwrite/settings.h), a font rendering engine for Windows
|
||||
- [**magicplan**](https://help.magicplan.app/acknowledgments), a mobile application for creating floor plans and interior designs using augmented reality
|
||||
- [**Microsoft Office for Mac**](https://officecdnmac.microsoft.com/pr/legal/mac/OfficeforMacAttributions.html), a suite of productivity applications developed by Microsoft for macOS, including tools for word processing, spreadsheets, and presentations
|
||||
- [**Microsoft Teams**](https://www.microsoft.com/microsoft-teams/), a team collaboration application offering workspace chat and video conferencing, file storage, and integration of proprietary and third-party applications and services
|
||||
- [**MuseScore**](https://github.com/musescore/MuseScore), a free and open-source music notation and composition application
|
||||
- [**NanaZip**](https://github.com/M2Team/NanaZip/blob/main/NanaZip.Codecs/NanaZip.Codecs.Archive.ElectronAsar.cpp), a 7-Zip derivative built for modern Windows
|
||||
- [**Nexthink Infinity**](https://docs.nexthink.com/legal/services-terms/experience-open-source-software-licenses/infinity-2022.8-software-licenses), a digital employee experience management platform for monitoring and improving IT performance
|
||||
- [**Sophos Connect Client**](https://docs.sophos.com/nsg/licenses/SophosConnect/SophosConnectAttribution.html), a secure VPN client from Sophos that allows remote users to connect to their corporate network, ensuring secure access to resources and data
|
||||
- [**Stonebranch**](https://stonebranchdocs.atlassian.net/wiki/spaces/UA77/pages/799545647/Licenses+for+Third-Party+Libraries), a cloud-based cybersecurity solution that integrates backup, disaster recovery, and cybersecurity features to protect data and ensure business continuity for organizations
|
||||
- [**Tablecruncher**](https://tablecruncher.com/), a data analysis tool that allows users to import, analyze, and visualize spreadsheet data, offering interactive features for better insights and decision-making
|
||||
- [**magicplan**](https://help.magicplan.app/acknowledgments), a mobile application for creating floor plans and interior designs using augmented reality
|
||||
- [**VNote**](https://github.com/vnotex/vnote/blob/master/src/core/services/notebookcoreservice.cpp), a Markdown-based note-taking application written in C++
|
||||
|
||||
## Databases and Big Data
|
||||
|
||||
- [**ADIOS2**](https://code.ornl.gov/ecpcitest/adios2/-/tree/pr4285_FFSUpstream/thirdparty/nlohmann_json?ref_type=heads), a data management framework designed for high-performance input and output operations
|
||||
- [**Apache Doris**](https://github.com/apache/doris/blob/master/be/src/runtime/be_proc_monitor.cpp), a real-time analytical database for high-concurrency queries
|
||||
- [**Claris FileMaker Server**](https://www.claris.com/company/legal/docs/acknowledgements/filemaker-server-macwin/claris_fms2025_acknowledgements_en.pdf), the server platform hosting FileMaker custom apps and databases, developed by Apple subsidiary Claris
|
||||
- [**ClickHouse**](https://github.com/ClickHouse/ClickHouse), a column-oriented database management system for real-time analytical queries
|
||||
- [**Cribl Stream**](https://docs.cribl.io/stream/third-party-current-list/), a real-time data processing platform that enables organizations to collect, route, and transform observability data, enhancing visibility and insights into their systems
|
||||
- [**DB Browser for SQLite**](https://github.com/sqlitebrowser/sqlitebrowser), a visual open-source tool for creating, designing, and editing SQLite database files
|
||||
- [**Manticore Search**](https://github.com/manticoresoftware/manticoresearch/blob/main/src/searchdhttpcompat.cpp), a database for search, offering full-text and vector queries
|
||||
- [**Milvus**](https://github.com/milvus-io/milvus/blob/master/internal/core/src/query/PlanImpl.h), a cloud-native vector database built for embedding similarity search
|
||||
- [**MongoDB**](https://github.com/mongodb/mongo/blob/master/src/mongo/replay/config_handler.cpp), a general-purpose document database
|
||||
- [**MySQL Connector/C++**](https://docs.oracle.com/cd/E17952_01/connector-cpp-9.1-license-com-en/license-opentelemetry-cpp-com.html), a C++ library for connecting and interacting with MySQL databases
|
||||
- [**MySQL NDB Cluster**](https://downloads.mysql.com/docs/licenses/cluster-9.0-com-en.pdf), a distributed database system that provides high availability and scalability for MySQL databases
|
||||
- [**MySQL Shell**](https://downloads.mysql.com/docs/licenses/mysql-shell-8.0-gpl-en.pdf), an advanced client and code editor for interacting with MySQL servers, supporting SQL, Python, and JavaScript
|
||||
- [**PrestoDB**](https://github.com/prestodb/presto), a distributed SQL query engine designed for large-scale data analytics, originally developed by Facebook
|
||||
- [**PrestoDB**](https://github.com/prestodb/presto/blob/master/presto-native-execution/presto_cpp/main/Announcer.cpp), a distributed SQL query engine designed for large-scale data analytics, originally developed by Facebook
|
||||
- [**ROOT Data Analysis Framework**](https://root.cern/doc/v614/classnlohmann_1_1basic__json.html), an open-source data analysis framework widely used in high-energy physics and other fields for data processing and visualization
|
||||
- [**Typesense**](https://github.com/typesense/typesense/blob/v31/include/join.h), an open source typo-tolerant search engine
|
||||
- [**Vearch**](https://github.com/jd-opensource/vearch), a distributed vector database developed at JD.com for similarity search and retrieval-augmented generation
|
||||
- [**WiredTiger**](https://github.com/wiredtiger/wiredtiger), a high-performance storage engine for databases, offering support for compression, concurrency, and checkpointing
|
||||
|
||||
## Simulation and Modeling
|
||||
|
||||
- [**Adobe Lagrange**](https://github.com/adobe/lagrange), a geometry processing library developed by Adobe for mesh manipulation and analysis
|
||||
- [**Arcturus HoloSuite**](https://www.datocms-assets.com/104353/1698904597-holosuite-third-party-software-credits-and-attributions-2.pdf), a software toolset for capturing, editing, and streaming volumetric video, featuring advanced compression technologies for high-quality 3D content creation
|
||||
- [**azul**](https://pure.tudelft.nl/ws/files/85338589/tgis.12673.pdf), a fast and efficient 3D city model viewer designed for visualizing urban environments and spatial data
|
||||
- [**Bambu Studio**](https://github.com/bambulab/BambuStudio), a slicing and print management application for Bambu Lab 3D printers
|
||||
- [**Blender**](https://projects.blender.org/blender/blender/search?q=nlohmann), a free and open-source 3D creation suite for modeling, animation, rendering, and more
|
||||
- [**cpplot**](https://cpplot.readthedocs.io/en/latest/library_api/function_eigen_8h_1ac080eac0541014c5892a55e41bf785e6.html), a library for creating interactive graphs and charts in C++, which can be viewed in web browsers
|
||||
- [**Foundry Nuke**](https://learn.foundry.com/nuke/content/misc/studio_third_party_libraries.html), a powerful node-based digital compositing and visual effects application used in film and television post-production
|
||||
- [**FreeCAD**](https://github.com/FreeCAD/FreeCAD), a free and open-source parametric 3D CAD modeler for product design and engineering
|
||||
- [**GAMS**](https://www.gams.com/47/docs/THIRDPARTY.html), a high-performance mathematical modeling system for optimization and decision support
|
||||
- [**Keysight WirelessPro**](https://docs.keysight.com/display/engdocwirelesspro/WirelessPro+2026+Release+Notes), a simulation platform for 5G, 5G-Advanced, and 6G cellular network research
|
||||
- [**Kitware SMTK**](https://github.com/Kitware/SMTK), a software toolkit for managing simulation models and workflows in scientific and engineering applications
|
||||
- [**M-Star**](https://docs.mstarcfd.com/3_Licensing/thirdparty-licenses.html), a computational fluid dynamics software for simulating and analyzing fluid flow
|
||||
- [**MapleSim CAD Toolbox**](https://www.maplesoft.com/support/help/MapleSim/view.aspx?path=CADToolbox/copyright), a software extension for MapleSim that integrates CAD models, allowing users to import, manipulate, and analyze 3D CAD data within the MapleSim environment for enhanced modeling and simulation
|
||||
- [**Microsoft AirSim**](https://github.com/microsoft/AirSim/blob/main/AirLib/include/common/Settings.hpp), a simulator for autonomous vehicles and drones built on Unreal Engine
|
||||
- [**NVIDIA Omniverse**](https://docs.omniverse.nvidia.com/composer/latest/common/product-licenses/usd-explorer/usd-explorer-2023.2.0-licenses-manifest.html), a platform for 3D content creation and collaboration that enables real-time simulations and interactive experiences across various industries
|
||||
- [**OpenSCAD**](https://github.com/openscad/openscad/blob/master/src/core/AIClient.cc), a script-driven solid 3D CAD modeller
|
||||
- [**OrcaSlicer**](https://github.com/SoftFever/OrcaSlicer), an open-source slicer supporting a wide range of consumer 3D printers
|
||||
- [**Pixar Renderman**](https://rmanwiki-26.pixar.com/space/REN26/19662083/Legal+Notice), a photorealistic 3D rendering software developed by Pixar, widely used in the film industry for creating high-quality visual effects and animations
|
||||
- [**PrusaSlicer**](https://github.com/prusa3d/PrusaSlicer), the slicing software developed by Prusa Research for its 3D printers
|
||||
- [**ROS - Robot Operating System**](http://docs.ros.org/en/noetic/api/behaviortree_cpp/html/json_8hpp_source.html), a set of software libraries and tools that assist in developing robot applications
|
||||
- [**UBS**](https://www.ubs.com/), a multinational financial services and banking company
|
||||
|
||||
@@ -161,18 +278,31 @@ the result of an internet search. If you know further customers of the library,
|
||||
- [**Acronis Cyber Protect Cloud**](https://care.acronis.com/s/article/59533-Third-party-software-used-in-Acronis-Cyber-Protect-Cloud?language=en_US), an all-in-one data protection solution that combines backup, disaster recovery, and cybersecurity to safeguard business data from threats like ransomware
|
||||
- [**Baereos**](https://gitlab.tiger-computing.co.uk/packages/bareos/-/blob/tiger/bullseye/third-party/CLI11/examples/json.cpp), a backup solution that provides data protection and recovery options for various environments, including physical and virtual systems
|
||||
- [**Bitdefender Home Scanner**](https://www.bitdefender.de/site/Main/view/home-scanner-open-source.html), a tool from Bitdefender that scans devices for malware and security threats, providing a safeguard against potential online dangers
|
||||
- [**Cisco MLS++**](https://github.com/cisco/mlspp), an implementation of the Messaging Layer Security protocol for end-to-end encrypted group messaging
|
||||
- [**Citrix Provisioning**](https://docs.citrix.com/en-us/provisioning/2203-ltsr/downloads/pvs-third-party-notices-2203.pdf), a solution that streamlines the delivery of virtual desktops and applications by allowing administrators to manage and provision resources efficiently across multiple environments
|
||||
- [**Citrix Virtual Apps and Desktops**](https://docs.citrix.com/en-us/citrix-virtual-apps-desktops/2305/downloads/third-party-notices-apps-and-desktops.pdf), a solution from Citrix that delivers virtual apps and desktops
|
||||
- [**Cyberarc**](https://docs.cyberark.com/Downloads/Legal/Privileged%20Session%20Manager%20for%20SSH%20Third-Party%20Notices.pdf), a security solution that specializes in privileged access management, enabling organizations to control and monitor access to critical systems and data, thereby enhancing overall cybersecurity posture
|
||||
- [**Deutsche Telekom sysrepo-plugins**](https://github.com/telekom/sysrepo-plugins), a collection of YANG datastore plugins used to manage network devices
|
||||
- [**Egnyte Desktop**](https://helpdesk.egnyte.com/hc/en-us/articles/360007071732-Third-Party-Software-Acknowledgements), a secure cloud storage solution designed for businesses, enabling file sharing, collaboration, and data management across teams while ensuring compliance and data protection
|
||||
- [**Elster**](https://www.secunet.com/en/about-us/press/article/elstersecure-bietet-komfortablen-login-ohne-passwort-dank-secunet-protect4use), a digital platform developed by German tax authorities for secure and efficient electronic tax filing and management using secunet protect4use
|
||||
- [**Envoy**](https://github.com/envoyproxy/envoy), a cloud-native edge and service proxy that forms the data plane of many service meshes
|
||||
- [**Ethereum Solidity**](https://github.com/ethereum/solidity), a high-level, object-oriented programming language designed for implementing smart contracts on the Ethereum platform
|
||||
- [**gVisor**](https://github.com/google/gvisor), an application kernel that provides a secure sandbox for running untrusted containers
|
||||
- [**IBM Storage Virtualize**](https://public.dhe.ibm.com/systems/support/warranty/pdfs/stgoilc/SV_for_FS_7300_v8_7_0_Base_OILC.pdf), the software powering IBM FlashSystem enterprise storage arrays
|
||||
- [**Inciga**](https://fossies.org/linux/icinga2/third-party/nlohmann_json/json.hpp), a monitoring tool for IT infrastructure, designed to provide insights into system performance and availability through customizable dashboards and alerts
|
||||
- [**Intel Accelerator Management Daemon for VMware ESXi**](https://downloadmirror.intel.com/772507/THIRD-PARTY.txt), a management tool designed for monitoring and controlling Intel hardware accelerators within VMware ESXi environments, optimizing performance and resource allocation
|
||||
- [**Juniper Identity Management Service**](https://www.juniper.net/documentation/us/en/software/jims/jims-guide/jims-guide.pdf)
|
||||
- [**Meta FBOSS**](https://github.com/facebook/fboss), the software stack that controls the network switches in Meta's data centers
|
||||
- [**Microsoft Azure IoT SDK**](https://library.e.abb.com/public/2779c5f85f30484192eb3cb3f666a201/IP%20Gateway%20Open%20License%20Declaration_9AKK108467A4095_Rev_C.pdf), a collection of tools and libraries to help developers connect, build, and deploy Internet of Things (IoT) solutions on the Azure cloud platform
|
||||
- [**Microsoft Confidential Consortium Framework**](https://github.com/microsoft/CCF), a framework for building secure, highly available applications on trusted execution environments
|
||||
- [**Microsoft WinGet**](https://github.com/microsoft/winget-cli), a command-line utility included in the Windows Package Manager
|
||||
- [**Mitsubishi Electric SECS/GEM**](https://dl.mitsubishielectric.com/dl/fa/document/manual/plc/sh082483eng/sh082483engi.pdf), the semiconductor equipment communication software running on Mitsubishi Electric C Controller and C intelligent function modules
|
||||
- [**Moxa**](https://www.moxa.com/getmedia/fbe2a0c7-8dda-4b5b-a501-15e45adebb1f/moxa-foss-statement-for-da-720-series-win-10-ltsc-21h2-declaration-v1.0.pdf), a provider of industrial networking, computing, and automation infrastructure
|
||||
- [**plexusAV**](https://www.sisme.com/media/10994/manual_plexusav-p-avn-4-form8244-c.pdf), a high-performance AV-over-IP transceiver device capable of video encoding and decoding using the IPMX standard
|
||||
- [**Pointr**](https://docs-dev.pointr.tech/docs/8.x/Developer%20Portal/Open%20Source%20Licenses/), a platform for indoor positioning and navigation solutions, offering tools and SDKs for developers to create location-based applications
|
||||
- [**secunet protect4use**](https://www.secunet.com/en/about-us/press/article/elstersecure-bietet-komfortablen-login-ohne-passwort-dank-secunet-protect4use), a secure, passwordless multifactor authentication solution that transforms smartphones into digital keyrings, ensuring high security for online services and digital identities
|
||||
- [**Sencore MRD 7000**](https://www.foccusdigital.com/wp-content/uploads/2025/03/MRD-7000-Manual-8175V.pdf), a professional multi-channel receiver and decoder supporting UHD and HD stream decoding
|
||||
- [**Siemens SINEC**](https://cache.industry.siemens.com/dl/files/917/109974917/att_1298783/v2/OSS_SINEC-NMS_99.pdf), a family of network management and infrastructure services for industrial networks
|
||||
- [**Toshiba Industrial Servers**](https://www.global.toshiba/content/dam/toshiba/jp/products-solutions/industrial/computer/product/server/fs20000r/pdf/FS20000R_OSS_License_6E8C5817_rev0.pdf), the FS20000R series of industrial servers for factory automation and control systems
|
||||
- [**Wazuh**](https://github.com/wazuh/wazuh/blob/main/src/data_provider/src/sysInfo.cpp), a security platform for threat detection, integrity monitoring and incident response
|
||||
- [**ZeroTier**](https://github.com/zerotier/ZeroTierOne/blob/dev/osdep/OSUtils.hpp), a software-defined networking service that creates virtual Ethernet networks
|
||||
|
||||
Binary file not shown.
|
Before Width: | Height: | Size: 1.3 MiB After Width: | Height: | Size: 1001 KiB |
@@ -296,7 +296,6 @@ nav:
|
||||
- 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md
|
||||
- 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md
|
||||
- 'JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON': api/macros/json_use_legacy_discarded_value_comparison.md
|
||||
- 'JSON_USE_SIMDUTF': api/macros/json_use_simdutf.md
|
||||
- 'NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_derived_type.md
|
||||
- 'NLOHMANN_DEFINE_TYPE_INTRUSIVE, NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_intrusive.md
|
||||
- 'NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_non_intrusive.md
|
||||
|
||||
@@ -155,31 +155,11 @@ class input_stream_adapter
|
||||
|
||||
// General-purpose iterator-based adapter. It might not be as fast as
|
||||
// theoretically possible for some containers, but it is extremely versatile.
|
||||
// SentinelType defaults to IteratorType for backward compatibility, but may be
|
||||
// a different type, e.g. a C++20 sentinel such as std::default_sentinel_t when
|
||||
// IteratorType is a std::counted_iterator.
|
||||
// SentinelType defaults to IteratorType for backward compatibility, but may
|
||||
// be a different type (e.g., a C++20 sentinel or counted_iterator).
|
||||
template<typename IteratorType, typename SentinelType = IteratorType>
|
||||
class iterator_input_adapter
|
||||
{
|
||||
// Whether the number of elements between two positions can be computed in
|
||||
// O(1): either the iterator and the sentinel have the same type (plain
|
||||
// std::distance) or, in C++20, the sentinel is a sized sentinel for the
|
||||
// iterator (std::ranges::distance), e.g. std::default_sentinel_t paired
|
||||
// with std::counted_iterator.
|
||||
//
|
||||
// JSON_HAS_RANGES gates the C++20 branch: on standard libraries with an
|
||||
// incomplete <ranges> (libstdc++ < 11, see #4440) evaluating
|
||||
// std::contiguous_iterator on a std::counted_iterator is a hard error
|
||||
// instead of yielding false, and these traits are instantiated for every
|
||||
// adapter. Such toolchains fall back to the pointer-only test and simply
|
||||
// use the byte-at-a-time scanner.
|
||||
static constexpr bool sentinel_is_sized =
|
||||
#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||
std::is_same<IteratorType, SentinelType>::value || std::sized_sentinel_for<SentinelType, IteratorType>;
|
||||
#else
|
||||
std::is_same<IteratorType, SentinelType>::value;
|
||||
#endif
|
||||
|
||||
public:
|
||||
using char_type = typename std::iterator_traits<IteratorType>::value_type;
|
||||
|
||||
@@ -191,7 +171,7 @@ class iterator_input_adapter
|
||||
// in wide_string_input_adapter, which does not expose this).
|
||||
static constexpr bool supports_seek =
|
||||
std::is_same<typename std::iterator_traits<IteratorType>::iterator_category, std::random_access_iterator_tag>::value
|
||||
&& sentinel_is_sized
|
||||
&& std::is_same<IteratorType, SentinelType>::value
|
||||
&& sizeof(char_type) == 1;
|
||||
|
||||
iterator_input_adapter(IteratorType first, SentinelType last)
|
||||
@@ -239,60 +219,30 @@ class iterator_input_adapter
|
||||
private:
|
||||
// whether IteratorType refers to a contiguous range and therefore supports
|
||||
// a std::memcpy fast path (pointers always do; in C++20 we can also detect
|
||||
// library iterators such as those of std::vector and std::string). The
|
||||
// available element count must also be computable in O(1), hence
|
||||
// sentinel_is_sized.
|
||||
static constexpr bool iterator_is_contiguous = sentinel_is_sized &&
|
||||
#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||
(std::contiguous_iterator<IteratorType> || std::is_pointer<IteratorType>::value);
|
||||
// library iterators such as those of std::vector and std::string).
|
||||
// Computing the available element count needs either same-type iterators
|
||||
// (plain std::distance) or, in C++20, a sized sentinel (std::ranges::distance),
|
||||
// e.g. std::counted_iterator paired with std::default_sentinel_t.
|
||||
static constexpr bool iterator_is_contiguous =
|
||||
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||
(std::is_same<IteratorType, SentinelType>::value || std::sized_sentinel_for<SentinelType, IteratorType>)
|
||||
&& (std::contiguous_iterator<IteratorType> || std::is_pointer<IteratorType>::value);
|
||||
#else
|
||||
std::is_pointer<IteratorType>::value;
|
||||
std::is_same<IteratorType, SentinelType>::value && std::is_pointer<IteratorType>::value;
|
||||
#endif
|
||||
|
||||
// number of unread elements in [current, end)
|
||||
std::size_t remaining_count() const
|
||||
{
|
||||
#if JSON_HAS_RANGES && defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||
// std::ranges::distance also supports sized sentinels of a different
|
||||
// type (e.g. std::counted_iterator + std::default_sentinel_t)
|
||||
return static_cast<std::size_t>(std::ranges::distance(current, end));
|
||||
#else
|
||||
return static_cast<std::size_t>(std::distance(current, end));
|
||||
#endif
|
||||
}
|
||||
|
||||
public:
|
||||
// Whether the remaining input is a single contiguous block of 1-byte
|
||||
// elements that the lexer can inspect directly (used for the SWAR string
|
||||
// fast path).
|
||||
static constexpr bool supports_bulk_scan =
|
||||
iterator_is_contiguous && sizeof(char_type) == 1;
|
||||
|
||||
// Pointer to the next unread element; only valid when bulk_remaining() > 0.
|
||||
const char_type* bulk_data() const
|
||||
{
|
||||
return &*current;
|
||||
}
|
||||
|
||||
// Number of unread elements available as one contiguous block.
|
||||
std::size_t bulk_remaining() const
|
||||
{
|
||||
return remaining_count();
|
||||
}
|
||||
|
||||
// Consume @a n elements previously inspected via bulk_data().
|
||||
void bulk_skip(std::size_t n)
|
||||
{
|
||||
std::advance(current, static_cast<typename std::iterator_traits<IteratorType>::difference_type>(n));
|
||||
}
|
||||
|
||||
private:
|
||||
// contiguous fast path: bulk copy the remaining range with std::memcpy
|
||||
template<class T>
|
||||
std::size_t get_elements_impl(T* dest, std::size_t count, std::true_type /*contiguous*/)
|
||||
{
|
||||
const std::size_t wanted = count * sizeof(T);
|
||||
const std::size_t available = remaining_count() * sizeof(char_type);
|
||||
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||
// std::ranges::distance also supports sized sentinels of a different
|
||||
// type (e.g. std::counted_iterator + std::default_sentinel_t)
|
||||
const std::size_t available = static_cast<std::size_t>(std::ranges::distance(current, end)) * sizeof(char_type);
|
||||
#else
|
||||
const std::size_t available = static_cast<std::size_t>(std::distance(current, end)) * sizeof(char_type);
|
||||
#endif
|
||||
const std::size_t copied = (std::min)(wanted, available);
|
||||
if (JSON_HEDLEY_LIKELY(copied != 0))
|
||||
{
|
||||
@@ -620,46 +570,6 @@ typename iterator_input_adapter_factory<IteratorType, SentinelType>::adapter_typ
|
||||
return factory_type::create(first, last);
|
||||
}
|
||||
|
||||
// The element type a container's data() points at, cv-qualifiers removed.
|
||||
// Ill-formed - and therefore SFINAE-friendly - for types without data().
|
||||
template<typename ContainerType>
|
||||
using container_data_t = typename std::remove_cv<typename std::remove_pointer <
|
||||
decltype(std::declval<const ContainerType&>().data()) >::type >::type;
|
||||
|
||||
// The container's own element type, cv-qualifiers removed. It is looked up on
|
||||
// the bare type so it is also found when ContainerType is deduced as a
|
||||
// reference by the forwarding-reference overload below.
|
||||
template<typename ContainerType>
|
||||
using container_value_t = typename std::remove_cv <
|
||||
typename std::remove_cv<typename std::remove_reference<ContainerType>::type>::type::value_type >::type;
|
||||
|
||||
// Detect a container that stores its elements contiguously as single bytes
|
||||
// (std::string, std::vector<char/unsigned char>, std::array<char, N>,
|
||||
// std::string_view, ...). Such inputs are wrapped in a pointer-based adapter so
|
||||
// they benefit from the contiguous fast paths (bulk string scanning, memcpy for
|
||||
// binary formats) in every C++ standard - not only in C++20, where the standard
|
||||
// library iterators model std::contiguous_iterator and are detected directly.
|
||||
//
|
||||
// data() and size() on their own would be duck typing: they say nothing about
|
||||
// size() counting the units data() points at, and reading [data(), data() +
|
||||
// size()) as bytes would be wrong for a type where it does not. Requiring the
|
||||
// container's own value_type to be that same single-byte element ties the two
|
||||
// together; every contiguous standard container satisfies it. Anything else
|
||||
// keeps the iterator-based adapter, which is always correct - only slower.
|
||||
template<typename ContainerType, typename = void>
|
||||
struct is_contiguous_byte_container : std::false_type {};
|
||||
|
||||
template<typename ContainerType>
|
||||
struct is_contiguous_byte_container < ContainerType, void_t <
|
||||
container_data_t<ContainerType>,
|
||||
container_value_t<ContainerType>,
|
||||
decltype(std::declval<const ContainerType&>().size()) >>
|
||||
: std::integral_constant < bool,
|
||||
std::is_pointer<decltype(std::declval<const ContainerType&>().data())>::value&&
|
||||
std::is_integral<container_data_t<ContainerType>>::value&&
|
||||
sizeof(container_data_t<ContainerType>) == 1 &&
|
||||
std::is_same<container_data_t<ContainerType>, container_value_t<ContainerType>>::value > {};
|
||||
|
||||
// Convenience shorthand from container to iterator
|
||||
// Enables ADL on begin(container) and end(container)
|
||||
// Encloses the using declarations in namespace for not to leak them to outside scope
|
||||
@@ -687,32 +597,12 @@ struct container_input_adapter_factory< ContainerType,
|
||||
|
||||
} // namespace container_input_adapter_factory_impl
|
||||
|
||||
// General container path (iterator-based). Contiguous single-byte containers
|
||||
// are excluded here and routed through the pointer-based overload below.
|
||||
template < typename ContainerType,
|
||||
enable_if_t < !is_contiguous_byte_container<ContainerType>::value, int > = 0 >
|
||||
typename container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::adapter_type input_adapter(ContainerType && container)
|
||||
template<typename ContainerType>
|
||||
typename container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::adapter_type input_adapter(ContainerType&& container)
|
||||
{
|
||||
return container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::create(std::forward<ContainerType>(container));
|
||||
}
|
||||
|
||||
// Contiguous single-byte containers (std::string, std::vector<char>, ...) are
|
||||
// wrapped in a pointer-based adapter so the contiguous fast paths apply in every
|
||||
// standard. The pointer keeps the container's own element type (const char* for
|
||||
// std::string, const std::uint8_t* for std::vector<std::uint8_t>, ...), so the
|
||||
// resulting char_type - and therefore the parsing behavior - is byte-for-byte
|
||||
// identical to the iterator-based path; only the raw pointer additionally
|
||||
// enables the bulk fast paths. The container outlives the adapter for the whole
|
||||
// parse (temporaries live until the end of the full expression), exactly as the
|
||||
// iterators it replaces did.
|
||||
template < typename ContainerType,
|
||||
enable_if_t < is_contiguous_byte_container<ContainerType>::value, int > = 0 >
|
||||
auto input_adapter(const ContainerType& container)
|
||||
-> decltype(input_adapter(container.data(), container.data() + container.size()))
|
||||
{
|
||||
return input_adapter(container.data(), container.data() + container.size());
|
||||
}
|
||||
|
||||
// specialization for std::string
|
||||
using string_input_adapter_type = decltype(input_adapter(std::declval<std::string>()));
|
||||
|
||||
|
||||
@@ -222,16 +222,12 @@ class json_sax_dom_parser
|
||||
|
||||
bool string(string_t& val)
|
||||
{
|
||||
// json_sax documents that the passed value may be moved from,
|
||||
// so hand the buffer over instead of copying it
|
||||
handle_value(std::move(val));
|
||||
handle_value(val);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool binary(binary_t& val)
|
||||
{
|
||||
// json_sax documents that the passed value may be moved from,
|
||||
// so hand the buffer over instead of copying it
|
||||
handle_value(std::move(val));
|
||||
return true;
|
||||
}
|
||||
@@ -536,16 +532,12 @@ class json_sax_dom_callback_parser
|
||||
|
||||
bool string(string_t& val)
|
||||
{
|
||||
// json_sax documents that the passed value may be moved from,
|
||||
// so hand the buffer over instead of copying it
|
||||
handle_value(std::move(val));
|
||||
handle_value(val);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool binary(binary_t& val)
|
||||
{
|
||||
// json_sax documents that the passed value may be moved from,
|
||||
// so hand the buffer over instead of copying it
|
||||
handle_value(std::move(val));
|
||||
return true;
|
||||
}
|
||||
@@ -556,11 +548,6 @@ class json_sax_dom_callback_parser
|
||||
const bool keep = callback(static_cast<int>(ref_stack.size()), parse_event_t::object_start, discarded);
|
||||
keep_stack.push_back(keep);
|
||||
|
||||
// the key this object will be stored under, read before handle_value()
|
||||
// may consume it; kept in lockstep with ref_stack so end_object() can
|
||||
// find the object in its parent again
|
||||
container_key_stack.push_back(current_key());
|
||||
|
||||
auto val = handle_value(BasicJsonType::value_t::object, true);
|
||||
ref_stack.push_back(val.second);
|
||||
|
||||
@@ -594,9 +581,6 @@ class json_sax_dom_callback_parser
|
||||
// check callback for the key
|
||||
const bool keep = callback(static_cast<int>(ref_stack.size()), parse_event_t::key, k);
|
||||
key_keep_stack.push_back(keep);
|
||||
// remember the key so a rejected value can be erased without searching
|
||||
// the object for it (kept in lockstep with key_keep_stack)
|
||||
key_stack.push_back(val);
|
||||
|
||||
// add discarded value at the given key and store the reference for later
|
||||
if (keep && ref_stack.back())
|
||||
@@ -638,16 +622,13 @@ class json_sax_dom_callback_parser
|
||||
|
||||
JSON_ASSERT(!ref_stack.empty());
|
||||
JSON_ASSERT(!keep_stack.empty());
|
||||
JSON_ASSERT(!container_key_stack.empty());
|
||||
ref_stack.pop_back();
|
||||
keep_stack.pop_back();
|
||||
const string_t object_key = std::move(container_key_stack.back());
|
||||
container_key_stack.pop_back();
|
||||
|
||||
if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_structured())
|
||||
{
|
||||
// remove discarded value
|
||||
remove_discarded_value(*ref_stack.back(), object_key);
|
||||
remove_discarded_value(*ref_stack.back());
|
||||
}
|
||||
|
||||
return true;
|
||||
@@ -658,9 +639,6 @@ class json_sax_dom_callback_parser
|
||||
const bool keep = callback(static_cast<int>(ref_stack.size()), parse_event_t::array_start, discarded);
|
||||
keep_stack.push_back(keep);
|
||||
|
||||
// see start_object()
|
||||
container_key_stack.push_back(current_key());
|
||||
|
||||
auto val = handle_value(BasicJsonType::value_t::array, true);
|
||||
ref_stack.push_back(val.second);
|
||||
|
||||
@@ -723,11 +701,8 @@ class json_sax_dom_callback_parser
|
||||
|
||||
JSON_ASSERT(!ref_stack.empty());
|
||||
JSON_ASSERT(!keep_stack.empty());
|
||||
JSON_ASSERT(!container_key_stack.empty());
|
||||
ref_stack.pop_back();
|
||||
keep_stack.pop_back();
|
||||
const string_t object_key = std::move(container_key_stack.back());
|
||||
container_key_stack.pop_back();
|
||||
|
||||
// remove discarded value
|
||||
if (!ref_stack.empty() && ref_stack.back())
|
||||
@@ -741,7 +716,7 @@ class json_sax_dom_callback_parser
|
||||
// the array is either still stored under its key or was never
|
||||
// stored, leaving the placeholder key() wrote; both show up as
|
||||
// a discarded member of the parent object
|
||||
remove_discarded_value(*ref_stack.back(), object_key);
|
||||
remove_discarded_value(*ref_stack.back());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -834,56 +809,15 @@ class json_sax_dom_callback_parser
|
||||
}
|
||||
#endif
|
||||
|
||||
/*!
|
||||
@brief the key the value now being handled will be stored under
|
||||
|
||||
Empty unless the enclosing container is an object, in which case it is the
|
||||
key of the pending key() event. Read before handle_value() consumes that
|
||||
key, so it is also correct when the value never reaches its parent.
|
||||
*/
|
||||
string_t current_key() const
|
||||
/// remove the discarded value the callback rejected from its parent
|
||||
static void remove_discarded_value(BasicJsonType& parent)
|
||||
{
|
||||
if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_object()
|
||||
&& !key_stack.empty())
|
||||
for (auto it = parent.begin(); it != parent.end(); ++it)
|
||||
{
|
||||
return key_stack.back();
|
||||
}
|
||||
return string_t{};
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief remove the discarded value the callback rejected from its parent
|
||||
|
||||
A rejected value can only ever be the one most recently added to @a parent:
|
||||
the last element of an array, or the placeholder key() stored under @a key
|
||||
in an object. Looking there directly makes this O(1) resp. O(log n), where
|
||||
searching @a parent for it made a filtering parse quadratic in the number of
|
||||
members of a single container.
|
||||
|
||||
Finding no discarded value there means none was stored in the first place -
|
||||
the callback rejected the value before it reached its parent - so there is
|
||||
nothing to remove.
|
||||
|
||||
@param[in,out] parent the container to remove the rejected value from
|
||||
@param[in] key the key the value was stored under; unused for arrays
|
||||
*/
|
||||
static void remove_discarded_value(BasicJsonType& parent, const string_t& key)
|
||||
{
|
||||
if (parent.is_array())
|
||||
{
|
||||
auto& array = *parent.m_data.m_value.array;
|
||||
if (!array.empty() && array.back().is_discarded())
|
||||
if (it->is_discarded())
|
||||
{
|
||||
array.pop_back();
|
||||
}
|
||||
}
|
||||
else if (parent.is_object())
|
||||
{
|
||||
auto& object = *parent.m_data.m_value.object;
|
||||
const auto it = object.find(key);
|
||||
if (it != object.end() && it->second.is_discarded())
|
||||
{
|
||||
object.erase(it);
|
||||
parent.erase(it);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -933,14 +867,11 @@ class json_sax_dom_callback_parser
|
||||
if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_object())
|
||||
{
|
||||
JSON_ASSERT(!key_keep_stack.empty());
|
||||
JSON_ASSERT(!key_stack.empty());
|
||||
const bool placeholder_stored = key_keep_stack.back();
|
||||
key_keep_stack.pop_back();
|
||||
const string_t key = std::move(key_stack.back());
|
||||
key_stack.pop_back();
|
||||
if (placeholder_stored)
|
||||
{
|
||||
remove_discarded_value(*ref_stack.back(), key);
|
||||
remove_discarded_value(*ref_stack.back());
|
||||
}
|
||||
}
|
||||
return {false, nullptr};
|
||||
@@ -973,10 +904,8 @@ class json_sax_dom_callback_parser
|
||||
JSON_ASSERT(ref_stack.back()->is_object());
|
||||
// check if we should store an element for the current key
|
||||
JSON_ASSERT(!key_keep_stack.empty());
|
||||
JSON_ASSERT(!key_stack.empty());
|
||||
const bool store_element = key_keep_stack.back();
|
||||
key_keep_stack.pop_back();
|
||||
key_stack.pop_back();
|
||||
|
||||
if (!store_element)
|
||||
{
|
||||
@@ -996,12 +925,6 @@ class json_sax_dom_callback_parser
|
||||
std::vector<bool> keep_stack {}; // NOLINT(readability-redundant-member-init)
|
||||
/// stack to manage which object keys to keep
|
||||
std::vector<bool> key_keep_stack {}; // NOLINT(readability-redundant-member-init)
|
||||
/// the keys key() stored a placeholder for, in lockstep with key_keep_stack
|
||||
std::vector<string_t> key_stack {}; // NOLINT(readability-redundant-member-init)
|
||||
/// for each open container, the key it is stored under in its parent
|
||||
/// object, in lockstep with ref_stack; unused where the parent is not an
|
||||
/// object
|
||||
std::vector<string_t> container_key_stack {}; // NOLINT(readability-redundant-member-init)
|
||||
/// helper to hold the reference for the next object element
|
||||
BasicJsonType* object_element = nullptr;
|
||||
/// whether a syntax error occurred
|
||||
|
||||
@@ -19,9 +19,7 @@
|
||||
#include <vector> // vector
|
||||
|
||||
#include <nlohmann/detail/input/input_adapters.hpp>
|
||||
#include <nlohmann/detail/input/number_parse.hpp>
|
||||
#include <nlohmann/detail/input/position_t.hpp>
|
||||
#include <nlohmann/detail/input/string_scan.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||
|
||||
@@ -127,25 +125,6 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/)
|
||||
return false;
|
||||
}
|
||||
|
||||
// Detect whether an input adapter exposes a contiguous byte block that the
|
||||
// lexer can scan directly (see iterator_input_adapter::supports_bulk_scan).
|
||||
// Adapters without the flag - file, stream, wide-string, user-defined - fall
|
||||
// back to the character-at-a-time string scanner.
|
||||
template<typename InputAdapterType>
|
||||
using detect_supports_bulk_scan = decltype(InputAdapterType::supports_bulk_scan);
|
||||
|
||||
template<typename InputAdapterType>
|
||||
constexpr bool input_adapter_supports_bulk_scan(std::true_type /*detected*/)
|
||||
{
|
||||
return InputAdapterType::supports_bulk_scan;
|
||||
}
|
||||
|
||||
template<typename InputAdapterType>
|
||||
constexpr bool input_adapter_supports_bulk_scan(std::false_type /*detected*/)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief lexical analysis
|
||||
|
||||
@@ -167,14 +146,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
static constexpr bool lazy_token_string =
|
||||
input_adapter_supports_seek<InputAdapterType>(is_detected<detect_supports_seek, InputAdapterType> {});
|
||||
|
||||
/// whether string scanning may bulk-consume runs of ordinary characters
|
||||
/// directly from a contiguous input buffer (SWAR fast path). This requires
|
||||
/// the token to be reconstructible lazily (lazy_token_string), so bypassing
|
||||
/// the per-character capture in get() cannot lose error diagnostics.
|
||||
static constexpr bool bulk_scan =
|
||||
lazy_token_string
|
||||
&& input_adapter_supports_bulk_scan<InputAdapterType>(is_detected<detect_supports_bulk_scan, InputAdapterType> {});
|
||||
|
||||
public:
|
||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||
|
||||
@@ -294,40 +265,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
return true;
|
||||
}
|
||||
|
||||
/// contiguous input: bulk-append the run of ordinary characters and complete
|
||||
/// well-formed UTF-8 sequences starting at the current read position, leaving
|
||||
/// the first byte that needs individual handling (the closing quote, an
|
||||
/// escape, a control character, or an ill-formed UTF-8 byte) for get()
|
||||
void scan_string_bulk(std::true_type /*bulk*/)
|
||||
{
|
||||
// a pending unget must be consumed through the normal path first
|
||||
if (next_unget)
|
||||
{
|
||||
return;
|
||||
}
|
||||
const std::size_t remaining = ia.bulk_remaining();
|
||||
if (remaining == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
const auto* const data = reinterpret_cast<const unsigned char*>(ia.bulk_data());
|
||||
|
||||
const std::size_t pos = string_bulk_run(data, remaining);
|
||||
if (pos == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
token_buffer.append(reinterpret_cast<const typename string_t::value_type*>(data), pos);
|
||||
ia.bulk_skip(pos);
|
||||
// the run contains no newline (all bytes < 0x20 are treated as special),
|
||||
// so only the flat character counters advance
|
||||
position.chars_read_total += pos;
|
||||
position.chars_read_current_line += pos;
|
||||
}
|
||||
|
||||
/// streaming input: no bulk fast path
|
||||
void scan_string_bulk(std::false_type /*bulk*/) const noexcept {}
|
||||
|
||||
/*!
|
||||
@brief scan a string literal
|
||||
|
||||
@@ -353,10 +290,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
|
||||
while (true)
|
||||
{
|
||||
// bulk-consume ordinary characters from contiguous input, then
|
||||
// handle the next special byte through the switch below
|
||||
scan_string_bulk(std::integral_constant<bool, bulk_scan> {});
|
||||
|
||||
// get the next character
|
||||
switch (get())
|
||||
{
|
||||
@@ -1075,12 +1008,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
// changed if minus sign, decimal point, or exponent is read
|
||||
token_type number_type = token_type::value_unsigned;
|
||||
|
||||
// offset just past the last mantissa byte in token_buffer (i.e. the
|
||||
// index of 'e'/'E', or the whole token when there is no exponent).
|
||||
// convert_number() uses it to count significant digits; npos means
|
||||
// "not seen an exponent yet" and is resolved at scan_number_done
|
||||
std::size_t mantissa_end = std::string::npos;
|
||||
|
||||
// state (init): we just found out we need to scan a number
|
||||
switch (current)
|
||||
{
|
||||
@@ -1266,9 +1193,6 @@ scan_number_decimal2:
|
||||
scan_number_exponent:
|
||||
// we just parsed an exponent
|
||||
number_type = token_type::value_float;
|
||||
// this label is reached only right after the 'e'/'E' was appended (from
|
||||
// the zero, any1, and decimal2 states), so the mantissa ends before it
|
||||
mantissa_end = token_buffer.size() - 1;
|
||||
switch (get())
|
||||
{
|
||||
case '+':
|
||||
@@ -1355,147 +1279,45 @@ scan_number_done:
|
||||
// we are done scanning a number)
|
||||
unget();
|
||||
|
||||
// no exponent was scanned: the mantissa spans the whole token
|
||||
if (mantissa_end == std::string::npos)
|
||||
{
|
||||
mantissa_end = token_buffer.size();
|
||||
}
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
errno = 0;
|
||||
|
||||
return convert_number(number_type, mantissa_end);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert an already-validated integer token to its value
|
||||
|
||||
The digit sequence in [first, last) has been validated by the caller, so a
|
||||
dedicated parser can avoid the locale/errno overhead of std::strtoull.
|
||||
|
||||
@return the token type on success; token_type::uninitialized if @a
|
||||
number_type is not an integer type or the value does not fit, in
|
||||
which case the caller falls back to the floating-point conversion
|
||||
(matching the previous std::strtoull/std::strtoll behavior)
|
||||
*/
|
||||
token_type convert_integer(token_type number_type, const char* first, const char* last)
|
||||
{
|
||||
// try to parse integers first and fall back to floats
|
||||
if (number_type == token_type::value_unsigned)
|
||||
{
|
||||
if (parse_integer_unsigned(first, last, value_unsigned))
|
||||
const auto x = std::strtoull(token_buffer.data(), &endptr, 10);
|
||||
|
||||
// we checked the number format before
|
||||
JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size());
|
||||
|
||||
if (errno != ERANGE)
|
||||
{
|
||||
return token_type::value_unsigned;
|
||||
value_unsigned = static_cast<number_unsigned_t>(x);
|
||||
if (value_unsigned == x)
|
||||
{
|
||||
return token_type::value_unsigned;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (number_type == token_type::value_integer)
|
||||
{
|
||||
if (parse_integer_signed(first, last, value_integer))
|
||||
const auto x = std::strtoll(token_buffer.data(), &endptr, 10);
|
||||
|
||||
// we checked the number format before
|
||||
JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size());
|
||||
|
||||
if (errno != ERANGE)
|
||||
{
|
||||
return token_type::value_integer;
|
||||
}
|
||||
}
|
||||
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief check whether Clinger's fast path can still succeed for this token
|
||||
|
||||
parse_float_fast() needs a significand below 2^53. A mantissa with 17 or
|
||||
more significant digits is at least 10^16 and therefore always exceeds it,
|
||||
so calling the fast path would walk the token one extra time only to
|
||||
decline before strtod has to run anyway.
|
||||
|
||||
Significant digits are the mantissa's digits from the first nonzero one on;
|
||||
the sign, the decimal point, leading zeros, and the exponent do not count.
|
||||
The answer is derived from indices - the digits are not scanned again - so
|
||||
this stays off the hot path of the number scanners.
|
||||
|
||||
@param[in] mantissa_end offset just past the last mantissa byte in
|
||||
token_buffer
|
||||
@return false if parse_float_fast() is guaranteed to decline
|
||||
*/
|
||||
bool mantissa_fits_clinger(std::size_t mantissa_end) const
|
||||
{
|
||||
// 10^16 already exceeds 2^53, so 17 digits can never fit
|
||||
constexpr std::size_t limit = 17;
|
||||
|
||||
const std::size_t neg = (!token_buffer.empty() && token_buffer[0] == '-') ? 1u : 0u;
|
||||
const std::size_t has_dot = (decimal_point_position != std::string::npos) ? 1u : 0u;
|
||||
// the JSON grammar restricts the integer part to "0" or [1-9][0-9]*, so
|
||||
// a leading zero can only be a lone "0", which is not significant
|
||||
const std::size_t lead_zero = (token_buffer[neg] == '0') ? 1u : 0u;
|
||||
JSON_ASSERT(mantissa_end >= neg + has_dot + lead_zero);
|
||||
std::size_t digits = mantissa_end - neg - has_dot - lead_zero;
|
||||
|
||||
if (JSON_HEDLEY_LIKELY(digits < limit))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// Only a number below 1 can carry further insignificant zeros, and only
|
||||
// while the count stays at the limit does removing them change the
|
||||
// answer - so this loop is skipped for all but a few tokens. Note
|
||||
// token_buffer holds the locale's decimal point, so the fraction is
|
||||
// located through decimal_point_position rather than by searching '.'.
|
||||
if (lead_zero != 0)
|
||||
{
|
||||
JSON_ASSERT(has_dot != 0); // an integer "0" cannot reach the limit
|
||||
for (std::size_t i = decimal_point_position + 1;
|
||||
digits >= limit && i < mantissa_end && token_buffer[i] == '0'; ++i)
|
||||
{
|
||||
--digits;
|
||||
}
|
||||
}
|
||||
|
||||
return digits < limit;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the number text in token_buffer to its value and token type
|
||||
|
||||
The digit sequence in token_buffer has already been validated (by the
|
||||
scan_number() state machine or by the contiguous fast path) and holds the
|
||||
locale decimal point in place of '.'. Integers are parsed first and fall
|
||||
back to floating point on overflow. This is shared so both scanners produce
|
||||
identical results.
|
||||
|
||||
@param[in] mantissa_end offset just past the last mantissa byte in
|
||||
token_buffer (the index of 'e'/'E', or
|
||||
token_buffer.size() when there is no exponent);
|
||||
used to skip Clinger's fast path when it cannot
|
||||
possibly succeed - see mantissa_fits_clinger()
|
||||
*/
|
||||
token_type convert_number(token_type number_type, std::size_t mantissa_end)
|
||||
{
|
||||
const char* const num_begin = token_buffer.data();
|
||||
const char* const num_end = num_begin + token_buffer.size();
|
||||
|
||||
if (number_type != token_type::value_float)
|
||||
{
|
||||
const token_type integer_result = convert_integer(number_type, num_begin, num_end);
|
||||
if (integer_result != token_type::uninitialized)
|
||||
{
|
||||
return integer_result;
|
||||
value_integer = static_cast<number_integer_t>(x);
|
||||
if (value_integer == x)
|
||||
{
|
||||
return token_type::value_integer;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// this code is reached if we parse a floating-point number or if an
|
||||
// integer conversion above overflowed. Prefer std::from_chars
|
||||
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
|
||||
// otherwise the exact Clinger fast path (double only); otherwise the
|
||||
// locale-aware strtof/strtod.
|
||||
if (parse_float_from_chars(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
// Skipping a fast path that cannot succeed is lossless and saves a full
|
||||
// extra pass over the token's bytes, which otherwise shows up on
|
||||
// high-precision inputs such as canada.json
|
||||
if (mantissa_fits_clinger(mantissa_end)
|
||||
&& parse_float_fast(num_begin, num_end, decimal_point_char, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
// integer conversion above failed
|
||||
strtof(value_float, token_buffer.data(), &endptr);
|
||||
|
||||
// we checked the number format before
|
||||
@@ -1504,158 +1326,6 @@ scan_number_done:
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief contiguous fast path for scanning a number
|
||||
|
||||
Parses the whole number token straight from the input buffer, avoiding the
|
||||
per-character get()/add() of scan_number(). On success it fills token_buffer
|
||||
(with the locale decimal point substituted, as scan_number() does) and
|
||||
returns the token type. On anything it does not fully recognize as a
|
||||
well-formed number it makes no state change and returns
|
||||
token_type::uninitialized, so the caller falls back to scan_number(), which
|
||||
then produces the exact diagnostic. @a current is the first digit or the
|
||||
leading minus (already read); the remaining bytes are taken from the adapter.
|
||||
*/
|
||||
token_type scan_number_bulk_contiguous()
|
||||
{
|
||||
// a pending unget offsets the buffer position from current; fall back
|
||||
if (next_unget)
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
const std::size_t rem = ia.bulk_remaining();
|
||||
if (rem == 0)
|
||||
{
|
||||
// the first digit is the last input byte; let scan_number() finish
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
// the byte before the next unread one is current (contiguous input)
|
||||
const char* const data = reinterpret_cast<const char*>(ia.bulk_data()) - 1;
|
||||
const std::size_t avail = rem + 1;
|
||||
|
||||
// validate + classify the number extent (mirrors scan_number()'s grammar)
|
||||
std::size_t i = 0;
|
||||
std::size_t dot_index = std::string::npos;
|
||||
token_type number_type = token_type::value_unsigned;
|
||||
if (data[0] == '-')
|
||||
{
|
||||
number_type = token_type::value_integer;
|
||||
i = 1;
|
||||
if (i >= avail)
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
}
|
||||
if (data[i] == '0')
|
||||
{
|
||||
++i;
|
||||
}
|
||||
else if (data[i] >= '1' && data[i] <= '9')
|
||||
{
|
||||
++i;
|
||||
while (i < avail && data[i] >= '0' && data[i] <= '9')
|
||||
{
|
||||
++i;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
if (i < avail && data[i] == '.')
|
||||
{
|
||||
number_type = token_type::value_float;
|
||||
dot_index = i;
|
||||
++i;
|
||||
if (i >= avail || !(data[i] >= '0' && data[i] <= '9'))
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
while (i < avail && data[i] >= '0' && data[i] <= '9')
|
||||
{
|
||||
++i;
|
||||
}
|
||||
}
|
||||
// the mantissa ends here, whether or not an exponent part follows
|
||||
const std::size_t mantissa_end = i;
|
||||
if (i < avail && (data[i] == 'e' || data[i] == 'E'))
|
||||
{
|
||||
number_type = token_type::value_float;
|
||||
++i;
|
||||
if (i < avail && (data[i] == '+' || data[i] == '-'))
|
||||
{
|
||||
++i;
|
||||
}
|
||||
if (i >= avail || !(data[i] >= '0' && data[i] <= '9'))
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
while (i < avail && data[i] >= '0' && data[i] <= '9')
|
||||
{
|
||||
++i;
|
||||
}
|
||||
}
|
||||
const std::size_t len = i;
|
||||
|
||||
// reset() records where this token starts (for diagnostics), so it has
|
||||
// to run before the input position advances below
|
||||
reset();
|
||||
|
||||
// An integer token needs no token_buffer: the SAX callbacks for
|
||||
// number_integer/number_unsigned take only the value, and the overflow
|
||||
// diagnostic rebuilds the text from the input. Convert straight from the
|
||||
// input buffer and leave token_buffer empty. (JSON_DIAGNOSTIC_POSITIONS
|
||||
// derives a number's start position from get_string().size(), so there
|
||||
// the token still has to be materialized.)
|
||||
#if !JSON_DIAGNOSTIC_POSITIONS
|
||||
if (number_type != token_type::value_float)
|
||||
{
|
||||
const token_type integer_result = convert_integer(number_type, data, data + len);
|
||||
if (JSON_HEDLEY_LIKELY(integer_result != token_type::uninitialized))
|
||||
{
|
||||
ia.bulk_skip(len - 1);
|
||||
position.chars_read_total += (len - 1);
|
||||
position.chars_read_current_line += (len - 1);
|
||||
return integer_result;
|
||||
}
|
||||
// The value does not fit an integer, so this token converts as a
|
||||
// float. Recording that here keeps convert_number() below from
|
||||
// repeating the integer attempt that just failed.
|
||||
number_type = token_type::value_float;
|
||||
}
|
||||
#endif
|
||||
|
||||
// materialize the token exactly as scan_number() would, substituting the
|
||||
// locale decimal point so convert_number()'s strtof fallback stays valid.
|
||||
// reset() already cleared token_buffer, so append() fills it (assign() is
|
||||
// avoided because custom string_t types need not provide it)
|
||||
token_buffer.append(reinterpret_cast<const typename string_t::value_type*>(data), len);
|
||||
if (dot_index != std::string::npos)
|
||||
{
|
||||
token_buffer[dot_index] = static_cast<typename string_t::value_type>(decimal_point_char);
|
||||
decimal_point_position = dot_index;
|
||||
}
|
||||
|
||||
ia.bulk_skip(len - 1);
|
||||
position.chars_read_total += (len - 1);
|
||||
position.chars_read_current_line += (len - 1);
|
||||
|
||||
return convert_number(number_type, mantissa_end);
|
||||
}
|
||||
|
||||
/// contiguous input: try the number fast path, else the byte-path scanner
|
||||
token_type scan_number_dispatch(std::true_type /*bulk*/)
|
||||
{
|
||||
const token_type t = scan_number_bulk_contiguous();
|
||||
return (t != token_type::uninitialized) ? t : scan_number();
|
||||
}
|
||||
|
||||
/// streaming input: always use the byte-path scanner
|
||||
token_type scan_number_dispatch(std::false_type /*bulk*/)
|
||||
{
|
||||
return scan_number();
|
||||
}
|
||||
|
||||
/*!
|
||||
@param[in] literal_text the literal text to expect
|
||||
@param[in] length the length of the passed literal text
|
||||
@@ -1743,9 +1413,6 @@ scan_number_done:
|
||||
if (current == '\n')
|
||||
{
|
||||
++position.lines_read;
|
||||
// remember the column the newline was read at: chars_read_current_line
|
||||
// is about to be cleared, and a matching unget() cannot reconstruct it
|
||||
chars_read_before_newline = position.chars_read_current_line;
|
||||
position.chars_read_current_line = 0;
|
||||
}
|
||||
|
||||
@@ -1779,20 +1446,12 @@ scan_number_done:
|
||||
--position.chars_read_total;
|
||||
|
||||
// in case we "unget" a newline, we have to also decrement the lines_read
|
||||
// and restore the column that get() cleared when it saw the newline;
|
||||
// chars_read_current_line == 0 can only mean the last get() read one
|
||||
if (position.chars_read_current_line == 0)
|
||||
{
|
||||
if (position.lines_read > 0)
|
||||
{
|
||||
--position.lines_read;
|
||||
}
|
||||
|
||||
// chars_read_before_newline counts the newline itself, which is the
|
||||
// character being ungotten, hence the -1
|
||||
position.chars_read_current_line = (chars_read_before_newline > 0)
|
||||
? chars_read_before_newline - 1
|
||||
: 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -2035,7 +1694,7 @@ scan_number_done:
|
||||
case '7':
|
||||
case '8':
|
||||
case '9':
|
||||
return scan_number_dispatch(std::integral_constant<bool, bulk_scan> {});
|
||||
return scan_number();
|
||||
|
||||
// end of input (the null byte is needed when parsing from
|
||||
// string literals)
|
||||
@@ -2066,10 +1725,6 @@ scan_number_done:
|
||||
/// the start position of the current token
|
||||
position_t position {};
|
||||
|
||||
/// the value chars_read_current_line had when the last newline was read, so
|
||||
/// that unget() can restore the column instead of leaving it at 0
|
||||
std::size_t chars_read_before_newline = 0;
|
||||
|
||||
/// raw input token string for error messages; only populated for streaming
|
||||
/// adapters (seekable adapters reconstruct it lazily via token_string_start)
|
||||
std::vector<char_type> token_string {};
|
||||
|
||||
@@ -1,302 +0,0 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <limits> // numeric_limits
|
||||
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
// std::from_chars lives in <charconv>, but being in C++17 mode does not
|
||||
// guarantee the header exists: GCC 7 sets __cplusplus to C++17 yet ships no
|
||||
// <charconv> (added in GCC 8; floating-point support in GCC 11). Guard the
|
||||
// include with __has_include so such toolchains fall back to the scalar path.
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__has_include)
|
||||
#if __has_include(<charconv>)
|
||||
#include <charconv> // from_chars (only used when __cpp_lib_to_chars is defined)
|
||||
#include <system_error> // errc
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// This file contains the value-conversion helpers used by the lexer to turn an
|
||||
// already-validated number token into a value, without the locale/errno
|
||||
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
|
||||
// stays focused on scanning; see lexer::convert_number().
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
|
||||
/*!
|
||||
@brief fast integer parser for an already-validated unsigned integer
|
||||
|
||||
The number scanner has already checked that [first, last) is a valid JSON
|
||||
integer, so this only needs to accumulate the digits and detect overflow. This
|
||||
avoids the locale/errno machinery of std::strtoull, which dominates
|
||||
integer-heavy inputs.
|
||||
|
||||
@param[in] first pointer to the first character (a digit)
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] value the parsed value on success
|
||||
@return true if the value fit into @a NumberUnsignedType; false on overflow, in
|
||||
which case the caller falls back to floating-point parsing (matching the
|
||||
previous std::strtoull behavior)
|
||||
*/
|
||||
template<typename NumberUnsignedType>
|
||||
bool parse_integer_unsigned(const char* first, const char* last, NumberUnsignedType& value) noexcept
|
||||
{
|
||||
// accumulate in the widest unsigned type used by the previous strtoull
|
||||
// path so the overflow behavior is unchanged for custom number types
|
||||
std::uint64_t x = 0;
|
||||
constexpr std::uint64_t cutoff = (std::numeric_limits<std::uint64_t>::max)() / 10u;
|
||||
constexpr std::uint64_t cutlim = (std::numeric_limits<std::uint64_t>::max)() % 10u;
|
||||
for (const char* p = first; p != last; ++p)
|
||||
{
|
||||
const auto digit = static_cast<std::uint64_t>(static_cast<unsigned char>(*p) - static_cast<unsigned char>('0'));
|
||||
if (JSON_HEDLEY_UNLIKELY(x > cutoff || (x == cutoff && digit > cutlim)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
x = (x * 10u) + digit;
|
||||
}
|
||||
value = static_cast<NumberUnsignedType>(x);
|
||||
// reject values that do not round-trip into a narrower NumberUnsignedType
|
||||
return static_cast<std::uint64_t>(value) == x;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief fast integer parser for an already-validated negative integer
|
||||
|
||||
@param[in] first pointer to the leading '-'
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] value the parsed (negative) value on success
|
||||
@return true on success; false on overflow (caller falls back to float)
|
||||
*/
|
||||
template<typename NumberIntegerType>
|
||||
bool parse_integer_signed(const char* first, const char* last, NumberIntegerType& value) noexcept
|
||||
{
|
||||
// the state machine only reaches the signed path via a leading '-'
|
||||
JSON_ASSERT(first != last && *first == '-');
|
||||
std::uint64_t magnitude = 0;
|
||||
// |INT64_MIN| == INT64_MAX + 1; this is the largest admissible magnitude
|
||||
constexpr std::uint64_t limit = static_cast<std::uint64_t>((std::numeric_limits<std::int64_t>::max)()) + 1u;
|
||||
for (const char* p = first + 1; p != last; ++p)
|
||||
{
|
||||
const auto digit = static_cast<std::uint64_t>(static_cast<unsigned char>(*p) - static_cast<unsigned char>('0'));
|
||||
if (JSON_HEDLEY_UNLIKELY(magnitude > (limit - digit) / 10u))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
magnitude = (magnitude * 10u) + digit;
|
||||
}
|
||||
const std::int64_t x = (magnitude == limit)
|
||||
? (std::numeric_limits<std::int64_t>::min)()
|
||||
: -static_cast<std::int64_t>(magnitude);
|
||||
value = static_cast<NumberIntegerType>(x);
|
||||
// reject values that do not round-trip into a narrower NumberIntegerType
|
||||
return static_cast<std::int64_t>(value) == x;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief exact fast path for parsing a `double` (Clinger's algorithm)
|
||||
|
||||
For the common case - at most 19 significant digits, a decimal exponent in
|
||||
[-22, 22], and a significand below 2^53 - the value equals significand *
|
||||
10^exp computed in IEEE-754 double arithmetic, which is exact under
|
||||
round-to-nearest because both operands are exactly representable. This is the
|
||||
same fast path used by fast_float/simdjson; the general cases are left to
|
||||
std::strtod. The parser only activates for number_float_t == double; float and
|
||||
long double keep the std::strtof/std::strtold paths (see the templated overload
|
||||
below).
|
||||
|
||||
@param[in] first pointer to the first character of the number
|
||||
@param[in] last pointer past the last character
|
||||
@param[in] decimal_point the (locale-dependent) decimal point character
|
||||
@param[out] out the parsed value on success
|
||||
@return true if the value was parsed exactly; false to fall back to strtod
|
||||
*/
|
||||
template<typename DecimalPointType>
|
||||
bool parse_float_fast(const char* first, const char* last, DecimalPointType decimal_point, double& out) noexcept
|
||||
{
|
||||
#if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0
|
||||
// Clinger's fast path is only exact when double operations are evaluated in
|
||||
// true double precision. On platforms that keep intermediates in extended
|
||||
// precision (e.g. the x87 FPU on 32-bit x86, where FLT_EVAL_METHOD == 2) the
|
||||
// single significand * 10^scale step is double-rounded and can be 1 ULP off,
|
||||
// so decline and let the caller fall back to the correctly-rounded
|
||||
// std::from_chars / std::strtod path.
|
||||
static_cast<void>(first);
|
||||
static_cast<void>(last);
|
||||
static_cast<void>(decimal_point);
|
||||
static_cast<void>(out);
|
||||
return false;
|
||||
#else
|
||||
static const std::array<double, 23> powers_of_ten =
|
||||
{
|
||||
{
|
||||
1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11,
|
||||
1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22
|
||||
}
|
||||
};
|
||||
|
||||
const char* p = first;
|
||||
bool negative = false;
|
||||
if (p != last && (*p == '-' || *p == '+'))
|
||||
{
|
||||
negative = (*p == '-');
|
||||
++p;
|
||||
}
|
||||
|
||||
std::uint64_t significand = 0;
|
||||
int num_digits = 0;
|
||||
int fractional_digits = 0;
|
||||
bool seen_dot = false;
|
||||
bool any_digit = false;
|
||||
for (; p != last; ++p)
|
||||
{
|
||||
const char c = *p;
|
||||
if (c >= '0' && c <= '9')
|
||||
{
|
||||
any_digit = true;
|
||||
if (JSON_HEDLEY_UNLIKELY(num_digits >= 19))
|
||||
{
|
||||
return false; // significand may not fit into uint64_t
|
||||
}
|
||||
significand = (significand * 10u) + static_cast<std::uint64_t>(c - '0');
|
||||
++num_digits;
|
||||
fractional_digits += static_cast<int>(seen_dot);
|
||||
}
|
||||
else if (static_cast<DecimalPointType>(c) == decimal_point)
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(seen_dot))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
seen_dot = true;
|
||||
}
|
||||
else if (c == 'e' || c == 'E')
|
||||
{
|
||||
++p;
|
||||
break;
|
||||
}
|
||||
else
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!any_digit))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
int exponent = 0;
|
||||
if (p != last) // an exponent part remains
|
||||
{
|
||||
bool exp_negative = false;
|
||||
if (p != last && (*p == '-' || *p == '+'))
|
||||
{
|
||||
exp_negative = (*p == '-');
|
||||
++p;
|
||||
}
|
||||
bool any_exp_digit = false;
|
||||
for (; p != last; ++p)
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(*p < '0' || *p > '9'))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
exponent = (exponent * 10) + (*p - '0');
|
||||
any_exp_digit = true;
|
||||
if (JSON_HEDLEY_UNLIKELY(exponent > 9999))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!any_exp_digit))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if (exp_negative)
|
||||
{
|
||||
exponent = -exponent;
|
||||
}
|
||||
}
|
||||
|
||||
const int scale = exponent - fractional_digits;
|
||||
if (JSON_HEDLEY_UNLIKELY(significand >= (static_cast<std::uint64_t>(1) << 53)))
|
||||
{
|
||||
return false; // significand not exactly representable as double
|
||||
}
|
||||
|
||||
auto result = static_cast<double>(significand);
|
||||
if (scale >= 0)
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(scale > 22))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
result *= powers_of_ten[static_cast<std::size_t>(scale)];
|
||||
}
|
||||
else
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(-scale > 22))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
result /= powers_of_ten[static_cast<std::size_t>(-scale)];
|
||||
}
|
||||
out = negative ? -result : result;
|
||||
return true;
|
||||
#endif
|
||||
}
|
||||
|
||||
/// fast float path is only exact for `double`; decline for float/long double
|
||||
template<typename DecimalPointType, typename FloatType>
|
||||
bool parse_float_fast(const char* /*first*/, const char* /*last*/, DecimalPointType /*decimal_point*/, FloatType& /*out*/) noexcept
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief parse a float with std::from_chars (Eisel-Lemire) when available
|
||||
|
||||
std::from_chars is locale-independent, correctly rounded, and - via the
|
||||
Eisel-Lemire algorithm in modern standard libraries - much faster than strtod
|
||||
over the whole value range (not just the Clinger subset). It is used only when
|
||||
__cpp_lib_to_chars indicates full floating-point support and only when it
|
||||
consumes the entire token ([first, last)); a partial parse means the buffer
|
||||
uses a non-'.' locale decimal point, in which case the caller falls back to the
|
||||
locale-aware path. An under-/overflow (result_out_of_range) also declines, so
|
||||
the caller's strtod fallback supplies the well-defined ±inf/0 result the parser
|
||||
expects (side-stepping the P4168 divergence between implementations).
|
||||
|
||||
@return true if the value was parsed exactly and fully; false to fall back
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept
|
||||
{
|
||||
// JSON_HAS_CPP_17 must gate the use as well as the <charconv> include above:
|
||||
// some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even
|
||||
// in C++14 mode, where <charconv> is not included.
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars)
|
||||
const auto result = std::from_chars(first, last, out);
|
||||
return result.ec == std::errc() && result.ptr == last;
|
||||
#else
|
||||
static_cast<void>(first);
|
||||
static_cast<void>(last);
|
||||
static_cast<void>(out);
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -1,293 +0,0 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint64_t
|
||||
#include <cstring> // memcpy
|
||||
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
// Optional SIMD backend for bulk UTF-8 validation. This is an opt-in external
|
||||
// dependency: nlohmann/json itself stays header-only and the C++11 scalar
|
||||
// validator below is always available; defining JSON_USE_SIMDUTF additionally
|
||||
// requires the simdutf headers on the include path and linking the simdutf
|
||||
// library. See string_bulk_run().
|
||||
//
|
||||
// simdutf.h itself requires C++17 - it rejects older standards with an #error -
|
||||
// so the backend is only compiled in from C++17 on. Below that the macro has no
|
||||
// effect and the scalar validator is used; it accepts and rejects exactly the
|
||||
// same input, so only throughput differs. macro_scope.hpp is included above to
|
||||
// have JSON_HAS_CPP_17 available for this test.
|
||||
#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17)
|
||||
#include <simdutf.h>
|
||||
#endif
|
||||
|
||||
// This file contains the byte-level string-scanning helpers used by the lexer's
|
||||
// contiguous fast path. They operate purely on raw bytes (no dependency on the
|
||||
// lexer's template parameters) so they are free functions, keeping the lexer
|
||||
// itself focused on the state machine; see lexer::scan_string_bulk().
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
|
||||
// classify a single byte as needing individual string handling: the closing
|
||||
// quote, an escape, a control character, or a non-ASCII (UTF-8)
|
||||
// lead/continuation byte. Ordinary bytes (0x20..0x7F except '"' and '\\') are
|
||||
// copied verbatim, which the bulk scanner does 8 bytes at a time.
|
||||
inline bool is_string_special(unsigned char c) noexcept
|
||||
{
|
||||
return c == '\"' || c == '\\' || c < 0x20u || c >= 0x80u;
|
||||
}
|
||||
|
||||
// SWAR helper: return a word whose high bit is set in every byte of @a v that
|
||||
// is_string_special(); zero if the 8 bytes are all ordinary.
|
||||
inline std::uint64_t swar_string_special(std::uint64_t v) noexcept
|
||||
{
|
||||
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
||||
constexpr std::uint64_t high = 0x8080808080808080ull;
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
|
||||
const std::uint64_t has_quote = (q - ones) & ~q & high;
|
||||
const std::uint64_t has_backslash = (b - ones) & ~b & high;
|
||||
const std::uint64_t has_control = (v - 0x2020202020202020ull) & ~v & high; // < 0x20
|
||||
const std::uint64_t has_non_ascii = v & high; // >= 0x80
|
||||
return has_quote | has_backslash | has_control | has_non_ascii;
|
||||
}
|
||||
|
||||
// return the index of the first is_string_special() byte in [data, data+n), or
|
||||
// n if every byte is ordinary; scans 8 bytes at a time
|
||||
inline std::size_t find_string_special(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t word = 0;
|
||||
std::memcpy(&word, data + i, sizeof(word));
|
||||
if (swar_string_special(word) != 0)
|
||||
{
|
||||
// a special byte is in this word; locate it (endian-agnostic)
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
if (is_string_special(data[i + j]))
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
{
|
||||
if (is_string_special(data[i]))
|
||||
{
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
// classify a byte as one the serializer must NOT copy verbatim when
|
||||
// ensure_ascii is requested: the closing quote, an escape, a control character
|
||||
// (< 0x20), DEL (0x7F), or any non-ASCII byte (>= 0x80). Everything else -
|
||||
// printable ASCII except '"' and '\\' - is emitted unchanged. Note this differs
|
||||
// from is_string_special() only in that 0x7F is also a stop (it is escaped as
|
||||
// \u007f under ensure_ascii).
|
||||
inline bool is_ascii_copyable(unsigned char c) noexcept
|
||||
{
|
||||
return c >= 0x20u && c < 0x7Fu && c != '\"' && c != '\\';
|
||||
}
|
||||
|
||||
// return the index of the first byte in [data, data+n) that is NOT
|
||||
// is_ascii_copyable(), or n if every byte can be copied verbatim; scans 8 bytes
|
||||
// at a time. Used by the serializer's ensure_ascii fast path.
|
||||
inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
||||
constexpr std::uint64_t high = 0x8080808080808080ull;
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t v = 0;
|
||||
std::memcpy(&v, data + i, sizeof(v));
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
|
||||
const std::uint64_t d = v ^ 0x7F7F7F7F7F7F7F7Full; // DEL (0x7F)
|
||||
const std::uint64_t stop = ((q - ones) & ~q & high) // == '"'
|
||||
| ((b - ones) & ~b & high) // == '\\'
|
||||
| ((d - ones) & ~d & high) // == 0x7F
|
||||
| ((v - 0x2020202020202020ull) & ~v & high) // < 0x20
|
||||
| (v & high); // >= 0x80
|
||||
if (stop != 0)
|
||||
{
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
if (!is_ascii_copyable(data[i + j]))
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
{
|
||||
if (!is_ascii_copyable(data[i]))
|
||||
{
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its
|
||||
// length (2..4) only when the bytes form a *well-formed* sequence using exactly
|
||||
// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts
|
||||
// precisely what the byte path accepts. Returns 0 for anything that is invalid,
|
||||
// incomplete, or that the byte path must diagnose (the caller then defers to
|
||||
// that path, keeping error messages unchanged). Lead bytes < 0x80 are handled
|
||||
// by the caller and never passed here.
|
||||
inline std::size_t validate_one_utf8(const unsigned char* data, std::size_t avail) noexcept
|
||||
{
|
||||
const unsigned char c0 = data[0];
|
||||
if (c0 >= 0xC2 && c0 <= 0xDF) // U+0080..U+07FF
|
||||
{
|
||||
if (avail >= 2 && data[1] >= 0x80 && data[1] <= 0xBF)
|
||||
{
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
else if (c0 == 0xE0) // U+0800..U+0FFF
|
||||
{
|
||||
if (avail >= 3 && data[1] >= 0xA0 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF)
|
||||
{
|
||||
return 3;
|
||||
}
|
||||
}
|
||||
else if ((c0 >= 0xE1 && c0 <= 0xEC) || c0 == 0xEE || c0 == 0xEF) // U+1000..U+CFFF, U+E000..U+FFFF
|
||||
{
|
||||
if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF)
|
||||
{
|
||||
return 3;
|
||||
}
|
||||
}
|
||||
else if (c0 == 0xED) // U+D000..U+D7FF (excludes surrogates)
|
||||
{
|
||||
if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0x9F && data[2] >= 0x80 && data[2] <= 0xBF)
|
||||
{
|
||||
return 3;
|
||||
}
|
||||
}
|
||||
else if (c0 == 0xF0) // U+10000..U+3FFFF
|
||||
{
|
||||
if (avail >= 4 && data[1] >= 0x90 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
||||
{
|
||||
return 4;
|
||||
}
|
||||
}
|
||||
else if (c0 >= 0xF1 && c0 <= 0xF3) // U+40000..U+FFFFF
|
||||
{
|
||||
if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
||||
{
|
||||
return 4;
|
||||
}
|
||||
}
|
||||
else if (c0 == 0xF4) // U+100000..U+10FFFF
|
||||
{
|
||||
if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0x8F && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
||||
{
|
||||
return 4;
|
||||
}
|
||||
}
|
||||
return 0; // invalid, incomplete, or must be diagnosed by the byte path
|
||||
}
|
||||
|
||||
// Scalar (C++11) computation of the bulk run length: the number of leading
|
||||
// bytes in [data, data+n) that are ordinary ASCII or complete well-formed UTF-8
|
||||
// sequences, stopping before the first byte that needs individual handling (the
|
||||
// closing quote, an escape, a control character, or an ill-formed/truncated
|
||||
// sequence). ASCII is skipped 8 bytes at a time.
|
||||
inline std::size_t scalar_string_bulk_run(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
std::size_t pos = 0;
|
||||
while (pos < n)
|
||||
{
|
||||
pos += find_string_special(data + pos, n - pos);
|
||||
if (pos >= n || data[pos] < 0x80u)
|
||||
{
|
||||
break; // end of buffer, or a quote/escape/control byte
|
||||
}
|
||||
const std::size_t seq = validate_one_utf8(data + pos, n - pos);
|
||||
if (seq == 0)
|
||||
{
|
||||
break; // ill-formed or truncated: let the byte path diagnose it
|
||||
}
|
||||
pos += seq;
|
||||
}
|
||||
return pos;
|
||||
}
|
||||
|
||||
#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17)
|
||||
// Index of the first quote/escape/control byte in [data, data+n) (non-ASCII
|
||||
// bytes are *not* stops here - the whole run is handed to simdutf), or n.
|
||||
inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
||||
constexpr std::uint64_t high = 0x8080808080808080ull;
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t v = 0;
|
||||
std::memcpy(&v, data + i, sizeof(v));
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull;
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull;
|
||||
const std::uint64_t hit = ((q - ones) & ~q & high)
|
||||
| ((b - ones) & ~b & high)
|
||||
| ((v - 0x2020202020202020ull) & ~v & high);
|
||||
if (hit != 0)
|
||||
{
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
const unsigned char c = data[i + j];
|
||||
if (c == '\"' || c == '\\' || c < 0x20u)
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
{
|
||||
const unsigned char c = data[i];
|
||||
if (c == '\"' || c == '\\' || c < 0x20u)
|
||||
{
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Backend-dispatched bulk run length. With JSON_USE_SIMDUTF the run up to the
|
||||
// next delimiter is validated in one shot by simdutf; on the rare failure the
|
||||
// scalar helper recomputes the exact valid prefix so the byte path still
|
||||
// produces the precise diagnostic. Without it, the pure scalar path is used.
|
||||
inline std::size_t string_bulk_run(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17)
|
||||
const std::size_t run = find_string_delimiter(data, n);
|
||||
if (run != 0 && simdutf::validate_utf8(reinterpret_cast<const char*>(data), run))
|
||||
{
|
||||
return run;
|
||||
}
|
||||
#endif
|
||||
return scalar_string_bulk_run(data, n);
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1341,12 +1341,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
const error_handler_t error_handler = error_handler_t::strict) const
|
||||
{
|
||||
string_t result;
|
||||
detail::output_string_adapter<char, string_t> string_adapter(result);
|
||||
serializer s(string_adapter, indent_char, error_handler);
|
||||
serializer s(detail::output_adapter<char, string_t>(result), indent_char, error_handler);
|
||||
|
||||
if (indent >= 0)
|
||||
{
|
||||
s.dump(*this, true, ensure_ascii, static_cast<std::size_t>(indent));
|
||||
s.dump(*this, true, ensure_ascii, static_cast<unsigned int>(indent));
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -3574,6 +3573,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
{
|
||||
using std::swap;
|
||||
swap(*(m_data.m_value.array), other);
|
||||
set_parents();
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -3590,6 +3590,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
{
|
||||
using std::swap;
|
||||
swap(*(m_data.m_value.object), other);
|
||||
set_parents();
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -4056,8 +4057,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec
|
||||
o.width(0);
|
||||
|
||||
// do the actual serialization
|
||||
detail::output_stream_adapter<char> stream_adapter(o);
|
||||
serializer s(stream_adapter, o.fill());
|
||||
serializer s(detail::output_adapter<char>(o), o.fill());
|
||||
s.dump(j, pretty_print, false, static_cast<unsigned int>(indentation));
|
||||
return o;
|
||||
}
|
||||
|
||||
+174
-1992
File diff suppressed because it is too large
Load Diff
@@ -2,9 +2,6 @@ cmake_minimum_required(VERSION 3.13...4.0)
|
||||
|
||||
option(JSON_Valgrind "Execute test suite with Valgrind." OFF)
|
||||
option(JSON_FastTests "Skip expensive/slow tests." OFF)
|
||||
option(JSON_TestSimdutf "Build the unit tests against the simdutf UTF-8 validation backend." OFF)
|
||||
|
||||
set(JSON_SIMDUTF_VERSION 9.1.0 CACHE STRING "The simdutf version used by JSON_TestSimdutf.")
|
||||
|
||||
set(JSON_32bitTest AUTO CACHE STRING "Enable the 32bit unit test (ON/OFF/AUTO/ONLY).")
|
||||
set(JSON_TestStandards "" CACHE STRING "The list of standards to test explicitly.")
|
||||
@@ -152,71 +149,6 @@ if(test_force)
|
||||
endif()
|
||||
message(STATUS "${msg}")
|
||||
|
||||
#############################################################################
|
||||
# optionally validate UTF-8 with simdutf (JSON_USE_SIMDUTF)
|
||||
#############################################################################
|
||||
|
||||
# The simdutf backend is opt-in and not vendored, so it is fetched here rather
|
||||
# than being a checked-in dependency. Everything below hangs off test_main,
|
||||
# whose usage requirements every test target inherits; the library target and
|
||||
# the installed CMake package are deliberately left untouched.
|
||||
if (JSON_TestSimdutf)
|
||||
# simdutf requires C++17, both to compile itself and to be reachable from
|
||||
# the library, which keeps its scalar validator below that. Find a tested
|
||||
# standard that satisfies it.
|
||||
set(simdutf_standard "")
|
||||
foreach(cxx_standard ${test_cxx_standards})
|
||||
if(NOT cxx_standard LESS 17 AND compiler_supports_cpp_${cxx_standard})
|
||||
set(simdutf_standard ${cxx_standard})
|
||||
break()
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
if("${simdutf_standard}" STREQUAL "")
|
||||
# Building simdutf would fail outright without a C++17 compiler, and
|
||||
# even with one it would go unused if no C++17-or-later standard is
|
||||
# tested. Say so and fall back to the scalar validator rather than
|
||||
# failing the build.
|
||||
if(NOT compiler_supports_cpp_17)
|
||||
set(simdutf_reason "the compiler does not support C++17")
|
||||
else()
|
||||
set(simdutf_reason "no tested standard is C++17 or later (testing ${msg_standards})")
|
||||
endif()
|
||||
message(WARNING
|
||||
"JSON_TestSimdutf is enabled, but ${simdutf_reason}. simdutf requires C++17, so it "
|
||||
"is not fetched and JSON_USE_SIMDUTF is not defined: the tests run against the "
|
||||
"built-in scalar UTF-8 validator instead. Set JSON_TestStandards to include 17 or "
|
||||
"later, or build with a compiler that supports C++17.")
|
||||
else()
|
||||
if (CMAKE_VERSION VERSION_LESS 3.18)
|
||||
message(FATAL_ERROR "JSON_TestSimdutf requires CMake 3.18 or later (simdutf's minimum).")
|
||||
endif()
|
||||
|
||||
include(FetchContent)
|
||||
|
||||
# simdutf builds its tests and tools by default, and its tests pull
|
||||
# further dependencies of their own; only the library is needed here
|
||||
set(SIMDUTF_TESTS OFF CACHE BOOL "" FORCE)
|
||||
set(SIMDUTF_TOOLS OFF CACHE BOOL "" FORCE)
|
||||
set(SIMDUTF_BENCHMARKS OFF CACHE BOOL "" FORCE)
|
||||
set(SIMDUTF_ICONV OFF CACHE BOOL "" FORCE)
|
||||
|
||||
FetchContent_Declare(simdutf
|
||||
URL https://github.com/simdutf/simdutf/archive/refs/tags/v${JSON_SIMDUTF_VERSION}.tar.gz
|
||||
DOWNLOAD_EXTRACT_TIMESTAMP TRUE
|
||||
)
|
||||
FetchContent_MakeAvailable(simdutf)
|
||||
|
||||
target_compile_definitions(test_main PUBLIC JSON_USE_SIMDUTF)
|
||||
target_link_libraries(test_main PUBLIC simdutf::simdutf)
|
||||
|
||||
# simdutf.h requires C++17; below that the library keeps its scalar
|
||||
# validator, so any C++11/14 test targets exercise the fallback and the
|
||||
# C++17-and-later ones exercise simdutf. Both must agree.
|
||||
message(STATUS "UTF-8 validation delegated to simdutf ${JSON_SIMDUTF_VERSION} for C++17 and later (JSON_USE_SIMDUTF)")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# *DO* use json_test_set_test_options() above this line
|
||||
|
||||
json_test_should_build_32bit_test(json_32bit_test json_32bit_test_only "${JSON_32bitTest}")
|
||||
|
||||
@@ -81,44 +81,6 @@ BENCHMARK_CAPTURE(ParseString, signed_ints, TEST_DATA_DIRECTORY "/regressi
|
||||
BENCHMARK_CAPTURE(ParseString, unsigned_ints, TEST_DATA_DIRECTORY "/regression/unsigned_ints.json");
|
||||
BENCHMARK_CAPTURE(ParseString, small_signed_ints, TEST_DATA_DIRECTORY "/regression/small_signed_ints.json");
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
// parse pretty-printed JSON from string
|
||||
//
|
||||
// Every file in the corpus above is minified or only lightly spaced, so none of
|
||||
// them exercise the lexer's whitespace handling. Real-world JSON is frequently
|
||||
// indented - configuration files, pretty-printed API responses, anything kept
|
||||
// under version control - where insignificant whitespace can outweigh the data.
|
||||
// Re-serializing a document with an indentation and parsing that keeps the
|
||||
// content identical to the ParseString row above, so the pair isolates the cost
|
||||
// of the whitespace alone.
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
static void ParseIndented(benchmark::State& state, const char* filename, int indent)
|
||||
{
|
||||
std::ifstream f(filename);
|
||||
std::string str((std::istreambuf_iterator<char>(f)), std::istreambuf_iterator<char>());
|
||||
const std::string indented = json::parse(str).dump(indent);
|
||||
|
||||
while (state.KeepRunning())
|
||||
{
|
||||
state.PauseTiming();
|
||||
auto* j = new json();
|
||||
state.ResumeTiming();
|
||||
|
||||
*j = json::parse(indented);
|
||||
|
||||
state.PauseTiming();
|
||||
delete j;
|
||||
state.ResumeTiming();
|
||||
}
|
||||
|
||||
state.SetBytesProcessed(state.iterations() * indented.size());
|
||||
}
|
||||
BENCHMARK_CAPTURE(ParseIndented, jeopardy / 4, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", 4);
|
||||
BENCHMARK_CAPTURE(ParseIndented, canada / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", 4);
|
||||
BENCHMARK_CAPTURE(ParseIndented, citm_catalog / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", 4);
|
||||
BENCHMARK_CAPTURE(ParseIndented, twitter / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", 4);
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
// serialize JSON
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
@@ -12,11 +12,6 @@
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
#include <cstdlib> // strtod
|
||||
#include <sstream> // stringstream
|
||||
#include <string> // string
|
||||
#include <vector> // vector
|
||||
|
||||
namespace
|
||||
{
|
||||
// shortcut to scan a string literal
|
||||
@@ -229,431 +224,3 @@ TEST_CASE("lexer class")
|
||||
CHECK((scan_string("/**//**//**/", true) == json::lexer::token_type::end_of_input));
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("lexer number fast path")
|
||||
{
|
||||
// The contiguous fast path (used for pointer/string input) must agree with
|
||||
// the streaming byte path (used for std::istream) on token type, numeric
|
||||
// value, and round-trip text for every well-formed number, and reject the
|
||||
// same malformed numbers with the same message.
|
||||
SECTION("contiguous vs streaming parity")
|
||||
{
|
||||
const std::vector<std::string> numbers =
|
||||
{
|
||||
"0", "-0", "1", "-1", "42", "-42", "10", "100", "1234567890",
|
||||
"0.0", "-0.0", "3.14", "-3.14", "0.5", "-0.001", "123.456789",
|
||||
"1e0", "1E0", "1e10", "1e-10", "1e+10", "1.5e3", "-2.5E-4",
|
||||
"9223372036854775807", // INT64_MAX -> unsigned
|
||||
"9223372036854775808", // INT64_MAX + 1 -> unsigned
|
||||
"18446744073709551615", // UINT64_MAX -> unsigned
|
||||
"18446744073709551616", // UINT64_MAX + 1 -> float
|
||||
"-9223372036854775808", // INT64_MIN -> integer
|
||||
"-9223372036854775809", // INT64_MIN - 1 -> float
|
||||
"123456789012345678901234567890", // huge -> float
|
||||
"0.30000000000000004", "2.2250738585072014e-308", "1e308",
|
||||
// high-precision / wide-exponent values that exercise the
|
||||
// std::from_chars (Eisel-Lemire) path beyond the Clinger subset
|
||||
"1.7976931348623157e308", "1.2345678901234567e-250",
|
||||
"9007199254740993", "5e-324", "1e-320"
|
||||
};
|
||||
|
||||
for (const auto& n : numbers)
|
||||
{
|
||||
const std::string doc = "[" + n + "]";
|
||||
|
||||
// contiguous fast path
|
||||
const json a = json::parse(doc);
|
||||
// streaming byte path
|
||||
std::stringstream ss(doc);
|
||||
const json b = json::parse(ss);
|
||||
|
||||
CAPTURE(n);
|
||||
CHECK(a == b);
|
||||
CHECK(a.dump() == b.dump());
|
||||
CHECK(a[0].type() == b[0].type());
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("significant-digit gate for the Clinger fast path")
|
||||
{
|
||||
// Clinger's fast path needs a significand below 2^53, so it cannot
|
||||
// succeed once the mantissa has 17 or more significant digits (the
|
||||
// significand would be at least 10^16). The lexer skips the attempt
|
||||
// there. That is only allowed to save work: every value must still come
|
||||
// out bit-exactly, and both scanners must agree. In particular the gate
|
||||
// must not fire for tokens whose leading zeros merely look like extra
|
||||
// digits - "0.1234567890123456" has 16 significant digits, not 17.
|
||||
const std::vector<std::string> numbers =
|
||||
{
|
||||
"1234567890123456", // 16 significant digits
|
||||
"12345678901234567", // 17 -> attempt skipped
|
||||
"123456789012345678", // 18 -> attempt skipped
|
||||
"0.1234567890123456", // 16: the leading "0" is not significant
|
||||
"0.12345678901234567", // 17
|
||||
"0.00000000000000001", // 1, in a long token
|
||||
"0.000000000000000012345678901234", // 14, in a long token
|
||||
"-0.0000000000000000000001", // 1, negative
|
||||
"1.0000000000000000", // 17: trailing zeros are significant here
|
||||
"10000000000000000", // 17
|
||||
"9007199254740992", // 2^53
|
||||
"9007199254740993", // 2^53 + 1
|
||||
"-65.613616999999977", // canada.json shape
|
||||
"1.2345678901234567e-250", // 17 with an exponent
|
||||
"1.234567890123456e-250", // 16 with an exponent
|
||||
"1e10", "0.0", "-0.0", "0e0", "0.000123"
|
||||
};
|
||||
|
||||
for (const auto& n : numbers)
|
||||
{
|
||||
CAPTURE(n);
|
||||
const std::string doc = "[" + n + "]";
|
||||
|
||||
const json a = json::parse(doc); // contiguous fast path
|
||||
std::stringstream ss(doc);
|
||||
const json b = json::parse(ss); // streaming byte path
|
||||
|
||||
CHECK(a[0].type() == b[0].type());
|
||||
CHECK(a == b);
|
||||
|
||||
if (a[0].is_number_float())
|
||||
{
|
||||
const double expected = std::strtod(n.c_str(), nullptr);
|
||||
CHECK(a[0].get<double>() == expected);
|
||||
CHECK(b[0].get<double>() == expected);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("token type classification")
|
||||
{
|
||||
CHECK((scan_string("0") == json::lexer::token_type::value_unsigned));
|
||||
CHECK((scan_string("-1") == json::lexer::token_type::value_integer));
|
||||
CHECK((scan_string("1.5") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("1e5") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("18446744073709551615") == json::lexer::token_type::value_unsigned));
|
||||
CHECK((scan_string("18446744073709551616") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("-9223372036854775808") == json::lexer::token_type::value_integer));
|
||||
CHECK((scan_string("-9223372036854775809") == json::lexer::token_type::value_float));
|
||||
}
|
||||
|
||||
SECTION("malformed numbers are rejected identically")
|
||||
{
|
||||
for (const char* bad :
|
||||
{"-", "1.", "1e", "1e+", "1.2e", "01", "-01", "1..2", "1.2.3"
|
||||
})
|
||||
{
|
||||
CAPTURE(bad);
|
||||
// the contiguous fast path must decline and let the byte path report
|
||||
const std::string doc = std::string("[") + bad + "]";
|
||||
CHECK_FALSE(json::accept(doc));
|
||||
std::stringstream ss(doc);
|
||||
CHECK_FALSE(json::accept(ss));
|
||||
}
|
||||
}
|
||||
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
// these sections parse invalid input, which aborts when exceptions are off
|
||||
SECTION("exhaustive grammar parity with the streaming path")
|
||||
{
|
||||
// The JSON number grammar is encoded twice: once as the scan_number()
|
||||
// state machine and once as the contiguous fast path. Enumerate every
|
||||
// short string over the number alphabet and require the two encodings to
|
||||
// agree exactly - on acceptance, on the reported error, and on the parsed
|
||||
// value - so they cannot drift apart.
|
||||
const std::string alphabet = "01.eE+-";
|
||||
|
||||
// full outcome of parsing @a doc, so a mismatch in type, value, or error
|
||||
// message is caught, not just a mismatch in acceptance
|
||||
const auto outcome = [](const std::string & doc, bool streaming) -> std::string
|
||||
{
|
||||
try
|
||||
{
|
||||
if (streaming)
|
||||
{
|
||||
std::stringstream ss(doc);
|
||||
const json j = json::parse(ss);
|
||||
return std::string(j[0].type_name()) + '|' + j.dump();
|
||||
}
|
||||
const json j = json::parse(doc);
|
||||
return std::string(j[0].type_name()) + '|' + j.dump();
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
return {e.what()};
|
||||
}
|
||||
};
|
||||
|
||||
std::vector<std::string> mismatches;
|
||||
std::vector<std::string> tokens{""};
|
||||
for (std::size_t length = 1; length <= 4; ++length)
|
||||
{
|
||||
std::vector<std::string> next;
|
||||
next.reserve(tokens.size() * alphabet.size());
|
||||
for (const auto& prefix : tokens)
|
||||
{
|
||||
for (const char c : alphabet)
|
||||
{
|
||||
next.push_back(prefix + c);
|
||||
}
|
||||
}
|
||||
tokens = next;
|
||||
|
||||
for (const auto& token : tokens)
|
||||
{
|
||||
const std::string doc = "[" + token + "]";
|
||||
if (outcome(doc, false) != outcome(doc, true))
|
||||
{
|
||||
mismatches.push_back(doc);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 7 + 49 + 343 + 2401 tokens
|
||||
CHECK(tokens.size() == 2401);
|
||||
CAPTURE(mismatches);
|
||||
CHECK(mismatches.empty());
|
||||
}
|
||||
|
||||
SECTION("error positions match the streaming path")
|
||||
{
|
||||
// Rejecting identically is not enough: the fast path must also report the
|
||||
// error at the same position as the byte path. A number directly followed
|
||||
// by a newline is the interesting case, because the byte path reaches the
|
||||
// newline (which resets the column) and then ungets it.
|
||||
// returns the parse_error message, or "" if the document parsed
|
||||
const auto contiguous_error = [](const std::string & doc) -> std::string
|
||||
{
|
||||
try
|
||||
{
|
||||
const json j = json::parse(doc);
|
||||
static_cast<void>(j);
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
return {e.what()};
|
||||
}
|
||||
return {};
|
||||
};
|
||||
const auto streaming_error = [](const std::string & doc) -> std::string
|
||||
{
|
||||
try
|
||||
{
|
||||
std::stringstream ss(doc);
|
||||
const json j = json::parse(ss);
|
||||
static_cast<void>(j);
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
return {e.what()};
|
||||
}
|
||||
return {};
|
||||
};
|
||||
|
||||
for (const char* bad :
|
||||
{"[01\n]", "[00\n]", "[-01\n]", "{1\n}", "[1\n2]", "[1.2.3\n]",
|
||||
"[1 \n2]", "[\n1\n2]", "1\n2", "[01\r\n]", "[1e\n]", "[-\n]"
|
||||
})
|
||||
{
|
||||
CAPTURE(bad);
|
||||
const std::string doc = bad;
|
||||
const std::string contiguous_what = contiguous_error(doc);
|
||||
|
||||
CHECK_FALSE(contiguous_what.empty());
|
||||
CHECK(contiguous_what == streaming_error(doc));
|
||||
}
|
||||
|
||||
// A number terminated by a newline must report the same position as the
|
||||
// same number terminated by anything else: scan_number() reads the
|
||||
// terminator and ungets it, so the reported column is the one reached
|
||||
// after the number's last character - not the 0 that an unget() across
|
||||
// the newline used to leave behind.
|
||||
CHECK(contiguous_error("[01\n]") == contiguous_error("[01 ]"));
|
||||
CHECK(contiguous_error("[01\n]") ==
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 3: "
|
||||
"syntax error while parsing array - unexpected number literal; expected ']'");
|
||||
|
||||
// the same for a multi-character token, where the column of the last
|
||||
// character (the '3' of "-2.5e3") differs from the column it starts at
|
||||
CHECK(contiguous_error("null -2.5e3\nfalse") == contiguous_error("null -2.5e3 false"));
|
||||
CHECK(contiguous_error("null -2.5e3\nfalse") ==
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 11: "
|
||||
"syntax error while parsing value - unexpected number literal; expected end of input");
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
TEST_CASE("lexer string fast path")
|
||||
{
|
||||
// Build a byte string from explicit values: a hex escape in a string
|
||||
// literal swallows every following hex digit, which makes sequences like
|
||||
// "\xC3\xA9b" mean something other than they look like.
|
||||
const auto bytes = [](std::initializer_list<int> values)
|
||||
{
|
||||
std::string result;
|
||||
for (const int value : values)
|
||||
{
|
||||
result.push_back(static_cast<char>(value));
|
||||
}
|
||||
return result;
|
||||
};
|
||||
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
// the full outcome of parsing @a doc: the parsed value, or the exact error
|
||||
// message, so a mismatch in either is caught. Only usable with exceptions
|
||||
// on: parsing invalid input aborts when they are off.
|
||||
const auto outcome = [](const std::string & doc, bool streaming) -> std::string
|
||||
{
|
||||
try
|
||||
{
|
||||
if (streaming)
|
||||
{
|
||||
std::stringstream ss(doc);
|
||||
const json j = json::parse(ss);
|
||||
return j.dump();
|
||||
}
|
||||
const json j = json::parse(doc);
|
||||
return j.dump();
|
||||
}
|
||||
// not just parse_error: if a bulk scanner ever let ill-formed UTF-8
|
||||
// through, dump() would throw type_error.316, and that has to surface
|
||||
// as a reported mismatch rather than as an uncaught exception
|
||||
catch (const json::exception& e)
|
||||
{
|
||||
return {e.what()};
|
||||
}
|
||||
};
|
||||
#endif
|
||||
|
||||
// once at the start of the string, once past the first 8-byte SWAR word, so
|
||||
// the bulk scanner sees each case with and without a run behind it
|
||||
const std::vector<std::size_t> offsets{0, 9};
|
||||
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
SECTION("exhaustive contiguous vs streaming parity")
|
||||
{
|
||||
// ordinary ASCII, both specials, a control byte, characters that make
|
||||
// the preceding backslash a valid escape, a UTF-8 lead byte of each
|
||||
// length, a continuation byte, and a byte that is never valid
|
||||
const std::vector<std::string> alphabet =
|
||||
{
|
||||
"a", "\"", "\\", "n", "u", "0", bytes({0x01}),
|
||||
bytes({0xC3}), bytes({0xA9}), bytes({0xE4}), bytes({0xF0}),
|
||||
bytes({0x80}), bytes({0xFF})
|
||||
};
|
||||
|
||||
std::vector<std::string> mismatches;
|
||||
std::vector<std::string> tokens{""};
|
||||
for (std::size_t length = 1; length <= 3; ++length)
|
||||
{
|
||||
std::vector<std::string> next;
|
||||
next.reserve(tokens.size() * alphabet.size());
|
||||
for (const auto& prefix : tokens)
|
||||
{
|
||||
for (const auto& symbol : alphabet)
|
||||
{
|
||||
next.push_back(prefix + symbol);
|
||||
}
|
||||
}
|
||||
tokens = next;
|
||||
|
||||
for (const auto& token : tokens)
|
||||
{
|
||||
for (const std::size_t offset : offsets)
|
||||
{
|
||||
const std::string doc = "[\"" + std::string(offset, 'a') + token + "\"]";
|
||||
if (outcome(doc, false) != outcome(doc, true))
|
||||
{
|
||||
mismatches.push_back(doc);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 13 + 169 + 2197 tokens, each at two offsets
|
||||
CHECK(tokens.size() == 2197);
|
||||
CAPTURE(mismatches);
|
||||
CHECK(mismatches.empty());
|
||||
}
|
||||
|
||||
SECTION("special bytes at every offset of the SWAR stride")
|
||||
{
|
||||
// The bulk scanner consumes 8 bytes at a time and then a tail; place
|
||||
// every kind of byte that ends a run at each offset across two words,
|
||||
// so multibyte sequences also straddle the word boundary.
|
||||
const std::vector<std::string> specials =
|
||||
{
|
||||
"\"", "\\", bytes({0x01}), bytes({0x1F}), bytes({0x7F}),
|
||||
bytes({0xC3, 0xA9}), bytes({0xE4, 0xB8, 0xAD}), bytes({0xF0, 0x9F, 0x98, 0x80}),
|
||||
bytes({0xFF}), bytes({0xC3}), bytes({0xE4, 0xB8})
|
||||
};
|
||||
|
||||
std::vector<std::string> mismatches;
|
||||
for (std::size_t offset = 0; offset <= 17; ++offset)
|
||||
{
|
||||
for (const auto& special : specials)
|
||||
{
|
||||
const std::string doc = "[\"" + std::string(offset, 'a') + special + "\"]";
|
||||
if (outcome(doc, false) != outcome(doc, true))
|
||||
{
|
||||
mismatches.push_back(doc);
|
||||
}
|
||||
}
|
||||
}
|
||||
CAPTURE(mismatches);
|
||||
CHECK(mismatches.empty());
|
||||
}
|
||||
#endif
|
||||
|
||||
// json::accept() never throws, so the ranges stay covered without exceptions
|
||||
SECTION("UTF-8 ranges are accepted and rejected as documented")
|
||||
{
|
||||
// The bulk validator must accept exactly what the byte-at-a-time
|
||||
// scanner accepts, so pin the boundaries of every range it recognizes.
|
||||
// aggregate, only ever brace-initialized below; default member
|
||||
// initializers would stop it being an aggregate in C++11
|
||||
struct utf8_case // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init)
|
||||
{
|
||||
std::string sequence;
|
||||
bool valid;
|
||||
const char* description;
|
||||
};
|
||||
const std::vector<utf8_case> cases =
|
||||
{
|
||||
{bytes({0xC2, 0x80}), true, "U+0080, shortest two-byte"},
|
||||
{bytes({0xDF, 0xBF}), true, "U+07FF, longest two-byte"},
|
||||
{bytes({0xC1, 0xBF}), false, "overlong two-byte"},
|
||||
{bytes({0xC2, 0x7F}), false, "two-byte with bad continuation"},
|
||||
{bytes({0xE0, 0xA0, 0x80}), true, "U+0800, shortest three-byte"},
|
||||
{bytes({0xE0, 0x9F, 0xBF}), false, "overlong three-byte"},
|
||||
{bytes({0xED, 0x9F, 0xBF}), true, "U+D7FF, just below the surrogates"},
|
||||
{bytes({0xED, 0xA0, 0x80}), false, "surrogate U+D800"},
|
||||
{bytes({0xED, 0xBF, 0xBF}), false, "surrogate U+DFFF"},
|
||||
{bytes({0xEE, 0x80, 0x80}), true, "U+E000, just above the surrogates"},
|
||||
{bytes({0xEF, 0xBF, 0xBF}), true, "U+FFFF"},
|
||||
{bytes({0xF0, 0x90, 0x80, 0x80}), true, "U+10000, shortest four-byte"},
|
||||
{bytes({0xF0, 0x8F, 0xBF, 0xBF}), false, "overlong four-byte"},
|
||||
{bytes({0xF4, 0x8F, 0xBF, 0xBF}), true, "U+10FFFF, highest code point"},
|
||||
{bytes({0xF4, 0x90, 0x80, 0x80}), false, "above U+10FFFF"},
|
||||
{bytes({0xF5, 0x80, 0x80, 0x80}), false, "lead byte out of range"},
|
||||
{bytes({0x80}), false, "bare continuation byte"},
|
||||
{bytes({0xFF}), false, "byte that never appears in UTF-8"},
|
||||
{bytes({0xC3}), false, "truncated two-byte"},
|
||||
{bytes({0xE4, 0xB8}), false, "truncated three-byte"},
|
||||
{bytes({0xF0, 0x9F, 0x98}), false, "truncated four-byte"}
|
||||
};
|
||||
|
||||
for (const auto& test_case : cases)
|
||||
{
|
||||
CAPTURE(test_case.description);
|
||||
for (const std::size_t offset : offsets)
|
||||
{
|
||||
CAPTURE(offset);
|
||||
const std::string doc = "[\"" + std::string(offset, 'a') + test_case.sequence + "\"]";
|
||||
CHECK(json::accept(doc) == test_case.valid);
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
CHECK(outcome(doc, false) == outcome(doc, true));
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1564,58 +1564,6 @@ TEST_CASE("parser class")
|
||||
CHECK (j_filtered2 == json({{"foo", {1, 2}}}));
|
||||
}
|
||||
|
||||
SECTION("filter many members of one container")
|
||||
{
|
||||
// Rejecting a value makes the parser remove the placeholder its key
|
||||
// event stored. Locating that placeholder used to be a scan of the
|
||||
// whole parent, which made filtering a large container quadratic:
|
||||
// 128k members took ~25 s. These cases keep many members alive
|
||||
// while discarding many others, so the removal cost is the whole
|
||||
// point; they run in milliseconds when the placeholder is erased
|
||||
// directly.
|
||||
constexpr int count = 20000;
|
||||
|
||||
std::string s = "{";
|
||||
for (int i = 0; i < count; ++i)
|
||||
{
|
||||
// "a<i>" is kept, "z<i>" is discarded
|
||||
s += "\"a" + std::to_string(i) + "\":" + std::to_string(i) + ",";
|
||||
s += "\"z" + std::to_string(i) + "\":-1,";
|
||||
}
|
||||
s.back() = '}';
|
||||
|
||||
const json j_values = json::parse(s, [](int /*unused*/, json::parse_event_t e, const json & parsed) noexcept
|
||||
{
|
||||
return !(e == json::parse_event_t::value && parsed == json(-1));
|
||||
});
|
||||
|
||||
CHECK(j_values.size() == count);
|
||||
CHECK(j_values.at("a0") == json(0));
|
||||
CHECK(j_values.at("a" + std::to_string(count - 1)) == json(count - 1));
|
||||
CHECK_FALSE(j_values.contains("z0"));
|
||||
CHECK_FALSE(j_values.contains("z" + std::to_string(count - 1)));
|
||||
|
||||
// the same, but discarding whole containers rather than values,
|
||||
// which takes the end_object()/end_array() removal path
|
||||
std::string s_nested = "{";
|
||||
for (int i = 0; i < count; ++i)
|
||||
{
|
||||
s_nested += "\"a" + std::to_string(i) + "\":" + std::to_string(i) + ",";
|
||||
s_nested += "\"z" + std::to_string(i) + "\":[1,2],";
|
||||
}
|
||||
s_nested.back() = '}';
|
||||
|
||||
const json j_arrays = json::parse(s_nested, [](int /*unused*/, json::parse_event_t e, const json& /*unused*/) noexcept
|
||||
{
|
||||
return e != json::parse_event_t::array_end;
|
||||
});
|
||||
|
||||
CHECK(j_arrays.size() == count);
|
||||
CHECK(j_arrays.at("a0") == json(0));
|
||||
CHECK_FALSE(j_arrays.contains("z0"));
|
||||
CHECK_FALSE(j_arrays.contains("z" + std::to_string(count - 1)));
|
||||
}
|
||||
|
||||
SECTION("filter specific events")
|
||||
{
|
||||
SECTION("first closing event")
|
||||
|
||||
@@ -98,10 +98,8 @@ void check_escaped(const char* original, const char* escaped = "", bool ensure_a
|
||||
void check_escaped(const char* original, const char* escaped, const bool ensure_ascii)
|
||||
{
|
||||
std::stringstream ss;
|
||||
nlohmann::detail::output_stream_adapter<char> adapter(ss);
|
||||
json::serializer s(adapter, ' ');
|
||||
json::serializer s(nlohmann::detail::output_adapter<char>(ss), ' ');
|
||||
s.dump_escaped(original, ensure_ascii);
|
||||
s.flush(); // dump_escaped writes into the serializer's internal buffer
|
||||
CHECK(ss.str() == escaped);
|
||||
}
|
||||
} // namespace
|
||||
|
||||
@@ -273,5 +273,36 @@ TEST_CASE("Regression tests for extended diagnostics")
|
||||
CHECK(j1["numbers"]["two"] == 2);
|
||||
CHECK(j1["string"] == "t");
|
||||
}
|
||||
|
||||
SECTION("Regression test - swap(array_t&)/swap(object_t&) must update JSON_DIAGNOSTICS parent pointers")
|
||||
{
|
||||
// swap(array_t&)
|
||||
{
|
||||
json j = json::array();
|
||||
json::array_t arr = {json::array({1})};
|
||||
j.swap(arr);
|
||||
|
||||
// parent pointers of the moved-in elements must point into j, not
|
||||
// into the now-defunct free-standing array_t
|
||||
CHECK_THROWS_WITH_AS(j[0][0].get<std::string>(), "[json.exception.type_error.302] (/0/0) type must be string, but is number", json::type_error);
|
||||
|
||||
// must not trigger assert_invariant() in a debug/assert-enabled build
|
||||
json const k = j;
|
||||
CHECK(k == j);
|
||||
}
|
||||
|
||||
// swap(object_t&)
|
||||
{
|
||||
json o = json::object();
|
||||
json::object_t obj = {{"a", json::array({1})}};
|
||||
o.swap(obj);
|
||||
|
||||
CHECK_THROWS_WITH_AS(o["a"][0].get<std::string>(), "[json.exception.type_error.302] (/a/0) type must be string, but is number", json::type_error);
|
||||
|
||||
// must not trigger assert_invariant() in a debug/assert-enabled build
|
||||
json const p = o;
|
||||
CHECK(p == o);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -382,232 +382,3 @@ TEST_CASE("dump for basic_json with long double number_float_t")
|
||||
check_same(100.0L, 100.0);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("serialization of strings (bulk fast path)")
|
||||
{
|
||||
// These cases exercise the SWAR bulk-copy fast path in dump_escaped and the
|
||||
// internal write buffer: long runs, escapes interrupting runs, 0x7F/DEL,
|
||||
// multibyte UTF-8 under both ensure_ascii settings, and payloads larger than
|
||||
// the write buffer.
|
||||
|
||||
SECTION("long unescaped ASCII exceeds the write buffer")
|
||||
{
|
||||
const std::string big(3000, 'a');
|
||||
const json j = big;
|
||||
CHECK(j.dump() == '"' + big + '"');
|
||||
CHECK(j.dump(-1, ' ', true) == '"' + big + '"');
|
||||
// round-trips
|
||||
CHECK(json::parse(j.dump()) == j);
|
||||
}
|
||||
|
||||
SECTION("runs interrupted by escapes")
|
||||
{
|
||||
const json j = std::string(500, 'x') + "\n\"\\" + std::string(500, 'y');
|
||||
const std::string out = j.dump();
|
||||
CHECK(out == '"' + std::string(500, 'x') + "\\n\\\"\\\\" + std::string(500, 'y') + '"');
|
||||
CHECK(json::parse(out) == j);
|
||||
}
|
||||
|
||||
SECTION("DEL (0x7F) depends on ensure_ascii")
|
||||
{
|
||||
const json j = std::string("a\x7f" "b");
|
||||
CHECK(j.dump(-1, ' ', false) == "\"a\x7f" "b\""); // copied verbatim
|
||||
CHECK(j.dump(-1, ' ', true) == "\"a\\u007fb\""); // escaped
|
||||
}
|
||||
|
||||
SECTION("multibyte UTF-8 under both ensure_ascii settings")
|
||||
{
|
||||
const json j = std::string("A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z"); // A é 你 😀 Z
|
||||
// not escaping non-ASCII: bytes are copied through the bulk validator
|
||||
CHECK(j.dump(-1, ' ', false) == "\"A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z\"");
|
||||
// ensure_ascii: escaped (with a surrogate pair for the emoji)
|
||||
CHECK(j.dump(-1, ' ', true) == "\"A\\u00e9\\u4f60\\ud83d\\ude00Z\"");
|
||||
CHECK(json::parse(j.dump(-1, ' ', true)) == j);
|
||||
}
|
||||
|
||||
SECTION("many small structural writes exceed the write buffer")
|
||||
{
|
||||
json arr = json::array();
|
||||
for (int i = 0; i < 2000; ++i)
|
||||
{
|
||||
arr.push_back(i);
|
||||
}
|
||||
const std::string out = arr.dump();
|
||||
CHECK(out.front() == '[');
|
||||
CHECK(out.back() == ']');
|
||||
CHECK(json::parse(out) == arr);
|
||||
|
||||
json obj = json::object();
|
||||
for (int i = 0; i < 500; ++i)
|
||||
{
|
||||
obj["key" + std::to_string(i)] = i;
|
||||
}
|
||||
CHECK(json::parse(obj.dump()) == obj);
|
||||
CHECK(json::parse(obj.dump(2)) == obj);
|
||||
|
||||
// an array of many empty strings emits a long run of single-character
|
||||
// writes ('"', '"', ',') at shallow nesting depth, so the write buffer
|
||||
// fills and flushes mid-run without the deep recursion that would
|
||||
// overflow the stack on some debug builds
|
||||
json many_empty = json::array();
|
||||
for (int i = 0; i < 500; ++i)
|
||||
{
|
||||
many_empty.push_back("");
|
||||
}
|
||||
const std::string out2 = many_empty.dump();
|
||||
CHECK(out2.size() > 1024); // spans multiple write-buffer flushes
|
||||
CHECK(out2.front() == '[');
|
||||
CHECK(out2.back() == ']');
|
||||
CHECK(json::parse(out2) == many_empty);
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 handling is unaffected by the fast path")
|
||||
{
|
||||
const json j = std::string("valid\xff" "more");
|
||||
CHECK_THROWS_WITH_AS(j.dump(), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0xFF", json::type_error&);
|
||||
CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"valid\xef\xbf\xbd" "more\"");
|
||||
CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"valid\\ufffdmore\"");
|
||||
CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"validmore\"");
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("indentation is written straight into the write buffer")
|
||||
{
|
||||
// put_indent() memsets the indentation into the write buffer instead of
|
||||
// copying it out of a pre-grown indentation string. These cases cover an
|
||||
// indentation wider than the buffer, a non-space indentation character, and
|
||||
// nesting deep enough that the accumulated indentation spans several
|
||||
// buffer-fulls - the situations the old grow-a-string approach got wrong.
|
||||
|
||||
SECTION("indent_step wider than the write buffer")
|
||||
{
|
||||
const json j = {{"a", 1}};
|
||||
// 2000 > the 1024-byte write buffer, and > the 512 the indentation
|
||||
// string used to start at
|
||||
CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"a\": 1\n}");
|
||||
// several whole buffer-fulls, so the buffer is refilled once and then
|
||||
// flushed repeatedly
|
||||
CHECK(j.dump(5000) == "{\n" + std::string(5000, ' ') + "\"a\": 1\n}");
|
||||
CHECK(j.dump(5000, '\t') == "{\n" + std::string(5000, '\t') + "\"a\": 1\n}");
|
||||
// an exact multiple of the buffer size
|
||||
CHECK(j.dump(4096) == "{\n" + std::string(4096, ' ') + "\"a\": 1\n}");
|
||||
}
|
||||
|
||||
SECTION("a non-space indentation character is used throughout")
|
||||
{
|
||||
const json j = {{"a", 1}};
|
||||
// 600 is past the point where the indentation used to be grown, which
|
||||
// is where a hard-coded space would have shown up
|
||||
CHECK(j.dump(600, '\t') == "{\n" + std::string(600, '\t') + "\"a\": 1\n}");
|
||||
CHECK(j.dump(3, '.') == "{\n...\"a\": 1\n}");
|
||||
}
|
||||
|
||||
SECTION("accumulated indentation spans several buffer-fulls")
|
||||
{
|
||||
// five levels deep at 400 per level: the innermost value is indented by
|
||||
// 2000 characters, reached in steps that each straddle the buffer end
|
||||
json j = json::array({1});
|
||||
for (int i = 0; i < 4; ++i)
|
||||
{
|
||||
j = json::array({j});
|
||||
}
|
||||
|
||||
const std::string out = j.dump(400);
|
||||
CHECK(out.find(std::string("\n") + std::string(2000, ' ') + "1\n") != std::string::npos);
|
||||
CHECK(json::parse(out) == j);
|
||||
}
|
||||
|
||||
SECTION("indentation is unchanged for ordinary widths")
|
||||
{
|
||||
const json j = {{"a", {1, 2}}, {"b", nullptr}};
|
||||
CHECK(j.dump(2) == "{\n \"a\": [\n 1,\n 2\n ],\n \"b\": null\n}");
|
||||
CHECK(j.dump(0) == "{\n\"a\": [\n1,\n2\n],\n\"b\": null\n}");
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("serialization of deeply nested values")
|
||||
{
|
||||
// dump() descends into a bounded number of levels and writes out whatever
|
||||
// is nested deeper than that without the call stack; see
|
||||
// https://github.com/nlohmann/json/issues/5387
|
||||
|
||||
SECTION("nested deeper than the call stack could follow")
|
||||
{
|
||||
// parsing is iterative, so building these costs little
|
||||
const std::size_t depth = 100000;
|
||||
|
||||
const std::string array_text = std::string(depth, '[') + '0' + std::string(depth, ']');
|
||||
CHECK(json::parse(array_text).dump() == array_text);
|
||||
|
||||
std::string object_text;
|
||||
object_text.reserve((6 * depth) + 1);
|
||||
for (std::size_t i = 0; i < depth; ++i)
|
||||
{
|
||||
object_text += "{\"a\":";
|
||||
}
|
||||
object_text += '1';
|
||||
object_text.append(depth, '}');
|
||||
CHECK(json::parse(object_text).dump() == object_text);
|
||||
}
|
||||
|
||||
SECTION("depths around the bound of the descent")
|
||||
{
|
||||
// Cover every depth around the bound, so that the two ways of writing a
|
||||
// value are known to meet cleanly - wherever the bound is set.
|
||||
for (std::size_t d = 1; d <= 300; ++d)
|
||||
{
|
||||
CAPTURE(d);
|
||||
|
||||
const std::string array_text = std::string(d, '[') + '7' + std::string(d, ']');
|
||||
CHECK(json::parse(array_text).dump() == array_text);
|
||||
|
||||
std::string object_text;
|
||||
for (std::size_t i = 0; i < d; ++i)
|
||||
{
|
||||
object_text += "{\"k\":";
|
||||
}
|
||||
object_text += '7';
|
||||
object_text.append(d, '}');
|
||||
CHECK(json::parse(object_text).dump() == object_text);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("pretty-printing across the bound")
|
||||
{
|
||||
for (std::size_t d = 120; d <= 140; ++d)
|
||||
{
|
||||
CAPTURE(d);
|
||||
|
||||
const json j = json::parse(std::string(d, '[') + '7' + std::string(d, ']'));
|
||||
|
||||
std::string expected;
|
||||
for (std::size_t i = 0; i < d; ++i)
|
||||
{
|
||||
expected += std::string(2 * i, ' ') + "[\n";
|
||||
}
|
||||
expected += std::string(2 * d, ' ') + '7';
|
||||
for (std::size_t i = d; i > 0; --i)
|
||||
{
|
||||
expected += '\n' + std::string(2 * (i - 1), ' ') + ']';
|
||||
}
|
||||
|
||||
CHECK(j.dump(2) == expected);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("an empty container below the bound")
|
||||
{
|
||||
// an empty container is written out in full and never descended into,
|
||||
// so it must not gain a newline when it is reached iteratively
|
||||
for (std::size_t d = 125; d <= 135; ++d)
|
||||
{
|
||||
CAPTURE(d);
|
||||
|
||||
const std::string compact = std::string(d, '[') + "[]" + std::string(d, ']');
|
||||
CHECK(json::parse(compact).dump() == compact);
|
||||
|
||||
const std::string with_object = std::string(d, '[') + "{}" + std::string(d, ']');
|
||||
CHECK(json::parse(with_object).dump() == with_object);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -18,12 +18,7 @@
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint8_t
|
||||
#include <list>
|
||||
#include <string> // string
|
||||
#include <vector> // vector
|
||||
|
||||
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||
#include <iterator>
|
||||
@@ -217,66 +212,6 @@ TEST_CASE("Parse with heterogeneous iterator and sentinel types")
|
||||
CHECK(j2.at(0) == 1);
|
||||
}
|
||||
|
||||
// A type whose data() hands out raw bytes but whose size() counts something
|
||||
// else - here fixed-size records. Reading [data(), data() + size()) as bytes
|
||||
// would silently truncate the input, so data() and size() alone must not be
|
||||
// taken as evidence of contiguous byte storage.
|
||||
struct record_buffer
|
||||
{
|
||||
using value_type = std::array<char, 4>;
|
||||
|
||||
std::string bytes;
|
||||
|
||||
const char* data() const noexcept
|
||||
{
|
||||
return bytes.data();
|
||||
}
|
||||
std::size_t size() const noexcept
|
||||
{
|
||||
return bytes.size() / sizeof(value_type);
|
||||
}
|
||||
const char* begin() const noexcept
|
||||
{
|
||||
return bytes.data();
|
||||
}
|
||||
const char* end() const noexcept
|
||||
{
|
||||
return bytes.data() + bytes.size();
|
||||
}
|
||||
};
|
||||
|
||||
TEST_CASE("Contiguous byte containers take the pointer adapter")
|
||||
{
|
||||
// Containers with contiguous single-byte storage are routed through the
|
||||
// pointer-based adapter so the bulk fast paths apply in every standard, not
|
||||
// only in C++20 where the library iterators model std::contiguous_iterator.
|
||||
CHECK(nlohmann::detail::is_contiguous_byte_container<std::string>::value);
|
||||
CHECK(nlohmann::detail::is_contiguous_byte_container<std::vector<char>>::value);
|
||||
CHECK(nlohmann::detail::is_contiguous_byte_container<std::vector<std::uint8_t>>::value);
|
||||
CHECK(nlohmann::detail::is_contiguous_byte_container<std::array<char, 4>>::value);
|
||||
|
||||
// input_adapter() takes its container by forwarding reference, so the trait
|
||||
// is also asked about reference types
|
||||
CHECK(nlohmann::detail::is_contiguous_byte_container<std::string&>::value);
|
||||
CHECK(nlohmann::detail::is_contiguous_byte_container<const std::string&>::value);
|
||||
|
||||
// everything else keeps the iterator-based adapter
|
||||
CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container<std::list<char>>::value);
|
||||
CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container<std::vector<int>>::value);
|
||||
CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container<const char*>::value);
|
||||
|
||||
// including a type that has data() and size() but whose size() does not
|
||||
// count the units data() points at: its value_type says so
|
||||
CHECK_FALSE(nlohmann::detail::is_contiguous_byte_container<record_buffer>::value);
|
||||
|
||||
// and such a container still parses through its iterators, in full - taking
|
||||
// it for a byte container would stop after data() + size() bytes
|
||||
const record_buffer buffer{"[1,2,3,4,5]"};
|
||||
CHECK(buffer.data() == buffer.bytes.data());
|
||||
CHECK(buffer.size() * sizeof(record_buffer::value_type) < buffer.bytes.size());
|
||||
CHECK(json::parse(buffer) == json({1, 2, 3, 4, 5}));
|
||||
}
|
||||
|
||||
#if defined(__cpp_lib_concepts) && defined(JSON_HAS_CPP_20)
|
||||
// JSON_HAS_CPP_20 (do not remove; see note at top of file)
|
||||
TEST_CASE("Parse with std::counted_iterator and std::default_sentinel_t")
|
||||
@@ -293,180 +228,6 @@ TEST_CASE("Parse with std::counted_iterator and std::default_sentinel_t")
|
||||
const std::counted_iterator<iterator_type> first2(json_str.begin(), len);
|
||||
CHECK(json::accept(first2, std::default_sentinel));
|
||||
}
|
||||
|
||||
TEST_CASE("std::counted_iterator reaches the contiguous fast paths")
|
||||
{
|
||||
// A sized sentinel makes the remaining element count computable in O(1), so
|
||||
// std::counted_iterator over a contiguous iterator must reach the same bulk
|
||||
// string/number scanners as a plain pointer - not just the byte-at-a-time
|
||||
// fallback (see #5268 for the equivalent memcpy fast path).
|
||||
#if JSON_HAS_RANGES
|
||||
// JSON_HAS_RANGES is 0 on standard libraries with an incomplete <ranges>
|
||||
// (libstdc++ < 11, libc++ < 16), where the adapter deliberately falls back
|
||||
// to the byte-at-a-time scanner; everything below still has to work there.
|
||||
using adapter_type = nlohmann::detail::iterator_input_adapter<std::counted_iterator<const char*>, std::default_sentinel_t>;
|
||||
CHECK(adapter_type::supports_bulk_scan);
|
||||
CHECK(adapter_type::supports_seek);
|
||||
#endif
|
||||
|
||||
// exercise every fast path: long ASCII run, multibyte UTF-8, escapes, and
|
||||
// integer/floating-point numbers
|
||||
const std::string json_str =
|
||||
R"({"ascii":"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",)"
|
||||
"\"utf8\":\"\xe4\xb8\xad\xe6\x96\x87\xf0\x9f\x98\x80\xc3\xa9\","
|
||||
R"("escaped":"aéb\n\\","ints":[0,-1,18446744073709551615,-9223372036854775808],)"
|
||||
R"("floats":[1.5,-2.25e3,0.30000000000000004]})";
|
||||
const auto len = static_cast<std::iter_difference_t<const char*>>(json_str.size());
|
||||
|
||||
const std::counted_iterator<const char*> first(json_str.data(), len);
|
||||
const json j = json::parse(first, std::default_sentinel);
|
||||
|
||||
// parsing through the pointer adapter must give exactly the same result
|
||||
CHECK(j == json::parse(json_str));
|
||||
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
// Diagnostics that quote the offending token are reconstructed from the
|
||||
// already-consumed input (supports_seek), a path a sized sentinel only
|
||||
// reaches now; check a few that include the "last read" text. Parsing
|
||||
// invalid input aborts when exceptions are off, hence the guard.
|
||||
// Raw strings and explicit bytes: an escaped literal and two literals
|
||||
// written next to each other both read as mistakes to static analysis.
|
||||
const auto byte = [](int value)
|
||||
{
|
||||
return std::string(1, static_cast<char>(value));
|
||||
};
|
||||
const std::vector<std::string> diagnostic_docs =
|
||||
{
|
||||
"1\nx",
|
||||
"truX",
|
||||
"[tru]",
|
||||
R"("abc)",
|
||||
R"(["\ud834"])",
|
||||
R"(["a)" + byte(0x01) + R"(b"])",
|
||||
R"([")" + byte(0xC3) + byte(0x28) + R"("])",
|
||||
"[1e]",
|
||||
R"(["aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaX)"
|
||||
};
|
||||
|
||||
for (const auto& text : diagnostic_docs)
|
||||
{
|
||||
CAPTURE(text);
|
||||
const std::counted_iterator<const char*> it(text.data(), static_cast<std::iter_difference_t<const char*>>(text.size()));
|
||||
std::string counted_message;
|
||||
std::string string_message;
|
||||
try
|
||||
{
|
||||
const json counted_result = json::parse(it, std::default_sentinel);
|
||||
static_cast<void>(counted_result);
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
counted_message = e.what();
|
||||
}
|
||||
try
|
||||
{
|
||||
const json string_result = json::parse(text);
|
||||
static_cast<void>(string_result);
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
string_message = e.what();
|
||||
}
|
||||
CHECK_FALSE(counted_message.empty());
|
||||
CHECK(counted_message == string_message);
|
||||
}
|
||||
|
||||
// and errors must still be reported identically
|
||||
const std::string bad = "[01\n]";
|
||||
const std::counted_iterator<const char*> bad_first(bad.data(), static_cast<std::iter_difference_t<const char*>>(bad.size()));
|
||||
std::string counted_what;
|
||||
std::string string_what;
|
||||
try
|
||||
{
|
||||
const json counted_result = json::parse(bad_first, std::default_sentinel);
|
||||
static_cast<void>(counted_result);
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
counted_what = e.what();
|
||||
}
|
||||
try
|
||||
{
|
||||
const json string_result = json::parse(bad);
|
||||
static_cast<void>(string_result);
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
string_what = e.what();
|
||||
}
|
||||
CHECK_FALSE(counted_what.empty());
|
||||
CHECK(counted_what == string_what);
|
||||
#endif
|
||||
}
|
||||
|
||||
#if !defined(JSON_NOEXCEPTION)
|
||||
// several cases below are truncated on purpose, and parsing invalid input
|
||||
// aborts when exceptions are off
|
||||
TEST_CASE("std::counted_iterator bulk scanning stops at the counted end")
|
||||
{
|
||||
// The count, not the size of the underlying buffer, is the end of the
|
||||
// input: the bulk scanners must never look at the bytes behind it, even
|
||||
// though they are readable. Each case is compared against parsing the
|
||||
// equivalent prefix as a std::string.
|
||||
const auto via_counted = [](const std::string & buf, std::size_t n) -> std::string
|
||||
{
|
||||
const std::counted_iterator<const char*> first(buf.data(), static_cast<std::iter_difference_t<const char*>>(n));
|
||||
try
|
||||
{
|
||||
const json j = json::parse(first, std::default_sentinel);
|
||||
return "OK|" + j.dump();
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
return {e.what()};
|
||||
}
|
||||
};
|
||||
const auto via_prefix = [](const std::string & buf, std::size_t n) -> std::string
|
||||
{
|
||||
try
|
||||
{
|
||||
const json j = json::parse(buf.substr(0, n));
|
||||
return "OK|" + j.dump();
|
||||
}
|
||||
catch (const json::parse_error& e)
|
||||
{
|
||||
return {e.what()};
|
||||
}
|
||||
};
|
||||
|
||||
struct testcase // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init)
|
||||
{
|
||||
const char* buffer;
|
||||
std::size_t count;
|
||||
};
|
||||
const std::vector<testcase> cases =
|
||||
{
|
||||
{"[\"abc\"]____TRAILING____", 7}, // exact fit, tail hidden
|
||||
{"[\"abcdefghijklmnop\"]____", 8}, // cut inside a string
|
||||
{"[\"abc\"]____", 6}, // cut just before the closing quote
|
||||
{"[12345]xxxxx", 4}, // cut inside a number
|
||||
{"[123]999999", 5}, // number ends exactly at the count
|
||||
{"[\"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\"]", 12}, // closing quote only behind the count
|
||||
{"[\"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\"]", 19}, // cut inside an 8-byte SWAR stride
|
||||
{"[\"\xe4\xb8\xad\xe6\x96\x87\"]", 5}, // cut inside a UTF-8 sequence
|
||||
{"[\"\xe4\xb8\xad\xe6\x96\x87\"]____", 10}, // complete UTF-8, tail hidden
|
||||
{"[1.25e3]TRAILINGDIGITS999", 7}, // number token reaches the count
|
||||
};
|
||||
|
||||
for (const auto& tc : cases)
|
||||
{
|
||||
CAPTURE(tc.buffer);
|
||||
CAPTURE(tc.count);
|
||||
const std::string buffer = tc.buffer;
|
||||
CHECK(via_counted(buffer, tc.count) == via_prefix(buffer, tc.count));
|
||||
}
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
|
||||
} // namespace
|
||||
|
||||
Reference in New Issue
Block a user