mirror of
https://github.com/nlohmann/json.git
synced 2026-07-28 05:14:54 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4ce130af4e | ||
|
|
5b5ba1ac35 | ||
|
|
85839d7408 | ||
|
|
6fb162a394 | ||
|
|
0bc01a066c |
@@ -38,14 +38,14 @@ jobs:
|
||||
|
||||
# Initializes the CodeQL tools for scanning.
|
||||
- name: Initialize CodeQL
|
||||
uses: github/codeql-action/init@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0
|
||||
uses: github/codeql-action/init@54f647b7e1bb85c95cddabcd46b0c578ec92bc1a # v4.36.3
|
||||
with:
|
||||
languages: c-cpp
|
||||
|
||||
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
|
||||
# If this step fails, then you should remove it and run the build manually (see below)
|
||||
- name: Autobuild
|
||||
uses: github/codeql-action/autobuild@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0
|
||||
uses: github/codeql-action/autobuild@54f647b7e1bb85c95cddabcd46b0c578ec92bc1a # v4.36.3
|
||||
|
||||
- name: Perform CodeQL Analysis
|
||||
uses: github/codeql-action/analyze@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0
|
||||
uses: github/codeql-action/analyze@54f647b7e1bb85c95cddabcd46b0c578ec92bc1a # v4.36.3
|
||||
|
||||
@@ -43,6 +43,6 @@ jobs:
|
||||
output: 'flawfinder_results.sarif'
|
||||
|
||||
- name: Upload analysis results to GitHub Security tab
|
||||
uses: github/codeql-action/upload-sarif@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0
|
||||
uses: github/codeql-action/upload-sarif@54f647b7e1bb85c95cddabcd46b0c578ec92bc1a # v4.36.3
|
||||
with:
|
||||
sarif_file: ${{github.workspace}}/flawfinder_results.sarif
|
||||
|
||||
@@ -76,6 +76,6 @@ jobs:
|
||||
|
||||
# Upload the results to GitHub's code scanning dashboard.
|
||||
- name: "Upload to code-scanning"
|
||||
uses: github/codeql-action/upload-sarif@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0
|
||||
uses: github/codeql-action/upload-sarif@54f647b7e1bb85c95cddabcd46b0c578ec92bc1a # v4.36.3
|
||||
with:
|
||||
sarif_file: results.sarif
|
||||
|
||||
@@ -61,7 +61,7 @@ jobs:
|
||||
|
||||
# Upload SARIF file generated in previous step
|
||||
- name: Upload SARIF file
|
||||
uses: github/codeql-action/upload-sarif@99df26d4f13ea111d4ec1a7dddef6063f76b97e9 # v4.37.0
|
||||
uses: github/codeql-action/upload-sarif@54f647b7e1bb85c95cddabcd46b0c578ec92bc1a # v4.36.3
|
||||
with:
|
||||
sarif_file: semgrep.sarif
|
||||
if: always()
|
||||
|
||||
@@ -20,7 +20,7 @@ jobs:
|
||||
with:
|
||||
egress-policy: audit
|
||||
|
||||
- uses: actions/stale@1e223db275d687790206a7acac4d1a11bd6fe629 # v10.4.0
|
||||
- uses: actions/stale@eb5cf3af3ac0a1aa4c9c45633dd1ae542a27a899 # v10.3.0
|
||||
with:
|
||||
stale-issue-label: 'state: stale'
|
||||
stale-pr-label: 'state: stale'
|
||||
|
||||
@@ -21,6 +21,7 @@ cc_library(
|
||||
"include/nlohmann/adl_serializer.hpp",
|
||||
"include/nlohmann/byte_container_with_subtype.hpp",
|
||||
"include/nlohmann/detail/abi_macros.hpp",
|
||||
"include/nlohmann/detail/conversions/from_chars.hpp",
|
||||
"include/nlohmann/detail/conversions/from_json.hpp",
|
||||
"include/nlohmann/detail/conversions/to_chars.hpp",
|
||||
"include/nlohmann/detail/conversions/to_json.hpp",
|
||||
|
||||
@@ -24,7 +24,6 @@ header. See also the [macro overview page](../../features/macros.md).
|
||||
- [**JSON_NO_IO**](json_no_io.md) - switch off functions relying on certain C++ I/O headers
|
||||
- [**JSON_SKIP_UNSUPPORTED_COMPILER_CHECK**](json_skip_unsupported_compiler_check.md) - do not warn about unsupported compilers
|
||||
- [**JSON_USE_GLOBAL_UDLS**](json_use_global_udls.md) - place user-defined string literals (UDLs) into the global namespace
|
||||
- [**JSON_USE_SIMDUTF**](json_use_simdutf.md) - use the simdutf library to accelerate UTF-8 validation
|
||||
|
||||
## Library version
|
||||
|
||||
|
||||
@@ -1,52 +0,0 @@
|
||||
# JSON_USE_SIMDUTF
|
||||
|
||||
```cpp
|
||||
#define JSON_USE_SIMDUTF
|
||||
```
|
||||
|
||||
When defined, the parser validates the UTF-8 content of JSON strings that come from a **contiguous byte input**
|
||||
(`std::string`, `std::vector<char>`/`<std::uint8_t>`, string literals, `const char*` ranges, …) using the
|
||||
[simdutf](https://github.com/simdutf/simdutf) library instead of the built-in scalar validator. On text with many
|
||||
non-ASCII characters (e.g. CJK or emoji) this can validate several times faster.
|
||||
|
||||
This is an **opt-in external dependency**. The library itself remains header-only and its behavior is unchanged: the
|
||||
same input is accepted or rejected either way, and every parse error is reported at the same position with the same
|
||||
message (simdutf is only used to fast-path *valid* runs; anything it flags falls back to the scalar path so the exact
|
||||
diagnostic is preserved). Streaming inputs (files, `std::istream`, wide strings, user-defined adapters) always use the
|
||||
scalar path.
|
||||
|
||||
When `JSON_USE_SIMDUTF` is defined you must make the `simdutf.h` header available on the include path and link the
|
||||
simdutf library. When it is not defined, no simdutf header is included and there is no dependency.
|
||||
|
||||
## Default definition
|
||||
|
||||
By default, `#!cpp JSON_USE_SIMDUTF` is not defined and the portable C++11 scalar validator is used.
|
||||
|
||||
```cpp
|
||||
#undef JSON_USE_SIMDUTF
|
||||
```
|
||||
|
||||
## Examples
|
||||
|
||||
??? example
|
||||
|
||||
The code below enables the simdutf backend for UTF-8 validation.
|
||||
|
||||
```cpp
|
||||
#define JSON_USE_SIMDUTF 1
|
||||
#include <simdutf.h>
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
...
|
||||
```
|
||||
|
||||
The project must also link against simdutf, e.g. with CMake:
|
||||
|
||||
```cmake
|
||||
target_compile_definitions(your_target PRIVATE JSON_USE_SIMDUTF)
|
||||
target_link_libraries(your_target PRIVATE simdutf::simdutf)
|
||||
```
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.12.1.
|
||||
@@ -296,7 +296,6 @@ nav:
|
||||
- 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md
|
||||
- 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md
|
||||
- 'JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON': api/macros/json_use_legacy_discarded_value_comparison.md
|
||||
- 'JSON_USE_SIMDUTF': api/macros/json_use_simdutf.md
|
||||
- 'NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_derived_type.md
|
||||
- 'NLOHMANN_DEFINE_TYPE_INTRUSIVE, NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_intrusive.md
|
||||
- 'NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_non_intrusive.md
|
||||
|
||||
@@ -0,0 +1,636 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
#if JSON_HAS_FROM_CHARS
|
||||
#include <algorithm> // find, find_if
|
||||
#include <charconv> // from_chars, from_chars_result
|
||||
#include <cmath> // isnan
|
||||
#else
|
||||
#include <cctype> // isspace
|
||||
#include <cerrno> // ERANGE
|
||||
#include <clocale> // localeconv, newlocale, freelocale
|
||||
#include <cstdlib> // strto*
|
||||
#include <memory> // unique_ptr
|
||||
#include <type_traits> // enable_if, is_unsigned, remove_pointer
|
||||
#include <utility> // declval
|
||||
|
||||
#ifdef __has_include
|
||||
#if __has_include(<xlocale.h>)
|
||||
#include <xlocale.h> // strto*_l, isspace_l
|
||||
#endif
|
||||
#endif
|
||||
#endif
|
||||
#include <limits> // numeric_limits<T>::[has_]infinity, quiet_NaN
|
||||
#include <system_error> // errc
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
|
||||
constexpr char get_locale_independent_decimal_point() noexcept
|
||||
{
|
||||
return '.';
|
||||
}
|
||||
|
||||
#if JSON_HAS_FROM_CHARS
|
||||
|
||||
//////////////////////////////////////////////
|
||||
// Delegate to std::from_chars if supported //
|
||||
//////////////////////////////////////////////
|
||||
|
||||
/*!
|
||||
@brief Number parsing implementation details
|
||||
|
||||
A traits class template that allows users of nlohmann::detail::from_chars to
|
||||
query details of the used implementation.
|
||||
|
||||
@tparam T numeric type to parse, matching the `value` argument of from_chars
|
||||
*/
|
||||
template <typename T>
|
||||
struct from_chars_traits
|
||||
{
|
||||
/// whether the from_chars in unaffected by the current global C locale
|
||||
static constexpr bool is_locale_independent = true;
|
||||
|
||||
/// getter for the character used as decimal point by from_chars
|
||||
static constexpr char(*get_decimal_point)() = &get_locale_independent_decimal_point;
|
||||
};
|
||||
|
||||
using std::from_chars_result;
|
||||
|
||||
/*!
|
||||
@brief Parse integer/floating-point numbers
|
||||
|
||||
Parses the string [first, last) into the numeric type T.
|
||||
|
||||
This implementation merely delegates to `std::from_chars` with one noteworthy
|
||||
difference: Different C++ standard library implementations do not agree on
|
||||
whether `value` should be set in the out-of-range case for floating-point
|
||||
numbers. Presently, libstdc++ leaves `value` unchanged whereas libc++ and MS STL
|
||||
set `value` to +/-0 or +/-inf which allows its users to distinguish between the
|
||||
various cases of over- and underflow. The C++ proposal [P4168][1] details this
|
||||
discrepancy and advocates to standardize the latter behavior.
|
||||
|
||||
Since we need to be able to distinguish absolute underflow (+/-0) from absolute
|
||||
overflow (+/-inf), this implementation "corrects" the out-of-range behavior
|
||||
accordingly by estimating the effective exponent through manual string parsing.
|
||||
|
||||
[1]: https://isocpp.org/files/papers/P4168R0.html
|
||||
|
||||
@tparam T numeric type to parse
|
||||
@param[in] first points to the beginning of the parsed string (inclusive)
|
||||
@param[in] end points to the end of the parsed string (exclusive)
|
||||
@param[out] value references the variable to set the parsed number
|
||||
@return structure consisting of `ptr` and `ec` such that:
|
||||
- If the string starting at `first` matches a number, `ptr` points to
|
||||
the address immediately after the last character that matches.
|
||||
* If the matched number can be represented by `T`, `value` is set and
|
||||
`ec` is value-initialized.
|
||||
* If the matched number cannot be represented by `T`, `ec` is set to
|
||||
`std::errc::result_out_of_range` and depending on the type of `T`:
|
||||
+ `value` is left unchanged for integral `T`,
|
||||
+ `value` is set to +/-0 or +/-inf for floating-point `T`.
|
||||
- If the string starting at `first` doesn't match a number, `ptr`
|
||||
points to `first`, `ec` is `std::errc::invalid_argument`, and `value`
|
||||
is left unchanged.
|
||||
*/
|
||||
template <typename T>
|
||||
JSON_HEDLEY_NON_NULL(1, 2)
|
||||
inline from_chars_result from_chars(const char* first, const char* last, T& value)
|
||||
{
|
||||
// No implementation of std::from_chars will set its value argument to NaN,
|
||||
// so by using NaN as the initial value, we can tell if it changed.
|
||||
T inner_value = std::numeric_limits<T>::quiet_NaN();
|
||||
auto res = std::from_chars(first, last, inner_value);
|
||||
|
||||
if constexpr (std::numeric_limits<T>::has_infinity)
|
||||
{
|
||||
if (res.ec == std::errc::result_out_of_range)
|
||||
{
|
||||
if (std::isnan(inner_value)) // inner_value was not set
|
||||
{
|
||||
// [first, res.ptr) matches a valid floating-point number that's
|
||||
// out-of-range and our std::from_chars did not set `value`, so we
|
||||
// need to distinguish the type of under-/overflow ourselves.
|
||||
const char* mantissa_begin = *first == '-' ? first + 1 : first;
|
||||
const char* mantissa_end = std::find_if(mantissa_begin, res.ptr, [](char c)
|
||||
{
|
||||
return c == 'e' || c == 'E';
|
||||
});
|
||||
const char* decimal_point = std::find(mantissa_begin, mantissa_end, '.');
|
||||
const char* significant_digit = std::find_if(mantissa_begin, mantissa_end, [](char d)
|
||||
{
|
||||
return d >= '1' && d <= '9';
|
||||
});
|
||||
|
||||
// Position of the significant digit relative to the decimal gives
|
||||
// the order of magnitude of the mantissa.
|
||||
auto effective_exponent = static_cast<int>(decimal_point - significant_digit);
|
||||
|
||||
// If the number includes an explicit exponent, add that to the the
|
||||
// total exponent.
|
||||
if (mantissa_end != res.ptr)
|
||||
{
|
||||
const char* exponent_begin = *(mantissa_end + 1) == '+'
|
||||
? mantissa_end + 2 // skip the exponent's + sign
|
||||
: mantissa_end + 1;
|
||||
int exponent = 0;
|
||||
auto exp_res = std::from_chars(exponent_begin, res.ptr, exponent);
|
||||
JSON_ASSERT(exp_res.ptr == res.ptr);
|
||||
|
||||
if (exp_res.ec == std::errc::result_out_of_range)
|
||||
{
|
||||
// NOTE: Parentheses around function names mitigate min/max macro collision on Windows
|
||||
effective_exponent = *exponent_begin == '-'
|
||||
? (std::numeric_limits<int>::min)()
|
||||
: (std::numeric_limits<int>::max)();
|
||||
}
|
||||
else
|
||||
{
|
||||
JSON_ASSERT(exp_res.ec == std::errc{});
|
||||
effective_exponent += exponent;
|
||||
}
|
||||
}
|
||||
|
||||
// Set `value` to the sign-correct non-finite value based on the
|
||||
// effective exponent.
|
||||
value = (*first == '-' ? -1 : 1) * (effective_exponent < 0
|
||||
? T{}
|
||||
: std::numeric_limits<T>::infinity());
|
||||
}
|
||||
else
|
||||
{
|
||||
value = inner_value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (res.ec == std::errc{})
|
||||
{
|
||||
value = inner_value;
|
||||
}
|
||||
|
||||
return res;
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
//////////////////////////////////////////////
|
||||
// Fallback implementation using strto*[_l] //
|
||||
//////////////////////////////////////////////
|
||||
|
||||
#if defined(_WIN32)
|
||||
|
||||
/*!
|
||||
@brief Extended locale functions on Windows
|
||||
|
||||
A trait object which provides wrappers for isspace and the strto* family of
|
||||
functions, based on the platform's support for extended locale APIs.
|
||||
|
||||
This is the Windows implementation. The primary class template definition
|
||||
falls back to the standard locale-dependent functions, which is used on
|
||||
MinGW, whose C runtime only provides an incomplete subset of the
|
||||
`_<funcname>_l` family (e.g., missing `_strtof_l`/`_strtold_l` depending on
|
||||
the toolchain version). A specialization for genuine MSVC (including
|
||||
clang-cl, which mimics the MSVC ABI and CRT) is used instead, as the full
|
||||
`_<funcname>_l` family has reliably been available there since at least
|
||||
Visual Studio 2005.
|
||||
*/
|
||||
template <typename = const char*, typename = float>
|
||||
struct extended_locale_traits
|
||||
{
|
||||
static constexpr bool is_locale_independent = false;
|
||||
|
||||
JSON_HEDLEY_PURE
|
||||
static char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
static int isspace(int c) noexcept
|
||||
{
|
||||
return std::isspace(c);
|
||||
}
|
||||
|
||||
// NOLINTNEXTLINE(runtime/int)
|
||||
static long long strtoll(const char* str, char** endptr) noexcept
|
||||
{
|
||||
return std::strtoll(str, endptr, 10);
|
||||
}
|
||||
|
||||
// NOLINTNEXTLINE(runtime/int)
|
||||
static unsigned long long strtoull(const char* str, char** endptr) noexcept
|
||||
{
|
||||
return std::strtoull(str, endptr, 10);
|
||||
}
|
||||
|
||||
static constexpr float(&strtof)(const char*, char**) = std::strtof;
|
||||
static constexpr double(&strtod)(const char*, char**) = std::strtod;
|
||||
static constexpr long double(&strtold)(const char*, char**) = std::strtold;
|
||||
};
|
||||
|
||||
#if defined(_MSC_VER)
|
||||
|
||||
/*!
|
||||
@brief Get a pointer to a globally shared instance of the "C" locale on Windows
|
||||
|
||||
This function uses _create_locale/_free_locale to allocate a "C" locale instance
|
||||
which can be passed to extended locale APIs. This instance is created on first
|
||||
use and has static storage duration, such that it can be used for the duration
|
||||
of the program.
|
||||
*/
|
||||
inline _locale_t get_c_locale_t() noexcept
|
||||
{
|
||||
struct LocaleTDeleter
|
||||
{
|
||||
void operator()(_locale_t loc) noexcept
|
||||
{
|
||||
::_free_locale(loc);
|
||||
}
|
||||
};
|
||||
using LocaleUPtr = std::unique_ptr<std::remove_pointer<_locale_t>::type, LocaleTDeleter>;
|
||||
static const LocaleUPtr c_locale = LocaleUPtr{::_create_locale(LC_ALL, "C"), LocaleTDeleter{}};
|
||||
return c_locale.get();
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
struct extended_locale_traits<T, float>
|
||||
{
|
||||
static constexpr bool is_locale_independent = true;
|
||||
static constexpr char(&get_decimal_point)() = get_locale_independent_decimal_point;
|
||||
|
||||
static int isspace(int c) noexcept
|
||||
{
|
||||
return _isspace_l(c, get_c_locale_t());
|
||||
}
|
||||
|
||||
// NOLINTNEXTLINE(runtime/int)
|
||||
static long long strtoll(const char* str, char** endptr) noexcept
|
||||
{
|
||||
return _strtoll_l(str, endptr, 10, get_c_locale_t());
|
||||
}
|
||||
|
||||
// NOLINTNEXTLINE(runtime/int)
|
||||
static unsigned long long strtoull(const char* str, char** endptr) noexcept
|
||||
{
|
||||
return _strtoull_l(str, endptr, 10, get_c_locale_t());
|
||||
}
|
||||
|
||||
static float strtof(T str, char** str_end)
|
||||
{
|
||||
return _strtof_l(str, str_end, get_c_locale_t());
|
||||
}
|
||||
|
||||
static double strtod(T str, char** str_end)
|
||||
{
|
||||
return _strtod_l(str, str_end, get_c_locale_t());
|
||||
}
|
||||
|
||||
static long double strtold(T str, char** str_end)
|
||||
{
|
||||
return _strtold_l(str, str_end, get_c_locale_t());
|
||||
}
|
||||
};
|
||||
|
||||
#endif
|
||||
|
||||
#else
|
||||
|
||||
/*!
|
||||
@brief Get a pointer to a globally shared instance of the "C" locale on POSIX
|
||||
|
||||
This function uses newlocale/freelocale to allocate a "C" locale instance
|
||||
which can be passed to extended locale APIs. This instance is created on first
|
||||
use and has static storage duration, such that it can be used for the duration
|
||||
of the program.
|
||||
|
||||
newlocale/freelocale themselves are in POSIX and are available independent of
|
||||
the platform's support for extended locale APIs.
|
||||
*/
|
||||
inline locale_t get_c_locale_t() noexcept
|
||||
{
|
||||
struct LocaleTDeleter
|
||||
{
|
||||
void operator()(locale_t loc) noexcept
|
||||
{
|
||||
::freelocale(loc);
|
||||
}
|
||||
};
|
||||
using LocaleUPtr = std::unique_ptr<std::remove_pointer<locale_t>::type, LocaleTDeleter>;
|
||||
|
||||
#if defined(__clang__)
|
||||
#pragma clang diagnostic push
|
||||
#pragma clang diagnostic ignored "-Wexit-time-destructors"
|
||||
#endif
|
||||
static const LocaleUPtr c_locale = LocaleUPtr {::newlocale(LC_ALL_MASK, "C", nullptr), LocaleTDeleter{}};
|
||||
#if defined(__clang__)
|
||||
#pragma clang diagnostic pop
|
||||
#endif
|
||||
|
||||
return c_locale.get();
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief Extended locale functions on POSIX
|
||||
|
||||
A trait object which provides wrappers for isspace and the strto* family of
|
||||
functions, based on the platform's support for extended locale APIs.
|
||||
|
||||
This is the POSIX implementation where extended locale support might not be
|
||||
available. The primary class template definition falls back to the standard
|
||||
locale-dependent functions. A partial specialization uses SFINAE to detect the
|
||||
availability of `strtof_l` (as a proxy for the whole `<funcname>_l` family) and
|
||||
provides access to locale-independent functions.
|
||||
*/
|
||||
template <typename = const char*, typename = float>
|
||||
struct extended_locale_traits
|
||||
{
|
||||
static constexpr bool is_locale_independent = false;
|
||||
|
||||
/*!
|
||||
@brief Query the decimal point used by the current C locale
|
||||
|
||||
Note that calling this function while switching the global C locale from
|
||||
another thread is undefined behavior.
|
||||
*/
|
||||
JSON_HEDLEY_PURE
|
||||
static char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
static int isspace(int c) noexcept
|
||||
{
|
||||
return std::isspace(c);
|
||||
}
|
||||
|
||||
// NOLINTNEXTLINE(runtime/int)
|
||||
static long long strtoll(const char* str, char** endptr) noexcept
|
||||
{
|
||||
return std::strtoll(str, endptr, 10);
|
||||
}
|
||||
|
||||
// NOLINTNEXTLINE(runtime/int)
|
||||
static unsigned long long strtoull(const char* str, char** endptr) noexcept
|
||||
{
|
||||
return std::strtoull(str, endptr, 10);
|
||||
}
|
||||
|
||||
static constexpr float(&strtof)(const char*, char**) = std::strtof;
|
||||
static constexpr double(&strtod)(const char*, char**) = std::strtod;
|
||||
static constexpr long double(&strtold)(const char*, char**) = std::strtold;
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
struct extended_locale_traits<T, decltype(strtof_l(
|
||||
std::declval<T>(),
|
||||
std::declval<char**>(),
|
||||
std::declval<locale_t>()))>
|
||||
{
|
||||
static constexpr bool is_locale_independent = true;
|
||||
static constexpr char(&get_decimal_point)() = get_locale_independent_decimal_point;
|
||||
|
||||
static int isspace(int c) noexcept
|
||||
{
|
||||
return isspace_l(c, get_c_locale_t());
|
||||
}
|
||||
|
||||
// NOLINTNEXTLINE(runtime/int)
|
||||
static long long strtoll(const char* str, char** endptr) noexcept
|
||||
{
|
||||
return ::strtoll_l(str, endptr, 10, get_c_locale_t());
|
||||
}
|
||||
|
||||
// NOLINTNEXTLINE(runtime/int)
|
||||
static unsigned long long strtoull(const char* str, char** endptr) noexcept
|
||||
{
|
||||
return ::strtoull_l(str, endptr, 10, get_c_locale_t());
|
||||
}
|
||||
|
||||
static float strtof(T str, char** str_end)
|
||||
{
|
||||
return ::strtof_l(str, str_end, get_c_locale_t());
|
||||
}
|
||||
|
||||
static double strtod(T str, char** str_end)
|
||||
{
|
||||
return ::strtod_l(str, str_end, get_c_locale_t());
|
||||
}
|
||||
|
||||
static long double strtold(T str, char** str_end)
|
||||
{
|
||||
return ::strtold_l(str, str_end, get_c_locale_t());
|
||||
}
|
||||
};
|
||||
|
||||
#endif
|
||||
|
||||
/*!
|
||||
@brief Number parsing implementation details
|
||||
|
||||
A traits class template that allows users of nlohmann::detail::from_chars to
|
||||
query details of the used implementation.
|
||||
|
||||
Each specialization provides the following static members:
|
||||
- is_locale_independent: whether the from_chars in unaffected by the current
|
||||
global C locale
|
||||
- get_decimal_point: getter for the character used as decimal point by from_chars
|
||||
- strto: dispatches to the C standard library strto* function for type T, or to
|
||||
the corresponding extended locale function, if available.
|
||||
|
||||
@tparam T numeric type to parse, matching the `value` argument of from_chars
|
||||
*/
|
||||
template <typename T>
|
||||
struct from_chars_traits;
|
||||
|
||||
template <>
|
||||
struct from_chars_traits<long long> // NOLINT(runtime/int)
|
||||
{
|
||||
static constexpr bool is_locale_independent = extended_locale_traits<>::is_locale_independent;
|
||||
|
||||
static char get_decimal_point()
|
||||
{
|
||||
return extended_locale_traits<>::get_decimal_point();
|
||||
}
|
||||
|
||||
// NOLINTNEXTLINE(runtime/int)
|
||||
static constexpr long long(*strto)(const char*, char**) = &extended_locale_traits<>::strtoll;
|
||||
};
|
||||
|
||||
template <>
|
||||
struct from_chars_traits<unsigned long long> // NOLINT(runtime/int)
|
||||
{
|
||||
static constexpr bool is_locale_independent = extended_locale_traits<>::is_locale_independent;
|
||||
|
||||
static char get_decimal_point()
|
||||
{
|
||||
return extended_locale_traits<>::get_decimal_point();
|
||||
}
|
||||
|
||||
// NOLINTNEXTLINE(runtime/int)
|
||||
static constexpr unsigned long long(*strto)(const char*, char**) = &extended_locale_traits<>::strtoull;
|
||||
};
|
||||
|
||||
template <>
|
||||
struct from_chars_traits<float>
|
||||
{
|
||||
static constexpr bool is_locale_independent = extended_locale_traits<>::is_locale_independent;
|
||||
|
||||
static char get_decimal_point()
|
||||
{
|
||||
return extended_locale_traits<>::get_decimal_point();
|
||||
}
|
||||
|
||||
static constexpr float(*strto)(const char*, char**) = &extended_locale_traits<>::strtof;
|
||||
};
|
||||
|
||||
template <>
|
||||
struct from_chars_traits<double>
|
||||
{
|
||||
static constexpr bool is_locale_independent = extended_locale_traits<>::is_locale_independent;
|
||||
|
||||
static char get_decimal_point()
|
||||
{
|
||||
return extended_locale_traits<>::get_decimal_point();
|
||||
}
|
||||
|
||||
static constexpr double(*strto)(const char*, char**) = &extended_locale_traits<>::strtod;
|
||||
};
|
||||
|
||||
template <>
|
||||
struct from_chars_traits<long double>
|
||||
{
|
||||
static constexpr bool is_locale_independent = extended_locale_traits<>::is_locale_independent;
|
||||
|
||||
static char get_decimal_point()
|
||||
{
|
||||
return extended_locale_traits<>::get_decimal_point();
|
||||
}
|
||||
|
||||
static constexpr long double(*strto)(const char*, char**) = &extended_locale_traits<>::strtold;
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
constexpr typename std::enable_if<std::numeric_limits<T>::has_infinity, bool>::type is_out_of_range_value(T value) noexcept
|
||||
{
|
||||
#ifdef __GNUC__
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wfloat-equal"
|
||||
#endif
|
||||
return value == T {} || value == std::numeric_limits<T>::infinity() || value == -std::numeric_limits<T>::infinity();
|
||||
#ifdef __GNUC__
|
||||
#pragma GCC diagnostic pop
|
||||
#endif
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
constexpr typename std::enable_if < !std::numeric_limits<T>::has_infinity, bool >::type is_out_of_range_value(T value) noexcept
|
||||
{
|
||||
// NOTE: Parentheses around function names mitigate min/max macro collision on Windows
|
||||
return value == (std::numeric_limits<T>::max)() || value == (std::numeric_limits<T>::min)();
|
||||
}
|
||||
|
||||
struct from_chars_result
|
||||
{
|
||||
const char* ptr;
|
||||
std::errc ec;
|
||||
};
|
||||
|
||||
/*!
|
||||
@brief Parse integer/floating-point numbers
|
||||
|
||||
Parses the string [first, last) into the numeric type T.
|
||||
|
||||
This implementation uses the C standard library functions strto*, or their
|
||||
extended locale counterparts strto*_l, to emulate the behavior of
|
||||
`std::from_chars` on platforms where it is not (fully) supported.
|
||||
|
||||
Regarding the under-/overflow behavior, this implementation adopts the behavior
|
||||
proposed by [P4168][1] and implemented by libc++ and MS STL of setting `value`
|
||||
to the corresponding non-finite floating-point value in case of an out-of-range
|
||||
result.
|
||||
|
||||
[1]: https://isocpp.org/files/papers/P4168R0.html
|
||||
|
||||
@pre Unlike std::from_chars, this implementation requires the string to be
|
||||
null-terminated, such that `last` dereferences into a NUL byte.
|
||||
|
||||
@tparam T numeric type to parse
|
||||
@param[in] first points to the beginning of the parsed string (inclusive)
|
||||
@param[in] end points to the end of the parsed string (exclusive)
|
||||
@param[out] value references the variable to set the parsed number
|
||||
@return structure consisting of `ptr` and `ec` such that:
|
||||
- If the string starting at `first` matches a number, `ptr` points to
|
||||
the address immediately after the last character that matches.
|
||||
* If the matched number can be represented by `T`, `value` is set and
|
||||
`ec` is value-initialized.
|
||||
* If the matched number cannot be represented by `T`, `ec` is set to
|
||||
`std::errc::result_out_of_range` and depending on the type of `T`:
|
||||
+ `value` is left unchanged for integral `T`,
|
||||
+ `value` is set to +/-0 or +/-inf for floating-point `T`.
|
||||
- If the string starting at `first` doesn't match a number, `ptr`
|
||||
points to `first`, `ec` is `std::errc::invalid_argument`, and `value`
|
||||
is left unchanged.
|
||||
*/
|
||||
template <typename T>
|
||||
JSON_HEDLEY_NON_NULL(1, 2)
|
||||
inline from_chars_result from_chars(const char* first, const char* last, T& value)
|
||||
{
|
||||
JSON_ASSERT(*last == '\0');
|
||||
|
||||
// Unlike strto*, from_chars does not accept leading whitespace or + signs
|
||||
if (first == last || *first == '+'
|
||||
|| (std::is_unsigned<T>::value && *first == '-')
|
||||
|| extended_locale_traits<>::isspace(*first) != 0)
|
||||
{
|
||||
return {first, std::errc::invalid_argument};
|
||||
}
|
||||
|
||||
errno = 0;
|
||||
char* ptr = nullptr; // NOLINT(misc-const-correctness)
|
||||
T result = from_chars_traits<T>::strto(first, &ptr);
|
||||
|
||||
if (ptr == first)
|
||||
{
|
||||
return {ptr, std::errc::invalid_argument};
|
||||
}
|
||||
|
||||
// Upon under-/overflow, strto* returns a marginal value and sets errno.
|
||||
// Note that it is NOT sufficient to just check errno: strto* only clears
|
||||
// errno if the parsed string actually parses into 0/[U]LLONG_MIN/-_MAX
|
||||
// and this result was returned without indicating an under-/overflow.
|
||||
// Otherwise, errno may not be relied upon to indicate the *absence* of
|
||||
// an out-of-range error.
|
||||
if (is_out_of_range_value(result) && errno == ERANGE)
|
||||
{
|
||||
if (std::numeric_limits<T>::has_infinity)
|
||||
{
|
||||
// ONLY for floating-point types, set `value` to the same non-finite
|
||||
// result returned by strto* to allow users to distinguish between
|
||||
// different types of under-/overflow.
|
||||
value = result;
|
||||
}
|
||||
return {ptr, std::errc::result_out_of_range};
|
||||
}
|
||||
|
||||
value = result;
|
||||
return {ptr, std::errc{}};
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -231,33 +231,6 @@ class iterator_input_adapter
|
||||
std::is_same<IteratorType, SentinelType>::value && std::is_pointer<IteratorType>::value;
|
||||
#endif
|
||||
|
||||
public:
|
||||
// Whether the remaining input is a single contiguous block of 1-byte
|
||||
// elements that the lexer can inspect directly (used for the SWAR string
|
||||
// fast path). Restricted to same-type iterator/sentinel pairs so that plain
|
||||
// std::distance/std::advance are well-defined in all standards.
|
||||
static constexpr bool supports_bulk_scan =
|
||||
iterator_is_contiguous && std::is_same<IteratorType, SentinelType>::value && sizeof(char_type) == 1;
|
||||
|
||||
// Pointer to the next unread element; only valid when bulk_remaining() > 0.
|
||||
const char_type* bulk_data() const
|
||||
{
|
||||
return &*current;
|
||||
}
|
||||
|
||||
// Number of unread elements available as one contiguous block.
|
||||
std::size_t bulk_remaining() const
|
||||
{
|
||||
return static_cast<std::size_t>(std::distance(current, end));
|
||||
}
|
||||
|
||||
// Consume @a n elements previously inspected via bulk_data().
|
||||
void bulk_skip(std::size_t n)
|
||||
{
|
||||
std::advance(current, static_cast<typename std::iterator_traits<IteratorType>::difference_type>(n));
|
||||
}
|
||||
|
||||
private:
|
||||
// contiguous fast path: bulk copy the remaining range with std::memcpy
|
||||
template<class T>
|
||||
std::size_t get_elements_impl(T* dest, std::size_t count, std::true_type /*contiguous*/)
|
||||
@@ -422,30 +395,17 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
|
||||
}
|
||||
else
|
||||
{
|
||||
// A supplementary code point is a high surrogate (0xD800..0xDBFF)
|
||||
// followed by a low surrogate (0xDC00..0xDFFF). A lone low
|
||||
// surrogate, a high surrogate at the end of the input, or a high
|
||||
// surrogate followed by any other unit is malformed UTF-16. In
|
||||
// that case the offending unit is passed through unchanged so the
|
||||
// UTF-8 decoder rejects it, matching how \uXXXX surrogate escapes
|
||||
// are handled in the lexer.
|
||||
bool valid_pair = false;
|
||||
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
|
||||
if (JSON_HEDLEY_UNLIKELY(!input.empty()))
|
||||
{
|
||||
const auto wc2 = static_cast<unsigned int>(input.get_character());
|
||||
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
|
||||
{
|
||||
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
|
||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu));
|
||||
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (charcode & 0x3Fu));
|
||||
utf8_bytes_filled = 4;
|
||||
valid_pair = true;
|
||||
}
|
||||
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
|
||||
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
|
||||
utf8_bytes[2] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu));
|
||||
utf8_bytes[3] = static_cast<std::char_traits<char>::int_type>(0x80u | (charcode & 0x3Fu));
|
||||
utf8_bytes_filled = 4;
|
||||
}
|
||||
|
||||
if (!valid_pair)
|
||||
else
|
||||
{
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
|
||||
utf8_bytes_filled = 1;
|
||||
@@ -593,24 +553,6 @@ typename iterator_input_adapter_factory<IteratorType, SentinelType>::adapter_typ
|
||||
return factory_type::create(first, last);
|
||||
}
|
||||
|
||||
// Detect a container that stores its elements contiguously as single bytes
|
||||
// (std::string, std::vector<char/unsigned char>, std::array<char, N>,
|
||||
// std::string_view, ...). Such inputs are wrapped in a pointer-based adapter so
|
||||
// they benefit from the contiguous fast paths (bulk string scanning, memcpy for
|
||||
// binary formats) in every C++ standard - not only in C++20, where the standard
|
||||
// library iterators model std::contiguous_iterator and are detected directly.
|
||||
template<typename ContainerType, typename = void>
|
||||
struct is_contiguous_byte_container : std::false_type {};
|
||||
|
||||
template<typename ContainerType>
|
||||
struct is_contiguous_byte_container < ContainerType, void_t <
|
||||
decltype(std::declval<const ContainerType&>().data()),
|
||||
decltype(std::declval<const ContainerType&>().size()) >>
|
||||
: std::integral_constant < bool,
|
||||
std::is_pointer<decltype(std::declval<const ContainerType&>().data())>::value&&
|
||||
std::is_integral<typename std::remove_pointer<decltype(std::declval<const ContainerType&>().data())>::type>::value&&
|
||||
sizeof(typename std::remove_pointer<decltype(std::declval<const ContainerType&>().data())>::type) == 1 > {};
|
||||
|
||||
// Convenience shorthand from container to iterator
|
||||
// Enables ADL on begin(container) and end(container)
|
||||
// Encloses the using declarations in namespace for not to leak them to outside scope
|
||||
@@ -638,32 +580,12 @@ struct container_input_adapter_factory< ContainerType,
|
||||
|
||||
} // namespace container_input_adapter_factory_impl
|
||||
|
||||
// General container path (iterator-based). Contiguous single-byte containers
|
||||
// are excluded here and routed through the pointer-based overload below.
|
||||
template < typename ContainerType,
|
||||
enable_if_t < !is_contiguous_byte_container<ContainerType>::value, int > = 0 >
|
||||
typename container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::adapter_type input_adapter(ContainerType && container)
|
||||
template<typename ContainerType>
|
||||
typename container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::adapter_type input_adapter(ContainerType&& container)
|
||||
{
|
||||
return container_input_adapter_factory_impl::container_input_adapter_factory<ContainerType>::create(std::forward<ContainerType>(container));
|
||||
}
|
||||
|
||||
// Contiguous single-byte containers (std::string, std::vector<char>, ...) are
|
||||
// wrapped in a pointer-based adapter so the contiguous fast paths apply in every
|
||||
// standard. The pointer keeps the container's own element type (const char* for
|
||||
// std::string, const std::uint8_t* for std::vector<std::uint8_t>, ...), so the
|
||||
// resulting char_type - and therefore the parsing behavior - is byte-for-byte
|
||||
// identical to the iterator-based path; only the raw pointer additionally
|
||||
// enables the bulk fast paths. The container outlives the adapter for the whole
|
||||
// parse (temporaries live until the end of the full expression), exactly as the
|
||||
// iterators it replaces did.
|
||||
template < typename ContainerType,
|
||||
enable_if_t < is_contiguous_byte_container<ContainerType>::value, int > = 0 >
|
||||
auto input_adapter(const ContainerType& container)
|
||||
-> decltype(input_adapter(container.data(), container.data() + container.size()))
|
||||
{
|
||||
return input_adapter(container.data(), container.data() + container.size());
|
||||
}
|
||||
|
||||
// specialization for std::string
|
||||
using string_input_adapter_type = decltype(input_adapter(std::declval<std::string>()));
|
||||
|
||||
|
||||
@@ -9,19 +9,17 @@
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <clocale> // localeconv
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdio> // snprintf
|
||||
#include <cstdlib> // strtof, strtod, strtold, strtoll, strtoull
|
||||
#include <initializer_list> // initializer_list
|
||||
#include <string> // char_traits, string
|
||||
#include <system_error> // errc
|
||||
#include <utility> // move
|
||||
#include <vector> // vector
|
||||
|
||||
#include <nlohmann/detail/conversions/from_chars.hpp>
|
||||
#include <nlohmann/detail/input/input_adapters.hpp>
|
||||
#include <nlohmann/detail/input/number_parse.hpp>
|
||||
#include <nlohmann/detail/input/position_t.hpp>
|
||||
#include <nlohmann/detail/input/string_scan.hpp>
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||
|
||||
@@ -127,25 +125,6 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/)
|
||||
return false;
|
||||
}
|
||||
|
||||
// Detect whether an input adapter exposes a contiguous byte block that the
|
||||
// lexer can scan directly (see iterator_input_adapter::supports_bulk_scan).
|
||||
// Adapters without the flag - file, stream, wide-string, user-defined - fall
|
||||
// back to the character-at-a-time string scanner.
|
||||
template<typename InputAdapterType>
|
||||
using detect_supports_bulk_scan = decltype(InputAdapterType::supports_bulk_scan);
|
||||
|
||||
template<typename InputAdapterType>
|
||||
constexpr bool input_adapter_supports_bulk_scan(std::true_type /*detected*/)
|
||||
{
|
||||
return InputAdapterType::supports_bulk_scan;
|
||||
}
|
||||
|
||||
template<typename InputAdapterType>
|
||||
constexpr bool input_adapter_supports_bulk_scan(std::false_type /*detected*/)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief lexical analysis
|
||||
|
||||
@@ -167,21 +146,13 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
static constexpr bool lazy_token_string =
|
||||
input_adapter_supports_seek<InputAdapterType>(is_detected<detect_supports_seek, InputAdapterType> {});
|
||||
|
||||
/// whether string scanning may bulk-consume runs of ordinary characters
|
||||
/// directly from a contiguous input buffer (SWAR fast path). This requires
|
||||
/// the token to be reconstructible lazily (lazy_token_string), so bypassing
|
||||
/// the per-character capture in get() cannot lose error diagnostics.
|
||||
static constexpr bool bulk_scan =
|
||||
lazy_token_string
|
||||
&& input_adapter_supports_bulk_scan<InputAdapterType>(is_detected<detect_supports_bulk_scan, InputAdapterType> {});
|
||||
|
||||
public:
|
||||
using token_type = typename lexer_base<BasicJsonType>::token_type;
|
||||
|
||||
explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept
|
||||
: ia(std::move(adapter))
|
||||
, ignore_comments(ignore_comments_)
|
||||
, decimal_point_char(static_cast<char_int_type>(get_decimal_point()))
|
||||
, decimal_point_char(static_cast<char_int_type>(from_chars_traits<number_float_t>::get_decimal_point()))
|
||||
{}
|
||||
|
||||
// deleted because of pointer members
|
||||
@@ -192,19 +163,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
~lexer() = default;
|
||||
|
||||
private:
|
||||
/////////////////////
|
||||
// locales
|
||||
/////////////////////
|
||||
|
||||
/// return the locale-dependent decimal point
|
||||
JSON_HEDLEY_PURE
|
||||
static char get_decimal_point() noexcept
|
||||
{
|
||||
const auto* loc = localeconv();
|
||||
JSON_ASSERT(loc != nullptr);
|
||||
return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point);
|
||||
}
|
||||
|
||||
/////////////////////
|
||||
// scan functions
|
||||
/////////////////////
|
||||
@@ -294,40 +252,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
return true;
|
||||
}
|
||||
|
||||
/// contiguous input: bulk-append the run of ordinary characters and complete
|
||||
/// well-formed UTF-8 sequences starting at the current read position, leaving
|
||||
/// the first byte that needs individual handling (the closing quote, an
|
||||
/// escape, a control character, or an ill-formed UTF-8 byte) for get()
|
||||
void scan_string_bulk(std::true_type /*bulk*/)
|
||||
{
|
||||
// a pending unget must be consumed through the normal path first
|
||||
if (next_unget)
|
||||
{
|
||||
return;
|
||||
}
|
||||
const std::size_t remaining = ia.bulk_remaining();
|
||||
if (remaining == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
const auto* const data = reinterpret_cast<const unsigned char*>(ia.bulk_data());
|
||||
|
||||
const std::size_t pos = string_bulk_run(data, remaining);
|
||||
if (pos == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
token_buffer.append(reinterpret_cast<const typename string_t::value_type*>(data), pos);
|
||||
ia.bulk_skip(pos);
|
||||
// the run contains no newline (all bytes < 0x20 are treated as special),
|
||||
// so only the flat character counters advance
|
||||
position.chars_read_total += pos;
|
||||
position.chars_read_current_line += pos;
|
||||
}
|
||||
|
||||
/// streaming input: no bulk fast path
|
||||
void scan_string_bulk(std::false_type /*bulk*/) const noexcept {}
|
||||
|
||||
/*!
|
||||
@brief scan a string literal
|
||||
|
||||
@@ -353,10 +277,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
|
||||
while (true)
|
||||
{
|
||||
// bulk-consume ordinary characters from contiguous input, then
|
||||
// handle the next special byte through the switch below
|
||||
scan_string_bulk(std::integral_constant<bool, bulk_scan> {});
|
||||
|
||||
// get the next character
|
||||
switch (get())
|
||||
{
|
||||
@@ -1008,24 +928,6 @@ class lexer : public lexer_base<BasicJsonType>
|
||||
}
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(float& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtof(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtod(str, endptr);
|
||||
}
|
||||
|
||||
JSON_HEDLEY_NON_NULL(2)
|
||||
static void strtof(long double& f, const char* str, char** endptr) noexcept
|
||||
{
|
||||
f = std::strtold(str, endptr);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief scan a number literal
|
||||
|
||||
@@ -1346,188 +1248,55 @@ scan_number_done:
|
||||
// we are done scanning a number)
|
||||
unget();
|
||||
|
||||
return convert_number(number_type);
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief convert the number text in token_buffer to its value and token type
|
||||
|
||||
The digit sequence in token_buffer has already been validated (by the
|
||||
scan_number() state machine or by the contiguous fast path) and holds the
|
||||
locale decimal point in place of '.'. Integers are parsed first and fall
|
||||
back to floating point on overflow. This is shared so both scanners produce
|
||||
identical results.
|
||||
*/
|
||||
token_type convert_number(token_type number_type)
|
||||
{
|
||||
const char* const num_begin = token_buffer.data();
|
||||
const char* const num_end = num_begin + token_buffer.size();
|
||||
|
||||
// try to parse integers first and fall back to floats; the digit
|
||||
// sequence has already been validated, so a dedicated parser can avoid
|
||||
// the locale/errno overhead of strtoull
|
||||
// try to parse integers first and fall back to floats
|
||||
if (number_type == token_type::value_unsigned)
|
||||
{
|
||||
if (parse_integer_unsigned(num_begin, num_end, value_unsigned))
|
||||
unsigned long long x{}; // NOLINT(runtime/int)
|
||||
const auto res = ::nlohmann::detail::from_chars(token_buffer.data(), token_buffer.data() + token_buffer.size(), x);
|
||||
|
||||
// we checked the number format before
|
||||
JSON_ASSERT(res.ptr == token_buffer.data() + token_buffer.size());
|
||||
|
||||
if (res.ec != std::errc::result_out_of_range)
|
||||
{
|
||||
return token_type::value_unsigned;
|
||||
JSON_ASSERT(res.ec == std::errc{});
|
||||
value_unsigned = static_cast<number_unsigned_t>(x);
|
||||
if (value_unsigned == x)
|
||||
{
|
||||
return token_type::value_unsigned;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (number_type == token_type::value_integer)
|
||||
{
|
||||
if (parse_integer_signed(num_begin, num_end, value_integer))
|
||||
long long x{}; // NOLINT(runtime/int)
|
||||
const auto res = ::nlohmann::detail::from_chars(token_buffer.data(), token_buffer.data() + token_buffer.size(), x);
|
||||
|
||||
// we checked the number format before
|
||||
JSON_ASSERT(res.ptr == token_buffer.data() + token_buffer.size());
|
||||
|
||||
if (res.ec != std::errc::result_out_of_range)
|
||||
{
|
||||
return token_type::value_integer;
|
||||
JSON_ASSERT(res.ec == std::errc{});
|
||||
value_integer = static_cast<number_integer_t>(x);
|
||||
if (value_integer == x)
|
||||
{
|
||||
return token_type::value_integer;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// this code is reached if we parse a floating-point number or if an
|
||||
// integer conversion above overflowed. Prefer std::from_chars
|
||||
// (Eisel-Lemire, locale-independent, correctly rounded) when available;
|
||||
// otherwise the exact Clinger fast path (double only); otherwise the
|
||||
// locale-aware strtof/strtod.
|
||||
if (parse_float_from_chars(num_begin, num_end, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
if (parse_float_fast(num_begin, num_end, decimal_point_char, value_float))
|
||||
{
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
||||
strtof(value_float, token_buffer.data(), &endptr);
|
||||
// integer conversion above failed
|
||||
const auto res = ::nlohmann::detail::from_chars(token_buffer.data(), token_buffer.data() + token_buffer.size(), value_float);
|
||||
|
||||
// we checked the number format before
|
||||
JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size());
|
||||
JSON_ASSERT(res.ptr == token_buffer.data() + token_buffer.size());
|
||||
JSON_ASSERT(res.ec != std::errc::invalid_argument);
|
||||
|
||||
return token_type::value_float;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief contiguous fast path for scanning a number
|
||||
|
||||
Parses the whole number token straight from the input buffer, avoiding the
|
||||
per-character get()/add() of scan_number(). On success it fills token_buffer
|
||||
(with the locale decimal point substituted, as scan_number() does) and
|
||||
returns the token type. On anything it does not fully recognize as a
|
||||
well-formed number it makes no state change and returns
|
||||
token_type::uninitialized, so the caller falls back to scan_number(), which
|
||||
then produces the exact diagnostic. @a current is the first digit or the
|
||||
leading minus (already read); the remaining bytes are taken from the adapter.
|
||||
*/
|
||||
token_type scan_number_bulk_contiguous()
|
||||
{
|
||||
// a pending unget offsets the buffer position from current; fall back
|
||||
if (next_unget)
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
const std::size_t rem = ia.bulk_remaining();
|
||||
if (rem == 0)
|
||||
{
|
||||
// the first digit is the last input byte; let scan_number() finish
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
// the byte before the next unread one is current (contiguous input)
|
||||
const char* const data = reinterpret_cast<const char*>(ia.bulk_data()) - 1;
|
||||
const std::size_t avail = rem + 1;
|
||||
|
||||
// validate + classify the number extent (mirrors scan_number()'s grammar)
|
||||
std::size_t i = 0;
|
||||
std::size_t dot_index = std::string::npos;
|
||||
token_type number_type = token_type::value_unsigned;
|
||||
if (data[0] == '-')
|
||||
{
|
||||
number_type = token_type::value_integer;
|
||||
i = 1;
|
||||
if (i >= avail)
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
}
|
||||
if (data[i] == '0')
|
||||
{
|
||||
++i;
|
||||
}
|
||||
else if (data[i] >= '1' && data[i] <= '9')
|
||||
{
|
||||
++i;
|
||||
while (i < avail && data[i] >= '0' && data[i] <= '9')
|
||||
{
|
||||
++i;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
if (i < avail && data[i] == '.')
|
||||
{
|
||||
number_type = token_type::value_float;
|
||||
dot_index = i;
|
||||
++i;
|
||||
if (i >= avail || !(data[i] >= '0' && data[i] <= '9'))
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
while (i < avail && data[i] >= '0' && data[i] <= '9')
|
||||
{
|
||||
++i;
|
||||
}
|
||||
}
|
||||
if (i < avail && (data[i] == 'e' || data[i] == 'E'))
|
||||
{
|
||||
number_type = token_type::value_float;
|
||||
++i;
|
||||
if (i < avail && (data[i] == '+' || data[i] == '-'))
|
||||
{
|
||||
++i;
|
||||
}
|
||||
if (i >= avail || !(data[i] >= '0' && data[i] <= '9'))
|
||||
{
|
||||
return token_type::uninitialized;
|
||||
}
|
||||
while (i < avail && data[i] >= '0' && data[i] <= '9')
|
||||
{
|
||||
++i;
|
||||
}
|
||||
}
|
||||
const std::size_t len = i;
|
||||
|
||||
// materialize the token exactly as scan_number() would, substituting the
|
||||
// locale decimal point so convert_number()'s strtof fallback stays valid.
|
||||
// reset() already cleared token_buffer, so append() fills it (assign() is
|
||||
// avoided because custom string_t types need not provide it)
|
||||
reset();
|
||||
token_buffer.append(reinterpret_cast<const typename string_t::value_type*>(data), len);
|
||||
if (dot_index != std::string::npos)
|
||||
{
|
||||
token_buffer[dot_index] = static_cast<typename string_t::value_type>(decimal_point_char);
|
||||
decimal_point_position = dot_index;
|
||||
}
|
||||
|
||||
// consume the remaining bytes of the number (current was already read)
|
||||
ia.bulk_skip(len - 1);
|
||||
position.chars_read_total += (len - 1);
|
||||
position.chars_read_current_line += (len - 1);
|
||||
|
||||
return convert_number(number_type);
|
||||
}
|
||||
|
||||
/// contiguous input: try the number fast path, else the byte-path scanner
|
||||
token_type scan_number_dispatch(std::true_type /*bulk*/)
|
||||
{
|
||||
const token_type t = scan_number_bulk_contiguous();
|
||||
return (t != token_type::uninitialized) ? t : scan_number();
|
||||
}
|
||||
|
||||
/// streaming input: always use the byte-path scanner
|
||||
token_type scan_number_dispatch(std::false_type /*bulk*/)
|
||||
{
|
||||
return scan_number();
|
||||
}
|
||||
|
||||
/*!
|
||||
@param[in] literal_text the literal text to expect
|
||||
@param[in] length the length of the passed literal text
|
||||
@@ -1882,7 +1651,7 @@ scan_number_done:
|
||||
case '7':
|
||||
case '8':
|
||||
case '9':
|
||||
return scan_number_dispatch(std::integral_constant<bool, bulk_scan> {});
|
||||
return scan_number();
|
||||
|
||||
// end of input (the null byte is needed when parsing from
|
||||
// string literals)
|
||||
|
||||
@@ -1,302 +0,0 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <cfloat> // FLT_EVAL_METHOD
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // int64_t, uint64_t
|
||||
#include <limits> // numeric_limits
|
||||
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
// std::from_chars lives in <charconv>, but being in C++17 mode does not
|
||||
// guarantee the header exists: GCC 7 sets __cplusplus to C++17 yet ships no
|
||||
// <charconv> (added in GCC 8; floating-point support in GCC 11). Guard the
|
||||
// include with __has_include so such toolchains fall back to the scalar path.
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__has_include)
|
||||
#if __has_include(<charconv>)
|
||||
#include <charconv> // from_chars (only used when __cpp_lib_to_chars is defined)
|
||||
#include <system_error> // errc
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// This file contains the value-conversion helpers used by the lexer to turn an
|
||||
// already-validated number token into a value, without the locale/errno
|
||||
// overhead of std::strtoull/std::strtod. They are free functions so the lexer
|
||||
// stays focused on scanning; see lexer::convert_number().
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
|
||||
/*!
|
||||
@brief fast integer parser for an already-validated unsigned integer
|
||||
|
||||
The number scanner has already checked that [first, last) is a valid JSON
|
||||
integer, so this only needs to accumulate the digits and detect overflow. This
|
||||
avoids the locale/errno machinery of std::strtoull, which dominates
|
||||
integer-heavy inputs.
|
||||
|
||||
@param[in] first pointer to the first character (a digit)
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] value the parsed value on success
|
||||
@return true if the value fit into @a NumberUnsignedType; false on overflow, in
|
||||
which case the caller falls back to floating-point parsing (matching the
|
||||
previous std::strtoull behavior)
|
||||
*/
|
||||
template<typename NumberUnsignedType>
|
||||
bool parse_integer_unsigned(const char* first, const char* last, NumberUnsignedType& value) noexcept
|
||||
{
|
||||
// accumulate in the widest unsigned type used by the previous strtoull
|
||||
// path so the overflow behavior is unchanged for custom number types
|
||||
std::uint64_t x = 0;
|
||||
constexpr std::uint64_t cutoff = (std::numeric_limits<std::uint64_t>::max)() / 10u;
|
||||
constexpr std::uint64_t cutlim = (std::numeric_limits<std::uint64_t>::max)() % 10u;
|
||||
for (const char* p = first; p != last; ++p)
|
||||
{
|
||||
const auto digit = static_cast<std::uint64_t>(static_cast<unsigned char>(*p) - static_cast<unsigned char>('0'));
|
||||
if (JSON_HEDLEY_UNLIKELY(x > cutoff || (x == cutoff && digit > cutlim)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
x = (x * 10u) + digit;
|
||||
}
|
||||
value = static_cast<NumberUnsignedType>(x);
|
||||
// reject values that do not round-trip into a narrower NumberUnsignedType
|
||||
return static_cast<std::uint64_t>(value) == x;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief fast integer parser for an already-validated negative integer
|
||||
|
||||
@param[in] first pointer to the leading '-'
|
||||
@param[in] last pointer past the last character
|
||||
@param[out] value the parsed (negative) value on success
|
||||
@return true on success; false on overflow (caller falls back to float)
|
||||
*/
|
||||
template<typename NumberIntegerType>
|
||||
bool parse_integer_signed(const char* first, const char* last, NumberIntegerType& value) noexcept
|
||||
{
|
||||
// the state machine only reaches the signed path via a leading '-'
|
||||
JSON_ASSERT(first != last && *first == '-');
|
||||
std::uint64_t magnitude = 0;
|
||||
// |INT64_MIN| == INT64_MAX + 1; this is the largest admissible magnitude
|
||||
constexpr std::uint64_t limit = static_cast<std::uint64_t>((std::numeric_limits<std::int64_t>::max)()) + 1u;
|
||||
for (const char* p = first + 1; p != last; ++p)
|
||||
{
|
||||
const auto digit = static_cast<std::uint64_t>(static_cast<unsigned char>(*p) - static_cast<unsigned char>('0'));
|
||||
if (JSON_HEDLEY_UNLIKELY(magnitude > (limit - digit) / 10u))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
magnitude = (magnitude * 10u) + digit;
|
||||
}
|
||||
const std::int64_t x = (magnitude == limit)
|
||||
? (std::numeric_limits<std::int64_t>::min)()
|
||||
: -static_cast<std::int64_t>(magnitude);
|
||||
value = static_cast<NumberIntegerType>(x);
|
||||
// reject values that do not round-trip into a narrower NumberIntegerType
|
||||
return static_cast<std::int64_t>(value) == x;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief exact fast path for parsing a `double` (Clinger's algorithm)
|
||||
|
||||
For the common case - at most 19 significant digits, a decimal exponent in
|
||||
[-22, 22], and a significand below 2^53 - the value equals significand *
|
||||
10^exp computed in IEEE-754 double arithmetic, which is exact under
|
||||
round-to-nearest because both operands are exactly representable. This is the
|
||||
same fast path used by fast_float/simdjson; the general cases are left to
|
||||
std::strtod. The parser only activates for number_float_t == double; float and
|
||||
long double keep the std::strtof/std::strtold paths (see the templated overload
|
||||
below).
|
||||
|
||||
@param[in] first pointer to the first character of the number
|
||||
@param[in] last pointer past the last character
|
||||
@param[in] decimal_point the (locale-dependent) decimal point character
|
||||
@param[out] out the parsed value on success
|
||||
@return true if the value was parsed exactly; false to fall back to strtod
|
||||
*/
|
||||
template<typename DecimalPointType>
|
||||
bool parse_float_fast(const char* first, const char* last, DecimalPointType decimal_point, double& out) noexcept
|
||||
{
|
||||
#if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0
|
||||
// Clinger's fast path is only exact when double operations are evaluated in
|
||||
// true double precision. On platforms that keep intermediates in extended
|
||||
// precision (e.g. the x87 FPU on 32-bit x86, where FLT_EVAL_METHOD == 2) the
|
||||
// single significand * 10^scale step is double-rounded and can be 1 ULP off,
|
||||
// so decline and let the caller fall back to the correctly-rounded
|
||||
// std::from_chars / std::strtod path.
|
||||
static_cast<void>(first);
|
||||
static_cast<void>(last);
|
||||
static_cast<void>(decimal_point);
|
||||
static_cast<void>(out);
|
||||
return false;
|
||||
#else
|
||||
static const std::array<double, 23> powers_of_ten =
|
||||
{
|
||||
{
|
||||
1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11,
|
||||
1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22
|
||||
}
|
||||
};
|
||||
|
||||
const char* p = first;
|
||||
bool negative = false;
|
||||
if (p != last && (*p == '-' || *p == '+'))
|
||||
{
|
||||
negative = (*p == '-');
|
||||
++p;
|
||||
}
|
||||
|
||||
std::uint64_t significand = 0;
|
||||
int num_digits = 0;
|
||||
int fractional_digits = 0;
|
||||
bool seen_dot = false;
|
||||
bool any_digit = false;
|
||||
for (; p != last; ++p)
|
||||
{
|
||||
const char c = *p;
|
||||
if (c >= '0' && c <= '9')
|
||||
{
|
||||
any_digit = true;
|
||||
if (JSON_HEDLEY_UNLIKELY(num_digits >= 19))
|
||||
{
|
||||
return false; // significand may not fit into uint64_t
|
||||
}
|
||||
significand = (significand * 10u) + static_cast<std::uint64_t>(c - '0');
|
||||
++num_digits;
|
||||
fractional_digits += static_cast<int>(seen_dot);
|
||||
}
|
||||
else if (static_cast<DecimalPointType>(c) == decimal_point)
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(seen_dot))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
seen_dot = true;
|
||||
}
|
||||
else if (c == 'e' || c == 'E')
|
||||
{
|
||||
++p;
|
||||
break;
|
||||
}
|
||||
else
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!any_digit))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
int exponent = 0;
|
||||
if (p != last) // an exponent part remains
|
||||
{
|
||||
bool exp_negative = false;
|
||||
if (p != last && (*p == '-' || *p == '+'))
|
||||
{
|
||||
exp_negative = (*p == '-');
|
||||
++p;
|
||||
}
|
||||
bool any_exp_digit = false;
|
||||
for (; p != last; ++p)
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(*p < '0' || *p > '9'))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
exponent = (exponent * 10) + (*p - '0');
|
||||
any_exp_digit = true;
|
||||
if (JSON_HEDLEY_UNLIKELY(exponent > 9999))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(!any_exp_digit))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if (exp_negative)
|
||||
{
|
||||
exponent = -exponent;
|
||||
}
|
||||
}
|
||||
|
||||
const int scale = exponent - fractional_digits;
|
||||
if (JSON_HEDLEY_UNLIKELY(significand >= (static_cast<std::uint64_t>(1) << 53)))
|
||||
{
|
||||
return false; // significand not exactly representable as double
|
||||
}
|
||||
|
||||
auto result = static_cast<double>(significand);
|
||||
if (scale >= 0)
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(scale > 22))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
result *= powers_of_ten[static_cast<std::size_t>(scale)];
|
||||
}
|
||||
else
|
||||
{
|
||||
if (JSON_HEDLEY_UNLIKELY(-scale > 22))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
result /= powers_of_ten[static_cast<std::size_t>(-scale)];
|
||||
}
|
||||
out = negative ? -result : result;
|
||||
return true;
|
||||
#endif
|
||||
}
|
||||
|
||||
/// fast float path is only exact for `double`; decline for float/long double
|
||||
template<typename DecimalPointType, typename FloatType>
|
||||
bool parse_float_fast(const char* /*first*/, const char* /*last*/, DecimalPointType /*decimal_point*/, FloatType& /*out*/) noexcept
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief parse a float with std::from_chars (Eisel-Lemire) when available
|
||||
|
||||
std::from_chars is locale-independent, correctly rounded, and - via the
|
||||
Eisel-Lemire algorithm in modern standard libraries - much faster than strtod
|
||||
over the whole value range (not just the Clinger subset). It is used only when
|
||||
__cpp_lib_to_chars indicates full floating-point support and only when it
|
||||
consumes the entire token ([first, last)); a partial parse means the buffer
|
||||
uses a non-'.' locale decimal point, in which case the caller falls back to the
|
||||
locale-aware path. An under-/overflow (result_out_of_range) also declines, so
|
||||
the caller's strtod fallback supplies the well-defined ±inf/0 result the parser
|
||||
expects (side-stepping the P4168 divergence between implementations).
|
||||
|
||||
@return true if the value was parsed exactly and fully; false to fall back
|
||||
*/
|
||||
template<typename FloatType>
|
||||
bool parse_float_from_chars(const char* first, const char* last, FloatType& out) noexcept
|
||||
{
|
||||
// JSON_HAS_CPP_17 must gate the use as well as the <charconv> include above:
|
||||
// some standard libraries (e.g. libstdc++ 15) define __cpp_lib_to_chars even
|
||||
// in C++14 mode, where <charconv> is not included.
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars)
|
||||
const auto result = std::from_chars(first, last, out);
|
||||
return result.ec == std::errc() && result.ptr == last;
|
||||
#else
|
||||
static_cast<void>(first);
|
||||
static_cast<void>(last);
|
||||
static_cast<void>(out);
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -1,237 +0,0 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint64_t
|
||||
#include <cstring> // memcpy
|
||||
|
||||
#if defined(JSON_USE_SIMDUTF)
|
||||
// Optional SIMD backend for bulk UTF-8 validation. This is an opt-in
|
||||
// external dependency: nlohmann/json itself stays header-only and the C++11
|
||||
// scalar validator below is always available; defining JSON_USE_SIMDUTF
|
||||
// additionally requires the simdutf headers on the include path and linking
|
||||
// the simdutf library. See string_bulk_run().
|
||||
#include <simdutf.h>
|
||||
#endif
|
||||
|
||||
#include <nlohmann/detail/macro_scope.hpp>
|
||||
|
||||
// This file contains the byte-level string-scanning helpers used by the lexer's
|
||||
// contiguous fast path. They operate purely on raw bytes (no dependency on the
|
||||
// lexer's template parameters) so they are free functions, keeping the lexer
|
||||
// itself focused on the state machine; see lexer::scan_string_bulk().
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
|
||||
// classify a single byte as needing individual string handling: the closing
|
||||
// quote, an escape, a control character, or a non-ASCII (UTF-8)
|
||||
// lead/continuation byte. Ordinary bytes (0x20..0x7F except '"' and '\\') are
|
||||
// copied verbatim, which the bulk scanner does 8 bytes at a time.
|
||||
inline bool is_string_special(unsigned char c) noexcept
|
||||
{
|
||||
return c == '\"' || c == '\\' || c < 0x20u || c >= 0x80u;
|
||||
}
|
||||
|
||||
// SWAR helper: return a word whose high bit is set in every byte of @a v that
|
||||
// is_string_special(); zero if the 8 bytes are all ordinary.
|
||||
inline std::uint64_t swar_string_special(std::uint64_t v) noexcept
|
||||
{
|
||||
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
||||
constexpr std::uint64_t high = 0x8080808080808080ull;
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
|
||||
const std::uint64_t has_quote = (q - ones) & ~q & high;
|
||||
const std::uint64_t has_backslash = (b - ones) & ~b & high;
|
||||
const std::uint64_t has_control = (v - 0x2020202020202020ull) & ~v & high; // < 0x20
|
||||
const std::uint64_t has_non_ascii = v & high; // >= 0x80
|
||||
return has_quote | has_backslash | has_control | has_non_ascii;
|
||||
}
|
||||
|
||||
// return the index of the first is_string_special() byte in [data, data+n), or
|
||||
// n if every byte is ordinary; scans 8 bytes at a time
|
||||
inline std::size_t find_string_special(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t word = 0;
|
||||
std::memcpy(&word, data + i, sizeof(word));
|
||||
if (swar_string_special(word) != 0)
|
||||
{
|
||||
// a special byte is in this word; locate it (endian-agnostic)
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
if (is_string_special(data[i + j]))
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
{
|
||||
if (is_string_special(data[i]))
|
||||
{
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its
|
||||
// length (2..4) only when the bytes form a *well-formed* sequence using exactly
|
||||
// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts
|
||||
// precisely what the byte path accepts. Returns 0 for anything that is invalid,
|
||||
// incomplete, or that the byte path must diagnose (the caller then defers to
|
||||
// that path, keeping error messages unchanged). Lead bytes < 0x80 are handled
|
||||
// by the caller and never passed here.
|
||||
inline std::size_t validate_one_utf8(const unsigned char* data, std::size_t avail) noexcept
|
||||
{
|
||||
const unsigned char c0 = data[0];
|
||||
if (c0 >= 0xC2 && c0 <= 0xDF) // U+0080..U+07FF
|
||||
{
|
||||
if (avail >= 2 && data[1] >= 0x80 && data[1] <= 0xBF)
|
||||
{
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
else if (c0 == 0xE0) // U+0800..U+0FFF
|
||||
{
|
||||
if (avail >= 3 && data[1] >= 0xA0 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF)
|
||||
{
|
||||
return 3;
|
||||
}
|
||||
}
|
||||
else if ((c0 >= 0xE1 && c0 <= 0xEC) || c0 == 0xEE || c0 == 0xEF) // U+1000..U+CFFF, U+E000..U+FFFF
|
||||
{
|
||||
if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF)
|
||||
{
|
||||
return 3;
|
||||
}
|
||||
}
|
||||
else if (c0 == 0xED) // U+D000..U+D7FF (excludes surrogates)
|
||||
{
|
||||
if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0x9F && data[2] >= 0x80 && data[2] <= 0xBF)
|
||||
{
|
||||
return 3;
|
||||
}
|
||||
}
|
||||
else if (c0 == 0xF0) // U+10000..U+3FFFF
|
||||
{
|
||||
if (avail >= 4 && data[1] >= 0x90 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
||||
{
|
||||
return 4;
|
||||
}
|
||||
}
|
||||
else if (c0 >= 0xF1 && c0 <= 0xF3) // U+40000..U+FFFFF
|
||||
{
|
||||
if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
||||
{
|
||||
return 4;
|
||||
}
|
||||
}
|
||||
else if (c0 == 0xF4) // U+100000..U+10FFFF
|
||||
{
|
||||
if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0x8F && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
||||
{
|
||||
return 4;
|
||||
}
|
||||
}
|
||||
return 0; // invalid, incomplete, or must be diagnosed by the byte path
|
||||
}
|
||||
|
||||
// Scalar (C++11) computation of the bulk run length: the number of leading
|
||||
// bytes in [data, data+n) that are ordinary ASCII or complete well-formed UTF-8
|
||||
// sequences, stopping before the first byte that needs individual handling (the
|
||||
// closing quote, an escape, a control character, or an ill-formed/truncated
|
||||
// sequence). ASCII is skipped 8 bytes at a time.
|
||||
inline std::size_t scalar_string_bulk_run(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
std::size_t pos = 0;
|
||||
while (pos < n)
|
||||
{
|
||||
pos += find_string_special(data + pos, n - pos);
|
||||
if (pos >= n || data[pos] < 0x80u)
|
||||
{
|
||||
break; // end of buffer, or a quote/escape/control byte
|
||||
}
|
||||
const std::size_t seq = validate_one_utf8(data + pos, n - pos);
|
||||
if (seq == 0)
|
||||
{
|
||||
break; // ill-formed or truncated: let the byte path diagnose it
|
||||
}
|
||||
pos += seq;
|
||||
}
|
||||
return pos;
|
||||
}
|
||||
|
||||
#if defined(JSON_USE_SIMDUTF)
|
||||
// Index of the first quote/escape/control byte in [data, data+n) (non-ASCII
|
||||
// bytes are *not* stops here - the whole run is handed to simdutf), or n.
|
||||
inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
||||
constexpr std::uint64_t high = 0x8080808080808080ull;
|
||||
std::size_t i = 0;
|
||||
for (; i + 8 <= n; i += 8)
|
||||
{
|
||||
std::uint64_t v = 0;
|
||||
std::memcpy(&v, data + i, sizeof(v));
|
||||
const std::uint64_t q = v ^ 0x2222222222222222ull;
|
||||
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull;
|
||||
const std::uint64_t hit = ((q - ones) & ~q & high)
|
||||
| ((b - ones) & ~b & high)
|
||||
| ((v - 0x2020202020202020ull) & ~v & high);
|
||||
if (hit != 0)
|
||||
{
|
||||
for (std::size_t j = 0; j < 8; ++j)
|
||||
{
|
||||
const unsigned char c = data[i + j];
|
||||
if (c == '\"' || c == '\\' || c < 0x20u)
|
||||
{
|
||||
return i + j;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (; i < n; ++i)
|
||||
{
|
||||
const unsigned char c = data[i];
|
||||
if (c == '\"' || c == '\\' || c < 0x20u)
|
||||
{
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Backend-dispatched bulk run length. With JSON_USE_SIMDUTF the run up to the
|
||||
// next delimiter is validated in one shot by simdutf; on the rare failure the
|
||||
// scalar helper recomputes the exact valid prefix so the byte path still
|
||||
// produces the precise diagnostic. Without it, the pure scalar path is used.
|
||||
inline std::size_t string_bulk_run(const unsigned char* data, std::size_t n) noexcept
|
||||
{
|
||||
#if defined(JSON_USE_SIMDUTF)
|
||||
const std::size_t run = find_string_delimiter(data, n);
|
||||
if (run != 0 && simdutf::validate_utf8(reinterpret_cast<const char*>(data), run))
|
||||
{
|
||||
return run;
|
||||
}
|
||||
return scalar_string_bulk_run(data, n);
|
||||
#else
|
||||
return scalar_string_bulk_run(data, n);
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -124,6 +124,20 @@
|
||||
#define JSON_HAS_FILESYSTEM 0
|
||||
#endif
|
||||
|
||||
#ifndef JSON_HAS_FROM_CHARS
|
||||
#if defined(JSON_HAS_CPP_17) && defined(__cpp_lib_to_chars)
|
||||
// The std::from_chars<float> implementation in libstdc++ 11 gets some of the corner cases wrong;
|
||||
// starting with libstdc++ 12, these are fixed by switching the implementation to fast_float.
|
||||
#if defined(_GLIBCXX_RELEASE) && _GLIBCXX_RELEASE < 12
|
||||
#define JSON_HAS_FROM_CHARS 0
|
||||
#else
|
||||
#define JSON_HAS_FROM_CHARS 1
|
||||
#endif
|
||||
#else
|
||||
#define JSON_HAS_FROM_CHARS 0
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifndef JSON_HAS_THREE_WAY_COMPARISON
|
||||
#if defined(__cpp_impl_three_way_comparison) && __cpp_impl_three_way_comparison >= 201907L \
|
||||
&& defined(__cpp_lib_three_way_comparison) && __cpp_lib_three_way_comparison >= 201907L
|
||||
|
||||
@@ -39,6 +39,7 @@
|
||||
#undef JSON_HAS_CPP_26
|
||||
#undef JSON_HAS_FILESYSTEM
|
||||
#undef JSON_HAS_EXPERIMENTAL_FILESYSTEM
|
||||
#undef JSON_HAS_FROM_CHARS
|
||||
#undef JSON_HAS_THREE_WAY_COMPARISON
|
||||
#undef JSON_HAS_RANGES
|
||||
#undef JSON_HAS_STD_FORMAT
|
||||
|
||||
+647
-846
File diff suppressed because it is too large
Load Diff
@@ -121,6 +121,10 @@ json_test_set_test_options(test-disabled_exceptions
|
||||
# raise timeout of expensive Unicode test
|
||||
json_test_set_test_options(test-unicode4 TEST_PROPERTIES TIMEOUT 3000)
|
||||
|
||||
# link pthreads to tests that need it
|
||||
find_package(Threads REQUIRED)
|
||||
json_test_set_test_options(test-regression2 LINK_LIBRARIES Threads::Threads)
|
||||
|
||||
#############################################################################
|
||||
# add unit tests
|
||||
#############################################################################
|
||||
|
||||
@@ -12,10 +12,6 @@
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
#include <sstream> // stringstream
|
||||
#include <string> // string
|
||||
#include <vector> // vector
|
||||
|
||||
namespace
|
||||
{
|
||||
// shortcut to scan a string literal
|
||||
@@ -59,6 +55,8 @@ TEST_CASE("lexer class")
|
||||
|
||||
SECTION("numbers")
|
||||
{
|
||||
// Number parsing implementation uses std::from_chars where available,
|
||||
// so run this test suite with JSON_HAS_CPP_17 as well.
|
||||
CHECK((scan_string("0") == json::lexer::token_type::value_unsigned));
|
||||
CHECK((scan_string("1") == json::lexer::token_type::value_unsigned));
|
||||
CHECK((scan_string("2") == json::lexer::token_type::value_unsigned));
|
||||
@@ -69,13 +67,20 @@ TEST_CASE("lexer class")
|
||||
CHECK((scan_string("7") == json::lexer::token_type::value_unsigned));
|
||||
CHECK((scan_string("8") == json::lexer::token_type::value_unsigned));
|
||||
CHECK((scan_string("9") == json::lexer::token_type::value_unsigned));
|
||||
CHECK((scan_string("18446744073709551615") == json::lexer::token_type::value_unsigned));
|
||||
|
||||
CHECK((scan_string("-0") == json::lexer::token_type::value_integer));
|
||||
CHECK((scan_string("-1") == json::lexer::token_type::value_integer));
|
||||
CHECK((scan_string("-9223372036854775808") == json::lexer::token_type::value_integer));
|
||||
|
||||
CHECK((scan_string("1.1") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("-1.1") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("1E10") == json::lexer::token_type::value_float));
|
||||
|
||||
// out-of-range integers/floats are treated as value_float tokens
|
||||
CHECK((scan_string("18446744073709551616") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("-9223372036854775809") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("1E400") == json::lexer::token_type::value_float));
|
||||
}
|
||||
|
||||
SECTION("whitespace")
|
||||
@@ -228,75 +233,3 @@ TEST_CASE("lexer class")
|
||||
CHECK((scan_string("/**//**//**/", true) == json::lexer::token_type::end_of_input));
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("lexer number fast path")
|
||||
{
|
||||
// The contiguous fast path (used for pointer/string input) must agree with
|
||||
// the streaming byte path (used for std::istream) on token type, numeric
|
||||
// value, and round-trip text for every well-formed number, and reject the
|
||||
// same malformed numbers with the same message.
|
||||
SECTION("contiguous vs streaming parity")
|
||||
{
|
||||
const std::vector<std::string> numbers =
|
||||
{
|
||||
"0", "-0", "1", "-1", "42", "-42", "10", "100", "1234567890",
|
||||
"0.0", "-0.0", "3.14", "-3.14", "0.5", "-0.001", "123.456789",
|
||||
"1e0", "1E0", "1e10", "1e-10", "1e+10", "1.5e3", "-2.5E-4",
|
||||
"9223372036854775807", // INT64_MAX -> unsigned
|
||||
"9223372036854775808", // INT64_MAX + 1 -> unsigned
|
||||
"18446744073709551615", // UINT64_MAX -> unsigned
|
||||
"18446744073709551616", // UINT64_MAX + 1 -> float
|
||||
"-9223372036854775808", // INT64_MIN -> integer
|
||||
"-9223372036854775809", // INT64_MIN - 1 -> float
|
||||
"123456789012345678901234567890", // huge -> float
|
||||
"0.30000000000000004", "2.2250738585072014e-308", "1e308",
|
||||
// high-precision / wide-exponent values that exercise the
|
||||
// std::from_chars (Eisel-Lemire) path beyond the Clinger subset
|
||||
"1.7976931348623157e308", "1.2345678901234567e-250",
|
||||
"9007199254740993", "5e-324", "1e-320"
|
||||
};
|
||||
|
||||
for (const auto& n : numbers)
|
||||
{
|
||||
const std::string doc = "[" + n + "]";
|
||||
|
||||
// contiguous fast path
|
||||
const json a = json::parse(doc);
|
||||
// streaming byte path
|
||||
std::stringstream ss(doc);
|
||||
const json b = json::parse(ss);
|
||||
|
||||
CAPTURE(n);
|
||||
CHECK(a == b);
|
||||
CHECK(a.dump() == b.dump());
|
||||
CHECK(a[0].type() == b[0].type());
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("token type classification")
|
||||
{
|
||||
CHECK((scan_string("0") == json::lexer::token_type::value_unsigned));
|
||||
CHECK((scan_string("-1") == json::lexer::token_type::value_integer));
|
||||
CHECK((scan_string("1.5") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("1e5") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("18446744073709551615") == json::lexer::token_type::value_unsigned));
|
||||
CHECK((scan_string("18446744073709551616") == json::lexer::token_type::value_float));
|
||||
CHECK((scan_string("-9223372036854775808") == json::lexer::token_type::value_integer));
|
||||
CHECK((scan_string("-9223372036854775809") == json::lexer::token_type::value_float));
|
||||
}
|
||||
|
||||
SECTION("malformed numbers are rejected identically")
|
||||
{
|
||||
for (const char* bad :
|
||||
{"-", "1.", "1e", "1e+", "1.2e", "01", "-01", "1..2", "1.2.3"
|
||||
})
|
||||
{
|
||||
CAPTURE(bad);
|
||||
// the contiguous fast path must decline and let the byte path report
|
||||
const std::string doc = std::string("[") + bad + "]";
|
||||
CHECK_FALSE(json::accept(doc));
|
||||
std::stringstream ss(doc);
|
||||
CHECK_FALSE(json::accept(ss));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,278 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstddef>
|
||||
#include <limits>
|
||||
#include <string>
|
||||
#include <system_error>
|
||||
|
||||
// nlohmann::details::from_chars is a wrapper around std::from_chars when it is
|
||||
// available (__cpp_lib_to_chars is defined), otherwise our fallback
|
||||
// implementation kicks in.
|
||||
//
|
||||
// Due to the incomplete implementation of this C++17 standard library feature
|
||||
// (for example, lack of support for long double even in current libc++),
|
||||
// JSON_HAS_CPP_17 does not guarantee that std::from_chars is available.
|
||||
// However, the reverse holds: If the test is compiled in C++11 mode, the
|
||||
// fallback implementation will be used for sure.
|
||||
// By mentioning the JSON_HAS_CPP_17 macro in this here comment, the test will
|
||||
// be compiled both using our fallback and, at least on some platforms,
|
||||
// the C++17 standard library version, as a way of ensuring that the
|
||||
// expectations formulated in this test align with `std::from_chars`.
|
||||
|
||||
namespace
|
||||
{
|
||||
struct init_val_t {};
|
||||
constexpr init_val_t init_val{};
|
||||
|
||||
template <typename T>
|
||||
void check_result(std::string& str, T& value, std::ptrdiff_t expected_ptr_offset, std::errc expected_ec)
|
||||
{
|
||||
// On platforms where neither `std::from_chars`, nor extended locale support (`strtof_l`) are
|
||||
// available, the `strtof`-based implementation is in fact locale-dependent and might use a
|
||||
// different decimal separator, so we need to adapt the test expectations accordingly.
|
||||
std::replace(str.begin(), str.end(), '.', nlohmann::detail::from_chars_traits<T>::get_decimal_point());
|
||||
|
||||
auto res = nlohmann::detail::from_chars(str.data(), str.data() + str.size(), value);
|
||||
CHECK_MESSAGE(res.ec == expected_ec, "Error code mismatch while parsing: \"", str,
|
||||
"\": Actual: ", make_error_code(res.ec).message(),
|
||||
"; expected: ", make_error_code(expected_ec).message());
|
||||
CHECK_MESSAGE(res.ptr - str.data() == expected_ptr_offset, "Ptr offset mismatch while parsing: \"", str, "\"");
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
typename std::enable_if<std::is_integral<T>::value, void>::type
|
||||
check(std::string str, T expected_value, std::ptrdiff_t expected_ptr_offset, std::errc expected_ec)
|
||||
{
|
||||
T value = 42;
|
||||
check_result(str, value, expected_ptr_offset, expected_ec);
|
||||
CHECK_MESSAGE(value == expected_value, "while parsing: \"", str, "\"");
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
typename std::enable_if<std::is_floating_point<T>::value, void>::type
|
||||
check(std::string str, T expected_value, std::ptrdiff_t expected_ptr_offset, std::errc expected_ec)
|
||||
{
|
||||
{
|
||||
T value = 42;
|
||||
check_result(str, value, expected_ptr_offset, expected_ec);
|
||||
CHECK_MESSAGE(value == expected_value, "while parsing: \"", str, "\"");
|
||||
CHECK_MESSAGE(std::signbit(value) == std::signbit(expected_value), "while parsing: \"", str, "\"");
|
||||
}
|
||||
{
|
||||
T value = std::numeric_limits<T>::quiet_NaN();
|
||||
check_result(str, value, expected_ptr_offset, expected_ec);
|
||||
CHECK_MESSAGE(value == expected_value, "while parsing: \"", str, "\"");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
typename std::enable_if<std::is_integral<T>::value, void>::type
|
||||
check(std::string str, init_val_t /*unused*/, std::ptrdiff_t expected_ptr_offset, std::errc expected_ec)
|
||||
{
|
||||
T const init_value = 42;
|
||||
T value = init_value;
|
||||
check_result(str, value, expected_ptr_offset, expected_ec);
|
||||
CHECK_MESSAGE(value == init_value, "while parsing: \"", str, "\"");
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
typename std::enable_if<std::is_floating_point<T>::value, void>::type
|
||||
check(std::string str, init_val_t /*unused*/, std::ptrdiff_t expected_ptr_offset, std::errc expected_ec)
|
||||
{
|
||||
{
|
||||
T const init_value = 42;
|
||||
T value = init_value;
|
||||
check_result(str, value, expected_ptr_offset, expected_ec);
|
||||
CHECK_MESSAGE(value == init_value, "while parsing: \"", str, "\"");
|
||||
}
|
||||
{
|
||||
T value = std::numeric_limits<T>::quiet_NaN();
|
||||
check_result(str, value, expected_ptr_offset, expected_ec);
|
||||
CHECK_MESSAGE(doctest::IsNaN<T>(value), "while parsing: \"", str, "\"");
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("integral results consistent with std::from_chars")
|
||||
{
|
||||
SECTION("unsigned long long")
|
||||
{
|
||||
check<unsigned long long>("", init_val, 0, std::errc::invalid_argument);
|
||||
check<unsigned long long>("0", 0ULL, 1, std::errc{});
|
||||
check<unsigned long long>(" 123", init_val, 0, std::errc::invalid_argument);
|
||||
check<unsigned long long>("123 ", 123ULL, 3, std::errc{});
|
||||
check<unsigned long long>("+123", init_val, 0, std::errc::invalid_argument);
|
||||
check<unsigned long long>("-123", init_val, 0, std::errc::invalid_argument);
|
||||
check<unsigned long long>("123", 123ULL, 3, std::errc{});
|
||||
check<unsigned long long>("123e10", 123ULL, 3, std::errc{});
|
||||
check<unsigned long long>("18446744073709551615", 18446744073709551615ULL, 20, std::errc{});
|
||||
check<unsigned long long>("18446744073709551616", init_val, 20, std::errc::result_out_of_range);
|
||||
}
|
||||
|
||||
SECTION("long long")
|
||||
{
|
||||
check<long long>("", init_val, 0, std::errc::invalid_argument);
|
||||
check<long long>("0", 0LL, 1, std::errc{});
|
||||
check<long long>(" 123", init_val, 0, std::errc::invalid_argument);
|
||||
check<long long>("123 ", 123LL, 3, std::errc{});
|
||||
check<long long>("+123", init_val, 0, std::errc::invalid_argument);
|
||||
check<long long>("-123", -123LL, 4, std::errc{});
|
||||
check<long long>("123", 123LL, 3, std::errc{});
|
||||
check<long long>("123e10", 123LL, 3, std::errc{});
|
||||
check<long long>("9223372036854775807", 9223372036854775807LL, 19, std::errc{});
|
||||
check<long long>("9223372036854775808", init_val, 19, std::errc::result_out_of_range);
|
||||
check<long long>("-9223372036854775808", -9223372036854775807LL - 1, 20, std::errc{});
|
||||
check<long long>("-9223372036854775809", init_val, 20, std::errc::result_out_of_range);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("floating point results consistent with std::from_chars")
|
||||
{
|
||||
INFO("Using decimal point '",
|
||||
std::string{1, nlohmann::detail::from_chars_traits<float>::get_decimal_point()},
|
||||
"' to match our possibly-locale-dependent from_chars fallback implementation");
|
||||
|
||||
SECTION("single precision")
|
||||
{
|
||||
check<float>("", init_val, 0, std::errc::invalid_argument);
|
||||
check<float>(" 123", init_val, 0, std::errc::invalid_argument);
|
||||
check<float>("123 ", 123.0f, 3, std::errc{});
|
||||
check<float>("+123", init_val, 0, std::errc::invalid_argument);
|
||||
check<float>("-123", -123.0f, 4, std::errc{});
|
||||
check<float>("123", 123.0f, 3, std::errc{});
|
||||
check<float>("123e10", 123e10f, 6, std::errc{});
|
||||
check<float>("123e+10", 123e10f, 7, std::errc{});
|
||||
check<float>("123e-10", 123e-10f, 7, std::errc{});
|
||||
check<float>("123.456", 123.456f, 7, std::errc{});
|
||||
check<float>("123;456", 123.0f, 3, std::errc{});
|
||||
check<float>("123.456 ", 123.456f, 7, std::errc{});
|
||||
check<float>("123;456 ", 123.0f, 3, std::errc{});
|
||||
check<float>("123456789.123456789", 123456789.123456789f, 19, std::errc{});
|
||||
check<float>("1e40", std::numeric_limits<float>::infinity(), 4, std::errc::result_out_of_range);
|
||||
check<float>("-1e40", -std::numeric_limits<float>::infinity(), 5, std::errc::result_out_of_range);
|
||||
check<float>("2e308", std::numeric_limits<float>::infinity(), 5, std::errc::result_out_of_range);
|
||||
check<float>("-2e308", -std::numeric_limits<float>::infinity(), 6, std::errc::result_out_of_range);
|
||||
check<float>("123.456e-789", 0.0f, 12, std::errc::result_out_of_range);
|
||||
check<float>("-123.456e-789", -0.0f, 13, std::errc::result_out_of_range);
|
||||
check<float>("1e-45", 1e-45f, 5, std::errc{});
|
||||
check<float>("1e-46", 0.0f, 5, std::errc::result_out_of_range);
|
||||
check<float>("1E-45", 1e-45f, 5, std::errc{});
|
||||
check<float>("1E-46", 0.0f, 5, std::errc::result_out_of_range);
|
||||
check<float>("10e-46", 1e-45f, 6, std::errc{});
|
||||
check<float>("0.1e-45", 0.0f, 7, std::errc::result_out_of_range);
|
||||
check<float>("100000000000000000000000000000000000000", 1e38f, 39, std::errc{});
|
||||
check<float>("100000000000000000000000000000000000000.0", 1e38f, 41, std::errc{});
|
||||
check<float>("1000000000000000000000000000000000000000e-1", 1e38f, 43, std::errc{});
|
||||
check<float>("0.0000000000000000000000000000000000000000000001e84", 1e38f, 51, std::errc{});
|
||||
check<float>("0.000000000000000000000000000000000000000000001", 1e-45f, 47, std::errc{});
|
||||
check<float>("00.000000000000000000000000000000000000000000001", 1e-45f, 48, std::errc{});
|
||||
check<float>("0.0000000000000000000000000000000000000000000001e1", 1e-45f, 50, std::errc{});
|
||||
check<float>("1000000000000000000000000000000000000000e-84", 1e-45f, 44, std::errc{});
|
||||
check<float>("1000000000000000000000000000000000000000", std::numeric_limits<float>::infinity(), 40, std::errc::result_out_of_range);
|
||||
check<float>("1000000000000000000000000000000000000000.0", std::numeric_limits<float>::infinity(), 42, std::errc::result_out_of_range);
|
||||
check<float>("100000000000000000000000000000000000000e1", std::numeric_limits<float>::infinity(), 41, std::errc::result_out_of_range);
|
||||
check<float>("0.0000000000000000000000000000000000000000000001e85", std::numeric_limits<float>::infinity(), 51, std::errc::result_out_of_range);
|
||||
check<float>("1000000000000000000000000000000000000000e-85", 0.0f, 44, std::errc::result_out_of_range);
|
||||
check<float>("0.0000000000000000000000000000000000000000000001", 0.0f, 48, std::errc::result_out_of_range);
|
||||
check<float>("0.000000000000000000000000000000000000000000001e-1", 0.0f, 50, std::errc::result_out_of_range);
|
||||
check<float>("1e-99999999999999999999", 0.0f, 23, std::errc::result_out_of_range);
|
||||
check<float>("1e+99999999999999999999", std::numeric_limits<float>::infinity(), 23, std::errc::result_out_of_range);
|
||||
check<float>("-1e-99999999999999999999", -0.0f, 24, std::errc::result_out_of_range);
|
||||
check<float>("-1e+99999999999999999999", -std::numeric_limits<float>::infinity(), 24, std::errc::result_out_of_range);
|
||||
}
|
||||
|
||||
SECTION("double precision")
|
||||
{
|
||||
check<double>("", init_val, 0, std::errc::invalid_argument);
|
||||
check<double>(" 123", init_val, 0, std::errc::invalid_argument);
|
||||
check<double>("123 ", 123.0, 3, std::errc{});
|
||||
check<double>("+123", init_val, 0, std::errc::invalid_argument);
|
||||
check<double>("-123", -123.0, 4, std::errc{});
|
||||
check<double>("123", 123.0, 3, std::errc{});
|
||||
check<double>("123e10", 123e10, 6, std::errc{});
|
||||
check<double>("123e+10", 123e10, 7, std::errc{});
|
||||
check<double>("123e-10", 123e-10, 7, std::errc{});
|
||||
check<double>("123.456", 123.456, 7, std::errc{});
|
||||
check<double>("123;456", 123.0, 3, std::errc{});
|
||||
check<double>("123.456 ", 123.456, 7, std::errc{});
|
||||
check<double>("123;456 ", 123.0, 3, std::errc{});
|
||||
check<double>("123456789.123456789", 123456789.123456789, 19, std::errc{});
|
||||
check<double>("1e40", 1e40, 4, std::errc{});
|
||||
check<double>("-1e40", -1e40, 5, std::errc{});
|
||||
check<double>("2e308", std::numeric_limits<double>::infinity(), 5, std::errc::result_out_of_range);
|
||||
check<double>("-2e308", -std::numeric_limits<double>::infinity(), 6, std::errc::result_out_of_range);
|
||||
check<double>("2e-324", 0.0, 6, std::errc::result_out_of_range);
|
||||
check<double>("-2e-324", -0.0, 7, std::errc::result_out_of_range);
|
||||
check<double>("123.456e-789", 0.0, 12, std::errc::result_out_of_range);
|
||||
check<double>("-123.456e-789", -0.0, 13, std::errc::result_out_of_range);
|
||||
check<double>("1e-99999999999999999999", 0.0, 23, std::errc::result_out_of_range);
|
||||
check<double>("1e+99999999999999999999", std::numeric_limits<double>::infinity(), 23, std::errc::result_out_of_range);
|
||||
check<double>("-1e-99999999999999999999", -0.0, 24, std::errc::result_out_of_range);
|
||||
check<double>("-1e+99999999999999999999", -std::numeric_limits<double>::infinity(), 24, std::errc::result_out_of_range);
|
||||
}
|
||||
|
||||
SECTION("long double precision")
|
||||
{
|
||||
check<long double>("", init_val, 0, std::errc::invalid_argument);
|
||||
check<long double>(" 123", init_val, 0, std::errc::invalid_argument);
|
||||
check<long double>("123 ", 123.0L, 3, std::errc{});
|
||||
check<long double>("+123", init_val, 0, std::errc::invalid_argument);
|
||||
check<long double>("-123", -123.0L, 4, std::errc{});
|
||||
check<long double>("123", 123.0L, 3, std::errc{});
|
||||
check<long double>("123e10", 123e10L, 6, std::errc{});
|
||||
check<long double>("123e+10", 123e10L, 7, std::errc{});
|
||||
check<long double>("123e-10", 123e-10L, 7, std::errc{});
|
||||
check<long double>("123.456", 123.456L, 7, std::errc{});
|
||||
check<long double>("123;456", 123.0L, 3, std::errc{});
|
||||
check<long double>("123.456 ", 123.456L, 7, std::errc{});
|
||||
check<long double>("123;456 ", 123.0L, 3, std::errc{});
|
||||
check<long double>("123456789.123456789", 123456789.123456789L, 19, std::errc{});
|
||||
check<long double>("1e40", 1e40L, 4, std::errc{});
|
||||
check<long double>("-1e40", -1e40L, 5, std::errc{});
|
||||
#if defined(_MSC_VER)
|
||||
#pragma warning(push)
|
||||
#pragma warning(disable: 4127)
|
||||
#endif
|
||||
if (sizeof(long double) > 8)
|
||||
#if defined(_MSC_VER)
|
||||
#pragma warning(pop)
|
||||
#endif
|
||||
{
|
||||
// Right-hand-side is calculated to avoid warning about literal
|
||||
// exceeding range on platforms where this branch is NOT taken.
|
||||
check<long double>("2e308", 1e308L * 2, 5, std::errc{});
|
||||
check<long double>("-2e308", -1e308L * 2, 6, std::errc{});
|
||||
check<long double>("2e-324", 8e-324L / 4, 6, std::errc{});
|
||||
check<long double>("-2e-324", -8e-324L / 4, 7, std::errc{});
|
||||
}
|
||||
else
|
||||
{
|
||||
check<long double>("2e308", std::numeric_limits<long double>::infinity(), 5, std::errc::result_out_of_range);
|
||||
check<long double>("-2e308", -std::numeric_limits<long double>::infinity(), 6, std::errc::result_out_of_range);
|
||||
check<long double>("2e-324", 0.0L, 6, std::errc::result_out_of_range);
|
||||
check<long double>("-2e-324", -0.0L, 7, std::errc::result_out_of_range);
|
||||
}
|
||||
check<long double>("1e5000", std::numeric_limits<long double>::infinity(), 6, std::errc::result_out_of_range);
|
||||
check<long double>("-1e5000", -std::numeric_limits<long double>::infinity(), 7, std::errc::result_out_of_range);
|
||||
check<long double>("123.456e-7890", 0.0L, 13, std::errc::result_out_of_range);
|
||||
check<long double>("-123.456e-7890", -0.0L, 14, std::errc::result_out_of_range);
|
||||
check<long double>("1e-99999999999999999999", 0.0L, 23, std::errc::result_out_of_range);
|
||||
check<long double>("1e+99999999999999999999", std::numeric_limits<long double>::infinity(), 23, std::errc::result_out_of_range);
|
||||
check<long double>("-1e-99999999999999999999", -0.0L, 24, std::errc::result_out_of_range);
|
||||
check<long double>("-1e+99999999999999999999", -std::numeric_limits<long double>::infinity(), 24, std::errc::result_out_of_range);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -26,8 +26,11 @@ using ordered_json = nlohmann::ordered_json;
|
||||
using namespace nlohmann::literals; // NOLINT(google-build-using-namespace)
|
||||
#endif
|
||||
|
||||
#include <atomic>
|
||||
#include <clocale>
|
||||
#include <cstdio>
|
||||
#include <list>
|
||||
#include <thread>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
|
||||
@@ -1239,6 +1242,44 @@ TEST_CASE("regression tests 2")
|
||||
CHECK(j == json({2, 4, 6}));
|
||||
}
|
||||
#endif
|
||||
|
||||
SECTION("issue #5198 - TOCTOU race between lexer construction and locale changes causes float truncation")
|
||||
{
|
||||
std::atomic<bool> stop_requested{};
|
||||
std::thread switcher([&stop_requested]
|
||||
{
|
||||
bool german = true;
|
||||
const std::string initial_locale = std::setlocale(LC_NUMERIC, nullptr);
|
||||
|
||||
while (!stop_requested)
|
||||
{
|
||||
const bool setlocale_res = std::setlocale(LC_NUMERIC, german ? "de_DE.UTF-8" : "C");
|
||||
WARN_MESSAGE(setlocale_res,
|
||||
"Setting locale failed, probably because de_DE.UTF-8 is not installed. "
|
||||
"This test was skipped as it would be inconclusive.");
|
||||
if (!setlocale_res)
|
||||
{
|
||||
stop_requested = true;
|
||||
}
|
||||
german = !german;
|
||||
}
|
||||
|
||||
(void)std::setlocale(LC_NUMERIC, initial_locale.c_str()); // restore original locale
|
||||
});
|
||||
|
||||
for (std::size_t i = 0; i < 10000 && !stop_requested; ++i)
|
||||
{
|
||||
const json j = json::parse("{\"val\": 99.123456789}");
|
||||
const double parsed = j.value("val", 0.0);
|
||||
if (!CHECK(parsed == 99.123456789))
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
stop_requested = true;
|
||||
switcher.join();
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE_TEMPLATE("issue #4798 - nlohmann::json::to_msgpack() encode float NaN as double", T, double, float) // NOLINT(readability-math-missing-parentheses, bugprone-throwing-static-initialization)
|
||||
|
||||
@@ -53,27 +53,6 @@ TEST_CASE("wide strings")
|
||||
std::wstring const w = L"\"\xDBFF";
|
||||
json _;
|
||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||
|
||||
// the exact message depends on the width of wchar_t: a 16-bit
|
||||
// wchar_t passes the lone surrogate to the UTF-8 decoder unchanged
|
||||
// (rejected as a single ill-formed byte at column 2), while a
|
||||
// 32-bit wchar_t first encodes it as an ill-formed three-byte
|
||||
// sequence (rejected one byte later, at column 3)
|
||||
const char* const error_low_surrogate = sizeof(wchar_t) == 2
|
||||
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
||||
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xB0'";
|
||||
const char* const error_high_surrogate = sizeof(wchar_t) == 2
|
||||
? "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'"
|
||||
: "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xED\xA0'";
|
||||
|
||||
// a lone low surrogate cannot start a pair
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'"'}), error_low_surrogate, json::parse_error&);
|
||||
// a high surrogate followed by a non-low-surrogate unit is invalid
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&);
|
||||
// a lone low surrogate must not swallow the following unit: pairing
|
||||
// it with any second unit would produce valid UTF-8, so the error
|
||||
// has to report an ill-formed byte at the surrogate's own position
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast<wchar_t>(0xDC00), L'a', L'"'}), error_low_surrogate, json::parse_error&);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -89,22 +68,11 @@ TEST_CASE("wide strings")
|
||||
|
||||
SECTION("invalid std::u16string")
|
||||
{
|
||||
if (u16string_is_utf16())
|
||||
if (wstring_is_utf16())
|
||||
{
|
||||
std::u16string const w = u"\"\xDBFF";
|
||||
json _;
|
||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||
|
||||
// a lone low surrogate cannot start a pair
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// a high surrogate followed by a non-low-surrogate unit is invalid
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// a lone low surrogate must not swallow the following unit: pairing
|
||||
// it with any second unit would produce valid UTF-8, so the error
|
||||
// has to report an ill-formed byte at the surrogate's own position
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
// a valid surrogate pair is still decoded (U+1F600)
|
||||
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user