mirror of
https://github.com/nlohmann/json.git
synced 2026-10-06 22:47:13 +00:00
An image is a document stored so that loading it needs no
parsing: save() writes the node index, the text and the decoded
strings; the static load() reads an image written by save().
load() takes a pointer and size, a borrowed vector, or an owned
rvalue vector; the nodes are copied so they are aligned and can
be edited, while the text and decoded strings stay in the image.
image_check controls how much load() trusts the input: full
checks structure, bounds, strings and numbers, the parser's own
guarantees; bounds checks structure and bounds only; none skips
all checks, for images from a trusted source.
Layout is little-endian only ("NJVI" header, nodes, text, decoded
strings), following the idea of zero-copy formats such as
FlatBuffers and YaFF; the check follows FlatBuffers' Verifier.
New errors: parse_error.116 for a malformed image or a failed
check, type_error.320 for a discarded document or a big-endian
target.
A dedicated fuzzer and 6,000 seeded corruptions, checked under
ASan/UBSan, found and fixed two gaps: unchecked reserved header
fields, and unbounded null/boolean offsets that could make
dump() throw std::length_error.
Signed-off-by: Niels Lohmann <mail@nlohmann.me>
220 lines
7.3 KiB
C++
220 lines
7.3 KiB
C++
// __ _____ _____ _____
|
|
// __| | __| | | | JSON for Modern C++
|
|
// | | |__ | | | | | | version 3.12.0
|
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
|
//
|
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
#pragma once
|
|
|
|
#include <cstddef> // size_t
|
|
#include <cstdint> // int64_t, uint64_t
|
|
#include <limits> // numeric_limits
|
|
#include <string> // string
|
|
#include <type_traits> // integral_constant
|
|
|
|
#include <nlohmann/json.hpp>
|
|
#include <nlohmann/detail/view/document_data.hpp>
|
|
#include <nlohmann/detail/view/macro_scope.hpp>
|
|
#include <nlohmann/detail/view/node.hpp>
|
|
#include <nlohmann/detail/view/scan.hpp>
|
|
|
|
NLOHMANN_JSON_NAMESPACE_BEGIN
|
|
namespace detail
|
|
{
|
|
namespace view
|
|
{
|
|
|
|
/*!
|
|
@brief locate the decimal point and the end of the mantissa of a float token
|
|
|
|
Also checks that the token is a JSON number. Tokens of the parser and of edits
|
|
always are; an image loaded with image_check::bounds can hold any bytes, which
|
|
must not reach the conversion (it expects a well-formed token).
|
|
*/
|
|
inline bool float_token_layout(const char* first, const char* last, std::size_t& dot, std::size_t& mantissa_end) noexcept
|
|
{
|
|
const auto digit = [last](const char* q)
|
|
{
|
|
return q != last && is_digit(static_cast<unsigned char>(*q));
|
|
};
|
|
const char* p = first;
|
|
p += (p != last && *p == '-') ? 1 : 0;
|
|
if (!digit(p) || (*p == '0' && digit(p + 1)))
|
|
{
|
|
return false;
|
|
}
|
|
while (digit(p))
|
|
{
|
|
++p;
|
|
}
|
|
dot = std::string::npos;
|
|
if (p != last && *p == '.')
|
|
{
|
|
dot = static_cast<std::size_t>(p - first);
|
|
if (!digit(++p))
|
|
{
|
|
return false;
|
|
}
|
|
while (digit(p))
|
|
{
|
|
++p;
|
|
}
|
|
}
|
|
mantissa_end = static_cast<std::size_t>(p - first);
|
|
if (p != last && (*p == 'e' || *p == 'E'))
|
|
{
|
|
++p;
|
|
p += (p != last && (*p == '+' || *p == '-')) ? 1 : 0;
|
|
if (!digit(p))
|
|
{
|
|
return false;
|
|
}
|
|
while (digit(p))
|
|
{
|
|
++p;
|
|
}
|
|
}
|
|
return p == last;
|
|
}
|
|
|
|
/*!
|
|
@brief the value of the float token of a node, as parse() converts it
|
|
|
|
Uses the lexer's conversion (detail::convert_float), so that the values are
|
|
bit-identical to parse(): float and double are converted without allocation
|
|
and independent of the locale. A token that is not a JSON number (only in a
|
|
damaged image loaded with image_check::bounds) yields 0.
|
|
*/
|
|
template<typename FloatType>
|
|
NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n)
|
|
{
|
|
const char* const last = first + n.len;
|
|
std::size_t dot = 0;
|
|
std::size_t mantissa_end = 0;
|
|
if (NLOHMANN_VIEW_UNLIKELY(!float_token_layout(first, last, dot, mantissa_end)))
|
|
{
|
|
return FloatType{};
|
|
}
|
|
return convert_float<FloatType>(first, last, dot, mantissa_end);
|
|
}
|
|
|
|
/*!
|
|
@brief the digits of a float token with at most 19 digits, from its layout
|
|
|
|
The digit layout recorded while parsing says where the integer digits, the
|
|
fraction digits, and the exponent are, so the digits are read eight at a
|
|
time without scanning.
|
|
|
|
@param[in] p first character of the token
|
|
@param[in] e end of the token
|
|
@param[in] limit end of the readable memory (the source text)
|
|
*/
|
|
NLOHMANN_VIEW_ALWAYS_INLINE float_significand layout_decimal(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept
|
|
{
|
|
const bool negative = *p == '-';
|
|
p += negative ? 1 : 0;
|
|
std::uint64_t w = parse_upto19(p, int_digits, limit);
|
|
p += int_digits;
|
|
std::int64_t q = 0;
|
|
if (frac_digits != 0)
|
|
{
|
|
w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit);
|
|
p += 1 + frac_digits;
|
|
q = -static_cast<std::int64_t>(frac_digits);
|
|
}
|
|
if (p != e)
|
|
{
|
|
// [eE][+-]digits; huge exponents saturate (the parser rejected
|
|
// overflow). The token is not read beyond e, and the digits are taken
|
|
// as unsigned, so that a token that is not well-formed (a damaged
|
|
// image loaded with image_check::bounds) yields a wrong value, but no
|
|
// overflow.
|
|
++p;
|
|
const bool exp_negative = p != e && *p == '-';
|
|
p += (p != e && (*p == '-' || *p == '+')) ? 1 : 0;
|
|
std::int64_t exp_value = 0;
|
|
for (; p != e; ++p)
|
|
{
|
|
if (exp_value < 0x10000000)
|
|
{
|
|
exp_value = (exp_value * 10) + static_cast<unsigned char>(*p - '0');
|
|
}
|
|
}
|
|
q += exp_negative ? -exp_value : exp_value;
|
|
}
|
|
|
|
float_significand d;
|
|
d.w = w;
|
|
d.exponent = q;
|
|
d.negative = negative;
|
|
return d;
|
|
}
|
|
|
|
/*!
|
|
@brief the value of a float token with at most 19 digits, from its layout
|
|
|
|
The result is correctly rounded by the lexer's conversion
|
|
(detail::decimal_to_float(): Clinger's fast path where both operands are
|
|
exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19
|
|
digits), so it is the value parse() produces.
|
|
*/
|
|
template<typename FloatType>
|
|
NLOHMANN_VIEW_ALWAYS_INLINE FloatType layout_float(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept
|
|
{
|
|
return decimal_to_float<FloatType>(layout_decimal(p, e, int_digits, frac_digits, limit));
|
|
}
|
|
|
|
/// the value of a float set by an edit: its token (the shortest round-trip
|
|
/// text, or "nan", "inf", "-inf") in the edit arena
|
|
template<typename FloatType>
|
|
NLOHMANN_VIEW_NOINLINE FloatType edited_float(const char* token, const node& n)
|
|
{
|
|
if (token[0] == 'n')
|
|
{
|
|
return std::numeric_limits<FloatType>::quiet_NaN();
|
|
}
|
|
if (token[0] == 'i' || (token[0] == '-' && token[1] == 'i'))
|
|
{
|
|
return token[0] == 'i' ? std::numeric_limits<FloatType>::infinity() : -std::numeric_limits<FloatType>::infinity();
|
|
}
|
|
return float_value<FloatType>(token, n);
|
|
}
|
|
|
|
/// the value of the float token of a node, as parse() converts it; floats and
|
|
/// doubles with at most 19 digits are converted from the digit layout
|
|
template<typename FloatType>
|
|
FloatType float_value(const document_data& d, const node& n)
|
|
{
|
|
if (NLOHMANN_VIEW_UNLIKELY((n.flags & node_flags::storage) == node_flags::edited))
|
|
{
|
|
return edited_float<FloatType>(d.str(n), n);
|
|
}
|
|
return float_value<FloatType>(d, n, std::integral_constant<bool, has_native_float_format<FloatType>::value> {});
|
|
}
|
|
|
|
template<typename FloatType>
|
|
FloatType float_value(const document_data& d, const node& n, std::true_type /*binary32 or binary64*/)
|
|
{
|
|
const unsigned int_digits = n.extra & 0xFFu;
|
|
const unsigned frac_digits = n.extra >> 8u;
|
|
if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many")
|
|
{
|
|
// (a float token not written by an edit is in the text)
|
|
const auto* const first = reinterpret_cast<const unsigned char*>(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
|
return layout_float<FloatType>(first, first + n.len, int_digits, frac_digits, reinterpret_cast<const unsigned char*>(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
|
}
|
|
return float_value<FloatType>(d.str(n), n);
|
|
}
|
|
|
|
template<typename FloatType>
|
|
FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/)
|
|
{
|
|
return float_value<FloatType>(d.str(n), n);
|
|
}
|
|
|
|
} // namespace view
|
|
} // namespace detail
|
|
NLOHMANN_JSON_NAMESPACE_END
|