mirror of
https://github.com/nlohmann/json.git
synced 2026-09-06 16:27:59 +00:00
Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d027f06a42 | ||
|
|
09b6b6b5ba |
@@ -125,6 +125,7 @@ The library uses the following mapping from JSON values types to BJData types ac
|
||||
|
||||
- `"_ArrayType_"` is one of `uint8`, `int8`, `uint16`, `int16`, `uint32`, `int32`, `uint64`, `int64`, `single`,
|
||||
`double`, `char`, or `byte`,
|
||||
- `"_ArraySize_"` is an array, since the dimensions are written as the ND-array header's length,
|
||||
- every entry of `"_ArraySize_"` is a non-negative integer, and their product is representable as a `std::size_t`,
|
||||
- `"_ArrayData_"` holds exactly that many elements, and
|
||||
- every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"` (a floating-point number for
|
||||
|
||||
@@ -1668,6 +1668,15 @@ class binary_writer
|
||||
CharType dtype = it->second;
|
||||
|
||||
key = "_ArraySize_";
|
||||
// the dimensions are written verbatim as the header length below, so a
|
||||
// value that is not an array cannot produce a valid one: null emits 'Z'
|
||||
// and an object emits '{', neither of which a reader accepts after '#'.
|
||||
// Such an object is not a valid ndarray and falls back to a plain object.
|
||||
if (!value.at(key).is_array())
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
std::size_t len = (value.at(key).empty() ? 0 : 1);
|
||||
for (const auto& el : value.at(key))
|
||||
{
|
||||
|
||||
@@ -71,7 +71,7 @@ class serializer
|
||||
, thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->thousands_sep)))
|
||||
, decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->decimal_point)))
|
||||
, indent_char(ichar)
|
||||
, indent_string(512, indent_char)
|
||||
, indent_string()
|
||||
, error_handler(error_handler_)
|
||||
{}
|
||||
|
||||
@@ -126,6 +126,10 @@ class serializer
|
||||
|
||||
// variable to hold indentation for recursive calls
|
||||
const auto new_indent = current_indent + indent_step;
|
||||
if (JSON_HEDLEY_UNLIKELY(indent_string.empty()))
|
||||
{
|
||||
indent_string.resize(512, indent_char);
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(indent_string.size() < new_indent))
|
||||
{
|
||||
indent_string.resize(indent_string.size() * 2, ' ');
|
||||
@@ -199,6 +203,10 @@ class serializer
|
||||
|
||||
// variable to hold indentation for recursive calls
|
||||
const auto new_indent = current_indent + indent_step;
|
||||
if (JSON_HEDLEY_UNLIKELY(indent_string.empty()))
|
||||
{
|
||||
indent_string.resize(512, indent_char);
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(indent_string.size() < new_indent))
|
||||
{
|
||||
indent_string.resize(indent_string.size() * 2, ' ');
|
||||
@@ -260,6 +268,10 @@ class serializer
|
||||
|
||||
// variable to hold indentation for recursive calls
|
||||
const auto new_indent = current_indent + indent_step;
|
||||
if (JSON_HEDLEY_UNLIKELY(indent_string.empty()))
|
||||
{
|
||||
indent_string.resize(512, indent_char);
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(indent_string.size() < new_indent))
|
||||
{
|
||||
indent_string.resize(indent_string.size() * 2, ' ');
|
||||
@@ -1010,7 +1022,7 @@ class serializer
|
||||
|
||||
/// the indentation character
|
||||
const char indent_char;
|
||||
/// the indentation string
|
||||
/// the indentation string (lazily allocated on first use by a pretty-print branch)
|
||||
string_t indent_string;
|
||||
|
||||
/// error_handler how to react on decoding errors
|
||||
|
||||
@@ -18676,6 +18676,15 @@ class binary_writer
|
||||
CharType dtype = it->second;
|
||||
|
||||
key = "_ArraySize_";
|
||||
// the dimensions are written verbatim as the header length below, so a
|
||||
// value that is not an array cannot produce a valid one: null emits 'Z'
|
||||
// and an object emits '{', neither of which a reader accepts after '#'.
|
||||
// Such an object is not a valid ndarray and falls back to a plain object.
|
||||
if (!value.at(key).is_array())
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
std::size_t len = (value.at(key).empty() ? 0 : 1);
|
||||
for (const auto& el : value.at(key))
|
||||
{
|
||||
@@ -20141,7 +20150,7 @@ class serializer
|
||||
, thousands_sep(loc->thousands_sep == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->thousands_sep)))
|
||||
, decimal_point(loc->decimal_point == nullptr ? '\0' : std::char_traits<char>::to_char_type(* (loc->decimal_point)))
|
||||
, indent_char(ichar)
|
||||
, indent_string(512, indent_char)
|
||||
, indent_string()
|
||||
, error_handler(error_handler_)
|
||||
{}
|
||||
|
||||
@@ -20196,6 +20205,10 @@ class serializer
|
||||
|
||||
// variable to hold indentation for recursive calls
|
||||
const auto new_indent = current_indent + indent_step;
|
||||
if (JSON_HEDLEY_UNLIKELY(indent_string.empty()))
|
||||
{
|
||||
indent_string.resize(512, indent_char);
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(indent_string.size() < new_indent))
|
||||
{
|
||||
indent_string.resize(indent_string.size() * 2, ' ');
|
||||
@@ -20269,6 +20282,10 @@ class serializer
|
||||
|
||||
// variable to hold indentation for recursive calls
|
||||
const auto new_indent = current_indent + indent_step;
|
||||
if (JSON_HEDLEY_UNLIKELY(indent_string.empty()))
|
||||
{
|
||||
indent_string.resize(512, indent_char);
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(indent_string.size() < new_indent))
|
||||
{
|
||||
indent_string.resize(indent_string.size() * 2, ' ');
|
||||
@@ -20330,6 +20347,10 @@ class serializer
|
||||
|
||||
// variable to hold indentation for recursive calls
|
||||
const auto new_indent = current_indent + indent_step;
|
||||
if (JSON_HEDLEY_UNLIKELY(indent_string.empty()))
|
||||
{
|
||||
indent_string.resize(512, indent_char);
|
||||
}
|
||||
if (JSON_HEDLEY_UNLIKELY(indent_string.size() < new_indent))
|
||||
{
|
||||
indent_string.resize(indent_string.size() * 2, ' ');
|
||||
@@ -21080,7 +21101,7 @@ class serializer
|
||||
|
||||
/// the indentation character
|
||||
const char indent_char;
|
||||
/// the indentation string
|
||||
/// the indentation string (lazily allocated on first use by a pretty-print branch)
|
||||
string_t indent_string;
|
||||
|
||||
/// error_handler how to react on decoding errors
|
||||
|
||||
@@ -2751,6 +2751,31 @@ TEST_CASE("BJData")
|
||||
CHECK(json::to_bjdata(j_ok) == std::vector<uint8_t>({'[', '$', 'U', '#', '[', 'i', 2, 'i', 3, ']', 1, 2, 3, 4, 5, 6}));
|
||||
CHECK(json::from_bjdata(json::to_bjdata(j_ok), true, true) == j_ok);
|
||||
}
|
||||
|
||||
SECTION("ndarray whose _ArraySize_ is not an array stays as object")
|
||||
{
|
||||
// the shape is written verbatim as the header length, so a
|
||||
// value that is not an array cannot produce a valid one: null
|
||||
// would emit 'Z' and an object '{', neither of which a reader
|
||||
// accepts after '#'. Both have to stay plain objects.
|
||||
json const j_null = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", nullptr}, {"_ArrayData_", json::array()}});
|
||||
const auto out_null = json::to_bjdata(j_null);
|
||||
CHECK(out_null.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_null) == j_null);
|
||||
|
||||
// an object shape passes the per-entry check by iterating its
|
||||
// values rather than dimensions, so it needs rejecting too
|
||||
json const j_obj = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {{"a", 1}}}, {"_ArrayData_", {1}}});
|
||||
const auto out_obj = json::to_bjdata(j_obj);
|
||||
CHECK(out_obj.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_obj) == j_obj);
|
||||
|
||||
// a scalar shape is not a dimension list either
|
||||
json const j_num = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", 1}, {"_ArrayData_", {1}}});
|
||||
const auto out_num = json::to_bjdata(j_num);
|
||||
CHECK(out_num.at(0) == '{');
|
||||
CHECK(json::from_bjdata(out_num) == j_num);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -14,6 +14,40 @@ using nlohmann::json;
|
||||
#include <array>
|
||||
#include <sstream>
|
||||
#include <iomanip>
|
||||
#include <cstdlib>
|
||||
#include <new>
|
||||
|
||||
namespace
|
||||
{
|
||||
// heap allocation counter used by the regression test for issue #5413
|
||||
// (https://github.com/nlohmann/json/issues/5413); disabled (and thus a
|
||||
// no-op besides the counting) unless explicitly toggled on
|
||||
bool count_heap_allocations = false; // NOLINT(cppcoreguidelines-avoid-non-const-global-variables)
|
||||
std::size_t heap_allocations = 0; // NOLINT(cppcoreguidelines-avoid-non-const-global-variables)
|
||||
} // namespace
|
||||
|
||||
void* operator new (std::size_t size) // NOLINT(cppcoreguidelines-owning-memory,misc-new-delete-overloads)
|
||||
{
|
||||
if (count_heap_allocations)
|
||||
{
|
||||
++heap_allocations;
|
||||
}
|
||||
if (void* ptr = std::malloc(size)) // NOLINT(cppcoreguidelines-no-malloc,cppcoreguidelines-owning-memory)
|
||||
{
|
||||
return ptr;
|
||||
}
|
||||
throw std::bad_alloc(); // NOLINT(hicpp-exception-baseclass)
|
||||
}
|
||||
|
||||
void operator delete (void* ptr) noexcept // NOLINT(cppcoreguidelines-owning-memory,misc-new-delete-overloads)
|
||||
{
|
||||
std::free(ptr); // NOLINT(cppcoreguidelines-no-malloc,cppcoreguidelines-owning-memory)
|
||||
}
|
||||
|
||||
void operator delete (void* ptr, std::size_t /*size*/) noexcept // NOLINT(cppcoreguidelines-owning-memory,misc-new-delete-overloads)
|
||||
{
|
||||
std::free(ptr); // NOLINT(cppcoreguidelines-no-malloc,cppcoreguidelines-owning-memory)
|
||||
}
|
||||
|
||||
TEST_CASE("serialization")
|
||||
{
|
||||
@@ -382,3 +416,76 @@ TEST_CASE("dump for basic_json with long double number_float_t")
|
||||
check_same(100.0L, 100.0);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("regression test for issue #5413 - lazily allocated indent_string")
|
||||
{
|
||||
// the serializer used to unconditionally allocate a 512-byte
|
||||
// indent_string in its constructor, even though it is only ever read
|
||||
// inside the pretty_print branches of dump(). This wasted a heap
|
||||
// allocation (and its matching deallocation) on every single compact
|
||||
// (i.e. non-pretty, the default) dump() call. indent_string is now
|
||||
// allocated lazily, the first time a pretty-print branch actually
|
||||
// needs it -- so a compact dump() must perform strictly fewer heap
|
||||
// allocations than a pretty dump() of the same value.
|
||||
const json j = {{"level", "info"}, {"msg", "hello world"}, {"id", 12345}};
|
||||
|
||||
// warm up anything unrelated to indentation (e.g., one-time locale
|
||||
// lookups) that might otherwise allocate on first use regardless of
|
||||
// pretty-printing, so it does not skew the counts measured below
|
||||
const auto warmup = j.dump();
|
||||
const auto warmup_pretty = j.dump(4);
|
||||
CHECK(!warmup.empty());
|
||||
CHECK(!warmup_pretty.empty());
|
||||
|
||||
SECTION("compact dump() has a stable, minimal allocation count")
|
||||
{
|
||||
count_heap_allocations = true;
|
||||
|
||||
heap_allocations = 0;
|
||||
const auto compact1 = j.dump();
|
||||
const auto allocs_compact1 = heap_allocations;
|
||||
|
||||
heap_allocations = 0;
|
||||
const auto compact2 = j.dump(-1);
|
||||
const auto allocs_compact2 = heap_allocations;
|
||||
|
||||
count_heap_allocations = false;
|
||||
|
||||
CHECK(compact1 == compact2);
|
||||
// dump() and dump(-1) both take the compact code path and must
|
||||
// never touch indent_string, so they allocate identically often
|
||||
CHECK(allocs_compact1 == allocs_compact2);
|
||||
}
|
||||
|
||||
SECTION("first pretty dump() allocates more than a compact dump()")
|
||||
{
|
||||
// use a tiny value whose compact ({"a":1}, 7 bytes) and pretty
|
||||
// ({"a": 1} with 1-space indent, 11 bytes) serializations both stay
|
||||
// well inside every common std::string small-string-optimization
|
||||
// buffer (>= 15 bytes on libstdc++/MSVC STL, >= 22 on libc++), so
|
||||
// building the result string itself causes no heap allocation
|
||||
// either way -- isolating indent_string as the only thing that can
|
||||
// possibly account for a difference in allocation count
|
||||
const json tiny = {{"a", 1}};
|
||||
|
||||
count_heap_allocations = true;
|
||||
|
||||
heap_allocations = 0;
|
||||
const auto compact = tiny.dump();
|
||||
const auto allocs_compact = heap_allocations;
|
||||
|
||||
heap_allocations = 0;
|
||||
const auto pretty = tiny.dump(1);
|
||||
const auto allocs_pretty = heap_allocations;
|
||||
|
||||
count_heap_allocations = false;
|
||||
|
||||
CHECK(compact == "{\"a\":1}");
|
||||
CHECK(pretty == "{\n \"a\": 1\n}");
|
||||
// a fresh serializer is created per dump() call; the pretty branch
|
||||
// lazily allocates indent_string on its first use, so it must
|
||||
// allocate at least once more than the compact branch, which never
|
||||
// touches indent_string at all
|
||||
CHECK(allocs_pretty > allocs_compact);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user