Files
json/tests/src/unit-alt-string.cpp
T
Niels Lohmann ccb290facf Reduce the string_t and array_t members the library requires
Several members were required only because of how the library happened to
be written, not because the functionality needs them. Dropping them widens
the set of usable string and array types, and one of them was also a
performance problem.

string_t:

- c_str() is gone. Every call site already knew the length and passed it
  along, so data() is enough. The one place that did not, the diagnostics
  path in exceptions.hpp, now builds the token from data() and size(),
  which also stops it from truncating keys that contain a null byte.
- back() is gone; the serializer indexes the last character instead.
- find(str, pos), replace(), and substr() are gone. escape() and
  unescape() rebuilt the string with one replace() per escaped character,
  which moves the tail every time: escaping a string of n characters that
  all need escaping cost O(n^2). Both now scan with find_first_of() -- a
  member the pointer parser already required -- and append whole runs, so
  the common case is one search and one copy. Escaping 64000 tildes drops
  from 717 ms to 20 ms; a string with nothing to escape gets faster too
  (8.4 ms to 5.8 ms), because the scan is still a single memchr per pass.
  json_pointer::split() takes its reference tokens with the
  (const char*, size_type) constructor rather than substr().
- json_pointer::to_string() accumulates with concat<string_t> instead of
  letting concat default to std::string and converting afterwards, so
  streaming a json_pointer no longer requires string_t to be assignable
  from a std::string.

array_t:

- at(size_type) is gone. basic_json::at(size_type) checked the index by
  calling array_t::at() and translating std::out_of_range, which also
  required the array type to throw that exact exception. It now compares
  against size() and uses operator[]. The thrown exception, its message,
  and the behaviour under JSON_NOEXCEPTION are unchanged.

The BSON writer wrote the terminating null byte out of the string's own
buffer (size() + 1). It now writes the byte itself, so string_t::data()
need not be null-terminated for to_bson().

The tests pin the reduced API: alt_string loses the five dropped members
and gains coverage of the escaping paths, and a std::vector whose at() is
hidden is used as an ArrayType.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-08-28 17:38:58 +00:00

377 lines
11 KiB
C++

// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++ (supporting code)
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-FileCopyrightText: 2018 Vitaliy Manushkin <agri@akamo.info>
// SPDX-License-Identifier: MIT
#include "doctest_compatibility.h"
#include <nlohmann/json.hpp>
#include <cstdint>
#include <string>
#include <utility>
#include <vector>
/* forward declarations */
class alt_string;
bool operator<(const char* op1, const alt_string& op2) noexcept; // NOLINT(misc-use-internal-linkage)
void int_to_string(alt_string& target, std::size_t value); // NOLINT(misc-use-internal-linkage)
/*
* This is virtually a string class.
* It covers std::string under the hood.
*
* It deliberately does not provide c_str(), back(), find(str, pos), replace(),
* or substr(): the library must not rely on them. Do not add members here
* without checking that the library actually needs them.
*/
class alt_string
{
public:
using value_type = std::string::value_type;
static constexpr auto npos = (std::numeric_limits<std::size_t>::max)();
alt_string(const char* str): str_impl(str) {}
alt_string(const char* str, std::size_t count): str_impl(str, count) {}
alt_string(size_t count, char chr): str_impl(count, chr) {}
alt_string() = default;
alt_string& append(char ch)
{
str_impl.push_back(ch);
return *this;
}
alt_string& append(const alt_string& str)
{
str_impl.append(str.str_impl);
return *this;
}
alt_string& append(const char* s, std::size_t length)
{
str_impl.append(s, length);
return *this;
}
void push_back(char c)
{
str_impl.push_back(c);
}
template <typename op_type>
bool operator==(const op_type& op) const
{
return str_impl == op;
}
bool operator==(const alt_string& op) const
{
return str_impl == op.str_impl;
}
template <typename op_type>
bool operator!=(const op_type& op) const
{
return str_impl != op;
}
bool operator!=(const alt_string& op) const
{
return str_impl != op.str_impl;
}
std::size_t size() const noexcept
{
return str_impl.size();
}
void resize (std::size_t n)
{
str_impl.resize(n);
}
void resize (std::size_t n, char c)
{
str_impl.resize(n, c);
}
template <typename op_type>
bool operator<(const op_type& op) const noexcept
{
return str_impl < op;
}
bool operator<(const alt_string& op) const noexcept
{
return str_impl < op.str_impl;
}
char& operator[](std::size_t index)
{
return str_impl[index];
}
const char& operator[](std::size_t index) const
{
return str_impl[index];
}
void clear()
{
str_impl.clear();
}
const value_type* data() const
{
return str_impl.data();
}
bool empty() const
{
return str_impl.empty();
}
std::size_t find_first_of(char c, std::size_t pos = 0) const
{
return str_impl.find_first_of(c, pos);
}
void reserve( std::size_t new_cap = 0 )
{
str_impl.reserve(new_cap);
}
private:
std::string str_impl {}; // NOLINT(readability-redundant-member-init)
friend bool operator<(const char* /*op1*/, const alt_string& /*op2*/) noexcept;
};
void int_to_string(alt_string& target, std::size_t value)
{
target = std::to_string(value).c_str();
}
using alt_json = nlohmann::basic_json <
std::map,
std::vector,
alt_string,
bool,
std::int64_t,
std::uint64_t,
double,
std::allocator,
nlohmann::adl_serializer >;
bool operator<(const char* op1, const alt_string& op2) noexcept
{
return op1 < op2.str_impl;
}
TEST_CASE("alternative string type")
{
SECTION("binary formats")
{
alt_json doc;
doc["pi"] = 3.141;
doc["happy"] = true;
doc["list"] = {1, 2, 3};
CHECK(alt_json::from_cbor(alt_json::to_cbor(doc)) == doc);
CHECK(alt_json::from_msgpack(alt_json::to_msgpack(doc)) == doc);
// BSON is not covered: it additionally needs string_t::find(value_type),
// which alt_string does not provide
CHECK(alt_json::from_ubjson(alt_json::to_ubjson(doc)) == doc);
// a UBJSON high-precision number is parsed into a std::string that the
// reader has to hand to the SAX interface as an alt_string
const std::vector<uint8_t> high_precision =
{
'H', 'i', 0x16, '3', '.', '1', '4', '1', '5', '9', '2', '6', '5', '3',
'5', '8', '9', '7', '9', '3', '2', '3', '8', '4', '6'
};
const auto number = alt_json::from_ubjson(high_precision);
CHECK(number.is_number_float());
CHECK(number.get<double>() == doctest::Approx(3.14159265358979323846));
}
SECTION("dump")
{
{
alt_json doc;
doc["pi"] = 3.141;
const alt_string dump = doc.dump();
CHECK(dump == R"({"pi":3.141})");
}
{
alt_json doc;
doc["happy"] = true;
const alt_string dump = doc.dump();
CHECK(dump == R"({"happy":true})");
}
{
alt_json doc;
doc["name"] = "I'm Batman";
const alt_string dump = doc.dump();
CHECK(dump == R"({"name":"I'm Batman"})");
}
{
alt_json doc;
doc["nothing"] = nullptr;
const alt_string dump = doc.dump();
CHECK(dump == R"({"nothing":null})");
}
{
alt_json doc;
doc["answer"]["everything"] = 42;
const alt_string dump = doc.dump();
CHECK(dump == R"({"answer":{"everything":42}})");
}
{
alt_json doc;
doc["list"] = { 1, 0, 2 };
const alt_string dump = doc.dump();
CHECK(dump == R"({"list":[1,0,2]})");
}
{
alt_json doc;
doc["object"] = { {"currency", "USD"}, {"value", 42.99} };
const alt_string dump = doc.dump();
CHECK(dump == R"({"object":{"currency":"USD","value":42.99}})");
}
}
SECTION("parse")
{
auto doc = alt_json::parse(R"({"foo": "bar"})");
const alt_string dump = doc.dump();
CHECK(dump == R"({"foo":"bar"})");
}
SECTION("items")
{
auto doc = alt_json::parse(R"({"foo": "bar"})");
for (const auto& item : doc.items())
{
CHECK(item.key() == "foo");
CHECK(item.value() == "bar");
}
auto doc_array = alt_json::parse(R"(["foo", "bar"])");
for (const auto& item : doc_array.items())
{
if (item.key() == "0" )
{
CHECK( item.value() == "foo" );
}
else if (item.key() == "1" )
{
CHECK(item.value() == "bar");
}
else
{
CHECK(false);
}
}
}
SECTION("equality")
{
alt_json doc;
doc["Who are you?"] = "I'm Batman";
CHECK("I'm Batman" == doc["Who are you?"]);
CHECK(doc["Who are you?"] == "I'm Batman");
CHECK_FALSE("I'm Batman" != doc["Who are you?"]);
CHECK_FALSE(doc["Who are you?"] != "I'm Batman");
CHECK("I'm Bruce Wayne" != doc["Who are you?"]);
CHECK(doc["Who are you?"] != "I'm Bruce Wayne");
CHECK_FALSE("I'm Bruce Wayne" == doc["Who are you?"]);
CHECK_FALSE(doc["Who are you?"] == "I'm Bruce Wayne");
{
const alt_json& const_doc = doc;
CHECK("I'm Batman" == const_doc["Who are you?"]);
CHECK(const_doc["Who are you?"] == "I'm Batman");
CHECK_FALSE("I'm Batman" != const_doc["Who are you?"]);
CHECK_FALSE(const_doc["Who are you?"] != "I'm Batman");
CHECK("I'm Bruce Wayne" != const_doc["Who are you?"]);
CHECK(const_doc["Who are you?"] != "I'm Bruce Wayne");
CHECK_FALSE("I'm Bruce Wayne" == const_doc["Who are you?"]);
CHECK_FALSE(const_doc["Who are you?"] == "I'm Bruce Wayne");
}
}
SECTION("JSON pointer")
{
// Direct conversion from a json literal to alt_json is not supported due to issue #3425:
// alt_json's string_t (alt_string) is not directly constructible from std::string, so the
// cross-basic_json conversion falls back to the array-conversion path, incorrectly representing
// objects as arrays of [key, value] pairs and strings as arrays of character codes.
// See https://github.com/nlohmann/json/issues/3425 for details.
// Workaround: use alt_json::parse() instead of implicit conversion.
auto j = alt_json::parse(R"({"foo": ["bar", "baz"]})");
CHECK(j.at(alt_json::json_pointer("/foo/0")) == j["foo"][0]);
CHECK(j.at(alt_json::json_pointer("/foo/1")) == j["foo"][1]);
// RFC 6901 escaping works without string_t::find(str, pos), replace(),
// and substr()
auto j2 = alt_json::parse(R"({"a/b": 1, "m~n": 2, "~/~~//": 3})");
CHECK(j2.at(alt_json::json_pointer("/a~1b")) == 1);
CHECK(j2.at(alt_json::json_pointer("/m~0n")) == 2);
CHECK(j2.at(alt_json::json_pointer("/~0~1~0~0~1~1")) == 3);
CHECK(alt_json::json_pointer("/~0~1~0~0~1~1").to_string() == alt_string("/~0~1~0~0~1~1"));
CHECK(j2.flatten().unflatten() == j2);
}
SECTION("patch")
{
alt_json const patch1 = alt_json::parse(R"([{ "op": "add", "path": "/a/b", "value": [ "foo", "bar" ] }])");
alt_json const doc1 = alt_json::parse(R"({ "a": { "foo": 1 } })");
CHECK_NOTHROW(doc1.patch(patch1));
alt_json doc1_ans = alt_json::parse(R"(
{
"a": {
"foo": 1,
"b": [ "foo", "bar" ]
}
}
)");
CHECK(doc1.patch(patch1) == doc1_ans);
}
SECTION("diff")
{
alt_json const j1 = {"foo", "bar", "baz"};
alt_json const j2 = {"foo", "bam"};
CHECK(alt_json::diff(j1, j2).dump() == "[{\"op\":\"replace\",\"path\":\"/1\",\"value\":\"bam\"},{\"op\":\"remove\",\"path\":\"/2\"}]");
}
SECTION("flatten")
{
// a JSON value
const alt_json j = alt_json::parse(R"({"foo": ["bar", "baz"]})");
const auto j2 = j.flatten();
CHECK(j2.dump() == R"({"/foo/0":"bar","/foo/1":"baz"})");
}
}