mirror of
https://github.com/nlohmann/json.git
synced 2026-10-04 21:50:33 +00:00
Merge remote-tracking branch 'origin/develop' into claude/fix-issue-3989-db7e45
Signed-off-by: Niels Lohmann <mail@nlohmann.me> # Conflicts: # include/nlohmann/detail/input/binary_reader.hpp # include/nlohmann/detail/string_utils.hpp # single_include/nlohmann/json.hpp
This commit is contained in:
@@ -44,6 +44,10 @@ TEST_CASE("default namespace")
|
||||
expected += "_snul";
|
||||
#endif
|
||||
|
||||
#if JSON_STRICT_BINARY_UTF8
|
||||
expected += "_sbu8";
|
||||
#endif
|
||||
|
||||
expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR);
|
||||
expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR);
|
||||
expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json";
|
||||
|
||||
@@ -45,6 +45,10 @@ TEST_CASE("default namespace without version component")
|
||||
expected += "_snul";
|
||||
#endif
|
||||
|
||||
#if JSON_STRICT_BINARY_UTF8
|
||||
expected += "_sbu8";
|
||||
#endif
|
||||
|
||||
expected += "::basic_json";
|
||||
|
||||
// fallback for Clang
|
||||
|
||||
@@ -0,0 +1,362 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
namespace
|
||||
{
|
||||
|
||||
struct ill_formed_case
|
||||
{
|
||||
const char* name;
|
||||
std::string bytes;
|
||||
};
|
||||
|
||||
// RFC 3629 ill-formed sequences used throughout this file, plus one
|
||||
// well-formed sequence for contrast
|
||||
const std::vector<ill_formed_case> ill_formed_cases =
|
||||
{
|
||||
{"overlong", "\xC0\xAE"},
|
||||
{"lone_0xFF", "\xFF"},
|
||||
{"truncated", "\xE2\x82"},
|
||||
{"surrogate", "\xED\xA0\x80"},
|
||||
};
|
||||
|
||||
const std::string valid_sequence = "\xC3\xA9"; // U+00E9, "é"
|
||||
|
||||
using eh = json::error_handler_t;
|
||||
const std::vector<eh> all_handlers = {eh::strict, eh::replace, eh::ignore, eh::keep};
|
||||
|
||||
// what dump()+parse() produces for a sanitizing error_handler; this is the
|
||||
// ground truth every binary writer/reader is checked against
|
||||
std::string dump_and_parse(const std::string& raw, eh error_handler)
|
||||
{
|
||||
return json::parse(json(raw).dump(-1, ' ', false, error_handler)).get<std::string>();
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
TEST_CASE("UTF-8 error_handler for the binary readers and writers")
|
||||
{
|
||||
SECTION("writers: string value")
|
||||
{
|
||||
for (const auto& c : ill_formed_cases)
|
||||
{
|
||||
CAPTURE(c.name);
|
||||
const json jval = c.bytes;
|
||||
|
||||
CHECK_THROWS_AS(json::to_cbor(jval, eh::strict), json::type_error&);
|
||||
CHECK_THROWS_AS(json::to_msgpack(jval, eh::strict), json::type_error&);
|
||||
CHECK_THROWS_AS(json::to_ubjson(jval, false, false, eh::strict), json::type_error&);
|
||||
CHECK_THROWS_AS(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::strict), json::type_error&);
|
||||
{
|
||||
json jobj;
|
||||
jobj["k"] = jval;
|
||||
CHECK_THROWS_AS(json::to_bson(jobj, eh::strict), json::type_error&);
|
||||
}
|
||||
|
||||
for (const auto h :
|
||||
{
|
||||
eh::replace, eh::ignore
|
||||
})
|
||||
{
|
||||
CAPTURE(static_cast<int>(h));
|
||||
const std::string expected = dump_and_parse(c.bytes, h);
|
||||
|
||||
CHECK(json::from_cbor(json::to_cbor(jval, h)).get<std::string>() == expected);
|
||||
CHECK(json::from_msgpack(json::to_msgpack(jval, h)).get<std::string>() == expected);
|
||||
CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, h)).get<std::string>() == expected);
|
||||
CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, h)).get<std::string>() == expected);
|
||||
{
|
||||
json jobj;
|
||||
jobj["k"] = jval;
|
||||
const auto bytes = json::to_bson(jobj, h);
|
||||
CHECK(json::from_bson(bytes)["k"].get<std::string>() == expected);
|
||||
}
|
||||
}
|
||||
|
||||
// keep: the writer passes the ill-formed bytes through unchanged,
|
||||
// exactly as every binary writer did before this parameter existed
|
||||
CHECK(json::from_cbor(json::to_cbor(jval, eh::keep)).get<std::string>() == c.bytes);
|
||||
CHECK(json::from_msgpack(json::to_msgpack(jval, eh::keep)).get<std::string>() == c.bytes);
|
||||
CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, eh::keep)).get<std::string>() == c.bytes);
|
||||
CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep)).get<std::string>() == c.bytes);
|
||||
{
|
||||
json jobj;
|
||||
jobj["k"] = jval;
|
||||
const auto bytes = json::to_bson(jobj, eh::keep);
|
||||
CHECK(json::from_bson(bytes)["k"].get<std::string>() == c.bytes);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("writers: object key")
|
||||
{
|
||||
for (const auto& c : ill_formed_cases)
|
||||
{
|
||||
CAPTURE(c.name);
|
||||
json jobj;
|
||||
jobj[c.bytes] = 1;
|
||||
|
||||
CHECK_THROWS_AS(json::to_cbor(jobj, eh::strict), json::type_error&);
|
||||
CHECK_THROWS_AS(json::to_msgpack(jobj, eh::strict), json::type_error&);
|
||||
CHECK_THROWS_AS(json::to_ubjson(jobj, false, false, eh::strict), json::type_error&);
|
||||
CHECK_THROWS_AS(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::strict), json::type_error&);
|
||||
CHECK_THROWS_AS(json::to_bson(jobj, eh::strict), json::type_error&);
|
||||
|
||||
for (const auto h :
|
||||
{
|
||||
eh::replace, eh::ignore
|
||||
})
|
||||
{
|
||||
CAPTURE(static_cast<int>(h));
|
||||
const std::string expected = dump_and_parse(c.bytes, h);
|
||||
|
||||
CHECK(json::from_cbor(json::to_cbor(jobj, h)).begin().key() == expected);
|
||||
CHECK(json::from_msgpack(json::to_msgpack(jobj, h)).begin().key() == expected);
|
||||
CHECK(json::from_ubjson(json::to_ubjson(jobj, false, false, h)).begin().key() == expected);
|
||||
CHECK(json::from_bjdata(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, h)).begin().key() == expected);
|
||||
CHECK(json::from_bson(json::to_bson(jobj, h)).begin().key() == expected);
|
||||
}
|
||||
|
||||
// keep: object keys round-trip unchanged too
|
||||
CHECK(json::from_cbor(json::to_cbor(jobj, eh::keep)).begin().key() == c.bytes);
|
||||
CHECK(json::from_msgpack(json::to_msgpack(jobj, eh::keep)).begin().key() == c.bytes);
|
||||
CHECK(json::from_ubjson(json::to_ubjson(jobj, false, false, eh::keep)).begin().key() == c.bytes);
|
||||
CHECK(json::from_bjdata(json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::keep)).begin().key() == c.bytes);
|
||||
CHECK(json::from_bson(json::to_bson(jobj, eh::keep)).begin().key() == c.bytes);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("readers: string value")
|
||||
{
|
||||
for (const auto& c : ill_formed_cases)
|
||||
{
|
||||
CAPTURE(c.name);
|
||||
|
||||
// bytes produced the lenient (keep) way, as any binary reader
|
||||
// accepted them before this parameter existed
|
||||
const auto cbor_bytes = json::to_cbor(json(c.bytes), eh::keep);
|
||||
const auto msgpack_bytes = json::to_msgpack(json(c.bytes)); // to_msgpack has no error_handler; always pass-through
|
||||
const auto ubjson_bytes = json::to_ubjson(json(c.bytes), false, false, eh::keep);
|
||||
const auto bjdata_bytes = json::to_bjdata(json(c.bytes), false, false, json::bjdata_version_t::draft2, eh::keep);
|
||||
const auto bson_bytes = [&c]
|
||||
{
|
||||
json jobj;
|
||||
jobj["k"] = c.bytes;
|
||||
return json::to_bson(jobj, eh::keep);
|
||||
}();
|
||||
|
||||
// keep (the default): bytes are kept unchanged
|
||||
CHECK(json::from_cbor(cbor_bytes).get<std::string>() == c.bytes);
|
||||
CHECK(json::from_msgpack(msgpack_bytes).get<std::string>() == c.bytes);
|
||||
CHECK(json::from_ubjson(ubjson_bytes).get<std::string>() == c.bytes);
|
||||
CHECK(json::from_bjdata(bjdata_bytes).get<std::string>() == c.bytes);
|
||||
CHECK(json::from_bson(bson_bytes)["k"].get<std::string>() == c.bytes);
|
||||
|
||||
// strict: parse_error.113, discarded (not thrown) when allow_exceptions is false
|
||||
CHECK_THROWS_AS(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, eh::strict), json::parse_error&);
|
||||
CHECK(json::from_cbor(cbor_bytes, true, false, json::cbor_tag_handler_t::error, eh::strict).is_discarded());
|
||||
CHECK_THROWS_AS(json::from_msgpack(msgpack_bytes, true, true, eh::strict), json::parse_error&);
|
||||
CHECK(json::from_msgpack(msgpack_bytes, true, false, eh::strict).is_discarded());
|
||||
CHECK_THROWS_AS(json::from_ubjson(ubjson_bytes, true, true, eh::strict), json::parse_error&);
|
||||
CHECK(json::from_ubjson(ubjson_bytes, true, false, eh::strict).is_discarded());
|
||||
CHECK_THROWS_AS(json::from_bjdata(bjdata_bytes, true, true, eh::strict), json::parse_error&);
|
||||
CHECK(json::from_bjdata(bjdata_bytes, true, false, eh::strict).is_discarded());
|
||||
CHECK_THROWS_AS(json::from_bson(bson_bytes, true, true, eh::strict), json::parse_error&);
|
||||
CHECK(json::from_bson(bson_bytes, true, false, eh::strict).is_discarded());
|
||||
|
||||
// replace / ignore: match what dump() would have sanitized the same bytes to
|
||||
for (const auto h :
|
||||
{
|
||||
eh::replace, eh::ignore
|
||||
})
|
||||
{
|
||||
CAPTURE(static_cast<int>(h));
|
||||
const std::string expected = dump_and_parse(c.bytes, h);
|
||||
|
||||
CHECK(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, h).get<std::string>() == expected);
|
||||
CHECK(json::from_msgpack(msgpack_bytes, true, true, h).get<std::string>() == expected);
|
||||
CHECK(json::from_ubjson(ubjson_bytes, true, true, h).get<std::string>() == expected);
|
||||
CHECK(json::from_bjdata(bjdata_bytes, true, true, h).get<std::string>() == expected);
|
||||
CHECK(json::from_bson(bson_bytes, true, true, h)["k"].get<std::string>() == expected);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("readers: object key")
|
||||
{
|
||||
for (const auto& c : ill_formed_cases)
|
||||
{
|
||||
CAPTURE(c.name);
|
||||
|
||||
json jobj;
|
||||
jobj[c.bytes] = 1;
|
||||
const auto cbor_bytes = json::to_cbor(jobj, eh::keep);
|
||||
const auto msgpack_bytes = json::to_msgpack(jobj);
|
||||
const auto ubjson_bytes = json::to_ubjson(jobj, false, false, eh::keep);
|
||||
const auto bjdata_bytes = json::to_bjdata(jobj, false, false, json::bjdata_version_t::draft2, eh::keep);
|
||||
const auto bson_bytes = json::to_bson(jobj, eh::keep);
|
||||
|
||||
CHECK(json::from_cbor(cbor_bytes).begin().key() == c.bytes);
|
||||
CHECK(json::from_msgpack(msgpack_bytes).begin().key() == c.bytes);
|
||||
CHECK(json::from_ubjson(ubjson_bytes).begin().key() == c.bytes);
|
||||
CHECK(json::from_bjdata(bjdata_bytes).begin().key() == c.bytes);
|
||||
CHECK(json::from_bson(bson_bytes).begin().key() == c.bytes);
|
||||
|
||||
CHECK_THROWS_AS(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, eh::strict), json::parse_error&);
|
||||
CHECK_THROWS_AS(json::from_msgpack(msgpack_bytes, true, true, eh::strict), json::parse_error&);
|
||||
CHECK_THROWS_AS(json::from_ubjson(ubjson_bytes, true, true, eh::strict), json::parse_error&);
|
||||
CHECK_THROWS_AS(json::from_bjdata(bjdata_bytes, true, true, eh::strict), json::parse_error&);
|
||||
CHECK_THROWS_AS(json::from_bson(bson_bytes, true, true, eh::strict), json::parse_error&);
|
||||
|
||||
for (const auto h :
|
||||
{
|
||||
eh::replace, eh::ignore
|
||||
})
|
||||
{
|
||||
CAPTURE(static_cast<int>(h));
|
||||
const std::string expected = dump_and_parse(c.bytes, h);
|
||||
|
||||
CHECK(json::from_cbor(cbor_bytes, true, true, json::cbor_tag_handler_t::error, h).begin().key() == expected);
|
||||
CHECK(json::from_msgpack(msgpack_bytes, true, true, h).begin().key() == expected);
|
||||
CHECK(json::from_ubjson(ubjson_bytes, true, true, h).begin().key() == expected);
|
||||
CHECK(json::from_bjdata(bjdata_bytes, true, true, h).begin().key() == expected);
|
||||
CHECK(json::from_bson(bson_bytes, true, true, h).begin().key() == expected);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("well-formed UTF-8 is unaffected by error_handler")
|
||||
{
|
||||
const json jval = valid_sequence;
|
||||
json jobj;
|
||||
jobj[valid_sequence] = valid_sequence;
|
||||
|
||||
for (const auto h : all_handlers)
|
||||
{
|
||||
CAPTURE(static_cast<int>(h));
|
||||
|
||||
CHECK(json::from_cbor(json::to_cbor(jval, h)).get<std::string>() == valid_sequence);
|
||||
CHECK(json::from_msgpack(json::to_msgpack(jval, h)).get<std::string>() == valid_sequence);
|
||||
CHECK(json::from_ubjson(json::to_ubjson(jval, false, false, h)).get<std::string>() == valid_sequence);
|
||||
CHECK(json::from_bjdata(json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, h)).get<std::string>() == valid_sequence);
|
||||
CHECK(json::from_bson(json::to_bson(jobj, h)).begin().key() == valid_sequence);
|
||||
|
||||
CHECK(json::from_cbor(json::to_cbor(jval, eh::keep), true, true, json::cbor_tag_handler_t::error, h).get<std::string>() == valid_sequence);
|
||||
CHECK(json::from_msgpack(json::to_msgpack(jval), true, true, h).get<std::string>() == valid_sequence);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("dump() with error_handler_t::keep writes raw bytes as is")
|
||||
{
|
||||
for (const auto& c : ill_formed_cases)
|
||||
{
|
||||
CAPTURE(c.name);
|
||||
|
||||
const json jval = c.bytes;
|
||||
const std::string dumped = jval.dump(-1, ' ', false, eh::keep);
|
||||
CHECK(dumped.find(c.bytes) != std::string::npos);
|
||||
|
||||
// even with ensure_ascii, the ill-formed bytes are written as is
|
||||
const std::string dumped_ascii = jval.dump(-1, ' ', true, eh::keep);
|
||||
CHECK(dumped_ascii.find(c.bytes) != std::string::npos);
|
||||
}
|
||||
|
||||
// well-formed characters around an ill-formed sequence are still
|
||||
// escaped as usual under ensure_ascii
|
||||
const json mixed = valid_sequence + ill_formed_cases[1].bytes; // "é" + lone 0xFF
|
||||
const std::string dumped_mixed = mixed.dump(-1, ' ', true, eh::keep);
|
||||
CHECK(dumped_mixed.find("\\u00e9") != std::string::npos);
|
||||
CHECK(dumped_mixed.find(ill_formed_cases[1].bytes) != std::string::npos);
|
||||
|
||||
// the byte that ends an ill-formed sequence is read again, so a quote,
|
||||
// a backslash, or a control character after it is still escaped, and
|
||||
// a well-formed code point after it is escaped under ensure_ascii
|
||||
for (const bool ensure_ascii :
|
||||
{
|
||||
false, true
|
||||
})
|
||||
{
|
||||
CAPTURE(ensure_ascii);
|
||||
CHECK(json("\xC3\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\\"\"");
|
||||
CHECK(json("\xC3\\").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\\\\"");
|
||||
CHECK(json("\xC3\n").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xC3\\n\"");
|
||||
CHECK(json("\xE2\x82\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xE2\x82\\\"\"");
|
||||
CHECK(json("\xFF\"").dump(-1, ' ', ensure_ascii, eh::keep) == "\"\xFF\\\"\"");
|
||||
CHECK(json("a\xE2\x82").dump(-1, ' ', ensure_ascii, eh::keep) == "\"a\xE2\x82\"");
|
||||
}
|
||||
CHECK(json("\xC3\xC3\xA9").dump(-1, ' ', false, eh::keep) == "\"\xC3\xC3\xA9\"");
|
||||
CHECK(json("\xC3\xC3\xA9").dump(-1, ' ', true, eh::keep) == "\"\xC3\\u00e9\"");
|
||||
}
|
||||
|
||||
SECTION("to_msgpack defaults to keep; to_bon8 is not affected by error_handler")
|
||||
{
|
||||
const json jval = ill_formed_cases[1].bytes; // lone 0xFF
|
||||
|
||||
// to_msgpack's error_handler defaults to keep, as MessagePack's spec
|
||||
// allows any bytes in a str, so the bytes are passed through
|
||||
CHECK(json::to_msgpack(jval) == json::to_msgpack(jval, eh::keep));
|
||||
CHECK(json::from_msgpack(json::to_msgpack(jval)).get<std::string>() == ill_formed_cases[1].bytes);
|
||||
|
||||
// the diagnostics context of an ill-formed key is the object
|
||||
json jobj;
|
||||
jobj["\xFF"] = 1;
|
||||
CHECK_THROWS_WITH_AS(json::to_msgpack(jobj, eh::strict), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
|
||||
// to_bon8 has no error_handler parameter; UTF-8 is structural for
|
||||
// BON8, so it always rejects ill-formed input
|
||||
CHECK_THROWS_AS(json::to_bon8(jval), json::type_error&);
|
||||
}
|
||||
|
||||
SECTION("allow_exceptions=false with error_handler_t::strict discards the value")
|
||||
{
|
||||
const auto bytes = json::to_cbor(json(ill_formed_cases[0].bytes), eh::keep);
|
||||
const json result = json::from_cbor(bytes, true, false, json::cbor_tag_handler_t::error, eh::strict);
|
||||
CHECK(result.is_discarded());
|
||||
}
|
||||
|
||||
SECTION("default parameters are unchanged")
|
||||
{
|
||||
const json jval = ill_formed_cases[0].bytes;
|
||||
|
||||
// to_*: the default error_handler is keep, so ill-formed bytes are
|
||||
// written unchanged, exactly as in release 3.12.0 (it is strict only
|
||||
// if JSON_STRICT_BINARY_UTF8 is enabled, see
|
||||
// unit-binary_utf8_strict.cpp)
|
||||
CHECK(json::to_cbor(jval) == json::to_cbor(jval, eh::keep));
|
||||
CHECK(json::to_ubjson(jval) == json::to_ubjson(jval, false, false, eh::keep));
|
||||
CHECK(json::to_bjdata(jval) == json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep));
|
||||
{
|
||||
json jobj;
|
||||
jobj["k"] = jval;
|
||||
CHECK(json::to_bson(jobj) == json::to_bson(jobj, eh::keep));
|
||||
}
|
||||
|
||||
// from_*: the default error_handler is keep, so ill-formed bytes are
|
||||
// still accepted unchanged, exactly as in release 3.12.0
|
||||
const auto cbor_bytes = json::to_cbor(jval, eh::keep);
|
||||
CHECK(json::from_cbor(cbor_bytes).get<std::string>() == ill_formed_cases[0].bytes);
|
||||
const auto ubjson_bytes = json::to_ubjson(jval, false, false, eh::keep);
|
||||
CHECK(json::from_ubjson(ubjson_bytes).get<std::string>() == ill_formed_cases[0].bytes);
|
||||
const auto bjdata_bytes = json::to_bjdata(jval, false, false, json::bjdata_version_t::draft2, eh::keep);
|
||||
CHECK(json::from_bjdata(bjdata_bytes).get<std::string>() == ill_formed_cases[0].bytes);
|
||||
const auto msgpack_bytes = json::to_msgpack(jval);
|
||||
CHECK(json::from_msgpack(msgpack_bytes).get<std::string>() == ill_formed_cases[0].bytes);
|
||||
json bson_obj;
|
||||
bson_obj["k"] = jval;
|
||||
const auto bson_bytes = json::to_bson(bson_obj, eh::keep);
|
||||
CHECK(json::from_bson(bson_bytes)["k"].get<std::string>() == ill_formed_cases[0].bytes);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
|
||||
// The binary writers check strings and object keys for valid UTF-8 only if
|
||||
// JSON_STRICT_BINARY_UTF8 is enabled (planned to be the default in 4.0.0).
|
||||
// Without it, they write the bytes unchanged, as before version 3.13.0; the
|
||||
// tests for that are next to the other tests of each format.
|
||||
#ifdef JSON_STRICT_BINARY_UTF8
|
||||
#undef JSON_STRICT_BINARY_UTF8
|
||||
#endif
|
||||
|
||||
#define JSON_STRICT_BINARY_UTF8 1
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
TEST_CASE("JSON_STRICT_BINARY_UTF8 (see #5529, #5651)")
|
||||
{
|
||||
SECTION("CBOR")
|
||||
{
|
||||
// a string value with ill-formed UTF-8 is rejected
|
||||
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
// a truncated multi-byte sequence
|
||||
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
|
||||
// an overlong encoding of '.'
|
||||
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
||||
|
||||
// an object key with ill-formed UTF-8 is rejected the same way
|
||||
CHECK_THROWS_WITH_AS(json::to_cbor(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
|
||||
// binary values are not text and are unaffected
|
||||
CHECK_NOTHROW(json::to_cbor(json::binary(std::vector<std::uint8_t>({0xFF}))));
|
||||
|
||||
// a value read back from CBOR with ill-formed bytes cannot be written
|
||||
// back either (the reader is lenient regardless of the macro)
|
||||
const json j = json::from_cbor(std::vector<std::uint8_t>({0x62, 0xc0, 0xae}));
|
||||
CHECK_THROWS_WITH_AS(json::to_cbor(j), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
||||
}
|
||||
|
||||
SECTION("UBJSON")
|
||||
{
|
||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
// a truncated multi-byte sequence
|
||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
|
||||
// an overlong encoding of '.'
|
||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
||||
|
||||
// an object key with ill-formed UTF-8 is rejected the same way
|
||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
}
|
||||
|
||||
SECTION("BJData")
|
||||
{
|
||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
// a truncated multi-byte sequence
|
||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
|
||||
// an overlong encoding of '.'
|
||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
||||
|
||||
// an object key with ill-formed UTF-8 is rejected the same way
|
||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
}
|
||||
|
||||
SECTION("BSON")
|
||||
{
|
||||
// to_bson() rejects the same kind of ill-formed string value, before
|
||||
// any bytes reach the output adapter (the BSON document length
|
||||
// prefix must be known up front, so nothing is written incrementally)
|
||||
std::vector<std::uint8_t> out{0x42}; // a sentinel byte the writer must not touch
|
||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}, nlohmann::detail::output_adapter<std::uint8_t>(out)), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
CHECK(out == std::vector<std::uint8_t> {0x42});
|
||||
|
||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
// a truncated multi-byte sequence
|
||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC3"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xED\xA0\x80"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
|
||||
// an overlong encoding of '.'
|
||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC0\xAF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
||||
|
||||
// an object key with ill-formed UTF-8 is rejected as well; unlike
|
||||
// the reader (which never validates element names), the writer
|
||||
// checks both string values and object keys
|
||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
||||
}
|
||||
|
||||
SECTION("an explicit error_handler overrides the default")
|
||||
{
|
||||
// the macro only changes the default of the error_handler parameter
|
||||
CHECK(json::to_cbor(json("\xFF"), json::error_handler_t::keep) == std::vector<std::uint8_t>({0x61, 0xff}));
|
||||
CHECK(json::to_ubjson(json("\xFF"), false, false, json::error_handler_t::keep) == std::vector<std::uint8_t>({'S', 'i', 1, 0xff}));
|
||||
CHECK(json::to_bjdata(json("\xFF"), false, false, json::bjdata_version_t::draft2, json::error_handler_t::keep) == std::vector<std::uint8_t>({'S', 'i', 1, 0xff}));
|
||||
CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}}, json::error_handler_t::keep)) == json{{"s", "\xFF"}});
|
||||
CHECK(json::to_cbor(json("\xFF"), json::error_handler_t::replace) == std::vector<std::uint8_t>({0x63, 0xef, 0xbf, 0xbd}));
|
||||
}
|
||||
|
||||
SECTION("MessagePack and BON8 are unaffected")
|
||||
{
|
||||
// MessagePack allows any bytes in a str, so to_msgpack() still
|
||||
// defaults to keep (strict only if passed explicitly); BON8 always
|
||||
// checks, because the lead bytes mark where strings end
|
||||
CHECK(json::to_msgpack(json("\xFF")) == std::vector<std::uint8_t>({0xa1, 0xff}));
|
||||
CHECK_THROWS_AS(json::to_msgpack(json("\xFF"), json::error_handler_t::strict), json::type_error&);
|
||||
CHECK_THROWS_AS(json::to_bon8(json("\xFF")), json::type_error&);
|
||||
}
|
||||
}
|
||||
@@ -3906,6 +3906,43 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
|
||||
CHECK(json::to_bjdata(j) == v);
|
||||
CHECK(json::from_bjdata(v) == j);
|
||||
}
|
||||
|
||||
SECTION("ill-formed UTF-8 (see #5529, #5651)")
|
||||
{
|
||||
// none of the binary format specs requires a decoder to reject
|
||||
// ill-formed UTF-8 in a text string, so a value whose bytes are
|
||||
// not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.')
|
||||
// round-trips byte for byte as a string value; to_bjdata() writes
|
||||
// the bytes unchanged, as before 3.13.0, unless
|
||||
// JSON_STRICT_BINARY_UTF8 is enabled (see
|
||||
// unit-binary_utf8_strict.cpp)
|
||||
const std::vector<uint8_t> v = {'S', 'i', 2, 0xc0, 0xae};
|
||||
json j;
|
||||
CHECK_NOTHROW(j = json::from_bjdata(v));
|
||||
REQUIRE(j.is_string());
|
||||
CHECK(j.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
CHECK_THROWS_AS(j.dump(), json::type_error&);
|
||||
CHECK(json::from_bjdata(json::to_bjdata(j)) == j);
|
||||
|
||||
// the same bytes as an object key round-trip as well
|
||||
const std::vector<uint8_t> v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'};
|
||||
json j_key;
|
||||
CHECK_NOTHROW(j_key = json::from_bjdata(v_key));
|
||||
REQUIRE(j_key.is_object());
|
||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
||||
CHECK(json::from_bjdata(json::to_bjdata(j_key)) == j_key);
|
||||
|
||||
CHECK(json::from_bjdata(json::to_bjdata(json("\xFF"))) == json("\xFF"));
|
||||
// a truncated multi-byte sequence
|
||||
CHECK(json::from_bjdata(json::to_bjdata(json("\xC3"))) == json("\xC3"));
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK(json::from_bjdata(json::to_bjdata(json("\xED\xA0\x80"))) == json("\xED\xA0\x80"));
|
||||
// an overlong encoding of '.'
|
||||
CHECK(json::from_bjdata(json::to_bjdata(json("\xC0\xAF"))) == json("\xC0\xAF"));
|
||||
|
||||
// an object key with ill-formed UTF-8 is kept the same way
|
||||
CHECK(json::from_bjdata(json::to_bjdata(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("Array Type")
|
||||
|
||||
@@ -154,6 +154,43 @@ TEST_CASE("BSON")
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("ill-formed UTF-8 (see #5529, #5651)")
|
||||
{
|
||||
// a BSON document {"s": "\xC0\xAE"} (0xC0 0xAE is an overlong
|
||||
// encoding of '.'); the BSON spec does not require a decoder to
|
||||
// reject ill-formed UTF-8 in a string value, so the reader hands the
|
||||
// bytes back unchanged
|
||||
const std::vector<uint8_t> v =
|
||||
{
|
||||
0x0F, 0x00, 0x00, 0x00, // document length
|
||||
0x02, 's', 0x00, // type 0x02 (string), key "s"
|
||||
0x03, 0x00, 0x00, 0x00, // string length (including null)
|
||||
0xc0, 0xae, 0x00, // string content and its null terminator
|
||||
0x00 // document terminator
|
||||
};
|
||||
json j;
|
||||
CHECK_NOTHROW(j = json::from_bson(v));
|
||||
REQUIRE(j.is_object());
|
||||
REQUIRE(j.contains("s"));
|
||||
CHECK(j["s"].get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
// dump() still requires valid UTF-8 and throws for such a value
|
||||
CHECK_THROWS_AS(j.dump(), json::type_error&);
|
||||
// to_bson() writes the bytes back unchanged, as before 3.13.0,
|
||||
// unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp)
|
||||
CHECK(json::from_bson(json::to_bson(j)) == j);
|
||||
|
||||
CHECK(json::from_bson(json::to_bson(json{{"s", "\xFF"}})) == json{{"s", "\xFF"}});
|
||||
// a truncated multi-byte sequence
|
||||
CHECK(json::from_bson(json::to_bson(json{{"s", "\xC3"}})) == json{{"s", "\xC3"}});
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK(json::from_bson(json::to_bson(json{{"s", "\xED\xA0\x80"}})) == json{{"s", "\xED\xA0\x80"}});
|
||||
// an overlong encoding of '.'
|
||||
CHECK(json::from_bson(json::to_bson(json{{"s", "\xC0\xAF"}})) == json{{"s", "\xC0\xAF"}});
|
||||
|
||||
// an object key with ill-formed UTF-8 is kept as well
|
||||
CHECK(json::from_bson(json::to_bson(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
|
||||
}
|
||||
|
||||
SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON")
|
||||
{
|
||||
// out_of_range.412 is thrown from a single shared helper
|
||||
|
||||
+67
-18
@@ -1801,19 +1801,41 @@ TEST_CASE("CBOR")
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&);
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in string (see #5529)")
|
||||
SECTION("ill-formed UTF-8 in string (see #5529, #5651)")
|
||||
{
|
||||
// RFC 8949 §3.1 leaves it up to the decoder whether to reject
|
||||
// ill-formed UTF-8 in a text string; this library does not, and
|
||||
// hands the original bytes back unchanged, matching the
|
||||
// MessagePack reader and the behavior before #5185/#5531 (not in
|
||||
// any release)
|
||||
|
||||
// a two-character text string (major type 3) whose bytes are not
|
||||
// valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be
|
||||
// rejected at decode time, matching every other kind of
|
||||
// malformed binary input, rather than only failing later when
|
||||
// the resulting value is dumped
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x62, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_cbor(std::vector<uint8_t>({0x62, 0xc0, 0xae}), true, false).is_discarded());
|
||||
// valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') round-trips
|
||||
// byte for byte as a string value
|
||||
const std::vector<uint8_t> ill_formed_value = {0x62, 0xc0, 0xae};
|
||||
json j_value;
|
||||
CHECK_NOTHROW(j_value = json::from_cbor(ill_formed_value));
|
||||
REQUIRE(j_value.is_string());
|
||||
CHECK(j_value.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
// dump() still requires valid UTF-8 and throws for such a value,
|
||||
// unless an error handler that replaces or ignores the bytes is
|
||||
// passed
|
||||
CHECK_THROWS_AS(j_value.dump(), json::type_error&);
|
||||
// to_cbor() writes the bytes back unchanged, as before 3.13.0,
|
||||
// unless JSON_STRICT_BINARY_UTF8 is enabled (see unit-binary_utf8_strict.cpp)
|
||||
CHECK(json::from_cbor(json::to_cbor(j_value)) == j_value);
|
||||
|
||||
// the same bytes as an object key round-trip as well
|
||||
const std::vector<uint8_t> ill_formed_key = {0xa1, 0x62, 0xc0, 0xae, 0x01};
|
||||
json j_key;
|
||||
CHECK_NOTHROW(j_key = json::from_cbor(ill_formed_key));
|
||||
REQUIRE(j_key.is_object());
|
||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
||||
CHECK(json::from_cbor(json::to_cbor(j_key)) == j_key);
|
||||
|
||||
// a CBOR byte string (major type 2) with the very same bytes is
|
||||
// NOT text and must still be accepted as-is
|
||||
json _;
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x42, 0xc0, 0xae})));
|
||||
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
|
||||
|
||||
@@ -1822,17 +1844,47 @@ TEST_CASE("CBOR")
|
||||
CHECK(json::from_cbor(json::to_cbor(j)) == j);
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in indefinite-length string")
|
||||
SECTION("to_cbor keeps ill-formed UTF-8 (see #5651)")
|
||||
{
|
||||
// to_cbor() writes the bytes unchanged, as before 3.13.0, unless
|
||||
// JSON_STRICT_BINARY_UTF8 is enabled (see
|
||||
// unit-binary_utf8_strict.cpp); from_cbor() reads them back as is
|
||||
CHECK(json::from_cbor(json::to_cbor(json("\xFF"))) == json("\xFF"));
|
||||
// a truncated multi-byte sequence
|
||||
CHECK(json::from_cbor(json::to_cbor(json("\xC3"))) == json("\xC3"));
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK(json::from_cbor(json::to_cbor(json("\xED\xA0\x80"))) == json("\xED\xA0\x80"));
|
||||
// an overlong encoding of '.'
|
||||
CHECK(json::from_cbor(json::to_cbor(json("\xC0\xAF"))) == json("\xC0\xAF"));
|
||||
|
||||
// an object key with ill-formed UTF-8 is kept the same way
|
||||
CHECK(json::from_cbor(json::to_cbor(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
|
||||
|
||||
// binary values are not text and are unaffected
|
||||
CHECK_NOTHROW(json::to_cbor(json::binary(std::vector<std::uint8_t>({0xFF}))));
|
||||
}
|
||||
|
||||
SECTION("ill-formed UTF-8 in indefinite-length string")
|
||||
{
|
||||
json _;
|
||||
|
||||
// every chunk must be valid UTF-8 on its own (RFC 8949, Section
|
||||
// 3.2.3), so a code point split across two chunks is rejected
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}), true, false).is_discarded());
|
||||
// the chunks are concatenated as is, without checking that each
|
||||
// chunk is valid UTF-8 on its own (RFC 8949, Section 3.2.3), so
|
||||
// a code point split across two chunks yields a valid string
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})));
|
||||
CHECK(_ == "\xc3\xa9");
|
||||
CHECK(_.dump() == "\"\xc3\xa9\"");
|
||||
|
||||
// an ill-formed later chunk is rejected after valid ones
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
// a truncated code point is kept as is
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0xff})));
|
||||
CHECK(_ == "\xc3");
|
||||
CHECK_THROWS_AS(_.dump(), json::type_error&);
|
||||
CHECK(json::from_cbor(json::to_cbor(_)) == _);
|
||||
|
||||
// an ill-formed later chunk is kept after valid ones
|
||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})));
|
||||
CHECK(_ == "\xc3\xa9\xc0\xae");
|
||||
CHECK_THROWS_AS(_.dump(), json::type_error&);
|
||||
|
||||
// valid multi-byte chunks are accepted
|
||||
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc3, 0xb6, 0xff})) == "\xc3\xa9\xc3\xb6");
|
||||
@@ -1840,9 +1892,6 @@ TEST_CASE("CBOR")
|
||||
|
||||
SECTION("many chunks in indefinite-length string")
|
||||
{
|
||||
// only the newly read chunk is validated, not the whole string
|
||||
// collected so far; validating the latter made this input take
|
||||
// quadratic time (about ten seconds for 100000 chunks)
|
||||
constexpr std::size_t chunks = 100000;
|
||||
std::vector<uint8_t> v{0x7f};
|
||||
for (std::size_t i = 0; i < chunks; ++i)
|
||||
|
||||
@@ -2686,12 +2686,10 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX")
|
||||
|
||||
SECTION("move constructor resets the moved-from value to npos")
|
||||
{
|
||||
// basic_json(basic_json&&) (json.hpp, around line 1951) copies
|
||||
// basic_json(basic_json&&) copies
|
||||
// other's start_position/end_position into *this and then resets
|
||||
// other's to npos (see the cppcheck-suppress[accessForwarded]
|
||||
// annotation there, which flags this reset as worth a second
|
||||
// look). Only the top-level moved-from value is affected; its
|
||||
// (moved-away) children are gone along with it.
|
||||
// other's to npos. Only the top-level moved-from value is
|
||||
// affected; its (moved-away) children are gone along with it.
|
||||
const std::string s = R"({"a":1,"b":[1,2,3]})";
|
||||
json a = json::parse(s);
|
||||
const auto a_start = a.start_pos();
|
||||
|
||||
@@ -430,6 +430,37 @@ TEST_CASE("value conversion")
|
||||
CHECK(std::equal(std::begin(nbs[0][0][0]), std::end(nbs[1][1][1]), std::begin(nbs2[0][0][0])));
|
||||
}
|
||||
|
||||
SECTION("built-in arrays: 5D")
|
||||
{
|
||||
// NOLINTBEGIN(misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
|
||||
const int nbs[1][1][1][2][2] = {{{{{0, 1}, {2, 3}}}}};
|
||||
int nbs2[1][1][1][2][2] = {{{{{0, 0}, {0, 0}}}}};
|
||||
// NOLINTEND(misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
|
||||
|
||||
const json j2 = nbs;
|
||||
j2.get_to(nbs2);
|
||||
CHECK(std::equal(std::begin(nbs[0][0][0][0]), std::end(nbs[0][0][0][1]), std::begin(nbs2[0][0][0][0])));
|
||||
}
|
||||
|
||||
SECTION("built-in arrays: mismatched shape")
|
||||
{
|
||||
// NOLINTBEGIN(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
|
||||
int nbs2[2][3] = {{0, 0, 0}, {0, 0, 0}};
|
||||
// NOLINTEND(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays)
|
||||
|
||||
SECTION("not an array")
|
||||
{
|
||||
const json j2 = 42;
|
||||
CHECK_THROWS_WITH_AS(j2.get_to(nbs2), "[json.exception.type_error.304] cannot use at() with number", json::type_error&);
|
||||
}
|
||||
|
||||
SECTION("too few elements")
|
||||
{
|
||||
const json j2 = {{0, 1, 2}};
|
||||
CHECK_THROWS_WITH_AS(j2.get_to(nbs2), "[json.exception.out_of_range.401] array index 1 is out of range", json::out_of_range&);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("std::deque<json>")
|
||||
{
|
||||
std::deque<json> a{"previous", "value"};
|
||||
@@ -1748,6 +1779,34 @@ NLOHMANN_JSON_SERIALIZE_ENUM_STRICT(StrictTaskState,
|
||||
{STRICT_TS_COMPLETED, "completed"},
|
||||
})
|
||||
|
||||
// regression test for #5708 item 2: NLOHMANN_JSON_SERIALIZE_ENUM_STRICT must not rely on
|
||||
// unqualified lookup of a helper name that a user's own namespace may also declare
|
||||
namespace ns_with_colliding_name
|
||||
{
|
||||
// NOLINTNEXTLINE(misc-use-internal-linkage) - used to shadow the library's internal helper name
|
||||
inline void templated_json_throw(int /*unused*/) {}
|
||||
|
||||
enum class colliding_enum { a, b };
|
||||
|
||||
// NOLINTNEXTLINE(misc-use-internal-linkage,misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - false positive
|
||||
NLOHMANN_JSON_SERIALIZE_ENUM_STRICT(colliding_enum,
|
||||
{
|
||||
{colliding_enum::a, "a"},
|
||||
{colliding_enum::b, "b"}
|
||||
})
|
||||
} // namespace ns_with_colliding_name
|
||||
|
||||
TEST_CASE("NLOHMANN_JSON_SERIALIZE_ENUM_STRICT in a namespace with a colliding name")
|
||||
{
|
||||
using ns_with_colliding_name::colliding_enum;
|
||||
|
||||
CHECK(json(colliding_enum::a) == "a");
|
||||
CHECK(colliding_enum::b == json("b"));
|
||||
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json("nope").get<colliding_enum>(), "[json.exception.out_of_range.410] enum value out of range for colliding_enum: \"nope\"", json::out_of_range&);
|
||||
}
|
||||
|
||||
TEST_CASE("Strict JSON to enum mapping")
|
||||
{
|
||||
SECTION("enum class")
|
||||
|
||||
@@ -10,6 +10,8 @@
|
||||
#include <set>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <type_traits>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
@@ -406,6 +408,74 @@ TEST_CASE("JSON Visit Node")
|
||||
CHECK(expected.empty());
|
||||
}
|
||||
|
||||
// Test accessing members of a custom base class that are hidden by members of nlohmann::basic_json
|
||||
class base_class_with_hidden_members
|
||||
{
|
||||
public:
|
||||
const char* type_name() const noexcept // NOLINT(readability-convert-member-functions-to-static)
|
||||
{
|
||||
return "custom type_name";
|
||||
}
|
||||
|
||||
std::size_t size() const noexcept
|
||||
{
|
||||
return m_size;
|
||||
}
|
||||
|
||||
std::size_t m_size = 42;
|
||||
};
|
||||
|
||||
using json_with_hidden_base_members =
|
||||
nlohmann::basic_json <
|
||||
std::map,
|
||||
std::vector,
|
||||
std::string,
|
||||
bool,
|
||||
std::int64_t,
|
||||
std::uint64_t,
|
||||
double,
|
||||
std::allocator,
|
||||
nlohmann::adl_serializer,
|
||||
std::vector<std::uint8_t>,
|
||||
base_class_with_hidden_members
|
||||
>;
|
||||
|
||||
TEST_CASE("JSON Node as_base_class")
|
||||
{
|
||||
using json = json_with_hidden_base_members;
|
||||
|
||||
static_assert(std::is_same<decltype(std::declval<json&>().as_base_class()), json::json_base_class_t&>::value, "");
|
||||
static_assert(std::is_same<decltype(std::declval<const json&>().as_base_class()), const json::json_base_class_t&>::value, "");
|
||||
static_assert(noexcept(std::declval<json&>().as_base_class()), "");
|
||||
static_assert(noexcept(std::declval<const json&>().as_base_class()), "");
|
||||
|
||||
SECTION("non-const")
|
||||
{
|
||||
json j = {1, 2, 3};
|
||||
|
||||
CHECK(std::string(j.type_name()) == "array");
|
||||
CHECK(j.size() == 3);
|
||||
CHECK(std::string(j.as_base_class().type_name()) == "custom type_name");
|
||||
CHECK(j.as_base_class().size() == 42);
|
||||
CHECK(&j.as_base_class() == &static_cast<json::json_base_class_t&>(j));
|
||||
|
||||
j.as_base_class().m_size = 7;
|
||||
CHECK(j.as_base_class().size() == 7);
|
||||
CHECK(j.size() == 3);
|
||||
}
|
||||
|
||||
SECTION("const")
|
||||
{
|
||||
const json j = {1, 2, 3};
|
||||
|
||||
CHECK(std::string(j.type_name()) == "array");
|
||||
CHECK(j.size() == 3);
|
||||
CHECK(std::string(j.as_base_class().type_name()) == "custom type_name");
|
||||
CHECK(j.as_base_class().size() == 42);
|
||||
CHECK(&j.as_base_class() == &static_cast<const json::json_base_class_t&>(j));
|
||||
}
|
||||
}
|
||||
|
||||
// A custom base class with a const member: copy-constructible (initializing a
|
||||
// const member works fine), but not copy-/move-assignable (assigning one does
|
||||
// not). Used to check that copy construction never requires more than that.
|
||||
|
||||
@@ -630,6 +630,49 @@ TEST_CASE("modifiers")
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("rvalue at position moves rather than copies")
|
||||
{
|
||||
// regression test: insert(pos, basic_json&&) used to forward to
|
||||
// insert(pos, const basic_json&) because the named rvalue
|
||||
// reference parameter is itself an lvalue, so it always
|
||||
// deep-copied its argument instead of moving it
|
||||
json j_big = std::string(1000, 'x');
|
||||
const auto* const original_buffer = j_big.get_ref<const std::string&>().data();
|
||||
|
||||
auto it = j_array.insert(j_array.begin(), std::move(j_big));
|
||||
CHECK(j_array.size() == 5);
|
||||
CHECK(*it == json(std::string(1000, 'x')));
|
||||
CHECK((*it).get_ref<const std::string&>().data() == original_buffer);
|
||||
|
||||
// the moved-from value is null, the same as after push_back(&&)
|
||||
CHECK(j_big.is_null()); // NOLINT(bugprone-use-after-move,hicpp-invalid-access-moved)
|
||||
}
|
||||
|
||||
SECTION("self-aliasing insertion")
|
||||
{
|
||||
SECTION("without reallocation")
|
||||
{
|
||||
json j_self = {1, 2, 3, 4};
|
||||
j_self.get_ref<json::array_t&>().reserve(j_self.size() + 1);
|
||||
|
||||
auto it = j_self.insert(j_self.begin(), std::move(j_self[1]));
|
||||
CHECK(j_self.size() == 5);
|
||||
CHECK(*it == json(2));
|
||||
CHECK(j_self == json({2, 1, nullptr, 3, 4}));
|
||||
}
|
||||
|
||||
SECTION("with reallocation")
|
||||
{
|
||||
json j_self = {1, 2, 3, 4};
|
||||
j_self.get_ref<json::array_t&>().shrink_to_fit();
|
||||
|
||||
auto it = j_self.insert(j_self.begin(), std::move(j_self[1]));
|
||||
CHECK(j_self.size() == 5);
|
||||
CHECK(*it == json(2));
|
||||
CHECK(j_self == json({2, 1, nullptr, 3, 4}));
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("copies at position")
|
||||
{
|
||||
SECTION("insert before begin()")
|
||||
|
||||
@@ -1540,19 +1540,39 @@ TEST_CASE("MessagePack")
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing MessagePack string: unexpected end of input", json::parse_error&);
|
||||
}
|
||||
|
||||
SECTION("invalid UTF-8 in string (see #5529)")
|
||||
SECTION("ill-formed UTF-8 in string (see #5529, #5651)")
|
||||
{
|
||||
// the MessagePack specification explicitly allows a str object to
|
||||
// contain a byte sequence that is not valid UTF-8 and expects a
|
||||
// deserializer to hand the original bytes back unchanged; this
|
||||
// library follows that, unlike CBOR/UBJSON/BJData/BSON, whose
|
||||
// specifications require text strings to be valid UTF-8
|
||||
|
||||
// a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8
|
||||
// (0xC0 0xAE is an overlong encoding of '.') must be rejected at
|
||||
// decode time, matching every other kind of malformed binary
|
||||
// input, rather than only failing later when the resulting
|
||||
// value is dumped
|
||||
json _;
|
||||
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0xa2, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing MessagePack string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||
CHECK(json::from_msgpack(std::vector<uint8_t>({0xa2, 0xc0, 0xae}), true, false).is_discarded());
|
||||
// (0xC0 0xAE is an overlong encoding of '.') round-trips byte for
|
||||
// byte as a string value
|
||||
const std::vector<uint8_t> ill_formed_value = {0xa2, 0xc0, 0xae};
|
||||
json j_value;
|
||||
CHECK_NOTHROW(j_value = json::from_msgpack(ill_formed_value));
|
||||
REQUIRE(j_value.is_string());
|
||||
CHECK(j_value.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
CHECK(json::from_msgpack(json::to_msgpack(j_value)) == j_value);
|
||||
// dump() still requires valid UTF-8 and throws for such a value,
|
||||
// unless an error handler that replaces or ignores the bytes is
|
||||
// passed
|
||||
CHECK_THROWS_AS(j_value.dump(), json::type_error&);
|
||||
|
||||
// the same bytes as an object key round-trip as well
|
||||
const std::vector<uint8_t> ill_formed_key = {0x81, 0xa2, 0xc0, 0xae, 0x01};
|
||||
json j_key;
|
||||
CHECK_NOTHROW(j_key = json::from_msgpack(ill_formed_key));
|
||||
REQUIRE(j_key.is_object());
|
||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
||||
CHECK(json::from_msgpack(json::to_msgpack(j_key)) == j_key);
|
||||
|
||||
// a MessagePack bin8 blob with the very same bytes is NOT text
|
||||
// and must still be accepted as-is
|
||||
json _;
|
||||
CHECK_NOTHROW(_ = json::from_msgpack(std::vector<uint8_t>({0xc4, 0x02, 0xc0, 0xae})));
|
||||
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
|
||||
|
||||
|
||||
@@ -808,6 +808,26 @@ TEST_CASE("regression tests 2")
|
||||
CHECK(j == k);
|
||||
}
|
||||
|
||||
#ifdef JSON_HAS_CPP_17
|
||||
SECTION("issue #5066 - MSVC converts json to std::variant<json> via the conversion operator")
|
||||
{
|
||||
// std::variant<json> must not be retrievable via get<>(), because otherwise the
|
||||
// implicit conversion operator becomes a candidate that MSVC picks over the variant's
|
||||
// converting constructor, routing a number through the string from_json overload
|
||||
static_assert(!nlohmann::detail::is_detected<nlohmann::detail::get_template_function, const json&, std::variant<json>>::value,
|
||||
"std::variant<json> must not be retrievable via get<>()");
|
||||
|
||||
// clang before 7 cannot instantiate libstdc++'s std::variant<json>
|
||||
#if !(defined(__clang__) && __clang_major__ < 7)
|
||||
// push_back, not emplace_back: #5066 needs the implicit conversion
|
||||
// from json to the vector's value type
|
||||
std::vector<std::variant<json>> v;
|
||||
v.push_back(json(1)); // NOLINT(hicpp-use-emplace,modernize-use-emplace)
|
||||
CHECK(std::get<0>(v[0]) == 1);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
SECTION("issue #3669 - invalid use of incomplete type with optional member and to_json")
|
||||
{
|
||||
const Issue3669Holder h{};
|
||||
@@ -1286,15 +1306,11 @@ TEST_CASE("regression test - #3989 SAX parse_error() returning true")
|
||||
{json::input_format_t::cbor, {0x82, 0xC1, 0x05, 0xD9, 0xD9, 0xF7, 0x06}, {5, 6}, 2},
|
||||
// CBOR: undefined and other simple values become null
|
||||
{json::input_format_t::cbor, {0x84, 0xF7, 0xE0, 0xF8, 0x20, 0x01}, {nullptr, nullptr, nullptr, 1}, 3},
|
||||
// CBOR: ill-formed UTF-8 becomes U+FFFD, also in keys
|
||||
{json::input_format_t::cbor, {0xA1, 0x61, 0xFF, 0x62, 0xC3, 0x28}, {{replacement_character(), replacement_character() + "("}}, 2},
|
||||
// CBOR: members whose key is not a string are skipped, whatever their key and value
|
||||
{json::input_format_t::cbor, {0xA4, 0x01, 0x02, 0x82, 0x01, 0x02, 0xA1, 0x61, 'x', 0x9F, 0xFF, 0xC1, 0x01, 0x5F, 0x41, 0x00, 0xFF, 0x61, 'a', 0x03}, {{"a", 3}}, 3},
|
||||
{json::input_format_t::cbor, {0xBF, 0xF5, 0xBF, 0x61, 'x', 0x7F, 0x61, 'y', 0xFF, 0xFF, 0x61, 'a', 0x03, 0xFF}, {{"a", 3}}, 1},
|
||||
// MessagePack: members whose key is not a string are skipped
|
||||
{json::input_format_t::msgpack, {0x84, 0x01, 0x02, 0x81, 0xA1, 'x', 0x01, 0x92, 0x01, 0x02, 0xD4, 0x01, 0x02, 0xC0, 0xA1, 'a', 0x04}, {{"a", 4}}, 3},
|
||||
// MessagePack: ill-formed UTF-8 becomes U+FFFD
|
||||
{json::input_format_t::msgpack, {0x92, 0xA2, 0xC3, 0x28, 0xA3, 0xE2, 0x82, 'x'}, {replacement_character() + "(", replacement_character() + "x"}, 2},
|
||||
// UBJSON: a char that is not ASCII becomes U+FFFD
|
||||
{json::input_format_t::ubjson, {'[', 'C', 0x80, 'C', 'A', ']'}, {replacement_character(), "A"}, 1},
|
||||
// UBJSON: the longest beginning of a high-precision number is kept
|
||||
|
||||
@@ -10,10 +10,14 @@
|
||||
|
||||
#if JSON_TEST_USING_MULTIPLE_HEADERS
|
||||
#include <nlohmann/detail/meta/type_traits.hpp>
|
||||
#include <nlohmann/ordered_map.hpp>
|
||||
#else
|
||||
#include <nlohmann/json.hpp>
|
||||
#endif
|
||||
|
||||
#include <map>
|
||||
#include <string>
|
||||
|
||||
TEST_CASE("type traits")
|
||||
{
|
||||
SECTION("is_c_string")
|
||||
@@ -83,4 +87,12 @@ TEST_CASE("type traits")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("is_ordered_map")
|
||||
{
|
||||
using nlohmann::detail::is_ordered_map;
|
||||
|
||||
CHECK(is_ordered_map<nlohmann::ordered_map<std::string, int>>::value);
|
||||
CHECK_FALSE(is_ordered_map<std::map<std::string, int>>::value);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2505,6 +2505,43 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
|
||||
CHECK(json::to_ubjson(j) == v);
|
||||
CHECK(json::from_ubjson(v) == j);
|
||||
}
|
||||
|
||||
SECTION("ill-formed UTF-8 (see #5529, #5651)")
|
||||
{
|
||||
// none of the binary format specs requires a decoder to reject
|
||||
// ill-formed UTF-8 in a text string, so a value whose bytes are
|
||||
// not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.')
|
||||
// round-trips byte for byte as a string value; to_ubjson() writes
|
||||
// the bytes unchanged, as before 3.13.0, unless
|
||||
// JSON_STRICT_BINARY_UTF8 is enabled (see
|
||||
// unit-binary_utf8_strict.cpp)
|
||||
const std::vector<uint8_t> v = {'S', 'i', 2, 0xc0, 0xae};
|
||||
json j;
|
||||
CHECK_NOTHROW(j = json::from_ubjson(v));
|
||||
REQUIRE(j.is_string());
|
||||
CHECK(j.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
||||
CHECK_THROWS_AS(j.dump(), json::type_error&);
|
||||
CHECK(json::from_ubjson(json::to_ubjson(j)) == j);
|
||||
|
||||
// the same bytes as an object key round-trip as well
|
||||
const std::vector<uint8_t> v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'};
|
||||
json j_key;
|
||||
CHECK_NOTHROW(j_key = json::from_ubjson(v_key));
|
||||
REQUIRE(j_key.is_object());
|
||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
||||
CHECK(json::from_ubjson(json::to_ubjson(j_key)) == j_key);
|
||||
|
||||
CHECK(json::from_ubjson(json::to_ubjson(json("\xFF"))) == json("\xFF"));
|
||||
// a truncated multi-byte sequence
|
||||
CHECK(json::from_ubjson(json::to_ubjson(json("\xC3"))) == json("\xC3"));
|
||||
// an encoded surrogate half (U+D800)
|
||||
CHECK(json::from_ubjson(json::to_ubjson(json("\xED\xA0\x80"))) == json("\xED\xA0\x80"));
|
||||
// an overlong encoding of '.'
|
||||
CHECK(json::from_ubjson(json::to_ubjson(json("\xC0\xAF"))) == json("\xC0\xAF"));
|
||||
|
||||
// an object key with ill-formed UTF-8 is kept the same way
|
||||
CHECK(json::from_ubjson(json::to_ubjson(json{{"\xFF", 1}})) == json{{"\xFF", 1}});
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("Array Type")
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
#include "doctest_compatibility.h"
|
||||
|
||||
#include <cwchar>
|
||||
#include <nlohmann/json.hpp>
|
||||
using nlohmann::json;
|
||||
|
||||
@@ -68,15 +69,15 @@ TEST_CASE("wide strings")
|
||||
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
|
||||
|
||||
// a lone low surrogate cannot start a pair
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
|
||||
// a high surrogate followed by a non-low-surrogate unit is invalid
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
|
||||
// ... also when the unit is above the low surrogates
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
|
||||
// a lone low surrogate must not swallow the following unit: pairing
|
||||
// it with any second unit would produce valid UTF-8, so the error
|
||||
// has to report an ill-formed byte at the surrogate's own position
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
|
||||
// a valid surrogate pair is still decoded (U+1F600)
|
||||
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
|
||||
}
|
||||
@@ -104,4 +105,55 @@ TEST_CASE("wide strings")
|
||||
// the same unit inside a string is reported as an ill-formed byte
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
|
||||
}
|
||||
|
||||
SECTION("malformed wide-string input outside strings (#5645)")
|
||||
{
|
||||
json _;
|
||||
|
||||
// a lone low surrogate inside a literal must not be truncated to its
|
||||
// low byte and mistaken for the letter the literal expects next
|
||||
// (0xDC72 truncates to 'r', which is what "true" expects after 't')
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast<char16_t>(0xDC72), u'u', u'e'}),
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
|
||||
|
||||
// ... also when the lone surrogate is the last unit of the input
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'f', u'a', u'l', u's', static_cast<char16_t>(0xDD65)}),
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'fals\xFF'", json::parse_error&);
|
||||
|
||||
// a high surrogate followed by a unit that is not its low surrogate
|
||||
// must not silently swallow that unit
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast<char16_t>(0xD872), u'X', u'u', u'e'}),
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
|
||||
|
||||
// ... in particular, if the swallowed unit is the newline that ends a
|
||||
// // comment, the comment must not extend over the following line
|
||||
CHECK(json::parse(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast<char16_t>(0xD800), u'\n',
|
||||
u',', u'2', u' ', u'/', u'/', u'\n', u']'},
|
||||
nullptr, true, /*ignore_comments*/true) == json::parse("[1,2]"));
|
||||
CHECK(json::accept(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast<char16_t>(0xD800), u'\n',
|
||||
u',', u'2', u' ', u'/', u'/', u'\n', u']'}, /*ignore_comments*/true));
|
||||
|
||||
// cases 5 and 6 use a 32-bit wchar_t (Linux, macOS, the BSDs) to reach
|
||||
// the UTF-32 helper tested above via u32string; the 16-bit wchar_t of
|
||||
// Windows goes through the UTF-16 helper instead, already covered by
|
||||
// the u16string cases above
|
||||
#if WCHAR_MAX > 0xFFFFu
|
||||
// a negative wchar_t must not be mistaken for
|
||||
// char_traits<char>::eof() and silently end the input, letting
|
||||
// trailing garbage pass the strict end-of-input check (only observable
|
||||
// where wint_t is signed, e.g. macOS/the BSDs; on Linux wint_t is
|
||||
// unsigned and this was already handled by #5348)
|
||||
std::wstring w = L"[1]";
|
||||
w.push_back(static_cast<wchar_t>(-1));
|
||||
w += L"garbage";
|
||||
CHECK(!json::accept(w));
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(w),
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
|
||||
|
||||
// other negative wchar_t units must not be truncated to their low
|
||||
// byte (0xFFFFFF72 truncates to 'r', as in the u16string case above)
|
||||
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L't', static_cast<wchar_t>(0xFFFFFF72), L'u', L'e'}),
|
||||
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user