Merge the duplicated UBJSON/BJData integer marker ladders

write_number_with_ubjson_prefix() (unsigned and signed overloads) and
ubjson_prefix() (number_integer and number_unsigned cases) each picked
the UBJSON/BJData integer marker (i, U, I, u, l, m, L, M, H) with their
own independent if/else ladder, and the values beyond 64 bits were
handled by a second, tag-dispatched pair of ladders. An optimized
container announces the marker of its first element via ubjson_prefix()
and then writes every element through write_number_with_ubjson_prefix(),
so the two had to be kept in lockstep by hand across four call sites.

Replace all of that with one ubjson_integer_prefix() built on
value_in_range_of<T>, and one write_ubjson_integer_payload() that
writes the value (or, for 'H', the decimal digits) for a given marker.
write_number_with_ubjson_prefix() and ubjson_prefix() keep their
signatures and now just call these two helpers.

Behavior, the public API and the ABI are unchanged. Verified with a
new regression test covering scalars and $-optimized arrays/objects at
every int8/uint8/int16/uint16/int32/uint32/int64/uint64 boundary for
to_ubjson/to_bjdata (both use_size/use_type settings), and by diffing
to_ubjson/to_bjdata output before and after over the json_test_data
corpus (bit-identical).

Part of #5710

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-30 10:19:54 +02:00
parent e158b080bd
commit 0ffe9ab4a8
3 changed files with 413 additions and 480 deletions
+221
View File
@@ -3033,3 +3033,224 @@ TEST_CASE("UBJSON optimized array of unsigned integers beyond int64")
CHECK(json::to_ubjson(j, true, true) == expected);
CHECK(json::from_ubjson(expected) == j);
}
namespace
{
// the bytes that follow the marker of an integer: the value in the width of
// the marker (big endian for UBJSON, little endian for BJData), or, for a
// high-precision number, the length and the decimal digits
std::vector<std::uint8_t> integer_payload(const char marker, const json& value, const bool little_endian)
{
std::size_t width = 0;
switch (marker)
{
case 'i':
case 'U':
width = 1;
break;
case 'I':
case 'u':
width = 2;
break;
case 'l':
case 'm':
width = 4;
break;
case 'L':
case 'M':
width = 8;
break;
default:
{
const std::string digits = value.dump();
std::vector<std::uint8_t> result = {'i', static_cast<std::uint8_t>(digits.size())};
for (const char c : digits)
{
result.push_back(static_cast<std::uint8_t>(c));
}
return result;
}
}
const std::uint64_t bits = value.is_number_unsigned()
? value.get<std::uint64_t>()
: static_cast<std::uint64_t>(value.get<std::int64_t>());
std::vector<std::uint8_t> result(width);
for (std::size_t i = 0; i < width; ++i)
{
result[little_endian ? i : width - 1 - i] = static_cast<std::uint8_t>(bits >> (8 * i));
}
return result;
}
json i64(const std::int64_t v)
{
return v;
}
json u64(const std::uint64_t v)
{
return v;
}
} // namespace
TEST_CASE("UBJSON and BJData integer markers at every range edge")
{
// An optimized container announces the marker of its values after `$` and
// then writes every value without a marker, so the marker the writer
// announces and the width it writes must match for every value. This
// checks both for the values around each edge of the integer types, as
// scalars and as the values of optimized arrays and objects.
struct integer_case
{
json value;
char ubjson; // expected UBJSON marker
char bjdata; // expected BJData marker
};
const std::int64_t int64_min = (std::numeric_limits<std::int64_t>::min)();
const std::int64_t int64_max = (std::numeric_limits<std::int64_t>::max)();
const std::uint64_t uint64_max = (std::numeric_limits<std::uint64_t>::max)();
const std::vector<integer_case> cases =
{
// int8
{i64(-129), 'I', 'I'},
{i64(-128), 'i', 'i'},
{i64(-127), 'i', 'i'},
{i64(-1), 'i', 'i'},
{i64(0), 'i', 'i'},
{u64(0), 'i', 'i'},
{i64(126), 'i', 'i'},
{i64(127), 'i', 'i'},
{u64(127), 'i', 'i'},
{i64(128), 'U', 'U'},
{u64(128), 'U', 'U'},
// uint8
{i64(254), 'U', 'U'},
{i64(255), 'U', 'U'},
{u64(255), 'U', 'U'},
{i64(256), 'I', 'I'},
{u64(256), 'I', 'I'},
// int16
{i64(-32769), 'l', 'l'},
{i64(-32768), 'I', 'I'},
{i64(-32767), 'I', 'I'},
{i64(32766), 'I', 'I'},
{i64(32767), 'I', 'I'},
{u64(32767), 'I', 'I'},
{i64(32768), 'l', 'u'},
{u64(32768), 'l', 'u'},
// uint16 (BJData only)
{i64(65534), 'l', 'u'},
{i64(65535), 'l', 'u'},
{u64(65535), 'l', 'u'},
{i64(65536), 'l', 'l'},
{u64(65536), 'l', 'l'},
// int32
{i64(-2147483649LL), 'L', 'L'},
{i64(-2147483648LL), 'l', 'l'},
{i64(-2147483647LL), 'l', 'l'},
{i64(2147483646LL), 'l', 'l'},
{i64(2147483647LL), 'l', 'l'},
{u64(2147483647ULL), 'l', 'l'},
{i64(2147483648LL), 'L', 'm'},
{u64(2147483648ULL), 'L', 'm'},
// uint32 (BJData only)
{i64(4294967294LL), 'L', 'm'},
{i64(4294967295LL), 'L', 'm'},
{u64(4294967295ULL), 'L', 'm'},
{i64(4294967296LL), 'L', 'L'},
{u64(4294967296ULL), 'L', 'L'},
// int64
{i64(int64_min), 'L', 'L'},
{i64(int64_min + 1), 'L', 'L'},
{i64(int64_max - 1), 'L', 'L'},
{i64(int64_max), 'L', 'L'},
{u64(static_cast<std::uint64_t>(int64_max)), 'L', 'L'},
// uint64 (BJData only; UBJSON writes a high-precision number)
{u64(static_cast<std::uint64_t>(int64_max) + 1), 'H', 'M'},
{u64(uint64_max - 1), 'H', 'M'},
{u64(uint64_max), 'H', 'M'},
};
for (const auto& c : cases)
{
for (const bool bjdata :
{
false, true
})
{
const char marker = bjdata ? c.bjdata : c.ubjson;
const std::vector<std::uint8_t> payload = integer_payload(marker, c.value, bjdata);
const auto to_binary = [bjdata](const json & j, const bool use_size, const bool use_type)
{
return bjdata ? json::to_bjdata(j, use_size, use_type) : json::to_ubjson(j, use_size, use_type);
};
const auto from_binary = [bjdata](const std::vector<std::uint8_t>& v)
{
return bjdata ? json::from_bjdata(v) : json::from_ubjson(v);
};
INFO("value = " << c.value.dump() << (c.value.is_number_unsigned() ? " (unsigned)" : "") << ", format = " << (bjdata ? "BJData" : "UBJSON"));
// scalar
std::vector<std::uint8_t> expected = {static_cast<std::uint8_t>(marker)};
expected.insert(expected.end(), payload.begin(), payload.end());
for (const bool use_size :
{
false, true
})
{
CHECK(to_binary(c.value, use_size, false) == expected);
}
CHECK(from_binary(expected) == c.value);
const json arr = {c.value, c.value, c.value};
// array without count or type: every value has its marker
expected = {'['};
for (int i = 0; i < 3; ++i)
{
expected.push_back(static_cast<std::uint8_t>(marker));
expected.insert(expected.end(), payload.begin(), payload.end());
}
expected.push_back(']');
CHECK(to_binary(arr, false, false) == expected);
CHECK(from_binary(expected) == arr);
// array with count: every value has its marker
expected = {'[', '#', 'i', 3};
for (int i = 0; i < 3; ++i)
{
expected.push_back(static_cast<std::uint8_t>(marker));
expected.insert(expected.end(), payload.begin(), payload.end());
}
CHECK(to_binary(arr, true, false) == expected);
CHECK(from_binary(expected) == arr);
// array with type and count: the marker once, then the payloads
expected = {'[', '$', static_cast<std::uint8_t>(marker), '#', 'i', 3};
for (int i = 0; i < 3; ++i)
{
expected.insert(expected.end(), payload.begin(), payload.end());
}
CHECK(to_binary(arr, true, true) == expected);
CHECK(from_binary(expected) == arr);
// object with type and count: the marker once, then key and payload
const json obj = {{"a", c.value}, {"b", c.value}};
expected = {'{', '$', static_cast<std::uint8_t>(marker), '#', 'i', 2};
for (const char key :
{'a', 'b'
})
{
expected.push_back('i');
expected.push_back(1);
expected.push_back(static_cast<std::uint8_t>(key));
expected.insert(expected.end(), payload.begin(), payload.end());
}
CHECK(to_binary(obj, true, true) == expected);
CHECK(from_binary(expected) == obj);
}
}
}