Store node integers by halves on big-endian targets

The non-little-endian path of the node writer wrote the integer's native
word over len and next, so len got the high half there. Compose and split
the value explicitly (len is the low half, next the high half); the
little-endian path stays a plain memcpy.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann committed 2026-10-09 16:26:32 +02:00
1 parent 14afaca083
commit 320a463af7
3 files changed
+73 -2

No files matched your search

+4 -1
View File
@@ -801,7 +801,10 @@ indent_done:
n->flags = flags;
n->extra = extra;
n->off = static_cast<std::uint32_t>(off);
set_integer_bits(*n, second);
// len is the low half of the second word, next the high half
// (not a native word over both, which swaps them on big-endian)
n->len = static_cast<std::uint32_t>(second);
n->next = static_cast<std::uint32_t>(second >> 32);
#endif
return n;
}
+11 -1
View File
@@ -57,17 +57,27 @@ NLOHMANN_VIEW_ALWAYS_INLINE bool is_container(const node& n) noexcept
return static_cast<unsigned>(n.kind) - 1u <= 1u;
}
/// the converted value of an integer node (stored in len/next)
/// the converted value of an integer node: len is its low half, next its high
/// half (on little-endian targets the two words are the value in memory)
NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept
{
#if NLOHMANN_VIEW_LITTLE_ENDIAN
std::uint64_t v = 0;
std::memcpy(&v, reinterpret_cast<const unsigned char*>(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
return v;
#else
return static_cast<std::uint64_t>(n.len) | (static_cast<std::uint64_t>(n.next) << 32);
#endif
}
NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept
{
#if NLOHMANN_VIEW_LITTLE_ENDIAN
std::memcpy(reinterpret_cast<unsigned char*>(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
#else
n.len = static_cast<std::uint32_t>(v);
n.next = static_cast<std::uint32_t>(v >> 32);
#endif
}
/// token length of a number node
+58
View File
@@ -401,3 +401,61 @@ TEST_CASE("json_view builder")
}
}
}
TEST_CASE("json_view node integer bits")
{
using nlohmann::detail::view::integer_bits;
using nlohmann::detail::view::set_integer_bits;
// an integer lives in len (low half) and next (high half), on any byte
// order; a big-endian target must not store the native word over both
SECTION("set_integer_bits and integer_bits")
{
node n = {};
for (const std::uint64_t v :
{
std::uint64_t{0}, std::uint64_t{1}, std::uint64_t{0xFFFFFFFFu}, std::uint64_t{0x100000000u},
std::uint64_t{0x0000000200000003u}, std::uint64_t{0x0123456789ABCDEFu}, std::uint64_t{0xFFFFFFFFFFFFFFFEu}
})
{
CAPTURE(v)
n.kind = 0x5A;
n.flags = 0xA5;
n.extra = 0x1234;
n.off = 0x89ABCDEFu;
set_integer_bits(n, v);
CHECK(integer_bits(n) == v);
CHECK(n.len == static_cast<std::uint32_t>(v));
CHECK(n.next == static_cast<std::uint32_t>(v >> 32));
// the other fields are untouched
CHECK(n.kind == 0x5A);
CHECK(n.flags == 0xA5);
CHECK(n.extra == 0x1234);
CHECK(n.off == 0x89ABCDEFu);
}
}
SECTION("parsed integers")
{
struct integer_case
{
const char* text;
std::uint64_t bits;
};
for (const integer_case c :
{
integer_case{"[8589934595]", 0x0000000200000003u}, integer_case{"[4294967296]", 0x100000000u}, integer_case{"[4294967295]", 0xFFFFFFFFu},
integer_case{"[-2]", 0xFFFFFFFFFFFFFFFEu}, integer_case{"[-4294967297]", 0xFFFFFFFEFFFFFFFFu}, integer_case{"[18446744073709551615]", 0xFFFFFFFFFFFFFFFFu},
integer_case{"[7]", 7u}
})
{
CAPTURE(c.text)
const built b = build(c.text, false, false, true);
REQUIRE(b.ok);
const node& n = b.data->tape[1];
CHECK(integer_bits(n) == c.bits);
CHECK(n.len == static_cast<std::uint32_t>(c.bits));
CHECK(n.next == static_cast<std::uint32_t>(c.bits >> 32));
}
}
}