From a269794db7ce14879dc07b0ad510e878b593d4d2 Mon Sep 17 00:00:00 2001 From: Suyog Verma Date: Thu, 8 Oct 2026 12:19:32 +0530 Subject: [PATCH] Use MSVC intrinsics for full multiplication (#5782) * Use MSVC intrinsics for full multiplication Signed-off-by: Suyog Verma * Fix formatting in unit-class_lexer Signed-off-by: Suyog Verma * Address review feedback Signed-off-by: Suyog Verma --------- Signed-off-by: Suyog Verma --- include/nlohmann/detail/bit_ops.hpp | 13 ++++++++-- single_include/nlohmann/json.hpp | 13 ++++++++-- tests/src/unit-class_lexer.cpp | 37 ++++++++++++++++++++++++++--- 3 files changed, 56 insertions(+), 7 deletions(-) diff --git a/include/nlohmann/detail/bit_ops.hpp b/include/nlohmann/detail/bit_ops.hpp index 9655ff18b..9ccb82c66 100644 --- a/include/nlohmann/detail/bit_ops.hpp +++ b/include/nlohmann/detail/bit_ops.hpp @@ -9,12 +9,15 @@ #pragma once #include // uint64_t +#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + #include // __umulh, _umul128 +#endif #include // Portable bit-level helpers for the number and string scanners. They use -// compiler builtins where available and plain C++ otherwise, so they need no -// platform headers and work regardless of byte order. +// compiler builtins or platform-specific intrinsics where available and plain +// C++ otherwise, so they work regardless of byte order. NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -52,6 +55,12 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc __extension__ using uint128 = unsigned __int128; const uint128 r = static_cast(a) * b; return {static_cast(r), static_cast(r >> 64u)}; +#elif defined(_MSC_VER) && defined(_M_X64) + std::uint64_t high = 0; + const std::uint64_t low = _umul128(a, b, &high); + return {low, high}; +#elif defined(_MSC_VER) && defined(_M_ARM64) + return {a * b, __umulh(a, b)}; #else const std::uint64_t a_lo = a & 0xFFFFFFFFu; const std::uint64_t a_hi = a >> 32u; diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4f1ded530..72c95feea 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -8791,13 +8791,16 @@ NLOHMANN_JSON_NAMESPACE_END #include // uint64_t +#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64)) + #include // __umulh, _umul128 +#endif // #include // Portable bit-level helpers for the number and string scanners. They use -// compiler builtins where available and plain C++ otherwise, so they need no -// platform headers and work regardless of byte order. +// compiler builtins or platform-specific intrinsics where available and plain +// C++ otherwise, so they work regardless of byte order. NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -8835,6 +8838,12 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc __extension__ using uint128 = unsigned __int128; const uint128 r = static_cast(a) * b; return {static_cast(r), static_cast(r >> 64u)}; +#elif defined(_MSC_VER) && defined(_M_X64) + std::uint64_t high = 0; + const std::uint64_t low = _umul128(a, b, &high); + return {low, high}; +#elif defined(_MSC_VER) && defined(_M_ARM64) + return {a * b, __umulh(a, b)}; #else const std::uint64_t a_lo = a & 0xFFFFFFFFu; const std::uint64_t a_hi = a >> 32u; diff --git a/tests/src/unit-class_lexer.cpp b/tests/src/unit-class_lexer.cpp index 1d20901d6..860af81d6 100644 --- a/tests/src/unit-class_lexer.cpp +++ b/tests/src/unit-class_lexer.cpp @@ -17,6 +17,7 @@ using nlohmann::json; #include // uint32_t, uint64_t #include // strtod #include // memcpy +#include // numeric_limits #include // stringstream #include // string #include // pair @@ -891,8 +892,39 @@ TEST_CASE("Eisel-Lemire float conversion") SECTION("128-bit products and leading zeros") { + const auto check_product = [](std::uint64_t a, std::uint64_t b) + { + const auto product = nlohmann::detail::full_multiplication(a, b); + CHECK(big_from(product.high, product.low) == big_mul(big_from(0, a), big_from(0, b))); + }; + + const std::uint64_t max = (std::numeric_limits::max)(); + const std::array, 13> edge_cases = + { + { + {0, 0}, + {0, 1}, + {1, 1}, + {1, max}, + {0xFFFFFFFFu, 0x100000000u}, + {0x100000000u, 0x100000000u}, + {0x100000001u, 0x100000001u}, + {max, max}, + {max, 2}, + {0xFFFFFFFF00000000u, 0x100000001u}, + {0x100000001u, 0xFFFFFFFF00000000u}, + {max, 1}, + {2, max}, + } + }; + + for (const auto& test : edge_cases) + { + check_product(test.first, test.second); + } + // whichever implementation the compiler gets (with or without a - // 128-bit integer type or a builtin) + // 128-bit integer type or a builtin / intrinsic) std::uint64_t state = 42; for (int i = 0; i < 10000; ++i) { @@ -901,8 +933,7 @@ TEST_CASE("Eisel-Lemire float conversion") state ^= state << 17u; const std::uint64_t a = state; const std::uint64_t b = (state * 0x9E3779B97F4A7C15u) >> (i % 64); - const auto product = nlohmann::detail::full_multiplication(a, b); - CHECK(big_from(product.high, product.low) == big_mul(big_from(0, a), big_from(0, b))); + check_product(a, b); const int k = i % 64; const std::uint64_t x = (std::uint64_t{1} << k) | (a & ((std::uint64_t{1} << k) - 1));