Use MSVC intrinsics for full multiplication (#5782)

* Use MSVC intrinsics for full multiplication

Signed-off-by: Suyog Verma <suyogverma0057@gmail.com>

* Fix formatting in unit-class_lexer

Signed-off-by: Suyog Verma <suyogverma0057@gmail.com>

* Address review feedback

Signed-off-by: Suyog Verma <suyogverma0057@gmail.com>

---------

Signed-off-by: Suyog Verma <suyogverma0057@gmail.com>
This commit is contained in:
Suyog Verma authored and GitHub committed 2026-10-08 08:49:32 +02:00
1 parent ed8ba0201f
commit a269794db7
3 files changed
+56 -7

No files matched your search

+11 -2
View File
@@ -9,12 +9,15 @@
#pragma once
#include <cstdint> // uint64_t
#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
#include <intrin0.h> // __umulh, _umul128
#endif
#include <nlohmann/detail/abi_macros.hpp>
// Portable bit-level helpers for the number and string scanners. They use
// compiler builtins where available and plain C++ otherwise, so they need no
// platform headers and work regardless of byte order.
// compiler builtins or platform-specific intrinsics where available and plain
// C++ otherwise, so they work regardless of byte order.
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -52,6 +55,12 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc
__extension__ using uint128 = unsigned __int128;
const uint128 r = static_cast<uint128>(a) * b;
return {static_cast<std::uint64_t>(r), static_cast<std::uint64_t>(r >> 64u)};
#elif defined(_MSC_VER) && defined(_M_X64)
std::uint64_t high = 0;
const std::uint64_t low = _umul128(a, b, &high);
return {low, high};
#elif defined(_MSC_VER) && defined(_M_ARM64)
return {a * b, __umulh(a, b)};
#else
const std::uint64_t a_lo = a & 0xFFFFFFFFu;
const std::uint64_t a_hi = a >> 32u;
+11 -2
View File
@@ -8791,13 +8791,16 @@ NLOHMANN_JSON_NAMESPACE_END
#include <cstdint> // uint64_t
#if !defined(__SIZEOF_INT128__) && defined(_MSC_VER) && (defined(_M_X64) || defined(_M_ARM64))
#include <intrin0.h> // __umulh, _umul128
#endif
// #include <nlohmann/detail/abi_macros.hpp>
// Portable bit-level helpers for the number and string scanners. They use
// compiler builtins where available and plain C++ otherwise, so they need no
// platform headers and work regardless of byte order.
// compiler builtins or platform-specific intrinsics where available and plain
// C++ otherwise, so they work regardless of byte order.
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -8835,6 +8838,12 @@ inline uint128_parts full_multiplication(std::uint64_t a, std::uint64_t b) noexc
__extension__ using uint128 = unsigned __int128;
const uint128 r = static_cast<uint128>(a) * b;
return {static_cast<std::uint64_t>(r), static_cast<std::uint64_t>(r >> 64u)};
#elif defined(_MSC_VER) && defined(_M_X64)
std::uint64_t high = 0;
const std::uint64_t low = _umul128(a, b, &high);
return {low, high};
#elif defined(_MSC_VER) && defined(_M_ARM64)
return {a * b, __umulh(a, b)};
#else
const std::uint64_t a_lo = a & 0xFFFFFFFFu;
const std::uint64_t a_hi = a >> 32u;
+34 -3
View File
@@ -17,6 +17,7 @@ using nlohmann::json;
#include <cstdint> // uint32_t, uint64_t
#include <cstdlib> // strtod
#include <cstring> // memcpy
#include <limits> // numeric_limits
#include <sstream> // stringstream
#include <string> // string
#include <utility> // pair
@@ -891,8 +892,39 @@ TEST_CASE("Eisel-Lemire float conversion")
SECTION("128-bit products and leading zeros")
{
const auto check_product = [](std::uint64_t a, std::uint64_t b)
{
const auto product = nlohmann::detail::full_multiplication(a, b);
CHECK(big_from(product.high, product.low) == big_mul(big_from(0, a), big_from(0, b)));
};
const std::uint64_t max = (std::numeric_limits<std::uint64_t>::max)();
const std::array<std::pair<std::uint64_t, std::uint64_t>, 13> edge_cases =
{
{
{0, 0},
{0, 1},
{1, 1},
{1, max},
{0xFFFFFFFFu, 0x100000000u},
{0x100000000u, 0x100000000u},
{0x100000001u, 0x100000001u},
{max, max},
{max, 2},
{0xFFFFFFFF00000000u, 0x100000001u},
{0x100000001u, 0xFFFFFFFF00000000u},
{max, 1},
{2, max},
}
};
for (const auto& test : edge_cases)
{
check_product(test.first, test.second);
}
// whichever implementation the compiler gets (with or without a
// 128-bit integer type or a builtin)
// 128-bit integer type or a builtin / intrinsic)
std::uint64_t state = 42;
for (int i = 0; i < 10000; ++i)
{
@@ -901,8 +933,7 @@ TEST_CASE("Eisel-Lemire float conversion")
state ^= state << 17u;
const std::uint64_t a = state;
const std::uint64_t b = (state * 0x9E3779B97F4A7C15u) >> (i % 64);
const auto product = nlohmann::detail::full_multiplication(a, b);
CHECK(big_from(product.high, product.low) == big_mul(big_from(0, a), big_from(0, b)));
check_product(a, b);
const int k = i % 64;
const std::uint64_t x = (std::uint64_t{1} << k) | (a & ((std::uint64_t{1} << k) - 1));