diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index eda47e2d1..a6e857778 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -114,7 +114,7 @@ jobs: dgb_gentx_coinbase_test dgb_connection_coinbase_test dgb_pplns_payout_split_test nmc_auxpow_merkle_test nmc_template_builder_test nmc_auxpow_wire_test nmc_reconstruct_won_block_test nmc_mempool_name_test nmc_block_broadcast_test nmc_host_dualpath_test nmc_fallback_path_conformance_test dgb_gentx_share_path_test dgb_conn_pplns_producer_test dgb_other_tx_resolver_test \ dgb_other_tx_assembler_test dgb_reconstruct_won_block_test dgb_reconstruct_closure_test dgb_gentx_unpack_test dgb_work_source_test dgb_template_builder_test dgb_embedded_coin_node_test dgb_embedded_tx_select_test dgb_template_other_txs_test dgb_coinbase_value_parity_test dgb_submit_classify_test dgb_aux_parent_coinbase_parity_test dgb_template_capture_test dgb_aux_doge_db_commitment_bind_test dgb_aux_doge_mm_commitment_test dgb_aux_doge_dc_proof_test dgb_aux_doge_bind_parsers_test dgb_compact_blocks_bip152_parity_test dgb_aux_dual_target_select_test dgb_aux_broadcast_path_election_test dgb_aux_doge_submit_test dgb_aux_doge_embed_livewire_test dgb_aux_doge_dc_layout_verifier_test \ rpc_request_test softfork_check_test genesis_check_test algo_select_test digishield_walk_test header_chain_test \ - dgb_coin_node_seam_test dgb_block_broadcast_test dgb_won_block_dispatch_test dgb_forced_won_share_dualpath_test dgb_scrypt_pow_test dgb_nonce_grinder_test dgb_regrind_block_test dgb_won_block_finalize_test dgb_share_target_genesis_test dgb_share_target_retarget_test dgb_share_bits_oracle_pin_test dgb_pool_msg_wire_test dgb_get_shares_walk_test dgb_download_stops_test dgb_think_p1_walk_bounds_test dgb_think_p1_desired_emit_test dgb_think_p6_desired_cutoff_test dgb_think_p4_head_keys_test dgb_think_p3_best_head_test dgb_g1_oracle_byte_parity_test dgb_think_p2_walk_bounds_test dgb_expected_time_to_block_test dgb_tail_score_endpoints_test dgb_pool_attempts_per_second_test dgb_pool_efficiency_test dgb_think_p5_best_share_punish_test dgb_auto_ratchet_tail_guard_test dgb_auto_ratchet_sim_test dgb_binomial_conf_interval_test dgb_desired_version_tally_test dgb_min_protocol_ratchet_test dgb_get_height_and_last_endpoints_test dgb_chain_walk_window_test dgb_redistribute_delegate_ghal_test dgb_share_weight_decay_test dgb_naughty_propagation_test dgb_hash_format_parity_test dgb_emergency_decay_saturation_test v37_test \ + dgb_coin_node_seam_test dgb_block_broadcast_test dgb_won_block_dispatch_test dgb_forced_won_share_dualpath_test dgb_scrypt_pow_test dgb_nonce_grinder_test dgb_regrind_block_test dgb_won_block_finalize_test dgb_share_target_genesis_test dgb_share_target_retarget_test dgb_share_bits_oracle_pin_test dgb_pool_msg_wire_test dgb_get_shares_walk_test dgb_download_stops_test dgb_think_p1_walk_bounds_test dgb_think_p1_desired_emit_test dgb_think_p6_desired_cutoff_test dgb_think_p4_head_keys_test dgb_think_p3_best_head_test dgb_g1_oracle_byte_parity_test dgb_think_p2_walk_bounds_test dgb_expected_time_to_block_test dgb_tail_score_endpoints_test dgb_pool_attempts_per_second_test dgb_pool_efficiency_test dgb_think_p5_best_share_punish_test dgb_auto_ratchet_tail_guard_test dgb_auto_ratchet_sim_test dgb_binomial_conf_interval_test dgb_desired_version_tally_test dgb_min_protocol_ratchet_test dgb_get_height_and_last_endpoints_test dgb_chain_walk_window_test dgb_redistribute_delegate_ghal_test dgb_share_weight_decay_test dgb_naughty_propagation_test dgb_hash_format_parity_test dgb_emergency_decay_saturation_test dgb_arith256_muldiv_kat_test v37_test \ -j8 - name: Run tests @@ -263,7 +263,7 @@ jobs: dgb_gentx_coinbase_test dgb_connection_coinbase_test dgb_pplns_payout_split_test nmc_auxpow_merkle_test nmc_template_builder_test nmc_auxpow_wire_test nmc_reconstruct_won_block_test nmc_mempool_name_test nmc_block_broadcast_test nmc_host_dualpath_test nmc_fallback_path_conformance_test dgb_gentx_share_path_test dgb_conn_pplns_producer_test dgb_other_tx_resolver_test \ dgb_other_tx_assembler_test dgb_reconstruct_won_block_test dgb_reconstruct_closure_test dgb_gentx_unpack_test dgb_work_source_test dgb_template_builder_test dgb_embedded_coin_node_test dgb_embedded_tx_select_test dgb_template_other_txs_test dgb_coinbase_value_parity_test dgb_submit_classify_test dgb_aux_parent_coinbase_parity_test dgb_template_capture_test dgb_aux_doge_db_commitment_bind_test dgb_aux_doge_mm_commitment_test dgb_aux_doge_dc_proof_test dgb_aux_doge_bind_parsers_test dgb_compact_blocks_bip152_parity_test dgb_aux_dual_target_select_test dgb_aux_broadcast_path_election_test dgb_aux_doge_submit_test dgb_aux_doge_embed_livewire_test dgb_aux_doge_dc_layout_verifier_test \ rpc_request_test softfork_check_test genesis_check_test algo_select_test digishield_walk_test header_chain_test \ - dgb_coin_node_seam_test dgb_block_broadcast_test dgb_won_block_dispatch_test dgb_forced_won_share_dualpath_test dgb_scrypt_pow_test dgb_nonce_grinder_test dgb_regrind_block_test dgb_won_block_finalize_test dgb_share_target_genesis_test dgb_share_target_retarget_test dgb_share_bits_oracle_pin_test dgb_pool_msg_wire_test dgb_get_shares_walk_test dgb_download_stops_test dgb_think_p1_walk_bounds_test dgb_think_p1_desired_emit_test dgb_think_p6_desired_cutoff_test dgb_think_p4_head_keys_test dgb_think_p3_best_head_test dgb_g1_oracle_byte_parity_test dgb_think_p2_walk_bounds_test dgb_expected_time_to_block_test dgb_tail_score_endpoints_test dgb_pool_attempts_per_second_test dgb_pool_efficiency_test dgb_think_p5_best_share_punish_test dgb_auto_ratchet_tail_guard_test dgb_auto_ratchet_sim_test dgb_binomial_conf_interval_test dgb_desired_version_tally_test dgb_min_protocol_ratchet_test dgb_get_height_and_last_endpoints_test dgb_chain_walk_window_test dgb_redistribute_delegate_ghal_test dgb_share_weight_decay_test dgb_naughty_propagation_test dgb_hash_format_parity_test dgb_emergency_decay_saturation_test test_coin_broadcaster test_multiaddress_pplns test_pplns_stress \ + dgb_coin_node_seam_test dgb_block_broadcast_test dgb_won_block_dispatch_test dgb_forced_won_share_dualpath_test dgb_scrypt_pow_test dgb_nonce_grinder_test dgb_regrind_block_test dgb_won_block_finalize_test dgb_share_target_genesis_test dgb_share_target_retarget_test dgb_share_bits_oracle_pin_test dgb_pool_msg_wire_test dgb_get_shares_walk_test dgb_download_stops_test dgb_think_p1_walk_bounds_test dgb_think_p1_desired_emit_test dgb_think_p6_desired_cutoff_test dgb_think_p4_head_keys_test dgb_think_p3_best_head_test dgb_g1_oracle_byte_parity_test dgb_think_p2_walk_bounds_test dgb_expected_time_to_block_test dgb_tail_score_endpoints_test dgb_pool_attempts_per_second_test dgb_pool_efficiency_test dgb_think_p5_best_share_punish_test dgb_auto_ratchet_tail_guard_test dgb_auto_ratchet_sim_test dgb_binomial_conf_interval_test dgb_desired_version_tally_test dgb_min_protocol_ratchet_test dgb_get_height_and_last_endpoints_test dgb_chain_walk_window_test dgb_redistribute_delegate_ghal_test dgb_share_weight_decay_test dgb_naughty_propagation_test dgb_hash_format_parity_test dgb_emergency_decay_saturation_test dgb_arith256_muldiv_kat_test test_coin_broadcaster test_multiaddress_pplns test_pplns_stress \ v37_test \ -j8 diff --git a/src/impl/dgb/coin/dgb_arith256.hpp b/src/impl/dgb/coin/dgb_arith256.hpp index ec34b09f5..285ab91b1 100644 --- a/src/impl/dgb/coin/dgb_arith256.hpp +++ b/src/impl/dgb/coin/dgb_arith256.hpp @@ -28,8 +28,137 @@ #include +#ifdef _MSC_VER +#include // _umul128 / _udiv128 -- MSVC has no __int128 keyword +#endif + namespace dgb::coin { +// --------------------------------------------------------------------------- +// Portable 128-bit primitives. GCC/Clang express the 64x64->128 multiply, +// 128/64 divide and 64+64 add through unsigned __int128; MSVC has no such +// type, so it uses the x64 intrinsics (_umul128 / _udiv128) that lower to the +// SAME mul/div/add-with-carry instructions. Both branches are bit-identical, +// so the DigiShield boundary vectors are unchanged. Mirrors the portable +// mul128_shift precedent in dgb/share_tracker.hpp. +// --------------------------------------------------------------------------- +namespace detail { + +// low 64 of (a*b + add); high 64 written to hi. a*b+add always fits 128 bits. +inline uint64_t mul_add_u128(uint64_t a, uint64_t b, uint64_t add, uint64_t& hi) { +#ifdef _MSC_VER + uint64_t lo = _umul128(a, b, &hi); + lo += add; + hi += (lo < add) ? 1u : 0u; // carry out of the low limb + return lo; +#else + unsigned __int128 p = (unsigned __int128)a * b + add; + hi = (uint64_t)(p >> 64); + return (uint64_t)p; +#endif +} + +// quotient of (hi:lo)/d; remainder written to rem. Requires hi < d (holds for +// schoolbook long division where hi is the running remainder), so the 64-bit +// quotient never overflows. +inline uint64_t div_u128(uint64_t hi, uint64_t lo, uint64_t d, uint64_t& rem) { +#ifdef _MSC_VER + return _udiv128(hi, lo, d, &rem); +#else + unsigned __int128 cur = ((unsigned __int128)hi << 64) | lo; + rem = (uint64_t)(cur % d); + return (uint64_t)(cur / d); +#endif +} + +// low 64 of (a + b + carry_in); carry out (0/1) written to carry_out. +inline uint64_t add_u128(uint64_t a, uint64_t b, uint64_t carry_in, uint64_t& carry_out) { +#ifdef _MSC_VER + uint64_t s = a + b; + uint64_t c = (s < a) ? 1u : 0u; + s += carry_in; + c += (s < carry_in) ? 1u : 0u; + carry_out = c; + return s; +#else + unsigned __int128 s = (unsigned __int128)a + b + carry_in; + carry_out = (uint64_t)(s >> 64); + return (uint64_t)s; +#endif +} + + +// --------------------------------------------------------------------------- +// PORTABLE-FORCED reference primitives (the #690 native==portable KAT oracle). +// +// The MSVC branches above route through _umul128 / _udiv128 -- x64 hardware +// instructions that CANNOT be compiled or run on the Linux/macOS CI, so the +// "both branches bit-identical" claim can't be diffed directly there. These +// twins reproduce the SAME two operations (64x64->128 multiply, 128/64 divide) +// in pure 32-bit-limb software -- NO __int128, NO intrinsic -- a genuinely +// INDEPENDENT computation. If the software path equals the native __int128 +// path across the DigiShield operand domain on Linux, and the x64 +// _umul128/_udiv128 are (by ISA definition) exactly the 64x64->128 product and +// 128/64 quotient this software reproduces, then the MSVC path we cannot run +// here is equal too. Compiled unconditionally on every platform; NEVER +// referenced by the shipping u256 / mul_div_u256 (those stay byte-identical). +// --------------------------------------------------------------------------- + +// software 64x64 -> 128 multiply (what _umul128 does in hardware). +inline uint64_t umul128_soft(uint64_t a, uint64_t b, uint64_t& hi) { + const uint64_t a_lo = (uint32_t)a, a_hi = a >> 32; + const uint64_t b_lo = (uint32_t)b, b_hi = b >> 32; + const uint64_t p0 = a_lo * b_lo; + const uint64_t p1 = a_lo * b_hi; + const uint64_t p2 = a_hi * b_lo; + const uint64_t p3 = a_hi * b_hi; + const uint64_t mid = (p0 >> 32) + (uint32_t)p1 + (uint32_t)p2; + hi = p3 + (p1 >> 32) + (p2 >> 32) + (mid >> 32); + return (p0 & 0xffffffffULL) | (mid << 32); +} + +// software 128/64 -> 64 divide (what _udiv128 does in hardware). Requires +// hi < d so the 64-bit quotient never overflows -- the SAME precondition the +// _udiv128 intrinsic imposes. Binary long division from the top bit down; the +// running remainder r is kept < d, and (top:r) is the conceptual 65-bit value +// so d up to 2^64-1 is handled without a wider type. +inline uint64_t udiv128_soft(uint64_t hi, uint64_t lo, uint64_t d, uint64_t& rem) { + uint64_t q = 0, r = hi; // r < d on entry (precondition) + for (int i = 63; i >= 0; --i) { + const uint64_t top = r >> 63; // bit shifted out of the 64-bit r + r = (r << 1) | ((lo >> i) & 1ULL); // conceptual 65-bit value == (top:r) + if (top || r >= d) { // (top:r) >= d -> subtract, set bit + r -= d; // uint64 wrap yields the true low64 + q |= 1ULL << i; + } + } + rem = r; + return q; +} + +// Portable-forced twins of the three primitives the shipping u256 uses; each +// mirrors the corresponding MSVC branch above with the intrinsic swapped for +// the software op (add_u128's MSVC branch is already intrinsic-free). +inline uint64_t mul_add_u128_portable(uint64_t a, uint64_t b, uint64_t add, uint64_t& hi) { + uint64_t lo = umul128_soft(a, b, hi); + lo += add; + hi += (lo < add) ? 1u : 0u; + return lo; +} +inline uint64_t div_u128_portable(uint64_t hi, uint64_t lo, uint64_t d, uint64_t& rem) { + return udiv128_soft(hi, lo, d, rem); +} +inline uint64_t add_u128_portable(uint64_t a, uint64_t b, uint64_t carry_in, uint64_t& carry_out) { + uint64_t s = a + b; + uint64_t c = (s < a) ? 1u : 0u; + s += carry_in; + c += (s < carry_in) ? 1u : 0u; + carry_out = c; + return s; +} + +} // namespace detail + // 256-bit unsigned, little-endian limbs (limb[0] least significant). Only the // operations the DigiShield damped multiply needs: scale-by-u64 (truncating at // 256 bits exactly as arith_uint256::operator*=), divide-by-u64, and compare. @@ -75,11 +204,11 @@ struct u256 { // does. This is the divergence point vs a 128/wider proxy. u256 mul_u64(uint64_t m) const { u256 r; - unsigned __int128 carry = 0; + uint64_t carry = 0; for (int i = 0; i < 4; ++i) { - unsigned __int128 prod = (unsigned __int128)limb[i] * m + carry; - r.limb[i] = (uint64_t)prod; - carry = prod >> 64; + uint64_t hi; + r.limb[i] = detail::mul_add_u128(limb[i], m, carry, hi); + carry = hi; } // carry beyond limb[3] dropped -> 256-bit truncation (consensus). return r; @@ -89,11 +218,9 @@ struct u256 { // Schoolbook long division from the most-significant limb down. u256 div_u64(uint64_t d) const { u256 r; - unsigned __int128 rem = 0; + uint64_t rem = 0; for (int i = 3; i >= 0; --i) { - unsigned __int128 cur = (rem << 64) | limb[i]; - r.limb[i] = (uint64_t)(cur / d); - rem = cur % d; + r.limb[i] = detail::div_u128(rem, limb[i], d, rem); } return r; } @@ -102,11 +229,9 @@ struct u256 { // same width discipline as mul_u64). Accumulates the retarget window's // target sum before the divide-by-count average. u256& operator+=(const u256& o) { - unsigned __int128 carry = 0; + uint64_t carry = 0; for (int i = 0; i < 4; ++i) { - unsigned __int128 s = (unsigned __int128)limb[i] + o.limb[i] + carry; - limb[i] = (uint64_t)s; - carry = s >> 64; + limb[i] = detail::add_u128(limb[i], o.limb[i], carry, carry); } return *this; } @@ -130,4 +255,52 @@ inline u256 mul_div_u256(const u256& avg, uint64_t mul, uint64_t div) { return avg.mul_u64(mul).div_u64(div); } + +// --------------------------------------------------------------------------- +// PORTABLE-FORCED twin of u256 -- the #690 KAT oracle. Same little-endian limb +// layout and 256-bit truncation discipline, but mul_u64 / div_u64 / operator+= +// route through the software detail::*_portable primitives instead of the +// native __int128 (or MSVC intrinsic) ones. The native==portable guard diffs +// this against u256 on Linux. NOT used by any shipping code path. +// --------------------------------------------------------------------------- +struct u256_portable { + uint64_t limb[4] = {0, 0, 0, 0}; + + u256_portable() = default; + explicit u256_portable(const u256& s) { for (int i = 0; i < 4; ++i) limb[i] = s.limb[i]; } + + u256_portable mul_u64(uint64_t m) const { + u256_portable r; + uint64_t carry = 0; + for (int i = 0; i < 4; ++i) { + uint64_t hi; + r.limb[i] = detail::mul_add_u128_portable(limb[i], m, carry, hi); + carry = hi; + } + return r; + } + u256_portable div_u64(uint64_t d) const { + u256_portable r; + uint64_t rem = 0; + for (int i = 3; i >= 0; --i) + r.limb[i] = detail::div_u128_portable(rem, limb[i], d, rem); + return r; + } + u256_portable& operator+=(const u256_portable& o) { + uint64_t carry = 0; + for (int i = 0; i < 4; ++i) + limb[i] = detail::add_u128_portable(limb[i], o.limb[i], carry, carry); + return *this; + } + // BIT-IDENTICAL check against a native u256 (all four limbs). + bool equals(const u256& s) const { + return limb[0] == s.limb[0] && limb[1] == s.limb[1] + && limb[2] == s.limb[2] && limb[3] == s.limb[3]; + } +}; + +inline u256_portable mul_div_u256_portable(const u256_portable& avg, uint64_t mul, uint64_t div) { + return avg.mul_u64(mul).div_u64(div); +} + } // namespace dgb::coin \ No newline at end of file diff --git a/src/impl/dgb/test/CMakeLists.txt b/src/impl/dgb/test/CMakeLists.txt index 25bd4c4b3..cfd287e89 100644 --- a/src/impl/dgb/test/CMakeLists.txt +++ b/src/impl/dgb/test/CMakeLists.txt @@ -357,6 +357,20 @@ if (BUILD_TESTING AND GTest_FOUND) GTest::gtest_main GTest::gtest nlohmann_json::nlohmann_json) gtest_add_tests(${dgb_guard} "" AUTO) endforeach() + + # dgb_arith256_muldiv_kat_test: FENCED, additive KAT for #690 -- PROVES the + # DigiShield 256-bit retarget primitives (detail mul_add_u128 / div_u128 / + # add_u128, and u256 mul_u64 / div_u64 / operator+=) are BIT-IDENTICAL on the + # native __int128 path vs the portable-forced software path + # (detail::*_portable / u256_portable) that certifies the MSVC + # _umul128/_udiv128 branch we cannot compile or run on Linux. Special + # coverage of the _udiv128 hi only) -> links only GTest, like the foreach guards above. MUST + # also be in BOTH build.yml --target allowlists (#143 NOT_BUILT trap). + add_executable(dgb_arith256_muldiv_kat_test arith256_muldiv_kat_test.cpp) + target_link_libraries(dgb_arith256_muldiv_kat_test PRIVATE + GTest::gtest_main GTest::gtest) + gtest_add_tests(dgb_arith256_muldiv_kat_test "" AUTO) # --- #82 broadcaster-gate: CoinNode submit/seam contract ----------------- # Links the dgb OBJECT lib (coin_node.cpp) like dgb_share_test so the live # CoinNode::submit_block_hex !m_rpc guard + embedded WorkView slice are diff --git a/src/impl/dgb/test/arith256_muldiv_kat_test.cpp b/src/impl/dgb/test/arith256_muldiv_kat_test.cpp new file mode 100644 index 000000000..0549f6e82 --- /dev/null +++ b/src/impl/dgb/test/arith256_muldiv_kat_test.cpp @@ -0,0 +1,210 @@ +// --------------------------------------------------------------------------- +// DGB dgb_arith256 128-bit primitives KAT -- the guard for #690 (MSVC +// portability rework of the DigiShield 256-bit retarget arithmetic). +// +// dgb_arith256.hpp evaluates the DigiShield damped multiply avg*mul/div at true +// 256-bit width via three 128-bit primitives (mul_add_u128 / div_u128 / +// add_u128). GCC/Clang express them with unsigned __int128; MSVC (no __int128, +// error C4235) uses the x64 intrinsics _umul128 / _udiv128 that lower to the +// SAME mul/div/add-with-carry. The claim "both branches bit-identical" is +// CONSENSUS-critical (a divergent limb would compute a different next target and +// fork from DigiByte Core / p2pool-merged-v36), so it is PROVEN here, not +// asserted -- mirroring the bch #688 muldiv_kat precedent. +// +// The MSVC intrinsics cannot be compiled/run on the Linux/macOS CI, so the +// header exposes a third, fully SOFTWARE implementation (detail::*_portable / +// u256_portable, pure 32-bit-limb, no __int128, no intrinsic) as an independent +// oracle. This test proves software == native across the DigiShield operand +// domain plus a deterministic wide fuzz; since the x64 _umul128/_udiv128 are by +// ISA definition the exact 64x64->128 product and 128/64 quotient the software +// reproduces, native == portable on Linux certifies the MSVC path too. +// +// Pure header math: depends only on . No node / RPC / consensus link. +// p2pool-merged-v36 surface: NONE (local DigiShield retarget compute only). +// --------------------------------------------------------------------------- + +#include +#include + +#include + +using dgb::coin::u256; +using dgb::coin::u256_portable; +using dgb::coin::mul_div_u256; +using dgb::coin::mul_div_u256_portable; +namespace detail = dgb::coin::detail; + +namespace { + +constexpr uint64_t U64MAX = ~uint64_t(0); + +// Deterministic 64-bit LCG (no , no Date/rand) -- reproducible vectors. +struct Lcg { + uint64_t s; + explicit Lcg(uint64_t seed) : s(seed) {} + uint64_t next() { s = s * 6364136223846793005ULL + 1442695040888963407ULL; return s; } +}; + +u256 make(uint64_t l3, uint64_t l2, uint64_t l1, uint64_t l0) { + u256 r; r.limb[3] = l3; r.limb[2] = l2; r.limb[1] = l1; r.limb[0] = l0; return r; +} + +// The DigiShield operand domain: avg_target shapes near pow_limit (~2^224) and +// across the 256-bit truncation boundary, plus small/identity values. +const u256 kTargets[] = { + make(0, 0, 0, 0), + make(0, 0, 0, 1), + make(0, 0, 0, 4096), + make(0, 0, 0, U64MAX), + make(0, 0, 1, 0), // 2^64 + make(0x00000000ffffffffULL, U64MAX, U64MAX, U64MAX),// pow_limit-ish (~2^224) + make(0x8000000000000000ULL, 0, 0, 0), // 2^255 (mul-by-2 truncates) + make(U64MAX, U64MAX, U64MAX, U64MAX), // 2^256 - 1 + make(0x0123456789abcdefULL, 0xfedcba9876543210ULL, + 0x0f1e2d3c4b5a6978ULL, 0x8796a5b4c3d2e1f0ULL), // dense mixed limbs +}; + +// Scalars the damped multiply / divide actually apply: DigiShield timespans +// (60s block, MultiShield windows) plus wide stressors up to 2^64-1. +const uint64_t kScalars[] = { + 1, 2, 3, 60, 75, 90, 150, 3600, 37938, + (1ULL << 32), (1ULL << 40) + 1, (1ULL << 62), + 0x8000000000000001ULL, U64MAX - 1, U64MAX, +}; + +} // namespace + +// ---- primitive layer: native vs portable, exact --------------------------- + +TEST(Arith256Muldiv, MulAddPrimitiveNativeEqualsPortable) { + for (uint64_t a : kScalars) + for (uint64_t b : kScalars) + for (uint64_t add : {uint64_t(0), uint64_t(1), U64MAX, uint64_t(0x1234567800000000ULL)}) { + uint64_t hn, hp; + uint64_t ln = detail::mul_add_u128(a, b, add, hn); + uint64_t lp = detail::mul_add_u128_portable(a, b, add, hp); + EXPECT_EQ(ln, lp) << "lo a=" << a << " b=" << b << " add=" << add; + EXPECT_EQ(hn, hp) << "hi a=" << a << " b=" << b << " add=" << add; + } +} + +TEST(Arith256Muldiv, AddPrimitiveNativeEqualsPortable) { + for (uint64_t a : kScalars) + for (uint64_t b : kScalars) + for (uint64_t cin : {uint64_t(0), uint64_t(1)}) { + uint64_t cn, cp; + uint64_t sn = detail::add_u128(a, b, cin, cn); + uint64_t sp = detail::add_u128_portable(a, b, cin, cp); + EXPECT_EQ(sn, sp); + EXPECT_EQ(cn, cp); + } +} + +// Direct stress of the _udiv128 precondition (hi < d): feed hi at the MAXIMAL +// running-remainder boundary (hi = d-1) so any mishandling of the top-bit / +// running-remainder in the software long division surfaces as a diff vs native +// rather than as MSVC UB in the field. +TEST(Arith256Muldiv, DivPrimitiveRunningRemainderBoundary) { + const uint64_t ds[] = {2, 3, 7, 60, 90, 4096, (1ULL << 40), + 0x8000000000000000ULL, 0x8000000000000001ULL, + U64MAX - 1, U64MAX}; + const uint64_t los[] = {0, 1, 2, U64MAX, 0x0f0f0f0f0f0f0f0fULL, + 0x8000000000000000ULL, 0xdeadbeefcafef00dULL}; + for (uint64_t d : ds) { + for (uint64_t hi : {uint64_t(0), uint64_t(1), d / 2, d - 1}) { // all < d + for (uint64_t lo : los) { + uint64_t rn, rp; + uint64_t qn = detail::div_u128(hi, lo, d, rn); + uint64_t qp = detail::div_u128_portable(hi, lo, d, rp); + EXPECT_EQ(qn, qp) << "q hi=" << hi << " lo=" << lo << " d=" << d; + EXPECT_EQ(rn, rp) << "r hi=" << hi << " lo=" << lo << " d=" << d; + EXPECT_LT(rp, d) << "remainder must be < divisor"; + } + } + } +} + +// ---- u256 layer: mul_u64 / div_u64 / operator+= native vs portable -------- + +TEST(Arith256Muldiv, U256OpsNativeEqualsPortable) { + for (const u256& t : kTargets) { + u256_portable tp(t); + for (uint64_t m : kScalars) { + EXPECT_TRUE(tp.mul_u64(m).equals(t.mul_u64(m))) + << "mul_u64 m=" << m; + EXPECT_TRUE(tp.div_u64(m).equals(t.div_u64(m))) // m != 0 for all kScalars + << "div_u64 d=" << m; + } + for (const u256& o : kTargets) { + u256 sn = t; sn += o; + u256_portable sp = tp; sp += u256_portable(o); + EXPECT_TRUE(sp.equals(sn)) << "operator+="; + } + } +} + +// Full damped-multiply pipeline avg*mul/div at 256 bits, native vs portable. +TEST(Arith256Muldiv, MulDivPipelineNativeEqualsPortable) { + for (const u256& t : kTargets) + for (uint64_t mul : kScalars) + for (uint64_t div : kScalars) { // div != 0 + u256 n = mul_div_u256(t, mul, div); + u256_portable p = mul_div_u256_portable(u256_portable(t), mul, div); + EXPECT_TRUE(p.equals(n)) + << "avg*mul/div mul=" << mul << " div=" << div; + } +} + +// ---- hand-verified cross-platform known answers (hold on MSVC too) --------- + +TEST(Arith256Muldiv, HandVerifiedKnownAnswers) { + // u64-range proxy identities (the pre-256-bit slices relied on these). + EXPECT_EQ(mul_div_u256(u256::from_u64(1000000), 90, 60).low64(), 1500000ULL); + EXPECT_EQ(mul_div_u256(u256::from_u64(4096), 1, 1).low64(), 4096ULL); + + // 256-bit truncation: 2^255 * 2 == 2^256 == 0 (mod 2^256). The DIVERGENCE + // point vs a 128/wider proxy -- carry past limb[3] is dropped. + EXPECT_TRUE(make(0x8000000000000000ULL,0,0,0).mul_u64(2).is_zero()); + // (2^256 - 1) * 2 == 2^256 - 2 (low 256 bits). + { + u256 r = make(U64MAX,U64MAX,U64MAX,U64MAX).mul_u64(2); + EXPECT_EQ(r.limb[0], U64MAX - 1); + EXPECT_EQ(r.limb[1], U64MAX); + EXPECT_EQ(r.limb[2], U64MAX); + EXPECT_EQ(r.limb[3], U64MAX); + } + // (2^256 - 1) / 2 == 2^255 - 1. + { + u256 r = make(U64MAX,U64MAX,U64MAX,U64MAX).div_u64(2); + EXPECT_EQ(r.limb[3], 0x7fffffffffffffffULL); + EXPECT_EQ(r.limb[2], U64MAX); + EXPECT_EQ(r.limb[1], U64MAX); + EXPECT_EQ(r.limb[0], U64MAX); + } + // portable oracle independently satisfies the same truncation KATs. + EXPECT_TRUE(u256_portable(make(0x8000000000000000ULL,0,0,0)).mul_u64(2) + .equals(make(0,0,0,0))); +} + +// ---- deterministic wide fuzz: native == portable over full-width operands -- + +TEST(Arith256Muldiv, WideDeterministicFuzz) { + Lcg rng(0x9E3779B97F4A7C15ULL); + int checked = 0; + for (int i = 0; i < 100000; ++i) { + u256 t = make(rng.next(), rng.next(), rng.next(), rng.next()); + u256_portable tp(t); + uint64_t m = rng.next(); + uint64_t d = rng.next(); if (d == 0) d = 1; + + ASSERT_TRUE(tp.mul_u64(m).equals(t.mul_u64(m))) << "mul fuzz i=" << i; + ASSERT_TRUE(tp.div_u64(d).equals(t.div_u64(d))) << "div fuzz i=" << i; + u256 sn = t; sn += t; + u256_portable sp = tp; sp += tp; + ASSERT_TRUE(sp.equals(sn)) << "add fuzz i=" << i; + ASSERT_TRUE(mul_div_u256_portable(tp, m, d).equals(mul_div_u256(t, m, d))) + << "pipeline fuzz i=" << i; + ++checked; + } + EXPECT_GT(checked, 1000); +}