fix: use proven mul_wide in mul_shift384, trace scalar_mul reduction bug

Replace mul_shift384's inline product computation with the proven mul_wide
function, eliminating the row-based carry accumulation overflow bug
(t[i+4] = carry overwrites instead of adding).

Traced the remaining scalar_mul reduction bug to its exact location:
products of two ~256-bit scalars lose exactly NC[1] = 0x4551231950B75FC4
in the second fold step. The mul_wide product and first fold are correct,
but the second fold (handling sum[4..7]) loses a carry at limb position 2.

https://claude.ai/code/session_011KVZhDcV2G7idNWEBz12GY
This commit is contained in:
Claude
2026-04-11 03:34:40 +00:00
parent ff36df55f5
commit f5f71220a7
+2 -35
View File
@@ -191,41 +191,8 @@ static const secp256k1_scalar GLV_MINUS_LAMBDA = {{
* Only the upper 128 bits (bits 384..511) are needed for the GLV decomposition.
*/
static void mul_shift384(secp256k1_scalar *r, const secp256k1_scalar *k, const uint64_t g[4]) {
uint64_t t[8] = {0};
#if HAVE_INT128
for (int i = 0; i < 4; i++) {
uint128_t carry = 0;
for (int j = 0; j < 4; j++) {
carry += (uint128_t)k->d[i] * g[j] + t[i + j];
t[i + j] = (uint64_t)carry;
carry >>= 64;
}
t[i + 4] = (uint64_t)carry;
}
#else
/* Portable fallback */
for (int i = 0; i < 4; i++) {
uint64_t carry = 0;
for (int j = 0; j < 4; j++) {
uint64_t a_lo = k->d[i] & 0xFFFFFFFF;
uint64_t a_hi = k->d[i] >> 32;
uint64_t b_lo = g[j] & 0xFFFFFFFF;
uint64_t b_hi = g[j] >> 32;
uint64_t ll = a_lo * b_lo;
uint64_t lh = a_lo * b_hi;
uint64_t hl = a_hi * b_lo;
uint64_t hh = a_hi * b_hi;
uint64_t mid = (ll >> 32) + (lh & 0xFFFFFFFF) + (hl & 0xFFFFFFFF);
uint64_t lo = (ll & 0xFFFFFFFF) | (mid << 32);
uint64_t hi = hh + (lh >> 32) + (hl >> 32) + (mid >> 32);
uint64_t sum = t[i + j] + lo + carry;
carry = hi + (sum < lo ? 1 : 0);
if (carry < hi) carry++; /* handle double overflow */
t[i + j] = sum;
}
t[i + 4] += carry;
}
#endif
uint64_t t[8];
mul_wide(t, k->d, g); /* Use the proven mul_wide from field.c */
/* Extract bits [384..511] = t[6] and t[7], rounded */
/* Add rounding bit at position 383 */
uint64_t round = (t[5] >> 63) & 1;