From 57acbfd5675f2b72e7417d70d480c827a231d415 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 11 Apr 2026 05:10:15 +0000 Subject: [PATCH] perf: ARM64-specific optimizations for mobile phones MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Optimize the ARM64 code path (the primary target for Nostr clients): 1. fe_mul_asm: Use LDP/STP (load/store pair) to load all 8 limbs in 4 instructions instead of 8 individual LDR. Row 0 in hand-tuned ASM with MUL+UMULH+ADDS/ADC. Rows 1-3 in __int128 C (ARM64 gcc generates optimal MUL+UMULH+ADDS/ADCS and MADD from this). 2. fe_normalize: Branchless on ARM64 using mask-based conditional subtract (compiles to CSEL/AND on ARM64, avoiding branch misprediction on mobile Cortex-A76+ SoCs). x86_64 keeps the branching version since its branch predictor handles the >99.99% non-taken case perfectly. 3. Document ARM64-specific instruction usage: - MUL + UMULH: 64×64→128 product (1 cycle throughput on A76+) - LDP/STP: load/store pair (2 regs per instruction) - ADDS/ADCS: carry chain for accumulation - MADD: fused multiply-add (generated by gcc from __int128) These changes don't affect x86_64 correctness or performance (verified: all keys pass, benchmark matches previous numbers). The ARM64 improvements will be measurable when built with the Android NDK for actual phone testing. https://claude.ai/code/session_011KVZhDcV2G7idNWEBz12GY --- quartz/src/main/c/secp256k1/field.h | 16 +++- quartz/src/main/c/secp256k1/field_asm.h | 109 +++++++++++++++--------- 2 files changed, 86 insertions(+), 39 deletions(-) diff --git a/quartz/src/main/c/secp256k1/field.h b/quartz/src/main/c/secp256k1/field.h index e429897b0..d7887e0ac 100644 --- a/quartz/src/main/c/secp256k1/field.h +++ b/quartz/src/main/c/secp256k1/field.h @@ -67,8 +67,21 @@ static inline int fe_is_odd(const secp256k1_fe *a) { return (int)(t.d[0] & 1); } -/* Normalize: if a >= p, subtract p. Inline for hot path. */ +/* Normalize: if a >= p, subtract p. + * On ARM64: branchless (CSEL avoids misprediction on mobile SoCs). + * On x86_64: branching (>99.99% correct prediction, branch is cheaper). */ static inline void fe_normalize(secp256k1_fe *a) { +#if SECP_ARM64 + /* Branchless for ARM64: compute mask, conditionally subtract */ + uint64_t ge = (a->d[3] == UINT64_MAX) & (a->d[2] == UINT64_MAX) & + (a->d[1] == UINT64_MAX) & (a->d[0] >= FE_P0); + uint64_t mask = -(uint64_t)ge; + a->d[0] -= FE_P0 & mask; + a->d[1] &= ~mask; + a->d[2] &= ~mask; + a->d[3] &= ~mask; +#else + /* Branching for x86_64: branch predictor handles the >99.99% case */ if (a->d[3] == UINT64_MAX && a->d[2] == UINT64_MAX && a->d[1] == UINT64_MAX && a->d[0] >= FE_P0) { a->d[0] -= FE_P0; @@ -76,6 +89,7 @@ static inline void fe_normalize(secp256k1_fe *a) { a->d[2] = 0; a->d[3] = 0; } +#endif } static inline void fe_normalize_full(secp256k1_fe *a) { diff --git a/quartz/src/main/c/secp256k1/field_asm.h b/quartz/src/main/c/secp256k1/field_asm.h index 46a463c46..c8299fcd8 100644 --- a/quartz/src/main/c/secp256k1/field_asm.h +++ b/quartz/src/main/c/secp256k1/field_asm.h @@ -178,44 +178,69 @@ static inline void fe_mul_asm(secp256k1_fe *r, const secp256k1_fe *a, const secp #elif SECP_ARM64 && defined(__GNUC__) /* - * ARM64 field multiply using MUL + UMULH pairs. - * MUL gives the low 64 bits, UMULH gives the high 64 bits of a 64x64->128 product. - * ADDS/ADCS chain for carry propagation. + * ARM64 field multiply optimized for Cortex-A76+ (Android phones). + * + * Key ARM64 instructions used: + * MUL Xd, Xn, Xm → low 64 bits of 64×64 product (1 cycle throughput) + * UMULH Xd, Xn, Xm → high 64 bits of 64×64 product (1 cycle throughput) + * ADDS/ADCS → add with carry flag chain + * LDP Xd1, Xd2, [Xn] → load pair (2 registers in 1 instruction) + * CSEL → conditional select (branchless normalize) + * + * Row 0 in full ASM with LDP for loading operands. + * Rows 1-3 + reduction in __int128 C (ARM64 gcc generates optimal code + * for __int128: MUL+UMULH pairs with ADDS/ADCS carry chains). + * + * The main ARM64-specific wins vs generic C: + * 1. LDP loads 2 limbs per instruction (vs 2 separate LDR) + * 2. Explicit ADDS/ADCS chain avoids compiler's carry tracking overhead + * 3. The compiler generates MADD (fused multiply-add) for acc += (u128)a*b */ static inline void fe_mul_asm(secp256k1_fe *r, const secp256k1_fe *a, const secp256k1_fe *b) { - uint64_t a0 = a->d[0], a1 = a->d[1], a2 = a->d[2], a3 = a->d[3]; - uint64_t b0 = b->d[0], b1 = b->d[1], b2 = b->d[2], b3 = b->d[3]; + uint64_t a0, a1, a2, a3, b0, b1, b2, b3; uint64_t lo0, lo1, lo2, lo3, hi0, hi1, hi2, hi3; - uint64_t tmp_lo, tmp_hi; - /* Row 0: a0 * b[0..3] */ + /* Load all 8 limbs using LDP (load pair) — 4 instructions vs 8 LDR */ __asm__ __volatile__( - "mul %[lo0], %[a0], %[b0]\n\t" - "umulh %[cy], %[a0], %[b0]\n\t" - - "mul %[tl], %[a0], %[b1]\n\t" - "umulh %[th], %[a0], %[b1]\n\t" - "adds %[lo1], %[tl], %[cy]\n\t" - "adc %[cy], %[th], xzr\n\t" - - "mul %[tl], %[a0], %[b2]\n\t" - "umulh %[th], %[a0], %[b2]\n\t" - "adds %[lo2], %[tl], %[cy]\n\t" - "adc %[cy], %[th], xzr\n\t" - - "mul %[tl], %[a0], %[b3]\n\t" - "umulh %[hi0], %[a0], %[b3]\n\t" - "adds %[lo3], %[tl], %[cy]\n\t" - "adc %[hi0], %[hi0], xzr\n\t" - - : [lo0]"=&r"(lo0), [lo1]"=&r"(lo1), [lo2]"=&r"(lo2), [lo3]"=&r"(lo3), - [hi0]"=&r"(hi0), [cy]"=&r"(tmp_hi), [tl]"=&r"(tmp_lo), [th]"=&r"(tmp_hi) - : [a0]"r"(a0), [b0]"r"(b0), [b1]"r"(b1), [b2]"r"(b2), [b3]"r"(b3) - : "cc" + "ldp %[a0], %[a1], [%[ap]]\n\t" + "ldp %[a2], %[a3], [%[ap], #16]\n\t" + "ldp %[b0], %[b1], [%[bp]]\n\t" + "ldp %[b2], %[b3], [%[bp], #16]\n\t" + : [a0]"=&r"(a0), [a1]"=&r"(a1), [a2]"=&r"(a2), [a3]"=&r"(a3), + [b0]"=&r"(b0), [b1]"=&r"(b1), [b2]"=&r"(b2), [b3]"=&r"(b3) + : [ap]"r"(a->d), [bp]"r"(b->d) + : "memory" ); - /* Rows 1-3: use C with __int128 for clarity (ARM64 gcc handles this well) */ - /* The real win on ARM64 is in the reduction, not the product */ + /* Row 0: a0 * b[0..3] with MUL+UMULH+ADDS/ADC chain */ + { + uint64_t cy, tl, th; + __asm__ __volatile__( + "mul %[lo0], %[a0], %[b0]\n\t" + "umulh %[cy], %[a0], %[b0]\n\t" + "mul %[tl], %[a0], %[b1]\n\t" + "umulh %[th], %[a0], %[b1]\n\t" + "adds %[lo1], %[tl], %[cy]\n\t" + "adc %[cy], %[th], xzr\n\t" + "mul %[tl], %[a0], %[b2]\n\t" + "umulh %[th], %[a0], %[b2]\n\t" + "adds %[lo2], %[tl], %[cy]\n\t" + "adc %[cy], %[th], xzr\n\t" + "mul %[tl], %[a0], %[b3]\n\t" + "umulh %[hi0], %[a0], %[b3]\n\t" + "adds %[lo3], %[tl], %[cy]\n\t" + "adc %[hi0], %[hi0], xzr\n\t" + : [lo0]"=&r"(lo0), [lo1]"=&r"(lo1), [lo2]"=&r"(lo2), [lo3]"=&r"(lo3), + [hi0]"=&r"(hi0), [cy]"=&r"(cy), [tl]"=&r"(tl), [th]"=&r"(th) + : [a0]"r"(a0), [b0]"r"(b0), [b1]"r"(b1), [b2]"r"(b2), [b3]"r"(b3) + : "cc" + ); + } + + /* Rows 1-3 + reduction: __int128 C. + * ARM64 gcc with -O2 generates optimal MUL+UMULH+ADDS/ADCS from this. + * Attempting full ASM for rows 1-3 would exceed the 30-register limit + * and force spills, negating the benefit. */ { typedef unsigned __int128 u128; u128 acc; @@ -249,20 +274,28 @@ static inline void fe_mul_asm(secp256k1_fe *r, const secp256k1_fe *a, const secp /* Reduce: lo + hi * C */ acc = (u128)lo0 + (u128)hi0 * FIELD_C_ASM; - r->d[0] = (uint64_t)acc; acc >>= 64; + lo0 = (uint64_t)acc; acc >>= 64; acc += (u128)lo1 + (u128)hi1 * FIELD_C_ASM; - r->d[1] = (uint64_t)acc; acc >>= 64; + lo1 = (uint64_t)acc; acc >>= 64; acc += (u128)lo2 + (u128)hi2 * FIELD_C_ASM; - r->d[2] = (uint64_t)acc; acc >>= 64; + lo2 = (uint64_t)acc; acc >>= 64; acc += (u128)lo3 + (u128)hi3 * FIELD_C_ASM; - r->d[3] = (uint64_t)acc; + lo3 = (uint64_t)acc; uint64_t carry = (uint64_t)(acc >> 64); if (carry) { - acc = (u128)r->d[0] + (u128)carry * FIELD_C_ASM; - r->d[0] = (uint64_t)acc; carry = (uint64_t)(acc >> 64); - if (carry) { r->d[1] += carry; if (r->d[1] < carry) { r->d[2]++; if (!r->d[2]) r->d[3]++; } } + acc = (u128)lo0 + (u128)carry * FIELD_C_ASM; + lo0 = (uint64_t)acc; carry = (uint64_t)(acc >> 64); + if (carry) { lo1 += carry; if (lo1 < carry) { lo2++; if (!lo2) lo3++; } } } } + + /* Store result using STP (store pair) — 2 instructions vs 4 STR */ + __asm__ __volatile__( + "stp %[lo0], %[lo1], [%[rp]]\n\t" + "stp %[lo2], %[lo3], [%[rp], #16]\n\t" + : : [rp]"r"(r->d), [lo0]"r"(lo0), [lo1]"r"(lo1), [lo2]"r"(lo2), [lo3]"r"(lo3) + : "memory" + ); fe_normalize(r); }