]> git.codecow.com Git - nano25519.git/commitdiff
Parallel vector squaring did not perform any better than sequential scalar squaring...
authorChris Duncan <chris@codecow.com>
Thu, 1 Oct 2026 14:48:52 +0000 (07:48 -0700)
committerChris Duncan <chris@codecow.com>
Thu, 1 Oct 2026 14:48:52 +0000 (07:48 -0700)
src/assembly/ed25519/fe.ts
src/assembly/ed25519/p.ts

index a239260d732cd4ce00e04a3cacb35b9e5e28311f..ff29d9941bbe2879a76d9781b92699409b6e438d 100644 (file)
@@ -890,291 +890,6 @@ export function fe_sq (h: FieldElement, f: FieldElement): void {
        store<i32>(h_ptr, h9, 36)
 }
 
-/**
- * Square a FieldElement and store the result.
- *
- * Some iterations of the inner loop can be skipped since
- * `f[x]*f[y] + f[y]*f[x] = 2*f[x]*f[y]`
- *
- * ```
- * // non-constant-time example
- * for (let i = 0; i < 10; i++) {
- *     for (let j = i; j < 10; j++) {
- *             h[i + j] += f[i] * g[j]
- *             if (i < j) {
- *                     h[i + j] *= 2
- *             }
- *     }
- * }
- * ```
- *
- * h = f * f
- * Can overlap h with f.
- *
- * Preconditions:
- * |f| bounded by 1.65x2²⁶,1.65x2²⁵,1.65x2²⁶,1.65x2²⁵,etc.
- *
- * Postconditions:
- * |h| bounded by 1.01x2²⁵,1.01x2²⁴,1.01x2²⁵,1.01x2²⁴,etc.
- *
- * @param hx FieldElement power destination
- * @param fx FieldElement base source
- */
-export function fe_sq_vec (hx: FieldElement, hy: FieldElement, fx: FieldElement, fy: FieldElement): void {
-       const fx_ptr: usize = changetype<usize>(fx)
-       const fy_ptr: usize = changetype<usize>(fy)
-       const f0: v128 = i32x4(load<i32>(fx_ptr, 0), load<i32>(fy_ptr, 0), 0, 0)
-       const f1: v128 = i32x4(load<i32>(fx_ptr, 4), load<i32>(fy_ptr, 4), 0, 0)
-       const f2: v128 = i32x4(load<i32>(fx_ptr, 8), load<i32>(fy_ptr, 8), 0, 0)
-       const f3: v128 = i32x4(load<i32>(fx_ptr, 12), load<i32>(fy_ptr, 12), 0, 0)
-       const f4: v128 = i32x4(load<i32>(fx_ptr, 16), load<i32>(fy_ptr, 16), 0, 0)
-       const f5: v128 = i32x4(load<i32>(fx_ptr, 20), load<i32>(fy_ptr, 20), 0, 0)
-       const f6: v128 = i32x4(load<i32>(fx_ptr, 24), load<i32>(fy_ptr, 24), 0, 0)
-       const f7: v128 = i32x4(load<i32>(fx_ptr, 28), load<i32>(fy_ptr, 28), 0, 0)
-       const f8: v128 = i32x4(load<i32>(fx_ptr, 32), load<i32>(fy_ptr, 32), 0, 0)
-       const f9: v128 = i32x4(load<i32>(fx_ptr, 36), load<i32>(fy_ptr, 36), 0, 0)
-
-       const v19: v128 = i32x4.splat(19)
-       const f5_19: v128 = i32x4.mul(f5, v19) /* 1.959375*2²⁹ */
-       const f6_19: v128 = i32x4.mul(f6, v19) /* 1.959375*2³⁰ */
-       const f7_19: v128 = i32x4.mul(f7, v19) /* 1.959375*2²⁹ */
-       const f8_19: v128 = i32x4.mul(f8, v19) /* 1.959375*2³⁰ */
-       const f9_19: v128 = i32x4.mul(f9, v19) /* 1.959375*2²⁹ */
-
-       // f[0]
-       let f_2: v128 = i32x4.shl(f0, 1)
-
-       let h0: v128 = i64x2.extmul_low_i32x4_s(f0, f0)
-       let h1: v128 = i64x2.extmul_low_i32x4_s(f_2, f1)
-       let h2: v128 = i64x2.extmul_low_i32x4_s(f_2, f2)
-       let h3: v128 = i64x2.extmul_low_i32x4_s(f_2, f3)
-       let h4: v128 = i64x2.extmul_low_i32x4_s(f_2, f4)
-       let h5: v128 = i64x2.extmul_low_i32x4_s(f_2, f5)
-       let h6: v128 = i64x2.extmul_low_i32x4_s(f_2, f6)
-       let h7: v128 = i64x2.extmul_low_i32x4_s(f_2, f7)
-       let h8: v128 = i64x2.extmul_low_i32x4_s(f_2, f8)
-       let h9: v128 = i64x2.extmul_low_i32x4_s(f_2, f9)
-
-       // f[1]
-       f_2 = i32x4.shl(f1, 1)
-       let f_4: v128 = i32x4.shl(f1, 2)
-
-       let t0: v128 = i64x2.extmul_low_i32x4_s(f_2, f1)
-       let t1: v128 = i64x2.extmul_low_i32x4_s(f_2, f2)
-       let t2: v128 = i64x2.extmul_low_i32x4_s(f_4, f3)
-       let t3: v128 = i64x2.extmul_low_i32x4_s(f_2, f4)
-       let t4: v128 = i64x2.extmul_low_i32x4_s(f_4, f5)
-       let t5: v128 = i64x2.extmul_low_i32x4_s(f_2, f6)
-       let t6: v128 = i64x2.extmul_low_i32x4_s(f_4, f7)
-       let t7: v128 = i64x2.extmul_low_i32x4_s(f_2, f8)
-       let t8: v128 = i64x2.extmul_low_i32x4_s(f_4, f9_19)
-
-       h2 = i64x2.add(h2, t0)
-       h3 = i64x2.add(h3, t1)
-       h4 = i64x2.add(h4, t2)
-       h5 = i64x2.add(h5, t3)
-       h6 = i64x2.add(h6, t4)
-       h7 = i64x2.add(h7, t5)
-       h8 = i64x2.add(h8, t6)
-       h9 = i64x2.add(h9, t7)
-       h0 = i64x2.add(h0, t8)
-
-       // f[2]
-       f_2 = i32x4.shl(f2, 1)
-
-       t0 = i64x2.extmul_low_i32x4_s(f2, f2)
-       t1 = i64x2.extmul_low_i32x4_s(f_2, f3)
-       t2 = i64x2.extmul_low_i32x4_s(f_2, f4)
-       t3 = i64x2.extmul_low_i32x4_s(f_2, f5)
-       t4 = i64x2.extmul_low_i32x4_s(f_2, f6)
-       t5 = i64x2.extmul_low_i32x4_s(f_2, f7)
-       t6 = i64x2.extmul_low_i32x4_s(f_2, f8_19)
-       t7 = i64x2.extmul_low_i32x4_s(f_2, f9_19)
-
-       h4 = i64x2.add(h4, t0)
-       h5 = i64x2.add(h5, t1)
-       h6 = i64x2.add(h6, t2)
-       h7 = i64x2.add(h7, t3)
-       h8 = i64x2.add(h8, t4)
-       h9 = i64x2.add(h9, t5)
-       h0 = i64x2.add(h0, t6)
-       h1 = i64x2.add(h1, t7)
-
-       // f[3]
-       f_2 = i32x4.shl(f3, 1)
-       f_4 = i32x4.shl(f3, 2)
-
-       t0 = i64x2.extmul_low_i32x4_s(f_2, f3)
-       t1 = i64x2.extmul_low_i32x4_s(f_2, f4)
-       t2 = i64x2.extmul_low_i32x4_s(f_4, f5)
-       t3 = i64x2.extmul_low_i32x4_s(f_2, f6)
-       t4 = i64x2.extmul_low_i32x4_s(f_4, f7_19) /* 1.959375*2^30 */
-       t5 = i64x2.extmul_low_i32x4_s(f_2, f8_19)
-       t6 = i64x2.extmul_low_i32x4_s(f_4, f9_19)
-
-       h6 = i64x2.add(h6, t0)
-       h7 = i64x2.add(h7, t1)
-       h8 = i64x2.add(h8, t2)
-       h9 = i64x2.add(h9, t3)
-       h0 = i64x2.add(h0, t4) /* 1.959375*2^30 */
-       h1 = i64x2.add(h1, t5)
-       h2 = i64x2.add(h2, t6)
-
-       // f[4]
-       f_2 = i32x4.shl(f4, 1)
-
-       t0 = i64x2.extmul_low_i32x4_s(f4, f4)
-       t1 = i64x2.extmul_low_i32x4_s(f_2, f5)
-       t2 = i64x2.extmul_low_i32x4_s(f_2, f6_19)
-       t3 = i64x2.extmul_low_i32x4_s(f_2, f7_19)
-       t4 = i64x2.extmul_low_i32x4_s(f_2, f8_19)
-       t5 = i64x2.extmul_low_i32x4_s(f_2, f9_19)
-
-       h8 = i64x2.add(h8, t0)
-       h9 = i64x2.add(h9, t1)
-       h0 = i64x2.add(h0, t2)
-       h1 = i64x2.add(h1, t3)
-       h2 = i64x2.add(h2, t4)
-       h3 = i64x2.add(h3, t5)
-
-       // f[5]
-       f_2 = i32x4.shl(f5, 1)
-       f_4 = i32x4.shl(f5, 2)
-
-       t0 = i64x2.extmul_low_i32x4_s(f_2, f5_19)
-       t1 = i64x2.extmul_low_i32x4_s(f_2, f6_19)
-       t2 = i64x2.extmul_low_i32x4_s(f_4, f7_19)
-       t3 = i64x2.extmul_low_i32x4_s(f_2, f8_19)
-       t4 = i64x2.extmul_low_i32x4_s(f_4, f9_19) /* 1.959375*2^30 */
-
-       h0 = i64x2.add(h0, t0)
-       h1 = i64x2.add(h1, t1)
-       h2 = i64x2.add(h2, t2)
-       h3 = i64x2.add(h3, t3)
-       h4 = i64x2.add(h4, t4) /* 1.959375*2^30 */
-
-       // f[6]
-       f_2 = i32x4.shl(f6, 1)
-
-       t0 = i64x2.extmul_low_i32x4_s(f6, f6_19)
-       t1 = i64x2.extmul_low_i32x4_s(f_2, f7_19)
-       t2 = i64x2.extmul_low_i32x4_s(f_2, f8_19)
-       t3 = i64x2.extmul_low_i32x4_s(f_2, f9_19)
-
-       h2 = i64x2.add(h2, t0)
-       h3 = i64x2.add(h3, t1)
-       h4 = i64x2.add(h4, t2)
-       h5 = i64x2.add(h5, t3)
-
-       // f[7]
-       f_2 = i32x4.shl(f7, 1)
-       f_4 = i32x4.shl(f7, 2)
-
-       t0 = i64x2.extmul_low_i32x4_s(f_2, f7_19)
-       t1 = i64x2.extmul_low_i32x4_s(f_2, f8_19)
-       t2 = i64x2.extmul_low_i32x4_s(f_4, f9_19)
-
-       h4 = i64x2.add(h4, t0)
-       h5 = i64x2.add(h5, t1)
-       h6 = i64x2.add(h6, t2)
-
-       // f[8]
-       f_2 = i32x4.shl(f8, 1)
-
-       t0 = i64x2.extmul_low_i32x4_s(f8, f8_19)
-       t1 = i64x2.extmul_low_i32x4_s(f_2, f9_19)
-
-       h6 = i64x2.add(h6, t0)
-       h7 = i64x2.add(h7, t1)
-
-       // f[9]
-       f_2 = i32x4.shl(f9, 1)
-
-       t0 = i64x2.extmul_low_i32x4_s(f_2, f9_19)
-
-       h8 = i64x2.add(h8, t0)
-
-       const v24: v128 = i64x2.splat(1 << 24)
-       const v25: v128 = i64x2.splat(1 << 25)
-       let carry0: v128
-       let carry1: v128
-       let carry2: v128
-       let carry3: v128
-       let carry4: v128
-       let carry5: v128
-       let carry6: v128
-       let carry7: v128
-       let carry8: v128
-       let carry9: v128
-
-       carry0 = i64x2.shr_s(((i64x2.add(h0, v25))), 26)
-       h1 = i64x2.add(h1, carry0)
-       h0 = i64x2.sub(h0, i64x2.shl(carry0, 26))
-       carry4 = i64x2.shr_s(((i64x2.add(h4, v25))), 26)
-       h5 = i64x2.add(h5, carry4)
-       h4 = i64x2.sub(h4, i64x2.shl(carry4, 26))
-
-       carry1 = i64x2.shr_s(((i64x2.add(h1, v24))), 25)
-       h2 = i64x2.add(h2, carry1)
-       h1 = i64x2.sub(h1, i64x2.shl(carry1, 25))
-       carry5 = i64x2.shr_s(((i64x2.add(h5, v24))), 25)
-       h6 = i64x2.add(h6, carry5)
-       h5 = i64x2.sub(h5, i64x2.shl(carry5, 25))
-
-       carry2 = i64x2.shr_s(((i64x2.add(h2, v25))), 26)
-       h3 = i64x2.add(h3, carry2)
-       h2 = i64x2.sub(h2, i64x2.shl(carry2, 26))
-       carry6 = i64x2.shr_s(((i64x2.add(h6, v25))), 26)
-       h7 = i64x2.add(h7, carry6)
-       h6 = i64x2.sub(h6, i64x2.shl(carry6, 26))
-
-       carry3 = i64x2.shr_s(((i64x2.add(h3, v24))), 25)
-       h4 = i64x2.add(h4, carry3)
-       h3 = i64x2.sub(h3, i64x2.shl(carry3, 25))
-       carry7 = i64x2.shr_s(((i64x2.add(h7, v24))), 25)
-       h8 = i64x2.add(h8, carry7)
-       h7 = i64x2.sub(h7, i64x2.shl(carry7, 25))
-
-       carry4 = i64x2.shr_s(((i64x2.add(h4, v25))), 26)
-       h5 = i64x2.add(h5, carry4)
-       h4 = i64x2.sub(h4, i64x2.shl(carry4, 26))
-       carry8 = i64x2.shr_s(((i64x2.add(h8, v25))), 26)
-       h9 = i64x2.add(h9, carry8)
-       h8 = i64x2.sub(h8, i64x2.shl(carry8, 26))
-
-       carry9 = i64x2.shr_s(((i64x2.add(h9, v24))), 25)
-       h0 = i64x2.add(h0, i64x2.mul(carry9, i64x2.splat(19)))
-       h9 = i64x2.sub(h9, i64x2.shl(carry9, 25))
-
-       carry0 = i64x2.shr_s(((i64x2.add(h0, v25))), 26)
-       h1 = i64x2.add(h1, carry0)
-       h0 = i64x2.sub(h0, i64x2.shl(carry0, 26))
-
-       const hx_ptr: usize = changetype<usize>(hx)
-       store<i32>(hx_ptr, i64x2.extract_lane(h0, 0), 0)
-       store<i32>(hx_ptr, i64x2.extract_lane(h1, 0), 4)
-       store<i32>(hx_ptr, i64x2.extract_lane(h2, 0), 8)
-       store<i32>(hx_ptr, i64x2.extract_lane(h3, 0), 12)
-       store<i32>(hx_ptr, i64x2.extract_lane(h4, 0), 16)
-       store<i32>(hx_ptr, i64x2.extract_lane(h5, 0), 20)
-       store<i32>(hx_ptr, i64x2.extract_lane(h6, 0), 24)
-       store<i32>(hx_ptr, i64x2.extract_lane(h7, 0), 28)
-       store<i32>(hx_ptr, i64x2.extract_lane(h8, 0), 32)
-       store<i32>(hx_ptr, i64x2.extract_lane(h9, 0), 36)
-
-       const hy_ptr: usize = changetype<usize>(hy)
-       store<i32>(hy_ptr, i64x2.extract_lane(h0, 1), 0)
-       store<i32>(hy_ptr, i64x2.extract_lane(h1, 1), 4)
-       store<i32>(hy_ptr, i64x2.extract_lane(h2, 1), 8)
-       store<i32>(hy_ptr, i64x2.extract_lane(h3, 1), 12)
-       store<i32>(hy_ptr, i64x2.extract_lane(h4, 1), 16)
-       store<i32>(hy_ptr, i64x2.extract_lane(h5, 1), 20)
-       store<i32>(hy_ptr, i64x2.extract_lane(h6, 1), 24)
-       store<i32>(hy_ptr, i64x2.extract_lane(h7, 1), 28)
-       store<i32>(hy_ptr, i64x2.extract_lane(h8, 1), 32)
-       store<i32>(hy_ptr, i64x2.extract_lane(h9, 1), 36)
-}
-
 /**
  * Square a FieldElement, double it, and store the result.
  *
index c7c4e6bf1f67cf70040d4248aabbe8a99e6f108e..4d036276c66535176fe29a3407b8b74d145ecb51 100644 (file)
@@ -1,7 +1,7 @@
 //! SPDX-FileCopyrightText: 2026 Chris Duncan <chris@codecow.com>
 //! SPDX-License-Identifier: GPL-3.0-or-later
 
-import { FieldElement, fe, fe_0, fe_1, fe_add, fe_copy, fe_dbl, fe_invert, fe_isnegative, fe_mul, fe_neg, fe_sq, fe_sq2, fe_sq_vec, fe_sub, fe_tobytes } from './fe'
+import { FieldElement, fe, fe_0, fe_1, fe_add, fe_copy, fe_dbl, fe_invert, fe_isnegative, fe_mul, fe_neg, fe_sq, fe_sq2, fe_sub, fe_tobytes } from './fe'
 
 /**
  * 2d = −121665 / 60833
@@ -59,7 +59,8 @@ const ge_p2_dbl_t: FieldElement = fe()
 @inline
 export function ge_p2_dbl (r: ge_p1p1, p: ge_p2): void {
        const t = ge_p2_dbl_t
-       fe_sq_vec(r.X, r.Z, p.X, p.Y)
+       fe_sq(r.X, p.X)
+       fe_sq(r.Z, p.Y)
        fe_sq2(r.T, p.Z)
        fe_add(r.Y, p.X, p.Y)
        fe_sq(t, r.Y)
@@ -128,7 +129,8 @@ const ge_p3_dbl_t: FieldElement = fe()
  */
 export function ge_p3_dbl (r: ge_p1p1, p: ge_p3): void {
        const t = ge_p3_dbl_t
-       fe_sq_vec(r.X, r.Z, p.X, p.Y)
+       fe_sq(r.X, p.X)
+       fe_sq(r.Z, p.Y)
        fe_sq2(r.T, p.Z)
        fe_add(r.Y, p.X, p.Y)
        fe_sq(t, r.Y)