From: Chris Duncan Date: Thu, 1 Oct 2026 14:48:52 +0000 (-0700) Subject: Parallel vector squaring did not perform any better than sequential scalar squaring... X-Git-Url: https://git.codecow.com/?a=commitdiff_plain;h=39e3b8c1169902017f3af6e6d005777a1ca4e957;p=nano25519.git Parallel vector squaring did not perform any better than sequential scalar squaring, so deprecate it to simplify and reduce module size. --- diff --git a/src/assembly/ed25519/fe.ts b/src/assembly/ed25519/fe.ts index a239260..ff29d99 100644 --- a/src/assembly/ed25519/fe.ts +++ b/src/assembly/ed25519/fe.ts @@ -890,291 +890,6 @@ export function fe_sq (h: FieldElement, f: FieldElement): void { store(h_ptr, h9, 36) } -/** - * Square a FieldElement and store the result. - * - * Some iterations of the inner loop can be skipped since - * `f[x]*f[y] + f[y]*f[x] = 2*f[x]*f[y]` - * - * ``` - * // non-constant-time example - * for (let i = 0; i < 10; i++) { - * for (let j = i; j < 10; j++) { - * h[i + j] += f[i] * g[j] - * if (i < j) { - * h[i + j] *= 2 - * } - * } - * } - * ``` - * - * h = f * f - * Can overlap h with f. - * - * Preconditions: - * |f| bounded by 1.65x2²⁶,1.65x2²⁵,1.65x2²⁶,1.65x2²⁵,etc. - * - * Postconditions: - * |h| bounded by 1.01x2²⁵,1.01x2²⁴,1.01x2²⁵,1.01x2²⁴,etc. - * - * @param hx FieldElement power destination - * @param fx FieldElement base source - */ -export function fe_sq_vec (hx: FieldElement, hy: FieldElement, fx: FieldElement, fy: FieldElement): void { - const fx_ptr: usize = changetype(fx) - const fy_ptr: usize = changetype(fy) - const f0: v128 = i32x4(load(fx_ptr, 0), load(fy_ptr, 0), 0, 0) - const f1: v128 = i32x4(load(fx_ptr, 4), load(fy_ptr, 4), 0, 0) - const f2: v128 = i32x4(load(fx_ptr, 8), load(fy_ptr, 8), 0, 0) - const f3: v128 = i32x4(load(fx_ptr, 12), load(fy_ptr, 12), 0, 0) - const f4: v128 = i32x4(load(fx_ptr, 16), load(fy_ptr, 16), 0, 0) - const f5: v128 = i32x4(load(fx_ptr, 20), load(fy_ptr, 20), 0, 0) - const f6: v128 = i32x4(load(fx_ptr, 24), load(fy_ptr, 24), 0, 0) - const f7: v128 = i32x4(load(fx_ptr, 28), load(fy_ptr, 28), 0, 0) - const f8: v128 = i32x4(load(fx_ptr, 32), load(fy_ptr, 32), 0, 0) - const f9: v128 = i32x4(load(fx_ptr, 36), load(fy_ptr, 36), 0, 0) - - const v19: v128 = i32x4.splat(19) - const f5_19: v128 = i32x4.mul(f5, v19) /* 1.959375*2²⁹ */ - const f6_19: v128 = i32x4.mul(f6, v19) /* 1.959375*2³⁰ */ - const f7_19: v128 = i32x4.mul(f7, v19) /* 1.959375*2²⁹ */ - const f8_19: v128 = i32x4.mul(f8, v19) /* 1.959375*2³⁰ */ - const f9_19: v128 = i32x4.mul(f9, v19) /* 1.959375*2²⁹ */ - - // f[0] - let f_2: v128 = i32x4.shl(f0, 1) - - let h0: v128 = i64x2.extmul_low_i32x4_s(f0, f0) - let h1: v128 = i64x2.extmul_low_i32x4_s(f_2, f1) - let h2: v128 = i64x2.extmul_low_i32x4_s(f_2, f2) - let h3: v128 = i64x2.extmul_low_i32x4_s(f_2, f3) - let h4: v128 = i64x2.extmul_low_i32x4_s(f_2, f4) - let h5: v128 = i64x2.extmul_low_i32x4_s(f_2, f5) - let h6: v128 = i64x2.extmul_low_i32x4_s(f_2, f6) - let h7: v128 = i64x2.extmul_low_i32x4_s(f_2, f7) - let h8: v128 = i64x2.extmul_low_i32x4_s(f_2, f8) - let h9: v128 = i64x2.extmul_low_i32x4_s(f_2, f9) - - // f[1] - f_2 = i32x4.shl(f1, 1) - let f_4: v128 = i32x4.shl(f1, 2) - - let t0: v128 = i64x2.extmul_low_i32x4_s(f_2, f1) - let t1: v128 = i64x2.extmul_low_i32x4_s(f_2, f2) - let t2: v128 = i64x2.extmul_low_i32x4_s(f_4, f3) - let t3: v128 = i64x2.extmul_low_i32x4_s(f_2, f4) - let t4: v128 = i64x2.extmul_low_i32x4_s(f_4, f5) - let t5: v128 = i64x2.extmul_low_i32x4_s(f_2, f6) - let t6: v128 = i64x2.extmul_low_i32x4_s(f_4, f7) - let t7: v128 = i64x2.extmul_low_i32x4_s(f_2, f8) - let t8: v128 = i64x2.extmul_low_i32x4_s(f_4, f9_19) - - h2 = i64x2.add(h2, t0) - h3 = i64x2.add(h3, t1) - h4 = i64x2.add(h4, t2) - h5 = i64x2.add(h5, t3) - h6 = i64x2.add(h6, t4) - h7 = i64x2.add(h7, t5) - h8 = i64x2.add(h8, t6) - h9 = i64x2.add(h9, t7) - h0 = i64x2.add(h0, t8) - - // f[2] - f_2 = i32x4.shl(f2, 1) - - t0 = i64x2.extmul_low_i32x4_s(f2, f2) - t1 = i64x2.extmul_low_i32x4_s(f_2, f3) - t2 = i64x2.extmul_low_i32x4_s(f_2, f4) - t3 = i64x2.extmul_low_i32x4_s(f_2, f5) - t4 = i64x2.extmul_low_i32x4_s(f_2, f6) - t5 = i64x2.extmul_low_i32x4_s(f_2, f7) - t6 = i64x2.extmul_low_i32x4_s(f_2, f8_19) - t7 = i64x2.extmul_low_i32x4_s(f_2, f9_19) - - h4 = i64x2.add(h4, t0) - h5 = i64x2.add(h5, t1) - h6 = i64x2.add(h6, t2) - h7 = i64x2.add(h7, t3) - h8 = i64x2.add(h8, t4) - h9 = i64x2.add(h9, t5) - h0 = i64x2.add(h0, t6) - h1 = i64x2.add(h1, t7) - - // f[3] - f_2 = i32x4.shl(f3, 1) - f_4 = i32x4.shl(f3, 2) - - t0 = i64x2.extmul_low_i32x4_s(f_2, f3) - t1 = i64x2.extmul_low_i32x4_s(f_2, f4) - t2 = i64x2.extmul_low_i32x4_s(f_4, f5) - t3 = i64x2.extmul_low_i32x4_s(f_2, f6) - t4 = i64x2.extmul_low_i32x4_s(f_4, f7_19) /* 1.959375*2^30 */ - t5 = i64x2.extmul_low_i32x4_s(f_2, f8_19) - t6 = i64x2.extmul_low_i32x4_s(f_4, f9_19) - - h6 = i64x2.add(h6, t0) - h7 = i64x2.add(h7, t1) - h8 = i64x2.add(h8, t2) - h9 = i64x2.add(h9, t3) - h0 = i64x2.add(h0, t4) /* 1.959375*2^30 */ - h1 = i64x2.add(h1, t5) - h2 = i64x2.add(h2, t6) - - // f[4] - f_2 = i32x4.shl(f4, 1) - - t0 = i64x2.extmul_low_i32x4_s(f4, f4) - t1 = i64x2.extmul_low_i32x4_s(f_2, f5) - t2 = i64x2.extmul_low_i32x4_s(f_2, f6_19) - t3 = i64x2.extmul_low_i32x4_s(f_2, f7_19) - t4 = i64x2.extmul_low_i32x4_s(f_2, f8_19) - t5 = i64x2.extmul_low_i32x4_s(f_2, f9_19) - - h8 = i64x2.add(h8, t0) - h9 = i64x2.add(h9, t1) - h0 = i64x2.add(h0, t2) - h1 = i64x2.add(h1, t3) - h2 = i64x2.add(h2, t4) - h3 = i64x2.add(h3, t5) - - // f[5] - f_2 = i32x4.shl(f5, 1) - f_4 = i32x4.shl(f5, 2) - - t0 = i64x2.extmul_low_i32x4_s(f_2, f5_19) - t1 = i64x2.extmul_low_i32x4_s(f_2, f6_19) - t2 = i64x2.extmul_low_i32x4_s(f_4, f7_19) - t3 = i64x2.extmul_low_i32x4_s(f_2, f8_19) - t4 = i64x2.extmul_low_i32x4_s(f_4, f9_19) /* 1.959375*2^30 */ - - h0 = i64x2.add(h0, t0) - h1 = i64x2.add(h1, t1) - h2 = i64x2.add(h2, t2) - h3 = i64x2.add(h3, t3) - h4 = i64x2.add(h4, t4) /* 1.959375*2^30 */ - - // f[6] - f_2 = i32x4.shl(f6, 1) - - t0 = i64x2.extmul_low_i32x4_s(f6, f6_19) - t1 = i64x2.extmul_low_i32x4_s(f_2, f7_19) - t2 = i64x2.extmul_low_i32x4_s(f_2, f8_19) - t3 = i64x2.extmul_low_i32x4_s(f_2, f9_19) - - h2 = i64x2.add(h2, t0) - h3 = i64x2.add(h3, t1) - h4 = i64x2.add(h4, t2) - h5 = i64x2.add(h5, t3) - - // f[7] - f_2 = i32x4.shl(f7, 1) - f_4 = i32x4.shl(f7, 2) - - t0 = i64x2.extmul_low_i32x4_s(f_2, f7_19) - t1 = i64x2.extmul_low_i32x4_s(f_2, f8_19) - t2 = i64x2.extmul_low_i32x4_s(f_4, f9_19) - - h4 = i64x2.add(h4, t0) - h5 = i64x2.add(h5, t1) - h6 = i64x2.add(h6, t2) - - // f[8] - f_2 = i32x4.shl(f8, 1) - - t0 = i64x2.extmul_low_i32x4_s(f8, f8_19) - t1 = i64x2.extmul_low_i32x4_s(f_2, f9_19) - - h6 = i64x2.add(h6, t0) - h7 = i64x2.add(h7, t1) - - // f[9] - f_2 = i32x4.shl(f9, 1) - - t0 = i64x2.extmul_low_i32x4_s(f_2, f9_19) - - h8 = i64x2.add(h8, t0) - - const v24: v128 = i64x2.splat(1 << 24) - const v25: v128 = i64x2.splat(1 << 25) - let carry0: v128 - let carry1: v128 - let carry2: v128 - let carry3: v128 - let carry4: v128 - let carry5: v128 - let carry6: v128 - let carry7: v128 - let carry8: v128 - let carry9: v128 - - carry0 = i64x2.shr_s(((i64x2.add(h0, v25))), 26) - h1 = i64x2.add(h1, carry0) - h0 = i64x2.sub(h0, i64x2.shl(carry0, 26)) - carry4 = i64x2.shr_s(((i64x2.add(h4, v25))), 26) - h5 = i64x2.add(h5, carry4) - h4 = i64x2.sub(h4, i64x2.shl(carry4, 26)) - - carry1 = i64x2.shr_s(((i64x2.add(h1, v24))), 25) - h2 = i64x2.add(h2, carry1) - h1 = i64x2.sub(h1, i64x2.shl(carry1, 25)) - carry5 = i64x2.shr_s(((i64x2.add(h5, v24))), 25) - h6 = i64x2.add(h6, carry5) - h5 = i64x2.sub(h5, i64x2.shl(carry5, 25)) - - carry2 = i64x2.shr_s(((i64x2.add(h2, v25))), 26) - h3 = i64x2.add(h3, carry2) - h2 = i64x2.sub(h2, i64x2.shl(carry2, 26)) - carry6 = i64x2.shr_s(((i64x2.add(h6, v25))), 26) - h7 = i64x2.add(h7, carry6) - h6 = i64x2.sub(h6, i64x2.shl(carry6, 26)) - - carry3 = i64x2.shr_s(((i64x2.add(h3, v24))), 25) - h4 = i64x2.add(h4, carry3) - h3 = i64x2.sub(h3, i64x2.shl(carry3, 25)) - carry7 = i64x2.shr_s(((i64x2.add(h7, v24))), 25) - h8 = i64x2.add(h8, carry7) - h7 = i64x2.sub(h7, i64x2.shl(carry7, 25)) - - carry4 = i64x2.shr_s(((i64x2.add(h4, v25))), 26) - h5 = i64x2.add(h5, carry4) - h4 = i64x2.sub(h4, i64x2.shl(carry4, 26)) - carry8 = i64x2.shr_s(((i64x2.add(h8, v25))), 26) - h9 = i64x2.add(h9, carry8) - h8 = i64x2.sub(h8, i64x2.shl(carry8, 26)) - - carry9 = i64x2.shr_s(((i64x2.add(h9, v24))), 25) - h0 = i64x2.add(h0, i64x2.mul(carry9, i64x2.splat(19))) - h9 = i64x2.sub(h9, i64x2.shl(carry9, 25)) - - carry0 = i64x2.shr_s(((i64x2.add(h0, v25))), 26) - h1 = i64x2.add(h1, carry0) - h0 = i64x2.sub(h0, i64x2.shl(carry0, 26)) - - const hx_ptr: usize = changetype(hx) - store(hx_ptr, i64x2.extract_lane(h0, 0), 0) - store(hx_ptr, i64x2.extract_lane(h1, 0), 4) - store(hx_ptr, i64x2.extract_lane(h2, 0), 8) - store(hx_ptr, i64x2.extract_lane(h3, 0), 12) - store(hx_ptr, i64x2.extract_lane(h4, 0), 16) - store(hx_ptr, i64x2.extract_lane(h5, 0), 20) - store(hx_ptr, i64x2.extract_lane(h6, 0), 24) - store(hx_ptr, i64x2.extract_lane(h7, 0), 28) - store(hx_ptr, i64x2.extract_lane(h8, 0), 32) - store(hx_ptr, i64x2.extract_lane(h9, 0), 36) - - const hy_ptr: usize = changetype(hy) - store(hy_ptr, i64x2.extract_lane(h0, 1), 0) - store(hy_ptr, i64x2.extract_lane(h1, 1), 4) - store(hy_ptr, i64x2.extract_lane(h2, 1), 8) - store(hy_ptr, i64x2.extract_lane(h3, 1), 12) - store(hy_ptr, i64x2.extract_lane(h4, 1), 16) - store(hy_ptr, i64x2.extract_lane(h5, 1), 20) - store(hy_ptr, i64x2.extract_lane(h6, 1), 24) - store(hy_ptr, i64x2.extract_lane(h7, 1), 28) - store(hy_ptr, i64x2.extract_lane(h8, 1), 32) - store(hy_ptr, i64x2.extract_lane(h9, 1), 36) -} - /** * Square a FieldElement, double it, and store the result. * diff --git a/src/assembly/ed25519/p.ts b/src/assembly/ed25519/p.ts index c7c4e6b..4d03627 100644 --- a/src/assembly/ed25519/p.ts +++ b/src/assembly/ed25519/p.ts @@ -1,7 +1,7 @@ //! SPDX-FileCopyrightText: 2026 Chris Duncan //! SPDX-License-Identifier: GPL-3.0-or-later -import { FieldElement, fe, fe_0, fe_1, fe_add, fe_copy, fe_dbl, fe_invert, fe_isnegative, fe_mul, fe_neg, fe_sq, fe_sq2, fe_sq_vec, fe_sub, fe_tobytes } from './fe' +import { FieldElement, fe, fe_0, fe_1, fe_add, fe_copy, fe_dbl, fe_invert, fe_isnegative, fe_mul, fe_neg, fe_sq, fe_sq2, fe_sub, fe_tobytes } from './fe' /** * 2d = −121665 / 60833 @@ -59,7 +59,8 @@ const ge_p2_dbl_t: FieldElement = fe() @inline export function ge_p2_dbl (r: ge_p1p1, p: ge_p2): void { const t = ge_p2_dbl_t - fe_sq_vec(r.X, r.Z, p.X, p.Y) + fe_sq(r.X, p.X) + fe_sq(r.Z, p.Y) fe_sq2(r.T, p.Z) fe_add(r.Y, p.X, p.Y) fe_sq(t, r.Y) @@ -128,7 +129,8 @@ const ge_p3_dbl_t: FieldElement = fe() */ export function ge_p3_dbl (r: ge_p1p1, p: ge_p3): void { const t = ge_p3_dbl_t - fe_sq_vec(r.X, r.Z, p.X, p.Y) + fe_sq(r.X, p.X) + fe_sq(r.Z, p.Y) fe_sq2(r.T, p.Z) fe_add(r.Y, p.X, p.Y) fe_sq(t, r.Y)