store<i32>(h_ptr, h9, 36)
}
-/**
- * Square a FieldElement and store the result.
- *
- * Some iterations of the inner loop can be skipped since
- * `f[x]*f[y] + f[y]*f[x] = 2*f[x]*f[y]`
- *
- * ```
- * // non-constant-time example
- * for (let i = 0; i < 10; i++) {
- * for (let j = i; j < 10; j++) {
- * h[i + j] += f[i] * g[j]
- * if (i < j) {
- * h[i + j] *= 2
- * }
- * }
- * }
- * ```
- *
- * h = f * f
- * Can overlap h with f.
- *
- * Preconditions:
- * |f| bounded by 1.65x2²⁶,1.65x2²⁵,1.65x2²⁶,1.65x2²⁵,etc.
- *
- * Postconditions:
- * |h| bounded by 1.01x2²⁵,1.01x2²⁴,1.01x2²⁵,1.01x2²⁴,etc.
- *
- * @param hx FieldElement power destination
- * @param fx FieldElement base source
- */
-export function fe_sq_vec (hx: FieldElement, hy: FieldElement, fx: FieldElement, fy: FieldElement): void {
- const fx_ptr: usize = changetype<usize>(fx)
- const fy_ptr: usize = changetype<usize>(fy)
- const f0: v128 = i32x4(load<i32>(fx_ptr, 0), load<i32>(fy_ptr, 0), 0, 0)
- const f1: v128 = i32x4(load<i32>(fx_ptr, 4), load<i32>(fy_ptr, 4), 0, 0)
- const f2: v128 = i32x4(load<i32>(fx_ptr, 8), load<i32>(fy_ptr, 8), 0, 0)
- const f3: v128 = i32x4(load<i32>(fx_ptr, 12), load<i32>(fy_ptr, 12), 0, 0)
- const f4: v128 = i32x4(load<i32>(fx_ptr, 16), load<i32>(fy_ptr, 16), 0, 0)
- const f5: v128 = i32x4(load<i32>(fx_ptr, 20), load<i32>(fy_ptr, 20), 0, 0)
- const f6: v128 = i32x4(load<i32>(fx_ptr, 24), load<i32>(fy_ptr, 24), 0, 0)
- const f7: v128 = i32x4(load<i32>(fx_ptr, 28), load<i32>(fy_ptr, 28), 0, 0)
- const f8: v128 = i32x4(load<i32>(fx_ptr, 32), load<i32>(fy_ptr, 32), 0, 0)
- const f9: v128 = i32x4(load<i32>(fx_ptr, 36), load<i32>(fy_ptr, 36), 0, 0)
-
- const v19: v128 = i32x4.splat(19)
- const f5_19: v128 = i32x4.mul(f5, v19) /* 1.959375*2²⁹ */
- const f6_19: v128 = i32x4.mul(f6, v19) /* 1.959375*2³⁰ */
- const f7_19: v128 = i32x4.mul(f7, v19) /* 1.959375*2²⁹ */
- const f8_19: v128 = i32x4.mul(f8, v19) /* 1.959375*2³⁰ */
- const f9_19: v128 = i32x4.mul(f9, v19) /* 1.959375*2²⁹ */
-
- // f[0]
- let f_2: v128 = i32x4.shl(f0, 1)
-
- let h0: v128 = i64x2.extmul_low_i32x4_s(f0, f0)
- let h1: v128 = i64x2.extmul_low_i32x4_s(f_2, f1)
- let h2: v128 = i64x2.extmul_low_i32x4_s(f_2, f2)
- let h3: v128 = i64x2.extmul_low_i32x4_s(f_2, f3)
- let h4: v128 = i64x2.extmul_low_i32x4_s(f_2, f4)
- let h5: v128 = i64x2.extmul_low_i32x4_s(f_2, f5)
- let h6: v128 = i64x2.extmul_low_i32x4_s(f_2, f6)
- let h7: v128 = i64x2.extmul_low_i32x4_s(f_2, f7)
- let h8: v128 = i64x2.extmul_low_i32x4_s(f_2, f8)
- let h9: v128 = i64x2.extmul_low_i32x4_s(f_2, f9)
-
- // f[1]
- f_2 = i32x4.shl(f1, 1)
- let f_4: v128 = i32x4.shl(f1, 2)
-
- let t0: v128 = i64x2.extmul_low_i32x4_s(f_2, f1)
- let t1: v128 = i64x2.extmul_low_i32x4_s(f_2, f2)
- let t2: v128 = i64x2.extmul_low_i32x4_s(f_4, f3)
- let t3: v128 = i64x2.extmul_low_i32x4_s(f_2, f4)
- let t4: v128 = i64x2.extmul_low_i32x4_s(f_4, f5)
- let t5: v128 = i64x2.extmul_low_i32x4_s(f_2, f6)
- let t6: v128 = i64x2.extmul_low_i32x4_s(f_4, f7)
- let t7: v128 = i64x2.extmul_low_i32x4_s(f_2, f8)
- let t8: v128 = i64x2.extmul_low_i32x4_s(f_4, f9_19)
-
- h2 = i64x2.add(h2, t0)
- h3 = i64x2.add(h3, t1)
- h4 = i64x2.add(h4, t2)
- h5 = i64x2.add(h5, t3)
- h6 = i64x2.add(h6, t4)
- h7 = i64x2.add(h7, t5)
- h8 = i64x2.add(h8, t6)
- h9 = i64x2.add(h9, t7)
- h0 = i64x2.add(h0, t8)
-
- // f[2]
- f_2 = i32x4.shl(f2, 1)
-
- t0 = i64x2.extmul_low_i32x4_s(f2, f2)
- t1 = i64x2.extmul_low_i32x4_s(f_2, f3)
- t2 = i64x2.extmul_low_i32x4_s(f_2, f4)
- t3 = i64x2.extmul_low_i32x4_s(f_2, f5)
- t4 = i64x2.extmul_low_i32x4_s(f_2, f6)
- t5 = i64x2.extmul_low_i32x4_s(f_2, f7)
- t6 = i64x2.extmul_low_i32x4_s(f_2, f8_19)
- t7 = i64x2.extmul_low_i32x4_s(f_2, f9_19)
-
- h4 = i64x2.add(h4, t0)
- h5 = i64x2.add(h5, t1)
- h6 = i64x2.add(h6, t2)
- h7 = i64x2.add(h7, t3)
- h8 = i64x2.add(h8, t4)
- h9 = i64x2.add(h9, t5)
- h0 = i64x2.add(h0, t6)
- h1 = i64x2.add(h1, t7)
-
- // f[3]
- f_2 = i32x4.shl(f3, 1)
- f_4 = i32x4.shl(f3, 2)
-
- t0 = i64x2.extmul_low_i32x4_s(f_2, f3)
- t1 = i64x2.extmul_low_i32x4_s(f_2, f4)
- t2 = i64x2.extmul_low_i32x4_s(f_4, f5)
- t3 = i64x2.extmul_low_i32x4_s(f_2, f6)
- t4 = i64x2.extmul_low_i32x4_s(f_4, f7_19) /* 1.959375*2^30 */
- t5 = i64x2.extmul_low_i32x4_s(f_2, f8_19)
- t6 = i64x2.extmul_low_i32x4_s(f_4, f9_19)
-
- h6 = i64x2.add(h6, t0)
- h7 = i64x2.add(h7, t1)
- h8 = i64x2.add(h8, t2)
- h9 = i64x2.add(h9, t3)
- h0 = i64x2.add(h0, t4) /* 1.959375*2^30 */
- h1 = i64x2.add(h1, t5)
- h2 = i64x2.add(h2, t6)
-
- // f[4]
- f_2 = i32x4.shl(f4, 1)
-
- t0 = i64x2.extmul_low_i32x4_s(f4, f4)
- t1 = i64x2.extmul_low_i32x4_s(f_2, f5)
- t2 = i64x2.extmul_low_i32x4_s(f_2, f6_19)
- t3 = i64x2.extmul_low_i32x4_s(f_2, f7_19)
- t4 = i64x2.extmul_low_i32x4_s(f_2, f8_19)
- t5 = i64x2.extmul_low_i32x4_s(f_2, f9_19)
-
- h8 = i64x2.add(h8, t0)
- h9 = i64x2.add(h9, t1)
- h0 = i64x2.add(h0, t2)
- h1 = i64x2.add(h1, t3)
- h2 = i64x2.add(h2, t4)
- h3 = i64x2.add(h3, t5)
-
- // f[5]
- f_2 = i32x4.shl(f5, 1)
- f_4 = i32x4.shl(f5, 2)
-
- t0 = i64x2.extmul_low_i32x4_s(f_2, f5_19)
- t1 = i64x2.extmul_low_i32x4_s(f_2, f6_19)
- t2 = i64x2.extmul_low_i32x4_s(f_4, f7_19)
- t3 = i64x2.extmul_low_i32x4_s(f_2, f8_19)
- t4 = i64x2.extmul_low_i32x4_s(f_4, f9_19) /* 1.959375*2^30 */
-
- h0 = i64x2.add(h0, t0)
- h1 = i64x2.add(h1, t1)
- h2 = i64x2.add(h2, t2)
- h3 = i64x2.add(h3, t3)
- h4 = i64x2.add(h4, t4) /* 1.959375*2^30 */
-
- // f[6]
- f_2 = i32x4.shl(f6, 1)
-
- t0 = i64x2.extmul_low_i32x4_s(f6, f6_19)
- t1 = i64x2.extmul_low_i32x4_s(f_2, f7_19)
- t2 = i64x2.extmul_low_i32x4_s(f_2, f8_19)
- t3 = i64x2.extmul_low_i32x4_s(f_2, f9_19)
-
- h2 = i64x2.add(h2, t0)
- h3 = i64x2.add(h3, t1)
- h4 = i64x2.add(h4, t2)
- h5 = i64x2.add(h5, t3)
-
- // f[7]
- f_2 = i32x4.shl(f7, 1)
- f_4 = i32x4.shl(f7, 2)
-
- t0 = i64x2.extmul_low_i32x4_s(f_2, f7_19)
- t1 = i64x2.extmul_low_i32x4_s(f_2, f8_19)
- t2 = i64x2.extmul_low_i32x4_s(f_4, f9_19)
-
- h4 = i64x2.add(h4, t0)
- h5 = i64x2.add(h5, t1)
- h6 = i64x2.add(h6, t2)
-
- // f[8]
- f_2 = i32x4.shl(f8, 1)
-
- t0 = i64x2.extmul_low_i32x4_s(f8, f8_19)
- t1 = i64x2.extmul_low_i32x4_s(f_2, f9_19)
-
- h6 = i64x2.add(h6, t0)
- h7 = i64x2.add(h7, t1)
-
- // f[9]
- f_2 = i32x4.shl(f9, 1)
-
- t0 = i64x2.extmul_low_i32x4_s(f_2, f9_19)
-
- h8 = i64x2.add(h8, t0)
-
- const v24: v128 = i64x2.splat(1 << 24)
- const v25: v128 = i64x2.splat(1 << 25)
- let carry0: v128
- let carry1: v128
- let carry2: v128
- let carry3: v128
- let carry4: v128
- let carry5: v128
- let carry6: v128
- let carry7: v128
- let carry8: v128
- let carry9: v128
-
- carry0 = i64x2.shr_s(((i64x2.add(h0, v25))), 26)
- h1 = i64x2.add(h1, carry0)
- h0 = i64x2.sub(h0, i64x2.shl(carry0, 26))
- carry4 = i64x2.shr_s(((i64x2.add(h4, v25))), 26)
- h5 = i64x2.add(h5, carry4)
- h4 = i64x2.sub(h4, i64x2.shl(carry4, 26))
-
- carry1 = i64x2.shr_s(((i64x2.add(h1, v24))), 25)
- h2 = i64x2.add(h2, carry1)
- h1 = i64x2.sub(h1, i64x2.shl(carry1, 25))
- carry5 = i64x2.shr_s(((i64x2.add(h5, v24))), 25)
- h6 = i64x2.add(h6, carry5)
- h5 = i64x2.sub(h5, i64x2.shl(carry5, 25))
-
- carry2 = i64x2.shr_s(((i64x2.add(h2, v25))), 26)
- h3 = i64x2.add(h3, carry2)
- h2 = i64x2.sub(h2, i64x2.shl(carry2, 26))
- carry6 = i64x2.shr_s(((i64x2.add(h6, v25))), 26)
- h7 = i64x2.add(h7, carry6)
- h6 = i64x2.sub(h6, i64x2.shl(carry6, 26))
-
- carry3 = i64x2.shr_s(((i64x2.add(h3, v24))), 25)
- h4 = i64x2.add(h4, carry3)
- h3 = i64x2.sub(h3, i64x2.shl(carry3, 25))
- carry7 = i64x2.shr_s(((i64x2.add(h7, v24))), 25)
- h8 = i64x2.add(h8, carry7)
- h7 = i64x2.sub(h7, i64x2.shl(carry7, 25))
-
- carry4 = i64x2.shr_s(((i64x2.add(h4, v25))), 26)
- h5 = i64x2.add(h5, carry4)
- h4 = i64x2.sub(h4, i64x2.shl(carry4, 26))
- carry8 = i64x2.shr_s(((i64x2.add(h8, v25))), 26)
- h9 = i64x2.add(h9, carry8)
- h8 = i64x2.sub(h8, i64x2.shl(carry8, 26))
-
- carry9 = i64x2.shr_s(((i64x2.add(h9, v24))), 25)
- h0 = i64x2.add(h0, i64x2.mul(carry9, i64x2.splat(19)))
- h9 = i64x2.sub(h9, i64x2.shl(carry9, 25))
-
- carry0 = i64x2.shr_s(((i64x2.add(h0, v25))), 26)
- h1 = i64x2.add(h1, carry0)
- h0 = i64x2.sub(h0, i64x2.shl(carry0, 26))
-
- const hx_ptr: usize = changetype<usize>(hx)
- store<i32>(hx_ptr, i64x2.extract_lane(h0, 0), 0)
- store<i32>(hx_ptr, i64x2.extract_lane(h1, 0), 4)
- store<i32>(hx_ptr, i64x2.extract_lane(h2, 0), 8)
- store<i32>(hx_ptr, i64x2.extract_lane(h3, 0), 12)
- store<i32>(hx_ptr, i64x2.extract_lane(h4, 0), 16)
- store<i32>(hx_ptr, i64x2.extract_lane(h5, 0), 20)
- store<i32>(hx_ptr, i64x2.extract_lane(h6, 0), 24)
- store<i32>(hx_ptr, i64x2.extract_lane(h7, 0), 28)
- store<i32>(hx_ptr, i64x2.extract_lane(h8, 0), 32)
- store<i32>(hx_ptr, i64x2.extract_lane(h9, 0), 36)
-
- const hy_ptr: usize = changetype<usize>(hy)
- store<i32>(hy_ptr, i64x2.extract_lane(h0, 1), 0)
- store<i32>(hy_ptr, i64x2.extract_lane(h1, 1), 4)
- store<i32>(hy_ptr, i64x2.extract_lane(h2, 1), 8)
- store<i32>(hy_ptr, i64x2.extract_lane(h3, 1), 12)
- store<i32>(hy_ptr, i64x2.extract_lane(h4, 1), 16)
- store<i32>(hy_ptr, i64x2.extract_lane(h5, 1), 20)
- store<i32>(hy_ptr, i64x2.extract_lane(h6, 1), 24)
- store<i32>(hy_ptr, i64x2.extract_lane(h7, 1), 28)
- store<i32>(hy_ptr, i64x2.extract_lane(h8, 1), 32)
- store<i32>(hy_ptr, i64x2.extract_lane(h9, 1), 36)
-}
-
/**
* Square a FieldElement, double it, and store the result.
*