From b65ccda4f58dd38b4953fdf07e59c7078be279f0 Mon Sep 17 00:00:00 2001 From: Chris Duncan Date: Wed, 30 Sep 2026 14:12:42 -0700 Subject: [PATCH] Start converting to generated arithmetic. --- scripts/fe.gen.mjs | 997 ++++++++++-- scripts/p.gen.mjs | 291 ++-- src/assembly/ed25519/fe.ts | 318 ++-- src/assembly/ed25519/p.ts | 3077 +++++++++++++++++++++++++++++++++--- 4 files changed, 3986 insertions(+), 697 deletions(-) diff --git a/scripts/fe.gen.mjs b/scripts/fe.gen.mjs index 5bd688b..6a36597 100644 --- a/scripts/fe.gen.mjs +++ b/scripts/fe.gen.mjs @@ -9,6 +9,117 @@ * AssemblyScript compiler and type checker. */ +/** + * Initialize common algorithm variables once per group element operation. + * + * @param {string[]} fe List of FieldElement variable names to initialize + * @returns {string} Variable initializing statements + */ +export function fe_init_v128 (...fe) { + return ` + let ${fe.map(fe => ` + ${fe.replaceAll('.', '_')}_0123: v128, + ${fe.replaceAll('.', '_')}_4567: v128, + ${fe.replaceAll('.', '_')}_89xx: v128 +`)} +` +} + +/** + * Initialize common algorithm variables once per group element operation. + * + * @param {string[]} fe List of FieldElement variable names to initialize + * @returns {string} Variable initializing statements + */ +export function fe_init (...fe) { + return ` + ${fe_init_v128(...fe)} + + // mask to select even lane from first vector and odd lane from second + const m: v128 = i32x4(-1, 0, -1, 0) + + // mask to double odd-on-odd indexes + const odds: v128 = i32x4(0, -1, 0, -1) + + // factor of 19 for wrap from h9 to h0 + const v19: v128 = i32x4.splat(19) + + let c0: i64 + let c1: i64 + let c2: i64 + let c3: i64 + let c4: i64 + let c5: i64 + let c6: i64 + let c7: i64 + let c8: i64 + let c9: i64 + + let f0: i64 + let f1: i64 + let f2: i64 + let f3: i64 + let f4: i64 + let f5: i64 + let f6: i64 + let f7: i64 + let f8: i64 + let f9: i64 + + let f0_2: i64 + let f1_2: i64 + let f2_2: i64 + let f3_2: i64 + let f4_2: i64 + let f5_2: i64 + let f6_2: i64 + let f7_2: i64 + let f8_2: i64 + let f9_2: i64 + + let f5_19: i64 + let f6_19: i64 + let f7_19: i64 + let f8_19: i64 + let f9_19: i64 + + let f5_38: i64 + let f7_38: i64 + let f9_38: i64 + + let fi: i32 + let fv: v128 + + let h0: i64 + let h1: i64 + let h2: i64 + let h3: i64 + let h4: i64 + let h5: i64 + let h6: i64 + let h7: i64 + let h8: i64 + let h9: i64 + + let h01: v128 + let h12: v128 + let h23: v128 + let h34: v128 + let h45: v128 + let h56: v128 + let h67: v128 + let h78: v128 + let h89: v128 + let h90: v128 + + let t0: v128 + let t1: v128 + let t2: v128 + let t3: v128 + let t4: v128 +` +} + /** * Add two FieldElements and store the sum. * @@ -18,44 +129,702 @@ * @returns {string} Three vector addition commands */ export function fe_add (h, f, g) { + const h_name = h.replaceAll('.', '_') + const f_name = f.replaceAll('.', '_') + const g_name = g.replaceAll('.', '_') return ` // ${h} = ${f} + ${g} - ${h}0 = v128.add(${f}0, ${g}0) - ${h}1 = v128.add(${f}1, ${g}1) - ${h}2 = v128.add(${f}2, ${g}2) + ${h_name}_0123 = v128.add(${f_name}_0123, ${g_name}_0123) + ${h_name}_4567 = v128.add(${f_name}_4567, ${g_name}_4567) + ${h_name}_89xx = v128.add(${f_name}_89xx, ${g_name}_89xx) ` } /** - * Subtract a FieldElement another and store the result. + * Add a FieldElement to itself and store the sum. * - * @param {string} h Variable name of FieldElement difference destination - * @param {string} f Variable name of FieldElement minuend source - * @param {string} g Variable name of FieldElement subtrahend source - * @returns {string} Three vector subtraction commands + * @param {string} h Variable name of FieldElement sum destination + * @param {string} f Variable name of FieldElement summand source + * @returns {string} Three vector doubling commands */ -export function fe_sub (h, f, g) { +export function fe_dbl (h, f) { + const h_name = h.replaceAll('.', '_') + const f_name = f.replaceAll('.', '_') return ` - // ${h} = ${f} - ${g} - ${h}0 = v128.sub(${f}0, ${g}0) - ${h}1 = v128.sub(${f}1, ${g}1) - ${h}2 = v128.sub(${f}2, ${g}2) + // ${h} = 2*${f} + ${h_name}_0123 = v128.shl(${f_name}_0123, 1) + ${h_name}_4567 = v128.shl(${f_name}_4567, 1) + ${h_name}_89xx = v128.shl(${f_name}_89xx, 1) ` } /** - * Add a FieldElement to itself and store the sum. + * Loads a FieldElement from memory into named vectors. * - * @param {string} h Variable name of FieldElement sum destination - * @param {string} f Variable name of FieldElement summand source - * @returns {string} Three vector doubling commands + * @param {string} f Variable name of FieldElement source + * @returns {string} Three vectors */ -export function fe_dbl (h, f) { +export function fe_load (f) { + const f_name = f.replaceAll('.', '_') + const f_ptr = `changetype(${f})` return ` - // ${h} = 2*${f} - ${h}0 = v128.shl(${f}0, 1) - ${h}1 = v128.shl(${f}1, 1) - ${h}2 = v128.shl(${f}2, 1) + // [0..3, 4..7, 8..11] = ${f} + ${f_name}_0123 = v128.load(${f_ptr}, 0) + ${f_name}_4567 = v128.load(${f_ptr}, 16) + ${f_name}_89xx = v128.load(${f_ptr}, 32) +` +} + +/** + * Multiply two FieldElements and store the product. + * + * @param {string} h Variable name of FieldElement product destination + * @param {string} f Variable name of FieldElement multiplicand source + * @param {string} g Variable name of FieldElement multiplicand source + * @returns {string} Field multiplication algorithm + */ +export function fe_mul (h, f, g) { + const h_name = h.replaceAll('.', '_') + const f_name = f.replaceAll('.', '_') + const g_name = g.replaceAll('.', '_') + return ` + // ${h} = ${f} * ${g} + const ${g_name}_0123_19: v128 = i32x4.mul(${g_name}_0123, v19) + const ${g_name}_4567_19: v128 = i32x4.mul(${g_name}_4567, v19) + const ${g_name}_89xx_19: v128 = i32x4.mul(${g_name}_89xx, v19) + + // f[0] + fi = v128.extract_lane(${f_name}_0123, 0) + fv = i32x4.splat(fi) + + h01 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123) + h23 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123) + h45 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567) + h67 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567) + h89 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx) + + // f[1] + fi = v128.extract_lane(${f_name}_0123, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + h12 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123) + h34 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123) + h56 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567) + h78 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567) + h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(${g_name}_89xx, ${g_name}_89xx_19, m)) + + // f[2] + fi = v128.extract_lane(${f_name}_0123, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567) + t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19) + + h23 = i64x2.add(h23, t0) + h45 = i64x2.add(h45, t1) + h67 = i64x2.add(h67, t2) + h89 = i64x2.add(h89, t3) + h01 = i64x2.add(h01, t4) + + // f[3] + fi = v128.extract_lane(${f_name}_0123, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(${g_name}_4567, ${g_name}_4567_19, m)) + t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19) + + h34 = i64x2.add(h34, t0) + h56 = i64x2.add(h56, t1) + h78 = i64x2.add(h78, t2) + h90 = i64x2.add(h90, t3) + h12 = i64x2.add(h12, t4) + + // f[4] + fi = v128.extract_lane(${f_name}_4567, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19) + + h45 = i64x2.add(h45, t0) + h67 = i64x2.add(h67, t1) + h89 = i64x2.add(h89, t2) + h01 = i64x2.add(h01, t3) + h23 = i64x2.add(h23, t4) + + // f[5] + fi = v128.extract_lane(${f_name}_4567, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(${g_name}_4567, ${g_name}_4567_19, m)) + t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19) + + h56 = i64x2.add(h56, t0) + h78 = i64x2.add(h78, t1) + h90 = i64x2.add(h90, t2) + h12 = i64x2.add(h12, t3) + h34 = i64x2.add(h34, t4) + + // f[6] + fi = v128.extract_lane(${f_name}_4567, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19) + + h67 = i64x2.add(h67, t0) + h89 = i64x2.add(h89, t1) + h01 = i64x2.add(h01, t2) + h23 = i64x2.add(h23, t3) + h45 = i64x2.add(h45, t4) + + // f[7] + fi = v128.extract_lane(${f_name}_4567, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(${g_name}_0123, ${g_name}_0123_19, m)) + t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19) + + h78 = i64x2.add(h78, t0) + h90 = i64x2.add(h90, t1) + h12 = i64x2.add(h12, t2) + h34 = i64x2.add(h34, t3) + h56 = i64x2.add(h56, t4) + + // f[8] + fi = v128.extract_lane(${f_name}_89xx, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19) + + h89 = i64x2.add(h89, t0) + h01 = i64x2.add(h01, t1) + h23 = i64x2.add(h23, t2) + h45 = i64x2.add(h45, t3) + h67 = i64x2.add(h67, t4) + + // f[9] + fi = v128.extract_lane(${f_name}_89xx, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(${g_name}_0123, ${g_name}_0123_19, m)) + t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19) + + h90 = i64x2.add(h90, t0) + h12 = i64x2.add(h12, t1) + h34 = i64x2.add(h34, t2) + h56 = i64x2.add(h56, t3) + h78 = i64x2.add(h78, t4) + + // extract scalars from vectors + h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0) + h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0) + h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0) + h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0) + h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0) + h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0) + h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0) + h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0) + h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0) + h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0) + + // carry + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + (1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + (1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + (1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + (1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + (1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + (1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + (1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + (1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + ${h_name}_0123 = i32x4(h0, h1, h2, h3) + ${h_name}_4567 = i32x4(h4, h5, h6, h7) + ${h_name}_89xx = i32x4(h8, h9, 0, 0) +` +} + +/** + * Negate the values of a FieldElement and store the result. + * + * @param {string} h Variable name of FieldElement result destination + * @param {string} f Variable name of FieldElement operand source + * @returns {string} Three vector negation commands + */ +export function fe_neg (h, f) { + const h_name = h.replaceAll('.', '_') + const f_name = f.replaceAll('.', '_') + return ` + // ${h} = -${f} + ${h_name}_0123 = v128.neg(${f_name}_0123) + ${h_name}_4567 = v128.neg(${f_name}_4567) + ${h_name}_89xx = v128.neg(${f_name}_89xx) +` +} + +/** + * Square a FieldElement and store the result. + * + * Some iterations of the inner loop can be skipped since + * \`f[x]*f[y] + f[y]*f[x] = 2*f[x]*f[y]\` + * + * \`\`\` + * // non-constant-time example + * for (let i = 0; i < 10; i++) { + * for (let j = i; j < 10; j++) { + * h[i + j] += f[i] * g[j] + * if (i < j) { + * h[i + j] *= 2 + * } + * } + * } + * \`\`\` + * + * h = f * f + * Can overlap h with f. + * + * Preconditions: + * |f| bounded by 1.65*2²⁶,1.65*2²⁵,1.65*2²⁶,1.65*2²⁵,etc. + * + * Postconditions: + * |h| bounded by 1.01*2²⁵,1.01*2²⁴4,1.01*2²⁵,1.01*2²⁴,etc. + * + * @param {string} h Variable name of FieldElement power destination + * @param {string} f Variable name of FieldElement base source + * @returns {string} Field squaring algorithm + */ +export function fe_sq (h, f) { + const h_name = h.replaceAll('.', '_') + const f_name = f.replaceAll('.', '_') + return ` + // ${h} = ${f}² + f0 = i64(v128.extract_lane(${f_name}_0123, 0)) + f1 = i64(v128.extract_lane(${f_name}_0123, 1)) + f2 = i64(v128.extract_lane(${f_name}_0123, 2)) + f3 = i64(v128.extract_lane(${f_name}_0123, 3)) + f4 = i64(v128.extract_lane(${f_name}_4567, 0)) + f5 = i64(v128.extract_lane(${f_name}_4567, 1)) + f6 = i64(v128.extract_lane(${f_name}_4567, 2)) + f7 = i64(v128.extract_lane(${f_name}_4567, 3)) + f8 = i64(v128.extract_lane(${f_name}_89xx, 0)) + f9 = i64(v128.extract_lane(${f_name}_89xx, 1)) + + f0_2 = f0 * 2 + f1_2 = f1 * 2 + f2_2 = f2 * 2 + f3_2 = f3 * 2 + f4_2 = f4 * 2 + f5_2 = f5 * 2 + f6_2 = f6 * 2 + f7_2 = f7 * 2 + + f5_19 = f5 * 19 /* 1.959375*2²⁹ */ + f6_19 = f6 * 19 /* 1.959375*2³⁰ */ + f7_19 = f7 * 19 /* 1.959375*2²⁹ */ + f8_19 = f8 * 19 /* 1.959375*2³⁰ */ + f9_19 = f9 * 19 /* 1.959375*2²⁹ */ + + f7_38 = f7 * 38 /* 1.959375*2³⁰ */ + f9_38 = f9 * 38 /* 1.959375*2³⁰ */ + + h0 = f0 * f0 + h1 = f0_2 * f1 + h2 = f0_2 * f2 + h3 = f0_2 * f3 + h4 = f0_2 * f4 + h5 = f0_2 * f5 + h6 = f0_2 * f6 + h7 = f0_2 * f7 + h8 = f0_2 * f8 + h9 = f0_2 * f9 + + h2 += f1_2 * f1 + h3 += f1_2 * f2 + h4 += f1_2 * f3_2 + h5 += f1_2 * f4 + h6 += f1_2 * f5_2 + h7 += f1_2 * f6 + h8 += f1_2 * f7_2 + h9 += f1_2 * f8 + h0 += f1_2 * f9_38 + + h4 += f2 * f2 + h5 += f2_2 * f3 + h6 += f2_2 * f4 + h7 += f2_2 * f5 + h8 += f2_2 * f6 + h9 += f2_2 * f7 + h0 += f2_2 * f8_19 + h1 += f2_2 * f9_19 + + h6 += f3_2 * f3 + h7 += f3_2 * f4 + h8 += f3_2 * f5_2 + h9 += f3_2 * f6 + h0 += f3_2 * f7_38 + h1 += f3_2 * f8_19 + h2 += f3_2 * f9_38 + + h8 += f4 * f4 + h9 += f4_2 * f5 + h0 += f4_2 * f6_19 + h1 += f4_2 * f7_19 + h2 += f4_2 * f8_19 + h3 += f4_2 * f9_19 + + h0 += f5_2 * f5_19 + h1 += f5_2 * f6_19 + h2 += f5_2 * f7_38 + h3 += f5_2 * f8_19 + h4 += f5_2 * f9_38 + + h2 += f6 * f6_19 + h3 += f6_2 * f7_19 + h4 += f6_2 * f8_19 + h5 += f6_2 * f9_19 + + h4 += f7_2 * f7_19 + h5 += f7_2 * f8_19 + h6 += f7_2 * f9_38 + + h6 += f8 * f8_19 + h7 += f8 * f9_38 + + h8 += f9 * f9_38 + + c0 = (h0 + i64(1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + i64(1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + i64(1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + i64(1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + i64(1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + i64(1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + i64(1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + i64(1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + i64(1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + i64(1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + i64(1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + i64(1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + ${h_name}_0123 = i32x4(h0, h1, h2, h3) + ${h_name}_4567 = i32x4(h4, h5, h6, h7) + ${h_name}_89xx = i32x4(h8, h9, 0, 0) +` +} + +/** + * Square a FieldElement, double it, and store the result. + * + * h = 2 * f * f + * Can overlap h with f. + * + * Preconditions: + * |f| bounded by 1.65*2^26,1.65*2^25,1.65*2^26,1.65*2^25,etc. + * + * Postconditions: + * |h| bounded by 1.01*2^25,1.01*2^24,1.01*2^25,1.01*2^24,etc. + * + * @param {string} h Variable name of FieldElement power destination + * @param {string} f Variable name of FieldElement base source + * @returns {string} Field squaring algorithm + */ +export function fe_sq2 (h, f) { + const h_name = h.replaceAll('.', '_') + const f_name = f.replaceAll('.', '_') + return ` + // ${h} = 2${f}² + f0 = i64(v128.extract_lane(${f_name}_0123, 0)) + f1 = i64(v128.extract_lane(${f_name}_0123, 1)) + f2 = i64(v128.extract_lane(${f_name}_0123, 2)) + f3 = i64(v128.extract_lane(${f_name}_0123, 3)) + f4 = i64(v128.extract_lane(${f_name}_4567, 0)) + f5 = i64(v128.extract_lane(${f_name}_4567, 1)) + f6 = i64(v128.extract_lane(${f_name}_4567, 2)) + f7 = i64(v128.extract_lane(${f_name}_4567, 3)) + f8 = i64(v128.extract_lane(${f_name}_89xx, 0)) + f9 = i64(v128.extract_lane(${f_name}_89xx, 1)) + + f0_2 = f0 * 2 + f1_2 = f1 * 2 + f2_2 = f2 * 2 + f3_2 = f3 * 2 + f4_2 = f4 * 2 + f5_2 = f5 * 2 + f6_2 = f6 * 2 + f7_2 = f7 * 2 + + f5_38 = f5 * 38 /* 1.959375*2^30 */ + f6_19 = f6 * 19 /* 1.959375*2^30 */ + f7_38 = f7 * 38/* 1.959375*2^30 */ + f8_19 = f8 * 19/* 1.959375*2^30 */ + f9_38 = f9 * 38/* 1.959375*2^30 */ + + const f0f0: i64 = f0 * f0 + const f0f1_2: i64 = f0_2 * f1 + const f0f2_2: i64 = f0_2 * f2 + const f0f3_2: i64 = f0_2 * f3 + const f0f4_2: i64 = f0_2 * f4 + const f0f5_2: i64 = f0_2 * f5 + const f0f6_2: i64 = f0_2 * f6 + const f0f7_2: i64 = f0_2 * f7 + const f0f8_2: i64 = f0_2 * f8 + const f0f9_2: i64 = f0_2 * f9 + + const f1f1_2: i64 = f1_2 * f1 + const f1f2_2: i64 = f1_2 * f2 + const f1f3_4: i64 = f1_2 * f3_2 + const f1f4_2: i64 = f1_2 * f4 + const f1f5_4: i64 = f1_2 * f5_2 + const f1f6_2: i64 = f1_2 * f6 + const f1f7_4: i64 = f1_2 * f7_2 + const f1f8_2: i64 = f1_2 * f8 + const f1f9_76: i64 = f1_2 * f9_38 + + const f2f2: i64 = f2 * f2 + const f2f3_2: i64 = f2_2 * f3 + const f2f4_2: i64 = f2_2 * f4 + const f2f5_2: i64 = f2_2 * f5 + const f2f6_2: i64 = f2_2 * f6 + const f2f7_2: i64 = f2_2 * f7 + const f2f8_38: i64 = f2_2 * f8_19 + const f2f9_38: i64 = f2 * f9_38 + + const f3f3_2: i64 = f3_2 * f3 + const f3f4_2: i64 = f3_2 * f4 + const f3f5_4: i64 = f3_2 * f5_2 + const f3f6_2: i64 = f3_2 * f6 + const f3f7_76: i64 = f3_2 * f7_38 + const f3f8_38: i64 = f3_2 * f8_19 + const f3f9_76: i64 = f3_2 * f9_38 + + const f4f4: i64 = f4 * f4 + const f4f5_2: i64 = f4_2 * f5 + const f4f6_38: i64 = f4_2 * f6_19 + const f4f7_38: i64 = f4 * f7_38 + const f4f8_38: i64 = f4_2 * f8_19 + const f4f9_38: i64 = f4 * f9_38 + + const f5f5_38: i64 = f5 * f5_38 + const f5f6_38: i64 = f5_2 * f6_19 + const f5f7_76: i64 = f5_2 * f7_38 + const f5f8_38: i64 = f5_2 * f8_19 + const f5f9_76: i64 = f5_2 * f9_38 + + const f6f6_19: i64 = f6 * f6_19 + const f6f7_38: i64 = f6 * f7_38 + const f6f8_38: i64 = f6_2 * f8_19 + const f6f9_38: i64 = f6 * f9_38 + + const f7f7_38: i64 = f7 * f7_38 + const f7f8_38: i64 = f7_2 * f8_19 + const f7f9_76: i64 = f7_2 * f9_38 + + const f8f8_19: i64 = f8 * f8_19 + const f8f9_38: i64 = f8 * f9_38 + + const f9f9_38: i64 = f9 * f9_38 + + h0 = f0f0 + f1f9_76 + f2f8_38 + f3f7_76 + f4f6_38 + f5f5_38 + h1 = f0f1_2 + f2f9_38 + f3f8_38 + f4f7_38 + f5f6_38 + h2 = f0f2_2 + f1f1_2 + f3f9_76 + f4f8_38 + f5f7_76 + f6f6_19 + h3 = f0f3_2 + f1f2_2 + f4f9_38 + f5f8_38 + f6f7_38 + h4 = f0f4_2 + f1f3_4 + f2f2 + f5f9_76 + f6f8_38 + f7f7_38 + h5 = f0f5_2 + f1f4_2 + f2f3_2 + f6f9_38 + f7f8_38 + h6 = f0f6_2 + f1f5_4 + f2f4_2 + f3f3_2 + f7f9_76 + f8f8_19 + h7 = f0f7_2 + f1f6_2 + f2f5_2 + f3f4_2 + f8f9_38 + h8 = f0f8_2 + f1f7_4 + f2f6_2 + f3f5_4 + f4f4 + f9f9_38 + h9 = f0f9_2 + f1f8_2 + f2f7_2 + f3f6_2 + f4f5_2 + + h0 *= 2 + h1 *= 2 + h2 *= 2 + h3 *= 2 + h4 *= 2 + h5 *= 2 + h6 *= 2 + h7 *= 2 + h8 *= 2 + h9 *= 2 + + c0 = (h0 + i64(1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + i64(1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + i64(1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + i64(1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + i64(1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + i64(1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + i64(1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + i64(1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + i64(1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + i64(1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + i64(1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + i64(1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + ${h_name}_0123 = i32x4(h0, h1, h2, h3) + ${h_name}_4567 = i32x4(h4, h5, h6, h7) + ${h_name}_89xx = i32x4(h8, h9, 0, 0) +` +} + +/** + * Stores a FieldElement from named vectors into memory. + * + * @param {string} f Variable name of FieldElement + * @returns {string} StaticArray in memory + */ +export function fe_store (f) { + const f_name = f.replaceAll('.', '_') + const f_ptr = `changetype(${f})` + return ` + // ${f} = [0..3, 4..7, 8..11] + v128.store(${f_ptr}, ${f_name}_0123, 0) + v128.store(${f_ptr}, ${f_name}_4567, 16) + v128.store_lane(${f_ptr}, ${f_name}_89xx, 0, 32) +` +} + +/** + * Subtract a FieldElement another and store the result. + * + * @param {string} h Variable name of FieldElement difference destination + * @param {string} f Variable name of FieldElement minuend source + * @param {string} g Variable name of FieldElement subtrahend source + * @returns {string} Three vector subtraction commands + */ +export function fe_sub (h, f, g) { + const h_name = h.replaceAll('.', '_') + const f_name = f.replaceAll('.', '_') + const g_name = g.replaceAll('.', '_') + return ` + // ${h} = ${f} - ${g} + ${h_name}_0123 = v128.sub(${f_name}_0123, ${g_name}_0123) + ${h_name}_4567 = v128.sub(${f_name}_4567, ${g_name}_4567) + ${h_name}_89xx = v128.sub(${f_name}_89xx, ${g_name}_89xx) ` } @@ -126,26 +895,14 @@ export function fe_1 (h: FieldElement): void { //@ts-expect-error @inline export function fe_add (h: FieldElement, f: FieldElement, g: FieldElement): void { - const h_ptr = changetype(h) - const f_ptr = changetype(f) - const g_ptr = changetype(g) - - let h0: v128, h1: v128, h2: v128 - - let f0 = v128.load(f_ptr, 0) - let f1 = v128.load(f_ptr, 16) - let f2 = v128.load(f_ptr, 32) - - let g0 = v128.load(g_ptr, 0) - let g1 = v128.load(g_ptr, 16) - let g2 = v128.load(g_ptr, 32) + ${fe_init_v128('h', 'f', 'g')} + ${fe_load('h')} + ${fe_load('f')} + ${fe_load('g')} ${fe_add('h', 'f', 'g')} - // store h - v128.store(h_ptr, h0, 0) - v128.store(h_ptr, h1, 16) - v128.store_lane(h_ptr, h2, 0, 32) + ${fe_store('h')} } /** @@ -187,21 +944,13 @@ export function fe_copy (h: FieldElement, f: FieldElement): void { //@ts-expect-error @inline export function fe_dbl (h: FieldElement, f: FieldElement): void { - const h_ptr = changetype(h) - const f_ptr = changetype(f) - - let h0: v128, h1: v128, h2: v128 - - let f0 = v128.load(f_ptr, 0) - let f1 = v128.load(f_ptr, 16) - let f2 = v128.load(f_ptr, 32) + ${fe_init_v128('h', 'f')} + ${fe_load('h')} + ${fe_load('f')} ${fe_dbl('h', 'f')} - // store h - v128.store(h_ptr, h0, 0) - v128.store(h_ptr, h1, 16) - v128.store_lane(h_ptr, h2, 0, 32) + ${fe_store('h')} } /** @@ -369,11 +1118,9 @@ export function fe_iszero (f: FieldElement): u8 { * @param g FieldElement multiplicand source */ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void { + ${fe_init_v128('h', 'f', 'g')} const f_ptr: usize = changetype(f) - const g_ptr: usize = changetype(g) - const g0123: v128 = v128.load(g_ptr, 0) - const g4567: v128 = v128.load(g_ptr, 16) - const g89xx: v128 = v128.load(g_ptr, 32) + ${fe_load('g')} // mask to select even lane from first vector and odd lane from second const m: v128 = i32x4(-1, 0, -1, 0) @@ -383,40 +1130,40 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void // factor of 19 for wrap from h9 to h0 const v19: v128 = i32x4.splat(19) - const g0123_19: v128 = i32x4.mul(g0123, v19) - const g4567_19: v128 = i32x4.mul(g4567, v19) - const g89xx_19: v128 = i32x4.mul(g89xx, v19) + const g_0123_19: v128 = i32x4.mul(g_0123, v19) + const g_4567_19: v128 = i32x4.mul(g_4567, v19) + const g_89xx_19: v128 = i32x4.mul(g_89xx, v19) // f[0] let fi: i32 = load(f_ptr, 0) let fv: v128 = i32x4.splat(fi) - let h01: v128 = i64x2.extmul_low_i32x4_s(fv, g0123) - let h23: v128 = i64x2.extmul_high_i32x4_s(fv, g0123) - let h45: v128 = i64x2.extmul_low_i32x4_s(fv, g4567) - let h67: v128 = i64x2.extmul_high_i32x4_s(fv, g4567) - let h89: v128 = i64x2.extmul_low_i32x4_s(fv, g89xx) + let h01: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123) + let h23: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123) + let h45: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567) + let h67: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567) + let h89: v128 = i64x2.extmul_low_i32x4_s(fv, g_89xx) // f[1] fi = load(f_ptr, 4) fv = i32x4.splat(fi) fv = i32x4.add(fv, v128.and(fv, odds)) - let h12: v128 = i64x2.extmul_low_i32x4_s(fv, g0123) - let h34: v128 = i64x2.extmul_high_i32x4_s(fv, g0123) - let h56: v128 = i64x2.extmul_low_i32x4_s(fv, g4567) - let h78: v128 = i64x2.extmul_high_i32x4_s(fv, g4567) - let h90: v128 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g89xx, g89xx_19, m)) + let h12: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123) + let h34: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123) + let h56: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567) + let h78: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567) + let h90: v128 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_89xx, g_89xx_19, m)) // f[2] fi = load(f_ptr, 8) fv = i32x4.splat(fi) - let t0: v128 = i64x2.extmul_low_i32x4_s(fv, g0123) - let t1: v128 = i64x2.extmul_high_i32x4_s(fv, g0123) - let t2: v128 = i64x2.extmul_low_i32x4_s(fv, g4567) - let t3: v128 = i64x2.extmul_high_i32x4_s(fv, g4567) - let t4: v128 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + let t0: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123) + let t1: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123) + let t2: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567) + let t3: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567) + let t4: v128 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h23 = i64x2.add(h23, t0) h45 = i64x2.add(h45, t1) @@ -429,11 +1176,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fv = i32x4.splat(fi) fv = i32x4.add(fv, v128.and(fv, odds)) - t0 = i64x2.extmul_low_i32x4_s(fv, g0123) - t1 = i64x2.extmul_high_i32x4_s(fv, g0123) - t2 = i64x2.extmul_low_i32x4_s(fv, g4567) - t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g4567, g4567_19, m)) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, g_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, g_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, g_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g_4567, g_4567_19, m)) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h34 = i64x2.add(h34, t0) h56 = i64x2.add(h56, t1) @@ -445,11 +1192,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fi = load(f_ptr, 16) fv = i32x4.splat(fi) - t0 = i64x2.extmul_low_i32x4_s(fv, g0123) - t1 = i64x2.extmul_high_i32x4_s(fv, g0123) - t2 = i64x2.extmul_low_i32x4_s(fv, g4567) - t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, g_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, g_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, g_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h45 = i64x2.add(h45, t0) h67 = i64x2.add(h67, t1) @@ -462,11 +1209,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fv = i32x4.splat(fi) fv = i32x4.add(fv, v128.and(fv, odds)) - t0 = i64x2.extmul_low_i32x4_s(fv, g0123) - t1 = i64x2.extmul_high_i32x4_s(fv, g0123) - t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g4567, g4567_19, m)) - t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, g_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, g_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_4567, g_4567_19, m)) + t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h56 = i64x2.add(h56, t0) h78 = i64x2.add(h78, t1) @@ -478,11 +1225,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fi = load(f_ptr, 24) fv = i32x4.splat(fi) - t0 = i64x2.extmul_low_i32x4_s(fv, g0123) - t1 = i64x2.extmul_high_i32x4_s(fv, g0123) - t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19) - t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, g_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, g_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h67 = i64x2.add(h67, t0) h89 = i64x2.add(h89, t1) @@ -495,11 +1242,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fv = i32x4.splat(fi) fv = i32x4.add(fv, v128.and(fv, odds)) - t0 = i64x2.extmul_low_i32x4_s(fv, g0123) - t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g0123, g0123_19, m)) - t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19) - t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, g_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g_0123, g_0123_19, m)) + t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h78 = i64x2.add(h78, t0) h90 = i64x2.add(h90, t1) @@ -511,11 +1258,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fi = load(f_ptr, 32) fv = i32x4.splat(fi) - t0 = i64x2.extmul_low_i32x4_s(fv, g0123) - t1 = i64x2.extmul_high_i32x4_s(fv, g0123_19) - t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19) - t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, g_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, g_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h89 = i64x2.add(h89, t0) h01 = i64x2.add(h01, t1) @@ -528,11 +1275,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fv = i32x4.splat(fi) fv = i32x4.add(fv, v128.and(fv, odds)) - t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g0123, g0123_19, m)) - t1 = i64x2.extmul_high_i32x4_s(fv, g0123_19) - t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19) - t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_0123, g_0123_19, m)) + t1 = i64x2.extmul_high_i32x4_s(fv, g_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h90 = i64x2.add(h90, t0) h12 = i64x2.add(h12, t1) @@ -630,11 +1377,13 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void //@ts-expect-error @inline export function fe_neg (h: FieldElement, f: FieldElement): void { - const h_ptr = changetype(h) - const f_ptr = changetype(f) - v128.store(h_ptr, v128.neg(v128.load(f_ptr))) - v128.store(h_ptr, v128.neg(v128.load(f_ptr, 16)), 16) - v128.store_lane(h_ptr, v128.neg(v128.load(f_ptr, 32)), 0, 32) + ${fe_init_v128('h', 'f')} + ${fe_load('h')} + ${fe_load('f')} + + ${fe_neg('h', 'f')} + + ${fe_store('h')} } const fe_pow22523_t0: FieldElement = fe() @@ -1477,26 +2226,14 @@ export function fe_sq2 (h: FieldElement, f: FieldElement): void { //@ts-expect-error @inline export function fe_sub (h: FieldElement, f: FieldElement, g: FieldElement): void { - const h_ptr = changetype(h) - const f_ptr = changetype(f) - const g_ptr = changetype(g) - - let h0: v128, h1: v128, h2: v128 - - let f0 = v128.load(f_ptr, 0) - let f1 = v128.load(f_ptr, 16) - let f2 = v128.load(f_ptr, 32) - - let g0 = v128.load(g_ptr, 0) - let g1 = v128.load(g_ptr, 16) - let g2 = v128.load(g_ptr, 32) + ${fe_init_v128('h', 'f', 'g')} + ${fe_load('h')} + ${fe_load('f')} + ${fe_load('g')} ${fe_sub('h', 'f', 'g')} - // store h - v128.store(h_ptr, h0, 0) - v128.store(h_ptr, h1, 16) - v128.store_lane(h_ptr, h2, 0, 32) + ${fe_store('h')} } const fe_tobytes_t: FieldElement = fe() diff --git a/scripts/p.gen.mjs b/scripts/p.gen.mjs index 25a3e4d..a4936e6 100644 --- a/scripts/p.gen.mjs +++ b/scripts/p.gen.mjs @@ -9,7 +9,7 @@ * AssemblyScript compiler and type checker. */ -import { fe_add, fe_dbl, fe_sub } from './fe.gen.mjs' +import { fe_add, fe_dbl, fe_init, fe_load, fe_mul, fe_sq, fe_sq2, fe_store, fe_sub } from './fe.gen.mjs' export const P = `//! SPDX-FileCopyrightText: 2026 Chris Duncan //! SPDX-License-Identifier: GPL-3.0-or-later @@ -60,70 +60,40 @@ export function ge_p2_0 (h: ge_p2): void { fe_1(h.Z) } -const ge_p2_dbl_t0: FieldElement = fe() +const ge_p2_dbl_t: FieldElement = fe() //@ts-expect-error @inline export function ge_p2_dbl (r: ge_p1p1, p: ge_p2): void { - const rX_ptr = changetype(r.X) - const rY_ptr = changetype(r.Y) - const rZ_ptr = changetype(r.Z) - const rT_ptr = changetype(r.T) - - const pX_ptr = changetype(p.X) - const pY_ptr = changetype(p.Y) - const pZ_ptr = changetype(p.Z) - - const rYsq = ge_p2_dbl_t0 - const rYsq_ptr = changetype(rYsq) - - fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rY = pY² - fe_sq2(r.T, p.Z) // rT = pZ² - - // Start converting to performant inlined ops with locals to avoid load/store - let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32) - let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32) - let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32) - let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32) - - const pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32) - const pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32) - const pZ0 = v128.load(pZ_ptr), pZ1 = v128.load(pZ_ptr, 16), pZ2 = v128.load(pZ_ptr, 32) - - // rY = pX + pY - v128.store(rY_ptr, v128.add(pX0, pY0)) - v128.store(rY_ptr, v128.add(pX1, pY1), 16) - v128.store_lane(rY_ptr, v128.add(pX2, pY2), 0, 32) - - // t0 = r.Y² - fe_sq(rYsq, r.Y) - const rYsq0 = v128.load(rYsq_ptr, 0) - const rYsq1 = v128.load(rYsq_ptr, 16) - const rYsq2 = v128.load(rYsq_ptr, 32) - - ${fe_add('rY', 'rZ', 'rX')} - ${fe_sub('rZ', 'rZ', 'rX')} - ${fe_sub('rX', 'rYsq', 'rY')} - ${fe_sub('rT', 'rT', 'rZ')} - - // store r.X - v128.store(rX_ptr, rX0) - v128.store(rX_ptr, rX1, 16) - v128.store_lane(rX_ptr, rX2, 0, 32) - - // store r.Y - v128.store(rY_ptr, rY0) - v128.store(rY_ptr, rY1, 16) - v128.store_lane(rY_ptr, rY2, 0, 32) - - // store r.Z - v128.store(rZ_ptr, rZ0) - v128.store(rZ_ptr, rZ1, 16) - v128.store_lane(rZ_ptr, rZ2, 0, 32) - - // store r.T - v128.store(rT_ptr, rT0) - v128.store(rT_ptr, rT1, 16) - v128.store_lane(rT_ptr, rT2, 0, 32) + ${fe_init('p.X', 'p.Y', 'p.Z', 'r.X', 'r.Y', 'r.Z', 'r.T', 'rY2')} + ${fe_load('p.X')} + ${fe_load('p.Y')} + ${fe_load('p.Z')} + + ${fe_load('r.X')} + ${fe_load('r.Y')} + ${fe_load('r.Z')} + ${fe_load('r.T')} + + const rY2 = ge_p2_dbl_t + ${fe_load('rY2')} + + // could swap below with fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rZ = pY² + ${fe_sq('r.X', 'p.X')} + ${fe_sq('r.Z', 'p.Y')} + // could swap above with fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rZ = pY² + + ${fe_sq2('r.T', 'p.Z')} + ${fe_add('r.Y', 'p.X', 'p.Y')} + ${fe_sq('rY2', 'r.Y')} + ${fe_add('r.Y', 'r.Z', 'r.X')} + ${fe_sub('r.Z', 'r.Z', 'r.X')} + ${fe_sub('r.X', 'rY2', 'r.Y')} + ${fe_sub('r.T', 'r.T', 'r.Z')} + + ${fe_store('r.X')} + ${fe_store('r.Y')} + ${fe_store('r.Z')} + ${fe_store('r.T')} } /** @@ -222,70 +192,42 @@ export function ge_precomp_0 (h: ge_precomp): void { * r = p + q */ export function ge_add_cached (r: ge_p1p1, p: ge_p3, q: ge_cached): void { - const rX_ptr = changetype(r.X) - const rY_ptr = changetype(r.Y) - const rZ_ptr = changetype(r.Z) - const rT_ptr = changetype(r.T) - - const pX_ptr = changetype(p.X) - const pY_ptr = changetype(p.Y) - - let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32) - let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32) - let pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32) - let pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32) - - ${fe_add('rX', 'pY', 'pX')} - ${fe_sub('rY', 'pY', 'pX')} - - // store r.X - v128.store(rX_ptr, rX0) - v128.store(rX_ptr, rX1, 16) - v128.store_lane(rX_ptr, rX2, 0, 32) - - // store r.Y - v128.store(rY_ptr, rY0) - v128.store(rY_ptr, rY1, 16) - v128.store_lane(rY_ptr, rY2, 0, 32) - - // not yet converted to inline for size and complexity - fe_mul(r.Z, r.X, q.YplusX) - fe_mul(r.Y, r.Y, q.YminusX) - fe_mul(r.T, q.T2d, p.T) - fe_mul(r.X, p.Z, q.Z) - - // remaining arithmetic done inline to avoid unnecessary load/store - rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32) - rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32) - let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32) - let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32) - let t0: v128, t1: v128, t2: v128 - - ${fe_dbl('t', 'rX')} - ${fe_sub('rX', 'rZ', 'rY')} - ${fe_add('rY', 'rZ', 'rY')} - ${fe_add('rZ', 't', 'rT')} - ${fe_sub('rT', 't', 'rT')} - - // store r.X - v128.store(rX_ptr, rX0) - v128.store(rX_ptr, rX1, 16) - v128.store_lane(rX_ptr, rX2, 0, 32) - - // store r.Y - v128.store(rY_ptr, rY0) - v128.store(rY_ptr, rY1, 16) - v128.store_lane(rY_ptr, rY2, 0, 32) - - // store r.Z - v128.store(rZ_ptr, rZ0) - v128.store(rZ_ptr, rZ1, 16) - v128.store_lane(rZ_ptr, rZ2, 0, 32) - - // store r.T - v128.store(rT_ptr, rT0) - v128.store(rT_ptr, rT1, 16) - v128.store_lane(rT_ptr, rT2, 0, 32) + ${fe_init('r.X', 'r.Y', 'r.Z', 'r.T', 'p.X', 'p.Y', 'p.Z', 'p.T', 'q.YplusX', 'q.YminusX', 'q.Z', 'q.T2d')} + ${fe_load('r.X')} + ${fe_load('r.Y')} + ${fe_load('r.Z')} + ${fe_load('r.T')} + + ${fe_load('p.X')} + ${fe_load('p.Y')} + ${fe_load('p.Z')} + ${fe_load('p.T')} + + ${fe_load('q.YplusX')} + ${fe_load('q.YminusX')} + ${fe_load('q.Z')} + ${fe_load('q.T2d')} + + ${fe_add('r.X', 'p.Y', 'p.X')} + ${fe_sub('r.Y', 'p.Y', 'p.X')} + + ${fe_mul('r.Z', 'r.X', 'q.YplusX')} + ${fe_mul('r.Y', 'r.Y', 'q.YminusX')} + ${fe_mul('r.T', 'q.T2d', 'p.T')} + ${fe_mul('r.X', 'p.Z', 'q.Z')} + + let t_0123: v128, t_4567: v128, t_89xx: v128 + + ${fe_dbl('t', 'r.X')} + ${fe_sub('r.X', 'r.Z', 'r.Y')} + ${fe_add('r.Y', 'r.Z', 'r.Y')} + ${fe_add('r.Z', 't', 'r.T')} + ${fe_sub('r.T', 't', 'r.T')} + + ${fe_store('r.X')} + ${fe_store('r.Y')} + ${fe_store('r.Z')} + ${fe_store('r.T')} } /** @@ -294,73 +236,36 @@ export function ge_add_cached (r: ge_p1p1, p: ge_p3, q: ge_cached): void { //@ts-expect-error @inline export function ge_add_precomp (r: ge_p1p1, p: ge_p3, q: ge_precomp): void { - const rX_ptr = changetype(r.X) - const rY_ptr = changetype(r.Y) - const rZ_ptr = changetype(r.Z) - const rT_ptr = changetype(r.T) - - const pX_ptr = changetype(p.X) - const pY_ptr = changetype(p.Y) - const pZ_ptr = changetype(p.Z) - - let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32) - let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32) - const pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32) - const pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32) - - ${fe_add('rX', 'pY', 'pX')} - ${fe_sub('rY', 'pY', 'pX')} - - // store r.X - v128.store(rX_ptr, rX0) - v128.store(rX_ptr, rX1, 16) - v128.store_lane(rX_ptr, rX2, 0, 32) - - // store r.Y - v128.store(rY_ptr, rY0) - v128.store(rY_ptr, rY1, 16) - v128.store_lane(rY_ptr, rY2, 0, 32) - - // not yet converted to inline for size and complexity - fe_mul(r.Z, r.X, q.yplusx) - fe_mul(r.Y, r.Y, q.yminusx) - fe_mul(r.T, q.xy2d, p.T) - - // remaining arithmetic done inline to avoid unnecessary load/store - rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32) - rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32) - let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32) - let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32) - - let pZ0 = v128.load(pZ_ptr, 0) - let pZ1 = v128.load(pZ_ptr, 16) - let pZ2 = v128.load(pZ_ptr, 32) - - ${fe_dbl('pZ', 'pZ')} - ${fe_sub('rX', 'rZ', 'rY')} - ${fe_add('rY', 'rZ', 'rY')} - ${fe_add('rZ', 'pZ', 'rT')} - ${fe_sub('rT', 'pZ', 'rT')} - - // store r.X - v128.store(rX_ptr, rX0) - v128.store(rX_ptr, rX1, 16) - v128.store_lane(rX_ptr, rX2, 0, 32) - - // store r.Y - v128.store(rY_ptr, rY0) - v128.store(rY_ptr, rY1, 16) - v128.store_lane(rY_ptr, rY2, 0, 32) - - // store r.Z - v128.store(rZ_ptr, rZ0) - v128.store(rZ_ptr, rZ1, 16) - v128.store_lane(rZ_ptr, rZ2, 0, 32) - - // store r.T - v128.store(rT_ptr, rT0) - v128.store(rT_ptr, rT1, 16) - v128.store_lane(rT_ptr, rT2, 0, 32) + ${fe_init('r.X', 'r.Y', 'r.Z', 'r.T', 'p.X', 'p.Y', 'p.Z', 'p.T', 'q.yplusx', 'q.yminusx', 'q.xy2d')} + ${fe_load('r.X')} + ${fe_load('r.Y')} + ${fe_load('r.Z')} + ${fe_load('r.T')} + + ${fe_load('p.X')} + ${fe_load('p.Y')} + ${fe_load('p.Z')} + ${fe_load('p.T')} + + ${fe_load('q.yplusx')} + ${fe_load('q.yminusx')} + ${fe_load('q.xy2d')} + + ${fe_add('r.X', 'p.Y', 'p.X')} + ${fe_sub('r.Y', 'p.Y', 'p.X')} + ${fe_mul('r.Z', 'r.X', 'q.yplusx')} + ${fe_mul('r.Y', 'r.Y', 'q.yminusx')} + ${fe_mul('r.T', 'q.xy2d', 'p.T')} + ${fe_dbl('p.Z', 'p.Z')} + ${fe_sub('r.X', 'r.Z', 'r.Y')} + ${fe_add('r.Y', 'r.Z', 'r.Y')} + ${fe_add('r.Z', 'p.Z', 'r.T')} + ${fe_sub('r.T', 'p.Z', 'r.T')} + + ${fe_store('r.X')} + ${fe_store('r.Y')} + ${fe_store('r.Z')} + ${fe_store('r.T')} } const ge_sub_p3_q_cached = new ge_cached() diff --git a/src/assembly/ed25519/fe.ts b/src/assembly/ed25519/fe.ts index 93e443a..f4ab1c3 100644 --- a/src/assembly/ed25519/fe.ts +++ b/src/assembly/ed25519/fe.ts @@ -65,29 +65,45 @@ export function fe_1 (h: FieldElement): void { //@ts-expect-error @inline export function fe_add (h: FieldElement, f: FieldElement, g: FieldElement): void { - const h_ptr = changetype(h) - const f_ptr = changetype(f) - const g_ptr = changetype(g) + + let + h_0123: v128, + h_4567: v128, + h_89xx: v128 +, + f_0123: v128, + f_4567: v128, + f_89xx: v128 +, + g_0123: v128, + g_4567: v128, + g_89xx: v128 + + // [0..3, 4..7, 8..11] = h + h_0123 = v128.load(changetype(h), 0) + h_4567 = v128.load(changetype(h), 16) + h_89xx = v128.load(changetype(h), 32) + + // [0..3, 4..7, 8..11] = f + f_0123 = v128.load(changetype(f), 0) + f_4567 = v128.load(changetype(f), 16) + f_89xx = v128.load(changetype(f), 32) + + // [0..3, 4..7, 8..11] = g + g_0123 = v128.load(changetype(g), 0) + g_4567 = v128.load(changetype(g), 16) + g_89xx = v128.load(changetype(g), 32) - let h0: v128, h1: v128, h2: v128 - - let f0 = v128.load(f_ptr, 0) - let f1 = v128.load(f_ptr, 16) - let f2 = v128.load(f_ptr, 32) + // h = f + g + h_0123 = v128.add(f_0123, g_0123) + h_4567 = v128.add(f_4567, g_4567) + h_89xx = v128.add(f_89xx, g_89xx) - let g0 = v128.load(g_ptr, 0) - let g1 = v128.load(g_ptr, 16) - let g2 = v128.load(g_ptr, 32) + // h = [0..3, 4..7, 8..11] + v128.store(changetype(h), h_0123, 0) + v128.store(changetype(h), h_4567, 16) + v128.store_lane(changetype(h), h_89xx, 0, 32) - // h = f + g - h0 = v128.add(f0, g0) - h1 = v128.add(f1, g1) - h2 = v128.add(f2, g2) - - // store h - v128.store(h_ptr, h0, 0) - v128.store(h_ptr, h1, 16) - v128.store_lane(h_ptr, h2, 0, 32) } /** @@ -129,24 +145,36 @@ export function fe_copy (h: FieldElement, f: FieldElement): void { //@ts-expect-error @inline export function fe_dbl (h: FieldElement, f: FieldElement): void { - const h_ptr = changetype(h) - const f_ptr = changetype(f) + + let + h_0123: v128, + h_4567: v128, + h_89xx: v128 +, + f_0123: v128, + f_4567: v128, + f_89xx: v128 + + // [0..3, 4..7, 8..11] = h + h_0123 = v128.load(changetype(h), 0) + h_4567 = v128.load(changetype(h), 16) + h_89xx = v128.load(changetype(h), 32) + + // [0..3, 4..7, 8..11] = f + f_0123 = v128.load(changetype(f), 0) + f_4567 = v128.load(changetype(f), 16) + f_89xx = v128.load(changetype(f), 32) - let h0: v128, h1: v128, h2: v128 + // h = 2*f + h_0123 = v128.shl(f_0123, 1) + h_4567 = v128.shl(f_4567, 1) + h_89xx = v128.shl(f_89xx, 1) - let f0 = v128.load(f_ptr, 0) - let f1 = v128.load(f_ptr, 16) - let f2 = v128.load(f_ptr, 32) + // h = [0..3, 4..7, 8..11] + v128.store(changetype(h), h_0123, 0) + v128.store(changetype(h), h_4567, 16) + v128.store_lane(changetype(h), h_89xx, 0, 32) - // h = 2*f - h0 = v128.shl(f0, 1) - h1 = v128.shl(f1, 1) - h2 = v128.shl(f2, 1) - - // store h - v128.store(h_ptr, h0, 0) - v128.store(h_ptr, h1, 16) - v128.store_lane(h_ptr, h2, 0, 32) } /** @@ -314,11 +342,26 @@ export function fe_iszero (f: FieldElement): u8 { * @param g FieldElement multiplicand source */ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void { + + let + h_0123: v128, + h_4567: v128, + h_89xx: v128 +, + f_0123: v128, + f_4567: v128, + f_89xx: v128 +, + g_0123: v128, + g_4567: v128, + g_89xx: v128 + const f_ptr: usize = changetype(f) - const g_ptr: usize = changetype(g) - const g0123: v128 = v128.load(g_ptr, 0) - const g4567: v128 = v128.load(g_ptr, 16) - const g89xx: v128 = v128.load(g_ptr, 32) + + // [0..3, 4..7, 8..11] = g + g_0123 = v128.load(changetype(g), 0) + g_4567 = v128.load(changetype(g), 16) + g_89xx = v128.load(changetype(g), 32) // mask to select even lane from first vector and odd lane from second const m: v128 = i32x4(-1, 0, -1, 0) @@ -328,40 +371,40 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void // factor of 19 for wrap from h9 to h0 const v19: v128 = i32x4.splat(19) - const g0123_19: v128 = i32x4.mul(g0123, v19) - const g4567_19: v128 = i32x4.mul(g4567, v19) - const g89xx_19: v128 = i32x4.mul(g89xx, v19) + const g_0123_19: v128 = i32x4.mul(g_0123, v19) + const g_4567_19: v128 = i32x4.mul(g_4567, v19) + const g_89xx_19: v128 = i32x4.mul(g_89xx, v19) // f[0] let fi: i32 = load(f_ptr, 0) let fv: v128 = i32x4.splat(fi) - let h01: v128 = i64x2.extmul_low_i32x4_s(fv, g0123) - let h23: v128 = i64x2.extmul_high_i32x4_s(fv, g0123) - let h45: v128 = i64x2.extmul_low_i32x4_s(fv, g4567) - let h67: v128 = i64x2.extmul_high_i32x4_s(fv, g4567) - let h89: v128 = i64x2.extmul_low_i32x4_s(fv, g89xx) + let h01: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123) + let h23: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123) + let h45: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567) + let h67: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567) + let h89: v128 = i64x2.extmul_low_i32x4_s(fv, g_89xx) // f[1] fi = load(f_ptr, 4) fv = i32x4.splat(fi) fv = i32x4.add(fv, v128.and(fv, odds)) - let h12: v128 = i64x2.extmul_low_i32x4_s(fv, g0123) - let h34: v128 = i64x2.extmul_high_i32x4_s(fv, g0123) - let h56: v128 = i64x2.extmul_low_i32x4_s(fv, g4567) - let h78: v128 = i64x2.extmul_high_i32x4_s(fv, g4567) - let h90: v128 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g89xx, g89xx_19, m)) + let h12: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123) + let h34: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123) + let h56: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567) + let h78: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567) + let h90: v128 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_89xx, g_89xx_19, m)) // f[2] fi = load(f_ptr, 8) fv = i32x4.splat(fi) - let t0: v128 = i64x2.extmul_low_i32x4_s(fv, g0123) - let t1: v128 = i64x2.extmul_high_i32x4_s(fv, g0123) - let t2: v128 = i64x2.extmul_low_i32x4_s(fv, g4567) - let t3: v128 = i64x2.extmul_high_i32x4_s(fv, g4567) - let t4: v128 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + let t0: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123) + let t1: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123) + let t2: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567) + let t3: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567) + let t4: v128 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h23 = i64x2.add(h23, t0) h45 = i64x2.add(h45, t1) @@ -374,11 +417,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fv = i32x4.splat(fi) fv = i32x4.add(fv, v128.and(fv, odds)) - t0 = i64x2.extmul_low_i32x4_s(fv, g0123) - t1 = i64x2.extmul_high_i32x4_s(fv, g0123) - t2 = i64x2.extmul_low_i32x4_s(fv, g4567) - t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g4567, g4567_19, m)) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, g_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, g_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, g_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g_4567, g_4567_19, m)) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h34 = i64x2.add(h34, t0) h56 = i64x2.add(h56, t1) @@ -390,11 +433,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fi = load(f_ptr, 16) fv = i32x4.splat(fi) - t0 = i64x2.extmul_low_i32x4_s(fv, g0123) - t1 = i64x2.extmul_high_i32x4_s(fv, g0123) - t2 = i64x2.extmul_low_i32x4_s(fv, g4567) - t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, g_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, g_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, g_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h45 = i64x2.add(h45, t0) h67 = i64x2.add(h67, t1) @@ -407,11 +450,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fv = i32x4.splat(fi) fv = i32x4.add(fv, v128.and(fv, odds)) - t0 = i64x2.extmul_low_i32x4_s(fv, g0123) - t1 = i64x2.extmul_high_i32x4_s(fv, g0123) - t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g4567, g4567_19, m)) - t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, g_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, g_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_4567, g_4567_19, m)) + t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h56 = i64x2.add(h56, t0) h78 = i64x2.add(h78, t1) @@ -423,11 +466,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fi = load(f_ptr, 24) fv = i32x4.splat(fi) - t0 = i64x2.extmul_low_i32x4_s(fv, g0123) - t1 = i64x2.extmul_high_i32x4_s(fv, g0123) - t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19) - t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, g_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, g_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h67 = i64x2.add(h67, t0) h89 = i64x2.add(h89, t1) @@ -440,11 +483,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fv = i32x4.splat(fi) fv = i32x4.add(fv, v128.and(fv, odds)) - t0 = i64x2.extmul_low_i32x4_s(fv, g0123) - t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g0123, g0123_19, m)) - t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19) - t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, g_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g_0123, g_0123_19, m)) + t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h78 = i64x2.add(h78, t0) h90 = i64x2.add(h90, t1) @@ -456,11 +499,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fi = load(f_ptr, 32) fv = i32x4.splat(fi) - t0 = i64x2.extmul_low_i32x4_s(fv, g0123) - t1 = i64x2.extmul_high_i32x4_s(fv, g0123_19) - t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19) - t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, g_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, g_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h89 = i64x2.add(h89, t0) h01 = i64x2.add(h01, t1) @@ -473,11 +516,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void fv = i32x4.splat(fi) fv = i32x4.add(fv, v128.and(fv, odds)) - t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g0123, g0123_19, m)) - t1 = i64x2.extmul_high_i32x4_s(fv, g0123_19) - t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19) - t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19) - t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19) + t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_0123, g_0123_19, m)) + t1 = i64x2.extmul_high_i32x4_s(fv, g_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19) h90 = i64x2.add(h90, t0) h12 = i64x2.add(h12, t1) @@ -575,11 +618,36 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void //@ts-expect-error @inline export function fe_neg (h: FieldElement, f: FieldElement): void { - const h_ptr = changetype(h) - const f_ptr = changetype(f) - v128.store(h_ptr, v128.neg(v128.load(f_ptr))) - v128.store(h_ptr, v128.neg(v128.load(f_ptr, 16)), 16) - v128.store_lane(h_ptr, v128.neg(v128.load(f_ptr, 32)), 0, 32) + + let + h_0123: v128, + h_4567: v128, + h_89xx: v128 +, + f_0123: v128, + f_4567: v128, + f_89xx: v128 + + // [0..3, 4..7, 8..11] = h + h_0123 = v128.load(changetype(h), 0) + h_4567 = v128.load(changetype(h), 16) + h_89xx = v128.load(changetype(h), 32) + + // [0..3, 4..7, 8..11] = f + f_0123 = v128.load(changetype(f), 0) + f_4567 = v128.load(changetype(f), 16) + f_89xx = v128.load(changetype(f), 32) + + // h = -f + h_0123 = v128.neg(f_0123) + h_4567 = v128.neg(f_4567) + h_89xx = v128.neg(f_89xx) + + // h = [0..3, 4..7, 8..11] + v128.store(changetype(h), h_0123, 0) + v128.store(changetype(h), h_4567, 16) + v128.store_lane(changetype(h), h_89xx, 0, 32) + } const fe_pow22523_t0: FieldElement = fe() @@ -1422,29 +1490,45 @@ export function fe_sq2 (h: FieldElement, f: FieldElement): void { //@ts-expect-error @inline export function fe_sub (h: FieldElement, f: FieldElement, g: FieldElement): void { - const h_ptr = changetype(h) - const f_ptr = changetype(f) - const g_ptr = changetype(g) - - let h0: v128, h1: v128, h2: v128 + + let + h_0123: v128, + h_4567: v128, + h_89xx: v128 +, + f_0123: v128, + f_4567: v128, + f_89xx: v128 +, + g_0123: v128, + g_4567: v128, + g_89xx: v128 + + // [0..3, 4..7, 8..11] = h + h_0123 = v128.load(changetype(h), 0) + h_4567 = v128.load(changetype(h), 16) + h_89xx = v128.load(changetype(h), 32) + + // [0..3, 4..7, 8..11] = f + f_0123 = v128.load(changetype(f), 0) + f_4567 = v128.load(changetype(f), 16) + f_89xx = v128.load(changetype(f), 32) + + // [0..3, 4..7, 8..11] = g + g_0123 = v128.load(changetype(g), 0) + g_4567 = v128.load(changetype(g), 16) + g_89xx = v128.load(changetype(g), 32) - let f0 = v128.load(f_ptr, 0) - let f1 = v128.load(f_ptr, 16) - let f2 = v128.load(f_ptr, 32) + // h = f - g + h_0123 = v128.sub(f_0123, g_0123) + h_4567 = v128.sub(f_4567, g_4567) + h_89xx = v128.sub(f_89xx, g_89xx) - let g0 = v128.load(g_ptr, 0) - let g1 = v128.load(g_ptr, 16) - let g2 = v128.load(g_ptr, 32) + // h = [0..3, 4..7, 8..11] + v128.store(changetype(h), h_0123, 0) + v128.store(changetype(h), h_4567, 16) + v128.store_lane(changetype(h), h_89xx, 0, 32) - // h = f - g - h0 = v128.sub(f0, g0) - h1 = v128.sub(f1, g1) - h2 = v128.sub(f2, g2) - - // store h - v128.store(h_ptr, h0, 0) - v128.store(h_ptr, h1, 16) - v128.store_lane(h_ptr, h2, 0, 32) } const fe_tobytes_t: FieldElement = fe() diff --git a/src/assembly/ed25519/p.ts b/src/assembly/ed25519/p.ts index 0472a56..9b030f2 100644 --- a/src/assembly/ed25519/p.ts +++ b/src/assembly/ed25519/p.ts @@ -47,85 +47,809 @@ export function ge_p2_0 (h: ge_p2): void { fe_1(h.Z) } -const ge_p2_dbl_t0: FieldElement = fe() +const ge_p2_dbl_t: FieldElement = fe() //@ts-expect-error @inline export function ge_p2_dbl (r: ge_p1p1, p: ge_p2): void { - const rX_ptr = changetype(r.X) - const rY_ptr = changetype(r.Y) - const rZ_ptr = changetype(r.Z) - const rT_ptr = changetype(r.T) - - const pX_ptr = changetype(p.X) - const pY_ptr = changetype(p.Y) - const pZ_ptr = changetype(p.Z) - - const rYsq = ge_p2_dbl_t0 - const rYsq_ptr = changetype(rYsq) - - fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rY = pY² - fe_sq2(r.T, p.Z) // rT = pZ² - - // Start converting to performant inlined ops with locals to avoid load/store - let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32) - let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32) - let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32) - let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32) - - const pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32) - const pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32) - const pZ0 = v128.load(pZ_ptr), pZ1 = v128.load(pZ_ptr, 16), pZ2 = v128.load(pZ_ptr, 32) - - // rY = pX + pY - v128.store(rY_ptr, v128.add(pX0, pY0)) - v128.store(rY_ptr, v128.add(pX1, pY1), 16) - v128.store_lane(rY_ptr, v128.add(pX2, pY2), 0, 32) - - // t0 = r.Y² - fe_sq(rYsq, r.Y) - const rYsq0 = v128.load(rYsq_ptr, 0) - const rYsq1 = v128.load(rYsq_ptr, 16) - const rYsq2 = v128.load(rYsq_ptr, 32) - - // rY = rZ + rX - rY0 = v128.add(rZ0, rX0) - rY1 = v128.add(rZ1, rX1) - rY2 = v128.add(rZ2, rX2) - - // rZ = rZ - rX - rZ0 = v128.sub(rZ0, rX0) - rZ1 = v128.sub(rZ1, rX1) - rZ2 = v128.sub(rZ2, rX2) - - // rX = rYsq - rY - rX0 = v128.sub(rYsq0, rY0) - rX1 = v128.sub(rYsq1, rY1) - rX2 = v128.sub(rYsq2, rY2) - - // rT = rT - rZ - rT0 = v128.sub(rT0, rZ0) - rT1 = v128.sub(rT1, rZ1) - rT2 = v128.sub(rT2, rZ2) - - // store r.X - v128.store(rX_ptr, rX0) - v128.store(rX_ptr, rX1, 16) - v128.store_lane(rX_ptr, rX2, 0, 32) - - // store r.Y - v128.store(rY_ptr, rY0) - v128.store(rY_ptr, rY1, 16) - v128.store_lane(rY_ptr, rY2, 0, 32) - - // store r.Z - v128.store(rZ_ptr, rZ0) - v128.store(rZ_ptr, rZ1, 16) - v128.store_lane(rZ_ptr, rZ2, 0, 32) - - // store r.T - v128.store(rT_ptr, rT0) - v128.store(rT_ptr, rT1, 16) - v128.store_lane(rT_ptr, rT2, 0, 32) + + let + p_X_0123: v128, + p_X_4567: v128, + p_X_89xx: v128 +, + p_Y_0123: v128, + p_Y_4567: v128, + p_Y_89xx: v128 +, + p_Z_0123: v128, + p_Z_4567: v128, + p_Z_89xx: v128 +, + r_X_0123: v128, + r_X_4567: v128, + r_X_89xx: v128 +, + r_Y_0123: v128, + r_Y_4567: v128, + r_Y_89xx: v128 +, + r_Z_0123: v128, + r_Z_4567: v128, + r_Z_89xx: v128 +, + r_T_0123: v128, + r_T_4567: v128, + r_T_89xx: v128 +, + rY2_0123: v128, + rY2_4567: v128, + rY2_89xx: v128 + + // mask to select even lane from first vector and odd lane from second + const m: v128 = i32x4(-1, 0, -1, 0) + + // mask to double odd-on-odd indexes + const odds: v128 = i32x4(0, -1, 0, -1) + + // factor of 19 for wrap from h9 to h0 + const v19: v128 = i32x4.splat(19) + + let c0: i64 + let c1: i64 + let c2: i64 + let c3: i64 + let c4: i64 + let c5: i64 + let c6: i64 + let c7: i64 + let c8: i64 + let c9: i64 + + let f0: i64 + let f1: i64 + let f2: i64 + let f3: i64 + let f4: i64 + let f5: i64 + let f6: i64 + let f7: i64 + let f8: i64 + let f9: i64 + + let f0_2: i64 + let f1_2: i64 + let f2_2: i64 + let f3_2: i64 + let f4_2: i64 + let f5_2: i64 + let f6_2: i64 + let f7_2: i64 + let f8_2: i64 + let f9_2: i64 + + let f5_19: i64 + let f6_19: i64 + let f7_19: i64 + let f8_19: i64 + let f9_19: i64 + + let f5_38: i64 + let f7_38: i64 + let f9_38: i64 + + let fi: i32 + let fv: v128 + + let h0: i64 + let h1: i64 + let h2: i64 + let h3: i64 + let h4: i64 + let h5: i64 + let h6: i64 + let h7: i64 + let h8: i64 + let h9: i64 + + let h01: v128 + let h12: v128 + let h23: v128 + let h34: v128 + let h45: v128 + let h56: v128 + let h67: v128 + let h78: v128 + let h89: v128 + let h90: v128 + + let t0: v128 + let t1: v128 + let t2: v128 + let t3: v128 + let t4: v128 + + // [0..3, 4..7, 8..11] = p.X + p_X_0123 = v128.load(changetype(p.X), 0) + p_X_4567 = v128.load(changetype(p.X), 16) + p_X_89xx = v128.load(changetype(p.X), 32) + + // [0..3, 4..7, 8..11] = p.Y + p_Y_0123 = v128.load(changetype(p.Y), 0) + p_Y_4567 = v128.load(changetype(p.Y), 16) + p_Y_89xx = v128.load(changetype(p.Y), 32) + + // [0..3, 4..7, 8..11] = p.Z + p_Z_0123 = v128.load(changetype(p.Z), 0) + p_Z_4567 = v128.load(changetype(p.Z), 16) + p_Z_89xx = v128.load(changetype(p.Z), 32) + + // [0..3, 4..7, 8..11] = r.X + r_X_0123 = v128.load(changetype(r.X), 0) + r_X_4567 = v128.load(changetype(r.X), 16) + r_X_89xx = v128.load(changetype(r.X), 32) + + // [0..3, 4..7, 8..11] = r.Y + r_Y_0123 = v128.load(changetype(r.Y), 0) + r_Y_4567 = v128.load(changetype(r.Y), 16) + r_Y_89xx = v128.load(changetype(r.Y), 32) + + // [0..3, 4..7, 8..11] = r.Z + r_Z_0123 = v128.load(changetype(r.Z), 0) + r_Z_4567 = v128.load(changetype(r.Z), 16) + r_Z_89xx = v128.load(changetype(r.Z), 32) + + // [0..3, 4..7, 8..11] = r.T + r_T_0123 = v128.load(changetype(r.T), 0) + r_T_4567 = v128.load(changetype(r.T), 16) + r_T_89xx = v128.load(changetype(r.T), 32) + + const rY2 = ge_p2_dbl_t + + // [0..3, 4..7, 8..11] = rY2 + rY2_0123 = v128.load(changetype(rY2), 0) + rY2_4567 = v128.load(changetype(rY2), 16) + rY2_89xx = v128.load(changetype(rY2), 32) + + // could swap below with fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rZ = pY² + + // r.X = p.X² + f0 = i64(v128.extract_lane(p_X_0123, 0)) + f1 = i64(v128.extract_lane(p_X_0123, 1)) + f2 = i64(v128.extract_lane(p_X_0123, 2)) + f3 = i64(v128.extract_lane(p_X_0123, 3)) + f4 = i64(v128.extract_lane(p_X_4567, 0)) + f5 = i64(v128.extract_lane(p_X_4567, 1)) + f6 = i64(v128.extract_lane(p_X_4567, 2)) + f7 = i64(v128.extract_lane(p_X_4567, 3)) + f8 = i64(v128.extract_lane(p_X_89xx, 0)) + f9 = i64(v128.extract_lane(p_X_89xx, 1)) + + f0_2 = f0 * 2 + f1_2 = f1 * 2 + f2_2 = f2 * 2 + f3_2 = f3 * 2 + f4_2 = f4 * 2 + f5_2 = f5 * 2 + f6_2 = f6 * 2 + f7_2 = f7 * 2 + + f5_19 = f5 * 19 /* 1.959375*2²⁹ */ + f6_19 = f6 * 19 /* 1.959375*2³⁰ */ + f7_19 = f7 * 19 /* 1.959375*2²⁹ */ + f8_19 = f8 * 19 /* 1.959375*2³⁰ */ + f9_19 = f9 * 19 /* 1.959375*2²⁹ */ + + f7_38 = f7 * 38 /* 1.959375*2³⁰ */ + f9_38 = f9 * 38 /* 1.959375*2³⁰ */ + + h0 = f0 * f0 + h1 = f0_2 * f1 + h2 = f0_2 * f2 + h3 = f0_2 * f3 + h4 = f0_2 * f4 + h5 = f0_2 * f5 + h6 = f0_2 * f6 + h7 = f0_2 * f7 + h8 = f0_2 * f8 + h9 = f0_2 * f9 + + h2 += f1_2 * f1 + h3 += f1_2 * f2 + h4 += f1_2 * f3_2 + h5 += f1_2 * f4 + h6 += f1_2 * f5_2 + h7 += f1_2 * f6 + h8 += f1_2 * f7_2 + h9 += f1_2 * f8 + h0 += f1_2 * f9_38 + + h4 += f2 * f2 + h5 += f2_2 * f3 + h6 += f2_2 * f4 + h7 += f2_2 * f5 + h8 += f2_2 * f6 + h9 += f2_2 * f7 + h0 += f2_2 * f8_19 + h1 += f2_2 * f9_19 + + h6 += f3_2 * f3 + h7 += f3_2 * f4 + h8 += f3_2 * f5_2 + h9 += f3_2 * f6 + h0 += f3_2 * f7_38 + h1 += f3_2 * f8_19 + h2 += f3_2 * f9_38 + + h8 += f4 * f4 + h9 += f4_2 * f5 + h0 += f4_2 * f6_19 + h1 += f4_2 * f7_19 + h2 += f4_2 * f8_19 + h3 += f4_2 * f9_19 + + h0 += f5_2 * f5_19 + h1 += f5_2 * f6_19 + h2 += f5_2 * f7_38 + h3 += f5_2 * f8_19 + h4 += f5_2 * f9_38 + + h2 += f6 * f6_19 + h3 += f6_2 * f7_19 + h4 += f6_2 * f8_19 + h5 += f6_2 * f9_19 + + h4 += f7_2 * f7_19 + h5 += f7_2 * f8_19 + h6 += f7_2 * f9_38 + + h6 += f8 * f8_19 + h7 += f8 * f9_38 + + h8 += f9 * f9_38 + + c0 = (h0 + i64(1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + i64(1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + i64(1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + i64(1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + i64(1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + i64(1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + i64(1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + i64(1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + i64(1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + i64(1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + i64(1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + i64(1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + r_X_0123 = i32x4(h0, h1, h2, h3) + r_X_4567 = i32x4(h4, h5, h6, h7) + r_X_89xx = i32x4(h8, h9, 0, 0) + + // r.Z = p.Y² + f0 = i64(v128.extract_lane(p_Y_0123, 0)) + f1 = i64(v128.extract_lane(p_Y_0123, 1)) + f2 = i64(v128.extract_lane(p_Y_0123, 2)) + f3 = i64(v128.extract_lane(p_Y_0123, 3)) + f4 = i64(v128.extract_lane(p_Y_4567, 0)) + f5 = i64(v128.extract_lane(p_Y_4567, 1)) + f6 = i64(v128.extract_lane(p_Y_4567, 2)) + f7 = i64(v128.extract_lane(p_Y_4567, 3)) + f8 = i64(v128.extract_lane(p_Y_89xx, 0)) + f9 = i64(v128.extract_lane(p_Y_89xx, 1)) + + f0_2 = f0 * 2 + f1_2 = f1 * 2 + f2_2 = f2 * 2 + f3_2 = f3 * 2 + f4_2 = f4 * 2 + f5_2 = f5 * 2 + f6_2 = f6 * 2 + f7_2 = f7 * 2 + + f5_19 = f5 * 19 /* 1.959375*2²⁹ */ + f6_19 = f6 * 19 /* 1.959375*2³⁰ */ + f7_19 = f7 * 19 /* 1.959375*2²⁹ */ + f8_19 = f8 * 19 /* 1.959375*2³⁰ */ + f9_19 = f9 * 19 /* 1.959375*2²⁹ */ + + f7_38 = f7 * 38 /* 1.959375*2³⁰ */ + f9_38 = f9 * 38 /* 1.959375*2³⁰ */ + + h0 = f0 * f0 + h1 = f0_2 * f1 + h2 = f0_2 * f2 + h3 = f0_2 * f3 + h4 = f0_2 * f4 + h5 = f0_2 * f5 + h6 = f0_2 * f6 + h7 = f0_2 * f7 + h8 = f0_2 * f8 + h9 = f0_2 * f9 + + h2 += f1_2 * f1 + h3 += f1_2 * f2 + h4 += f1_2 * f3_2 + h5 += f1_2 * f4 + h6 += f1_2 * f5_2 + h7 += f1_2 * f6 + h8 += f1_2 * f7_2 + h9 += f1_2 * f8 + h0 += f1_2 * f9_38 + + h4 += f2 * f2 + h5 += f2_2 * f3 + h6 += f2_2 * f4 + h7 += f2_2 * f5 + h8 += f2_2 * f6 + h9 += f2_2 * f7 + h0 += f2_2 * f8_19 + h1 += f2_2 * f9_19 + + h6 += f3_2 * f3 + h7 += f3_2 * f4 + h8 += f3_2 * f5_2 + h9 += f3_2 * f6 + h0 += f3_2 * f7_38 + h1 += f3_2 * f8_19 + h2 += f3_2 * f9_38 + + h8 += f4 * f4 + h9 += f4_2 * f5 + h0 += f4_2 * f6_19 + h1 += f4_2 * f7_19 + h2 += f4_2 * f8_19 + h3 += f4_2 * f9_19 + + h0 += f5_2 * f5_19 + h1 += f5_2 * f6_19 + h2 += f5_2 * f7_38 + h3 += f5_2 * f8_19 + h4 += f5_2 * f9_38 + + h2 += f6 * f6_19 + h3 += f6_2 * f7_19 + h4 += f6_2 * f8_19 + h5 += f6_2 * f9_19 + + h4 += f7_2 * f7_19 + h5 += f7_2 * f8_19 + h6 += f7_2 * f9_38 + + h6 += f8 * f8_19 + h7 += f8 * f9_38 + + h8 += f9 * f9_38 + + c0 = (h0 + i64(1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + i64(1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + i64(1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + i64(1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + i64(1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + i64(1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + i64(1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + i64(1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + i64(1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + i64(1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + i64(1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + i64(1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + r_Z_0123 = i32x4(h0, h1, h2, h3) + r_Z_4567 = i32x4(h4, h5, h6, h7) + r_Z_89xx = i32x4(h8, h9, 0, 0) + + // could swap above with fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rZ = pY² + + // r.T = 2p.Z² + f0 = i64(v128.extract_lane(p_Z_0123, 0)) + f1 = i64(v128.extract_lane(p_Z_0123, 1)) + f2 = i64(v128.extract_lane(p_Z_0123, 2)) + f3 = i64(v128.extract_lane(p_Z_0123, 3)) + f4 = i64(v128.extract_lane(p_Z_4567, 0)) + f5 = i64(v128.extract_lane(p_Z_4567, 1)) + f6 = i64(v128.extract_lane(p_Z_4567, 2)) + f7 = i64(v128.extract_lane(p_Z_4567, 3)) + f8 = i64(v128.extract_lane(p_Z_89xx, 0)) + f9 = i64(v128.extract_lane(p_Z_89xx, 1)) + + f0_2 = f0 * 2 + f1_2 = f1 * 2 + f2_2 = f2 * 2 + f3_2 = f3 * 2 + f4_2 = f4 * 2 + f5_2 = f5 * 2 + f6_2 = f6 * 2 + f7_2 = f7 * 2 + + f5_38 = f5 * 38 /* 1.959375*2^30 */ + f6_19 = f6 * 19 /* 1.959375*2^30 */ + f7_38 = f7 * 38/* 1.959375*2^30 */ + f8_19 = f8 * 19/* 1.959375*2^30 */ + f9_38 = f9 * 38/* 1.959375*2^30 */ + + const f0f0: i64 = f0 * f0 + const f0f1_2: i64 = f0_2 * f1 + const f0f2_2: i64 = f0_2 * f2 + const f0f3_2: i64 = f0_2 * f3 + const f0f4_2: i64 = f0_2 * f4 + const f0f5_2: i64 = f0_2 * f5 + const f0f6_2: i64 = f0_2 * f6 + const f0f7_2: i64 = f0_2 * f7 + const f0f8_2: i64 = f0_2 * f8 + const f0f9_2: i64 = f0_2 * f9 + + const f1f1_2: i64 = f1_2 * f1 + const f1f2_2: i64 = f1_2 * f2 + const f1f3_4: i64 = f1_2 * f3_2 + const f1f4_2: i64 = f1_2 * f4 + const f1f5_4: i64 = f1_2 * f5_2 + const f1f6_2: i64 = f1_2 * f6 + const f1f7_4: i64 = f1_2 * f7_2 + const f1f8_2: i64 = f1_2 * f8 + const f1f9_76: i64 = f1_2 * f9_38 + + const f2f2: i64 = f2 * f2 + const f2f3_2: i64 = f2_2 * f3 + const f2f4_2: i64 = f2_2 * f4 + const f2f5_2: i64 = f2_2 * f5 + const f2f6_2: i64 = f2_2 * f6 + const f2f7_2: i64 = f2_2 * f7 + const f2f8_38: i64 = f2_2 * f8_19 + const f2f9_38: i64 = f2 * f9_38 + + const f3f3_2: i64 = f3_2 * f3 + const f3f4_2: i64 = f3_2 * f4 + const f3f5_4: i64 = f3_2 * f5_2 + const f3f6_2: i64 = f3_2 * f6 + const f3f7_76: i64 = f3_2 * f7_38 + const f3f8_38: i64 = f3_2 * f8_19 + const f3f9_76: i64 = f3_2 * f9_38 + + const f4f4: i64 = f4 * f4 + const f4f5_2: i64 = f4_2 * f5 + const f4f6_38: i64 = f4_2 * f6_19 + const f4f7_38: i64 = f4 * f7_38 + const f4f8_38: i64 = f4_2 * f8_19 + const f4f9_38: i64 = f4 * f9_38 + + const f5f5_38: i64 = f5 * f5_38 + const f5f6_38: i64 = f5_2 * f6_19 + const f5f7_76: i64 = f5_2 * f7_38 + const f5f8_38: i64 = f5_2 * f8_19 + const f5f9_76: i64 = f5_2 * f9_38 + + const f6f6_19: i64 = f6 * f6_19 + const f6f7_38: i64 = f6 * f7_38 + const f6f8_38: i64 = f6_2 * f8_19 + const f6f9_38: i64 = f6 * f9_38 + + const f7f7_38: i64 = f7 * f7_38 + const f7f8_38: i64 = f7_2 * f8_19 + const f7f9_76: i64 = f7_2 * f9_38 + + const f8f8_19: i64 = f8 * f8_19 + const f8f9_38: i64 = f8 * f9_38 + + const f9f9_38: i64 = f9 * f9_38 + + h0 = f0f0 + f1f9_76 + f2f8_38 + f3f7_76 + f4f6_38 + f5f5_38 + h1 = f0f1_2 + f2f9_38 + f3f8_38 + f4f7_38 + f5f6_38 + h2 = f0f2_2 + f1f1_2 + f3f9_76 + f4f8_38 + f5f7_76 + f6f6_19 + h3 = f0f3_2 + f1f2_2 + f4f9_38 + f5f8_38 + f6f7_38 + h4 = f0f4_2 + f1f3_4 + f2f2 + f5f9_76 + f6f8_38 + f7f7_38 + h5 = f0f5_2 + f1f4_2 + f2f3_2 + f6f9_38 + f7f8_38 + h6 = f0f6_2 + f1f5_4 + f2f4_2 + f3f3_2 + f7f9_76 + f8f8_19 + h7 = f0f7_2 + f1f6_2 + f2f5_2 + f3f4_2 + f8f9_38 + h8 = f0f8_2 + f1f7_4 + f2f6_2 + f3f5_4 + f4f4 + f9f9_38 + h9 = f0f9_2 + f1f8_2 + f2f7_2 + f3f6_2 + f4f5_2 + + h0 *= 2 + h1 *= 2 + h2 *= 2 + h3 *= 2 + h4 *= 2 + h5 *= 2 + h6 *= 2 + h7 *= 2 + h8 *= 2 + h9 *= 2 + + c0 = (h0 + i64(1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + i64(1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + i64(1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + i64(1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + i64(1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + i64(1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + i64(1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + i64(1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + i64(1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + i64(1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + i64(1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + i64(1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + r_T_0123 = i32x4(h0, h1, h2, h3) + r_T_4567 = i32x4(h4, h5, h6, h7) + r_T_89xx = i32x4(h8, h9, 0, 0) + + // r.Y = p.X + p.Y + r_Y_0123 = v128.add(p_X_0123, p_Y_0123) + r_Y_4567 = v128.add(p_X_4567, p_Y_4567) + r_Y_89xx = v128.add(p_X_89xx, p_Y_89xx) + + // rY2 = r.Y² + f0 = i64(v128.extract_lane(r_Y_0123, 0)) + f1 = i64(v128.extract_lane(r_Y_0123, 1)) + f2 = i64(v128.extract_lane(r_Y_0123, 2)) + f3 = i64(v128.extract_lane(r_Y_0123, 3)) + f4 = i64(v128.extract_lane(r_Y_4567, 0)) + f5 = i64(v128.extract_lane(r_Y_4567, 1)) + f6 = i64(v128.extract_lane(r_Y_4567, 2)) + f7 = i64(v128.extract_lane(r_Y_4567, 3)) + f8 = i64(v128.extract_lane(r_Y_89xx, 0)) + f9 = i64(v128.extract_lane(r_Y_89xx, 1)) + + f0_2 = f0 * 2 + f1_2 = f1 * 2 + f2_2 = f2 * 2 + f3_2 = f3 * 2 + f4_2 = f4 * 2 + f5_2 = f5 * 2 + f6_2 = f6 * 2 + f7_2 = f7 * 2 + + f5_19 = f5 * 19 /* 1.959375*2²⁹ */ + f6_19 = f6 * 19 /* 1.959375*2³⁰ */ + f7_19 = f7 * 19 /* 1.959375*2²⁹ */ + f8_19 = f8 * 19 /* 1.959375*2³⁰ */ + f9_19 = f9 * 19 /* 1.959375*2²⁹ */ + + f7_38 = f7 * 38 /* 1.959375*2³⁰ */ + f9_38 = f9 * 38 /* 1.959375*2³⁰ */ + + h0 = f0 * f0 + h1 = f0_2 * f1 + h2 = f0_2 * f2 + h3 = f0_2 * f3 + h4 = f0_2 * f4 + h5 = f0_2 * f5 + h6 = f0_2 * f6 + h7 = f0_2 * f7 + h8 = f0_2 * f8 + h9 = f0_2 * f9 + + h2 += f1_2 * f1 + h3 += f1_2 * f2 + h4 += f1_2 * f3_2 + h5 += f1_2 * f4 + h6 += f1_2 * f5_2 + h7 += f1_2 * f6 + h8 += f1_2 * f7_2 + h9 += f1_2 * f8 + h0 += f1_2 * f9_38 + + h4 += f2 * f2 + h5 += f2_2 * f3 + h6 += f2_2 * f4 + h7 += f2_2 * f5 + h8 += f2_2 * f6 + h9 += f2_2 * f7 + h0 += f2_2 * f8_19 + h1 += f2_2 * f9_19 + + h6 += f3_2 * f3 + h7 += f3_2 * f4 + h8 += f3_2 * f5_2 + h9 += f3_2 * f6 + h0 += f3_2 * f7_38 + h1 += f3_2 * f8_19 + h2 += f3_2 * f9_38 + + h8 += f4 * f4 + h9 += f4_2 * f5 + h0 += f4_2 * f6_19 + h1 += f4_2 * f7_19 + h2 += f4_2 * f8_19 + h3 += f4_2 * f9_19 + + h0 += f5_2 * f5_19 + h1 += f5_2 * f6_19 + h2 += f5_2 * f7_38 + h3 += f5_2 * f8_19 + h4 += f5_2 * f9_38 + + h2 += f6 * f6_19 + h3 += f6_2 * f7_19 + h4 += f6_2 * f8_19 + h5 += f6_2 * f9_19 + + h4 += f7_2 * f7_19 + h5 += f7_2 * f8_19 + h6 += f7_2 * f9_38 + + h6 += f8 * f8_19 + h7 += f8 * f9_38 + + h8 += f9 * f9_38 + + c0 = (h0 + i64(1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + i64(1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + i64(1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + i64(1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + i64(1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + i64(1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + i64(1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + i64(1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + i64(1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + i64(1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + i64(1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + i64(1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + rY2_0123 = i32x4(h0, h1, h2, h3) + rY2_4567 = i32x4(h4, h5, h6, h7) + rY2_89xx = i32x4(h8, h9, 0, 0) + + // r.Y = r.Z + r.X + r_Y_0123 = v128.add(r_Z_0123, r_X_0123) + r_Y_4567 = v128.add(r_Z_4567, r_X_4567) + r_Y_89xx = v128.add(r_Z_89xx, r_X_89xx) + + // r.Z = r.Z - r.X + r_Z_0123 = v128.sub(r_Z_0123, r_X_0123) + r_Z_4567 = v128.sub(r_Z_4567, r_X_4567) + r_Z_89xx = v128.sub(r_Z_89xx, r_X_89xx) + + // r.X = rY2 - r.Y + r_X_0123 = v128.sub(rY2_0123, r_Y_0123) + r_X_4567 = v128.sub(rY2_4567, r_Y_4567) + r_X_89xx = v128.sub(rY2_89xx, r_Y_89xx) + + // r.T = r.T - r.Z + r_T_0123 = v128.sub(r_T_0123, r_Z_0123) + r_T_4567 = v128.sub(r_T_4567, r_Z_4567) + r_T_89xx = v128.sub(r_T_89xx, r_Z_89xx) + + // r.X = [0..3, 4..7, 8..11] + v128.store(changetype(r.X), r_X_0123, 0) + v128.store(changetype(r.X), r_X_4567, 16) + v128.store_lane(changetype(r.X), r_X_89xx, 0, 32) + + // r.Y = [0..3, 4..7, 8..11] + v128.store(changetype(r.Y), r_Y_0123, 0) + v128.store(changetype(r.Y), r_Y_4567, 16) + v128.store_lane(changetype(r.Y), r_Y_89xx, 0, 32) + + // r.Z = [0..3, 4..7, 8..11] + v128.store(changetype(r.Z), r_Z_0123, 0) + v128.store(changetype(r.Z), r_Z_4567, 16) + v128.store_lane(changetype(r.Z), r_Z_89xx, 0, 32) + + // r.T = [0..3, 4..7, 8..11] + v128.store(changetype(r.T), r_T_0123, 0) + v128.store(changetype(r.T), r_T_4567, 16) + v128.store_lane(changetype(r.T), r_T_89xx, 0, 32) + } /** @@ -224,96 +948,1132 @@ export function ge_precomp_0 (h: ge_precomp): void { * r = p + q */ export function ge_add_cached (r: ge_p1p1, p: ge_p3, q: ge_cached): void { - const rX_ptr = changetype(r.X) - const rY_ptr = changetype(r.Y) - const rZ_ptr = changetype(r.Z) - const rT_ptr = changetype(r.T) - - const pX_ptr = changetype(p.X) - const pY_ptr = changetype(p.Y) - - let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32) - let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32) - let pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32) - let pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32) - - // rX = pY + pX - rX0 = v128.add(pY0, pX0) - rX1 = v128.add(pY1, pX1) - rX2 = v128.add(pY2, pX2) - - // rY = pY - pX - rY0 = v128.sub(pY0, pX0) - rY1 = v128.sub(pY1, pX1) - rY2 = v128.sub(pY2, pX2) - - // store r.X - v128.store(rX_ptr, rX0) - v128.store(rX_ptr, rX1, 16) - v128.store_lane(rX_ptr, rX2, 0, 32) - - // store r.Y - v128.store(rY_ptr, rY0) - v128.store(rY_ptr, rY1, 16) - v128.store_lane(rY_ptr, rY2, 0, 32) - - // not yet converted to inline for size and complexity - fe_mul(r.Z, r.X, q.YplusX) - fe_mul(r.Y, r.Y, q.YminusX) - fe_mul(r.T, q.T2d, p.T) - fe_mul(r.X, p.Z, q.Z) - // remaining arithmetic done inline to avoid unnecessary load/store - rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32) - rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32) - let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32) - let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32) - let t0: v128, t1: v128, t2: v128 - - // t = 2*rX - t0 = v128.shl(rX0, 1) - t1 = v128.shl(rX1, 1) - t2 = v128.shl(rX2, 1) - - // rX = rZ - rY - rX0 = v128.sub(rZ0, rY0) - rX1 = v128.sub(rZ1, rY1) - rX2 = v128.sub(rZ2, rY2) - - // rY = rZ + rY - rY0 = v128.add(rZ0, rY0) - rY1 = v128.add(rZ1, rY1) - rY2 = v128.add(rZ2, rY2) - - // rZ = t + rT - rZ0 = v128.add(t0, rT0) - rZ1 = v128.add(t1, rT1) - rZ2 = v128.add(t2, rT2) - - // rT = t - rT - rT0 = v128.sub(t0, rT0) - rT1 = v128.sub(t1, rT1) - rT2 = v128.sub(t2, rT2) - - // store r.X - v128.store(rX_ptr, rX0) - v128.store(rX_ptr, rX1, 16) - v128.store_lane(rX_ptr, rX2, 0, 32) - - // store r.Y - v128.store(rY_ptr, rY0) - v128.store(rY_ptr, rY1, 16) - v128.store_lane(rY_ptr, rY2, 0, 32) - - // store r.Z - v128.store(rZ_ptr, rZ0) - v128.store(rZ_ptr, rZ1, 16) - v128.store_lane(rZ_ptr, rZ2, 0, 32) - - // store r.T - v128.store(rT_ptr, rT0) - v128.store(rT_ptr, rT1, 16) - v128.store_lane(rT_ptr, rT2, 0, 32) + let + r_X_0123: v128, + r_X_4567: v128, + r_X_89xx: v128 +, + r_Y_0123: v128, + r_Y_4567: v128, + r_Y_89xx: v128 +, + r_Z_0123: v128, + r_Z_4567: v128, + r_Z_89xx: v128 +, + r_T_0123: v128, + r_T_4567: v128, + r_T_89xx: v128 +, + p_X_0123: v128, + p_X_4567: v128, + p_X_89xx: v128 +, + p_Y_0123: v128, + p_Y_4567: v128, + p_Y_89xx: v128 +, + p_Z_0123: v128, + p_Z_4567: v128, + p_Z_89xx: v128 +, + p_T_0123: v128, + p_T_4567: v128, + p_T_89xx: v128 +, + q_YplusX_0123: v128, + q_YplusX_4567: v128, + q_YplusX_89xx: v128 +, + q_YminusX_0123: v128, + q_YminusX_4567: v128, + q_YminusX_89xx: v128 +, + q_Z_0123: v128, + q_Z_4567: v128, + q_Z_89xx: v128 +, + q_T2d_0123: v128, + q_T2d_4567: v128, + q_T2d_89xx: v128 + + // mask to select even lane from first vector and odd lane from second + const m: v128 = i32x4(-1, 0, -1, 0) + + // mask to double odd-on-odd indexes + const odds: v128 = i32x4(0, -1, 0, -1) + + // factor of 19 for wrap from h9 to h0 + const v19: v128 = i32x4.splat(19) + + let c0: i64 + let c1: i64 + let c2: i64 + let c3: i64 + let c4: i64 + let c5: i64 + let c6: i64 + let c7: i64 + let c8: i64 + let c9: i64 + + let f0: i64 + let f1: i64 + let f2: i64 + let f3: i64 + let f4: i64 + let f5: i64 + let f6: i64 + let f7: i64 + let f8: i64 + let f9: i64 + + let f0_2: i64 + let f1_2: i64 + let f2_2: i64 + let f3_2: i64 + let f4_2: i64 + let f5_2: i64 + let f6_2: i64 + let f7_2: i64 + let f8_2: i64 + let f9_2: i64 + + let f5_19: i64 + let f6_19: i64 + let f7_19: i64 + let f8_19: i64 + let f9_19: i64 + + let f5_38: i64 + let f7_38: i64 + let f9_38: i64 + + let fi: i32 + let fv: v128 + + let h0: i64 + let h1: i64 + let h2: i64 + let h3: i64 + let h4: i64 + let h5: i64 + let h6: i64 + let h7: i64 + let h8: i64 + let h9: i64 + + let h01: v128 + let h12: v128 + let h23: v128 + let h34: v128 + let h45: v128 + let h56: v128 + let h67: v128 + let h78: v128 + let h89: v128 + let h90: v128 + + let t0: v128 + let t1: v128 + let t2: v128 + let t3: v128 + let t4: v128 + + // [0..3, 4..7, 8..11] = r.X + r_X_0123 = v128.load(changetype(r.X), 0) + r_X_4567 = v128.load(changetype(r.X), 16) + r_X_89xx = v128.load(changetype(r.X), 32) + + // [0..3, 4..7, 8..11] = r.Y + r_Y_0123 = v128.load(changetype(r.Y), 0) + r_Y_4567 = v128.load(changetype(r.Y), 16) + r_Y_89xx = v128.load(changetype(r.Y), 32) + + // [0..3, 4..7, 8..11] = r.Z + r_Z_0123 = v128.load(changetype(r.Z), 0) + r_Z_4567 = v128.load(changetype(r.Z), 16) + r_Z_89xx = v128.load(changetype(r.Z), 32) + + // [0..3, 4..7, 8..11] = r.T + r_T_0123 = v128.load(changetype(r.T), 0) + r_T_4567 = v128.load(changetype(r.T), 16) + r_T_89xx = v128.load(changetype(r.T), 32) + + // [0..3, 4..7, 8..11] = p.X + p_X_0123 = v128.load(changetype(p.X), 0) + p_X_4567 = v128.load(changetype(p.X), 16) + p_X_89xx = v128.load(changetype(p.X), 32) + + // [0..3, 4..7, 8..11] = p.Y + p_Y_0123 = v128.load(changetype(p.Y), 0) + p_Y_4567 = v128.load(changetype(p.Y), 16) + p_Y_89xx = v128.load(changetype(p.Y), 32) + + // [0..3, 4..7, 8..11] = p.Z + p_Z_0123 = v128.load(changetype(p.Z), 0) + p_Z_4567 = v128.load(changetype(p.Z), 16) + p_Z_89xx = v128.load(changetype(p.Z), 32) + + // [0..3, 4..7, 8..11] = p.T + p_T_0123 = v128.load(changetype(p.T), 0) + p_T_4567 = v128.load(changetype(p.T), 16) + p_T_89xx = v128.load(changetype(p.T), 32) + + // [0..3, 4..7, 8..11] = q.YplusX + q_YplusX_0123 = v128.load(changetype(q.YplusX), 0) + q_YplusX_4567 = v128.load(changetype(q.YplusX), 16) + q_YplusX_89xx = v128.load(changetype(q.YplusX), 32) + + // [0..3, 4..7, 8..11] = q.YminusX + q_YminusX_0123 = v128.load(changetype(q.YminusX), 0) + q_YminusX_4567 = v128.load(changetype(q.YminusX), 16) + q_YminusX_89xx = v128.load(changetype(q.YminusX), 32) + + // [0..3, 4..7, 8..11] = q.Z + q_Z_0123 = v128.load(changetype(q.Z), 0) + q_Z_4567 = v128.load(changetype(q.Z), 16) + q_Z_89xx = v128.load(changetype(q.Z), 32) + + // [0..3, 4..7, 8..11] = q.T2d + q_T2d_0123 = v128.load(changetype(q.T2d), 0) + q_T2d_4567 = v128.load(changetype(q.T2d), 16) + q_T2d_89xx = v128.load(changetype(q.T2d), 32) + + // r.X = p.Y + p.X + r_X_0123 = v128.add(p_Y_0123, p_X_0123) + r_X_4567 = v128.add(p_Y_4567, p_X_4567) + r_X_89xx = v128.add(p_Y_89xx, p_X_89xx) + + // r.Y = p.Y - p.X + r_Y_0123 = v128.sub(p_Y_0123, p_X_0123) + r_Y_4567 = v128.sub(p_Y_4567, p_X_4567) + r_Y_89xx = v128.sub(p_Y_89xx, p_X_89xx) + + // r.Z = r.X * q.YplusX + const q_YplusX_0123_19: v128 = i32x4.mul(q_YplusX_0123, v19) + const q_YplusX_4567_19: v128 = i32x4.mul(q_YplusX_4567, v19) + const q_YplusX_89xx_19: v128 = i32x4.mul(q_YplusX_89xx, v19) + + // f[0] + fi = v128.extract_lane(r_X_0123, 0) + fv = i32x4.splat(fi) + + h01 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123) + h23 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123) + h45 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567) + h67 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567) + h89 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx) + + // f[1] + fi = v128.extract_lane(r_X_0123, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + h12 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123) + h34 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123) + h56 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567) + h78 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567) + h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YplusX_89xx, q_YplusX_89xx_19, m)) + + // f[2] + fi = v128.extract_lane(r_X_0123, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19) + + h23 = i64x2.add(h23, t0) + h45 = i64x2.add(h45, t1) + h67 = i64x2.add(h67, t2) + h89 = i64x2.add(h89, t3) + h01 = i64x2.add(h01, t4) + + // f[3] + fi = v128.extract_lane(r_X_0123, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_YplusX_4567, q_YplusX_4567_19, m)) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19) + + h34 = i64x2.add(h34, t0) + h56 = i64x2.add(h56, t1) + h78 = i64x2.add(h78, t2) + h90 = i64x2.add(h90, t3) + h12 = i64x2.add(h12, t4) + + // f[4] + fi = v128.extract_lane(r_X_4567, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19) + + h45 = i64x2.add(h45, t0) + h67 = i64x2.add(h67, t1) + h89 = i64x2.add(h89, t2) + h01 = i64x2.add(h01, t3) + h23 = i64x2.add(h23, t4) + + // f[5] + fi = v128.extract_lane(r_X_4567, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YplusX_4567, q_YplusX_4567_19, m)) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19) + + h56 = i64x2.add(h56, t0) + h78 = i64x2.add(h78, t1) + h90 = i64x2.add(h90, t2) + h12 = i64x2.add(h12, t3) + h34 = i64x2.add(h34, t4) + + // f[6] + fi = v128.extract_lane(r_X_4567, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19) + + h67 = i64x2.add(h67, t0) + h89 = i64x2.add(h89, t1) + h01 = i64x2.add(h01, t2) + h23 = i64x2.add(h23, t3) + h45 = i64x2.add(h45, t4) + + // f[7] + fi = v128.extract_lane(r_X_4567, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_YplusX_0123, q_YplusX_0123_19, m)) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19) + + h78 = i64x2.add(h78, t0) + h90 = i64x2.add(h90, t1) + h12 = i64x2.add(h12, t2) + h34 = i64x2.add(h34, t3) + h56 = i64x2.add(h56, t4) + + // f[8] + fi = v128.extract_lane(r_X_89xx, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19) + + h89 = i64x2.add(h89, t0) + h01 = i64x2.add(h01, t1) + h23 = i64x2.add(h23, t2) + h45 = i64x2.add(h45, t3) + h67 = i64x2.add(h67, t4) + + // f[9] + fi = v128.extract_lane(r_X_89xx, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YplusX_0123, q_YplusX_0123_19, m)) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19) + + h90 = i64x2.add(h90, t0) + h12 = i64x2.add(h12, t1) + h34 = i64x2.add(h34, t2) + h56 = i64x2.add(h56, t3) + h78 = i64x2.add(h78, t4) + + // extract scalars from vectors + h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0) + h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0) + h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0) + h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0) + h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0) + h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0) + h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0) + h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0) + h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0) + h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0) + + // carry + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + (1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + (1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + (1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + (1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + (1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + (1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + (1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + (1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + r_Z_0123 = i32x4(h0, h1, h2, h3) + r_Z_4567 = i32x4(h4, h5, h6, h7) + r_Z_89xx = i32x4(h8, h9, 0, 0) + + // r.Y = r.Y * q.YminusX + const q_YminusX_0123_19: v128 = i32x4.mul(q_YminusX_0123, v19) + const q_YminusX_4567_19: v128 = i32x4.mul(q_YminusX_4567, v19) + const q_YminusX_89xx_19: v128 = i32x4.mul(q_YminusX_89xx, v19) + + // f[0] + fi = v128.extract_lane(r_Y_0123, 0) + fv = i32x4.splat(fi) + + h01 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123) + h23 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123) + h45 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567) + h67 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567) + h89 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx) + + // f[1] + fi = v128.extract_lane(r_Y_0123, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + h12 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123) + h34 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123) + h56 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567) + h78 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567) + h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YminusX_89xx, q_YminusX_89xx_19, m)) + + // f[2] + fi = v128.extract_lane(r_Y_0123, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19) + + h23 = i64x2.add(h23, t0) + h45 = i64x2.add(h45, t1) + h67 = i64x2.add(h67, t2) + h89 = i64x2.add(h89, t3) + h01 = i64x2.add(h01, t4) + + // f[3] + fi = v128.extract_lane(r_Y_0123, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_YminusX_4567, q_YminusX_4567_19, m)) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19) + + h34 = i64x2.add(h34, t0) + h56 = i64x2.add(h56, t1) + h78 = i64x2.add(h78, t2) + h90 = i64x2.add(h90, t3) + h12 = i64x2.add(h12, t4) + + // f[4] + fi = v128.extract_lane(r_Y_4567, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19) + + h45 = i64x2.add(h45, t0) + h67 = i64x2.add(h67, t1) + h89 = i64x2.add(h89, t2) + h01 = i64x2.add(h01, t3) + h23 = i64x2.add(h23, t4) + + // f[5] + fi = v128.extract_lane(r_Y_4567, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YminusX_4567, q_YminusX_4567_19, m)) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19) + + h56 = i64x2.add(h56, t0) + h78 = i64x2.add(h78, t1) + h90 = i64x2.add(h90, t2) + h12 = i64x2.add(h12, t3) + h34 = i64x2.add(h34, t4) + + // f[6] + fi = v128.extract_lane(r_Y_4567, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19) + + h67 = i64x2.add(h67, t0) + h89 = i64x2.add(h89, t1) + h01 = i64x2.add(h01, t2) + h23 = i64x2.add(h23, t3) + h45 = i64x2.add(h45, t4) + + // f[7] + fi = v128.extract_lane(r_Y_4567, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_YminusX_0123, q_YminusX_0123_19, m)) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19) + + h78 = i64x2.add(h78, t0) + h90 = i64x2.add(h90, t1) + h12 = i64x2.add(h12, t2) + h34 = i64x2.add(h34, t3) + h56 = i64x2.add(h56, t4) + + // f[8] + fi = v128.extract_lane(r_Y_89xx, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19) + + h89 = i64x2.add(h89, t0) + h01 = i64x2.add(h01, t1) + h23 = i64x2.add(h23, t2) + h45 = i64x2.add(h45, t3) + h67 = i64x2.add(h67, t4) + + // f[9] + fi = v128.extract_lane(r_Y_89xx, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YminusX_0123, q_YminusX_0123_19, m)) + t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19) + + h90 = i64x2.add(h90, t0) + h12 = i64x2.add(h12, t1) + h34 = i64x2.add(h34, t2) + h56 = i64x2.add(h56, t3) + h78 = i64x2.add(h78, t4) + + // extract scalars from vectors + h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0) + h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0) + h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0) + h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0) + h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0) + h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0) + h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0) + h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0) + h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0) + h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0) + + // carry + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + (1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + (1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + (1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + (1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + (1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + (1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + (1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + (1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + r_Y_0123 = i32x4(h0, h1, h2, h3) + r_Y_4567 = i32x4(h4, h5, h6, h7) + r_Y_89xx = i32x4(h8, h9, 0, 0) + + // r.T = q.T2d * p.T + const p_T_0123_19: v128 = i32x4.mul(p_T_0123, v19) + const p_T_4567_19: v128 = i32x4.mul(p_T_4567, v19) + const p_T_89xx_19: v128 = i32x4.mul(p_T_89xx, v19) + + // f[0] + fi = v128.extract_lane(q_T2d_0123, 0) + fv = i32x4.splat(fi) + + h01 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + h23 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + h45 = i64x2.extmul_low_i32x4_s(fv, p_T_4567) + h67 = i64x2.extmul_high_i32x4_s(fv, p_T_4567) + h89 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx) + + // f[1] + fi = v128.extract_lane(q_T2d_0123, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + h12 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + h34 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + h56 = i64x2.extmul_low_i32x4_s(fv, p_T_4567) + h78 = i64x2.extmul_high_i32x4_s(fv, p_T_4567) + h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_89xx, p_T_89xx_19, m)) + + // f[2] + fi = v128.extract_lane(q_T2d_0123, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h23 = i64x2.add(h23, t0) + h45 = i64x2.add(h45, t1) + h67 = i64x2.add(h67, t2) + h89 = i64x2.add(h89, t3) + h01 = i64x2.add(h01, t4) + + // f[3] + fi = v128.extract_lane(q_T2d_0123, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(p_T_4567, p_T_4567_19, m)) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h34 = i64x2.add(h34, t0) + h56 = i64x2.add(h56, t1) + h78 = i64x2.add(h78, t2) + h90 = i64x2.add(h90, t3) + h12 = i64x2.add(h12, t4) + + // f[4] + fi = v128.extract_lane(q_T2d_4567, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h45 = i64x2.add(h45, t0) + h67 = i64x2.add(h67, t1) + h89 = i64x2.add(h89, t2) + h01 = i64x2.add(h01, t3) + h23 = i64x2.add(h23, t4) + + // f[5] + fi = v128.extract_lane(q_T2d_4567, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_4567, p_T_4567_19, m)) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h56 = i64x2.add(h56, t0) + h78 = i64x2.add(h78, t1) + h90 = i64x2.add(h90, t2) + h12 = i64x2.add(h12, t3) + h34 = i64x2.add(h34, t4) + + // f[6] + fi = v128.extract_lane(q_T2d_4567, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h67 = i64x2.add(h67, t0) + h89 = i64x2.add(h89, t1) + h01 = i64x2.add(h01, t2) + h23 = i64x2.add(h23, t3) + h45 = i64x2.add(h45, t4) + + // f[7] + fi = v128.extract_lane(q_T2d_4567, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(p_T_0123, p_T_0123_19, m)) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h78 = i64x2.add(h78, t0) + h90 = i64x2.add(h90, t1) + h12 = i64x2.add(h12, t2) + h34 = i64x2.add(h34, t3) + h56 = i64x2.add(h56, t4) + + // f[8] + fi = v128.extract_lane(q_T2d_89xx, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h89 = i64x2.add(h89, t0) + h01 = i64x2.add(h01, t1) + h23 = i64x2.add(h23, t2) + h45 = i64x2.add(h45, t3) + h67 = i64x2.add(h67, t4) + + // f[9] + fi = v128.extract_lane(q_T2d_89xx, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_0123, p_T_0123_19, m)) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h90 = i64x2.add(h90, t0) + h12 = i64x2.add(h12, t1) + h34 = i64x2.add(h34, t2) + h56 = i64x2.add(h56, t3) + h78 = i64x2.add(h78, t4) + + // extract scalars from vectors + h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0) + h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0) + h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0) + h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0) + h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0) + h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0) + h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0) + h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0) + h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0) + h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0) + + // carry + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + (1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + (1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + (1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + (1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + (1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + (1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + (1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + (1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + r_T_0123 = i32x4(h0, h1, h2, h3) + r_T_4567 = i32x4(h4, h5, h6, h7) + r_T_89xx = i32x4(h8, h9, 0, 0) + + // r.X = p.Z * q.Z + const q_Z_0123_19: v128 = i32x4.mul(q_Z_0123, v19) + const q_Z_4567_19: v128 = i32x4.mul(q_Z_4567, v19) + const q_Z_89xx_19: v128 = i32x4.mul(q_Z_89xx, v19) + + // f[0] + fi = v128.extract_lane(p_Z_0123, 0) + fv = i32x4.splat(fi) + + h01 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123) + h23 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123) + h45 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567) + h67 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567) + h89 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx) + + // f[1] + fi = v128.extract_lane(p_Z_0123, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + h12 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123) + h34 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123) + h56 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567) + h78 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567) + h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_Z_89xx, q_Z_89xx_19, m)) + + // f[2] + fi = v128.extract_lane(p_Z_0123, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567) + t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19) + + h23 = i64x2.add(h23, t0) + h45 = i64x2.add(h45, t1) + h67 = i64x2.add(h67, t2) + h89 = i64x2.add(h89, t3) + h01 = i64x2.add(h01, t4) + + // f[3] + fi = v128.extract_lane(p_Z_0123, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_Z_4567, q_Z_4567_19, m)) + t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19) + + h34 = i64x2.add(h34, t0) + h56 = i64x2.add(h56, t1) + h78 = i64x2.add(h78, t2) + h90 = i64x2.add(h90, t3) + h12 = i64x2.add(h12, t4) + + // f[4] + fi = v128.extract_lane(p_Z_4567, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19) + + h45 = i64x2.add(h45, t0) + h67 = i64x2.add(h67, t1) + h89 = i64x2.add(h89, t2) + h01 = i64x2.add(h01, t3) + h23 = i64x2.add(h23, t4) + + // f[5] + fi = v128.extract_lane(p_Z_4567, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_Z_4567, q_Z_4567_19, m)) + t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19) + + h56 = i64x2.add(h56, t0) + h78 = i64x2.add(h78, t1) + h90 = i64x2.add(h90, t2) + h12 = i64x2.add(h12, t3) + h34 = i64x2.add(h34, t4) + + // f[6] + fi = v128.extract_lane(p_Z_4567, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19) + + h67 = i64x2.add(h67, t0) + h89 = i64x2.add(h89, t1) + h01 = i64x2.add(h01, t2) + h23 = i64x2.add(h23, t3) + h45 = i64x2.add(h45, t4) + + // f[7] + fi = v128.extract_lane(p_Z_4567, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_Z_0123, q_Z_0123_19, m)) + t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19) + + h78 = i64x2.add(h78, t0) + h90 = i64x2.add(h90, t1) + h12 = i64x2.add(h12, t2) + h34 = i64x2.add(h34, t3) + h56 = i64x2.add(h56, t4) + + // f[8] + fi = v128.extract_lane(p_Z_89xx, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19) + + h89 = i64x2.add(h89, t0) + h01 = i64x2.add(h01, t1) + h23 = i64x2.add(h23, t2) + h45 = i64x2.add(h45, t3) + h67 = i64x2.add(h67, t4) + + // f[9] + fi = v128.extract_lane(p_Z_89xx, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_Z_0123, q_Z_0123_19, m)) + t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19) + + h90 = i64x2.add(h90, t0) + h12 = i64x2.add(h12, t1) + h34 = i64x2.add(h34, t2) + h56 = i64x2.add(h56, t3) + h78 = i64x2.add(h78, t4) + + // extract scalars from vectors + h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0) + h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0) + h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0) + h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0) + h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0) + h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0) + h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0) + h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0) + h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0) + h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0) + + // carry + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + (1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + (1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + (1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + (1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + (1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + (1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + (1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + (1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + r_X_0123 = i32x4(h0, h1, h2, h3) + r_X_4567 = i32x4(h4, h5, h6, h7) + r_X_89xx = i32x4(h8, h9, 0, 0) + + let t_0123: v128, t_4567: v128, t_89xx: v128 + + // t = 2*r.X + t_0123 = v128.shl(r_X_0123, 1) + t_4567 = v128.shl(r_X_4567, 1) + t_89xx = v128.shl(r_X_89xx, 1) + + // r.X = r.Z - r.Y + r_X_0123 = v128.sub(r_Z_0123, r_Y_0123) + r_X_4567 = v128.sub(r_Z_4567, r_Y_4567) + r_X_89xx = v128.sub(r_Z_89xx, r_Y_89xx) + + // r.Y = r.Z + r.Y + r_Y_0123 = v128.add(r_Z_0123, r_Y_0123) + r_Y_4567 = v128.add(r_Z_4567, r_Y_4567) + r_Y_89xx = v128.add(r_Z_89xx, r_Y_89xx) + + // r.Z = t + r.T + r_Z_0123 = v128.add(t_0123, r_T_0123) + r_Z_4567 = v128.add(t_4567, r_T_4567) + r_Z_89xx = v128.add(t_89xx, r_T_89xx) + + // r.T = t - r.T + r_T_0123 = v128.sub(t_0123, r_T_0123) + r_T_4567 = v128.sub(t_4567, r_T_4567) + r_T_89xx = v128.sub(t_89xx, r_T_89xx) + + // r.X = [0..3, 4..7, 8..11] + v128.store(changetype(r.X), r_X_0123, 0) + v128.store(changetype(r.X), r_X_4567, 16) + v128.store_lane(changetype(r.X), r_X_89xx, 0, 32) + + // r.Y = [0..3, 4..7, 8..11] + v128.store(changetype(r.Y), r_Y_0123, 0) + v128.store(changetype(r.Y), r_Y_4567, 16) + v128.store_lane(changetype(r.Y), r_Y_89xx, 0, 32) + + // r.Z = [0..3, 4..7, 8..11] + v128.store(changetype(r.Z), r_Z_0123, 0) + v128.store(changetype(r.Z), r_Z_4567, 16) + v128.store_lane(changetype(r.Z), r_Z_89xx, 0, 32) + + // r.T = [0..3, 4..7, 8..11] + v128.store(changetype(r.T), r_T_0123, 0) + v128.store(changetype(r.T), r_T_4567, 16) + v128.store_lane(changetype(r.T), r_T_89xx, 0, 32) + } /** @@ -322,99 +2082,902 @@ export function ge_add_cached (r: ge_p1p1, p: ge_p3, q: ge_cached): void { //@ts-expect-error @inline export function ge_add_precomp (r: ge_p1p1, p: ge_p3, q: ge_precomp): void { - const rX_ptr = changetype(r.X) - const rY_ptr = changetype(r.Y) - const rZ_ptr = changetype(r.Z) - const rT_ptr = changetype(r.T) - - const pX_ptr = changetype(p.X) - const pY_ptr = changetype(p.Y) - const pZ_ptr = changetype(p.Z) - - let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32) - let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32) - const pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32) - const pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32) - - // rX = pY + pX - rX0 = v128.add(pY0, pX0) - rX1 = v128.add(pY1, pX1) - rX2 = v128.add(pY2, pX2) - - // rY = pY - pX - rY0 = v128.sub(pY0, pX0) - rY1 = v128.sub(pY1, pX1) - rY2 = v128.sub(pY2, pX2) - - // store r.X - v128.store(rX_ptr, rX0) - v128.store(rX_ptr, rX1, 16) - v128.store_lane(rX_ptr, rX2, 0, 32) - - // store r.Y - v128.store(rY_ptr, rY0) - v128.store(rY_ptr, rY1, 16) - v128.store_lane(rY_ptr, rY2, 0, 32) - - // not yet converted to inline for size and complexity - fe_mul(r.Z, r.X, q.yplusx) - fe_mul(r.Y, r.Y, q.yminusx) - fe_mul(r.T, q.xy2d, p.T) - // remaining arithmetic done inline to avoid unnecessary load/store - rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32) - rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32) - let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32) - let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32) - - let pZ0 = v128.load(pZ_ptr, 0) - let pZ1 = v128.load(pZ_ptr, 16) - let pZ2 = v128.load(pZ_ptr, 32) - - // pZ = 2*pZ - pZ0 = v128.shl(pZ0, 1) - pZ1 = v128.shl(pZ1, 1) - pZ2 = v128.shl(pZ2, 1) - - // rX = rZ - rY - rX0 = v128.sub(rZ0, rY0) - rX1 = v128.sub(rZ1, rY1) - rX2 = v128.sub(rZ2, rY2) - - // rY = rZ + rY - rY0 = v128.add(rZ0, rY0) - rY1 = v128.add(rZ1, rY1) - rY2 = v128.add(rZ2, rY2) - - // rZ = pZ + rT - rZ0 = v128.add(pZ0, rT0) - rZ1 = v128.add(pZ1, rT1) - rZ2 = v128.add(pZ2, rT2) - - // rT = pZ - rT - rT0 = v128.sub(pZ0, rT0) - rT1 = v128.sub(pZ1, rT1) - rT2 = v128.sub(pZ2, rT2) - - // store r.X - v128.store(rX_ptr, rX0) - v128.store(rX_ptr, rX1, 16) - v128.store_lane(rX_ptr, rX2, 0, 32) - - // store r.Y - v128.store(rY_ptr, rY0) - v128.store(rY_ptr, rY1, 16) - v128.store_lane(rY_ptr, rY2, 0, 32) - - // store r.Z - v128.store(rZ_ptr, rZ0) - v128.store(rZ_ptr, rZ1, 16) - v128.store_lane(rZ_ptr, rZ2, 0, 32) - - // store r.T - v128.store(rT_ptr, rT0) - v128.store(rT_ptr, rT1, 16) - v128.store_lane(rT_ptr, rT2, 0, 32) + let + r_X_0123: v128, + r_X_4567: v128, + r_X_89xx: v128 +, + r_Y_0123: v128, + r_Y_4567: v128, + r_Y_89xx: v128 +, + r_Z_0123: v128, + r_Z_4567: v128, + r_Z_89xx: v128 +, + r_T_0123: v128, + r_T_4567: v128, + r_T_89xx: v128 +, + p_X_0123: v128, + p_X_4567: v128, + p_X_89xx: v128 +, + p_Y_0123: v128, + p_Y_4567: v128, + p_Y_89xx: v128 +, + p_Z_0123: v128, + p_Z_4567: v128, + p_Z_89xx: v128 +, + p_T_0123: v128, + p_T_4567: v128, + p_T_89xx: v128 +, + q_yplusx_0123: v128, + q_yplusx_4567: v128, + q_yplusx_89xx: v128 +, + q_yminusx_0123: v128, + q_yminusx_4567: v128, + q_yminusx_89xx: v128 +, + q_xy2d_0123: v128, + q_xy2d_4567: v128, + q_xy2d_89xx: v128 + + // mask to select even lane from first vector and odd lane from second + const m: v128 = i32x4(-1, 0, -1, 0) + + // mask to double odd-on-odd indexes + const odds: v128 = i32x4(0, -1, 0, -1) + + // factor of 19 for wrap from h9 to h0 + const v19: v128 = i32x4.splat(19) + + let c0: i64 + let c1: i64 + let c2: i64 + let c3: i64 + let c4: i64 + let c5: i64 + let c6: i64 + let c7: i64 + let c8: i64 + let c9: i64 + + let f0: i64 + let f1: i64 + let f2: i64 + let f3: i64 + let f4: i64 + let f5: i64 + let f6: i64 + let f7: i64 + let f8: i64 + let f9: i64 + + let f0_2: i64 + let f1_2: i64 + let f2_2: i64 + let f3_2: i64 + let f4_2: i64 + let f5_2: i64 + let f6_2: i64 + let f7_2: i64 + let f8_2: i64 + let f9_2: i64 + + let f5_19: i64 + let f6_19: i64 + let f7_19: i64 + let f8_19: i64 + let f9_19: i64 + + let f5_38: i64 + let f7_38: i64 + let f9_38: i64 + + let fi: i32 + let fv: v128 + + let h0: i64 + let h1: i64 + let h2: i64 + let h3: i64 + let h4: i64 + let h5: i64 + let h6: i64 + let h7: i64 + let h8: i64 + let h9: i64 + + let h01: v128 + let h12: v128 + let h23: v128 + let h34: v128 + let h45: v128 + let h56: v128 + let h67: v128 + let h78: v128 + let h89: v128 + let h90: v128 + + let t0: v128 + let t1: v128 + let t2: v128 + let t3: v128 + let t4: v128 + + // [0..3, 4..7, 8..11] = r.X + r_X_0123 = v128.load(changetype(r.X), 0) + r_X_4567 = v128.load(changetype(r.X), 16) + r_X_89xx = v128.load(changetype(r.X), 32) + + // [0..3, 4..7, 8..11] = r.Y + r_Y_0123 = v128.load(changetype(r.Y), 0) + r_Y_4567 = v128.load(changetype(r.Y), 16) + r_Y_89xx = v128.load(changetype(r.Y), 32) + + // [0..3, 4..7, 8..11] = r.Z + r_Z_0123 = v128.load(changetype(r.Z), 0) + r_Z_4567 = v128.load(changetype(r.Z), 16) + r_Z_89xx = v128.load(changetype(r.Z), 32) + + // [0..3, 4..7, 8..11] = r.T + r_T_0123 = v128.load(changetype(r.T), 0) + r_T_4567 = v128.load(changetype(r.T), 16) + r_T_89xx = v128.load(changetype(r.T), 32) + + // [0..3, 4..7, 8..11] = p.X + p_X_0123 = v128.load(changetype(p.X), 0) + p_X_4567 = v128.load(changetype(p.X), 16) + p_X_89xx = v128.load(changetype(p.X), 32) + + // [0..3, 4..7, 8..11] = p.Y + p_Y_0123 = v128.load(changetype(p.Y), 0) + p_Y_4567 = v128.load(changetype(p.Y), 16) + p_Y_89xx = v128.load(changetype(p.Y), 32) + + // [0..3, 4..7, 8..11] = p.Z + p_Z_0123 = v128.load(changetype(p.Z), 0) + p_Z_4567 = v128.load(changetype(p.Z), 16) + p_Z_89xx = v128.load(changetype(p.Z), 32) + + // [0..3, 4..7, 8..11] = p.T + p_T_0123 = v128.load(changetype(p.T), 0) + p_T_4567 = v128.load(changetype(p.T), 16) + p_T_89xx = v128.load(changetype(p.T), 32) + + // [0..3, 4..7, 8..11] = q.yplusx + q_yplusx_0123 = v128.load(changetype(q.yplusx), 0) + q_yplusx_4567 = v128.load(changetype(q.yplusx), 16) + q_yplusx_89xx = v128.load(changetype(q.yplusx), 32) + + // [0..3, 4..7, 8..11] = q.yminusx + q_yminusx_0123 = v128.load(changetype(q.yminusx), 0) + q_yminusx_4567 = v128.load(changetype(q.yminusx), 16) + q_yminusx_89xx = v128.load(changetype(q.yminusx), 32) + + // [0..3, 4..7, 8..11] = q.xy2d + q_xy2d_0123 = v128.load(changetype(q.xy2d), 0) + q_xy2d_4567 = v128.load(changetype(q.xy2d), 16) + q_xy2d_89xx = v128.load(changetype(q.xy2d), 32) + + // r.X = p.Y + p.X + r_X_0123 = v128.add(p_Y_0123, p_X_0123) + r_X_4567 = v128.add(p_Y_4567, p_X_4567) + r_X_89xx = v128.add(p_Y_89xx, p_X_89xx) + + // r.Y = p.Y - p.X + r_Y_0123 = v128.sub(p_Y_0123, p_X_0123) + r_Y_4567 = v128.sub(p_Y_4567, p_X_4567) + r_Y_89xx = v128.sub(p_Y_89xx, p_X_89xx) + + // r.Z = r.X * q.yplusx + const q_yplusx_0123_19: v128 = i32x4.mul(q_yplusx_0123, v19) + const q_yplusx_4567_19: v128 = i32x4.mul(q_yplusx_4567, v19) + const q_yplusx_89xx_19: v128 = i32x4.mul(q_yplusx_89xx, v19) + + // f[0] + fi = v128.extract_lane(r_X_0123, 0) + fv = i32x4.splat(fi) + + h01 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123) + h23 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123) + h45 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567) + h67 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567) + h89 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx) + + // f[1] + fi = v128.extract_lane(r_X_0123, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + h12 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123) + h34 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123) + h56 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567) + h78 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567) + h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yplusx_89xx, q_yplusx_89xx_19, m)) + + // f[2] + fi = v128.extract_lane(r_X_0123, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19) + + h23 = i64x2.add(h23, t0) + h45 = i64x2.add(h45, t1) + h67 = i64x2.add(h67, t2) + h89 = i64x2.add(h89, t3) + h01 = i64x2.add(h01, t4) + + // f[3] + fi = v128.extract_lane(r_X_0123, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_yplusx_4567, q_yplusx_4567_19, m)) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19) + + h34 = i64x2.add(h34, t0) + h56 = i64x2.add(h56, t1) + h78 = i64x2.add(h78, t2) + h90 = i64x2.add(h90, t3) + h12 = i64x2.add(h12, t4) + + // f[4] + fi = v128.extract_lane(r_X_4567, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19) + + h45 = i64x2.add(h45, t0) + h67 = i64x2.add(h67, t1) + h89 = i64x2.add(h89, t2) + h01 = i64x2.add(h01, t3) + h23 = i64x2.add(h23, t4) + + // f[5] + fi = v128.extract_lane(r_X_4567, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yplusx_4567, q_yplusx_4567_19, m)) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19) + + h56 = i64x2.add(h56, t0) + h78 = i64x2.add(h78, t1) + h90 = i64x2.add(h90, t2) + h12 = i64x2.add(h12, t3) + h34 = i64x2.add(h34, t4) + + // f[6] + fi = v128.extract_lane(r_X_4567, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19) + + h67 = i64x2.add(h67, t0) + h89 = i64x2.add(h89, t1) + h01 = i64x2.add(h01, t2) + h23 = i64x2.add(h23, t3) + h45 = i64x2.add(h45, t4) + + // f[7] + fi = v128.extract_lane(r_X_4567, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_yplusx_0123, q_yplusx_0123_19, m)) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19) + + h78 = i64x2.add(h78, t0) + h90 = i64x2.add(h90, t1) + h12 = i64x2.add(h12, t2) + h34 = i64x2.add(h34, t3) + h56 = i64x2.add(h56, t4) + + // f[8] + fi = v128.extract_lane(r_X_89xx, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19) + + h89 = i64x2.add(h89, t0) + h01 = i64x2.add(h01, t1) + h23 = i64x2.add(h23, t2) + h45 = i64x2.add(h45, t3) + h67 = i64x2.add(h67, t4) + + // f[9] + fi = v128.extract_lane(r_X_89xx, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yplusx_0123, q_yplusx_0123_19, m)) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19) + + h90 = i64x2.add(h90, t0) + h12 = i64x2.add(h12, t1) + h34 = i64x2.add(h34, t2) + h56 = i64x2.add(h56, t3) + h78 = i64x2.add(h78, t4) + + // extract scalars from vectors + h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0) + h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0) + h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0) + h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0) + h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0) + h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0) + h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0) + h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0) + h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0) + h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0) + + // carry + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + (1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + (1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + (1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + (1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + (1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + (1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + (1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + (1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + r_Z_0123 = i32x4(h0, h1, h2, h3) + r_Z_4567 = i32x4(h4, h5, h6, h7) + r_Z_89xx = i32x4(h8, h9, 0, 0) + + // r.Y = r.Y * q.yminusx + const q_yminusx_0123_19: v128 = i32x4.mul(q_yminusx_0123, v19) + const q_yminusx_4567_19: v128 = i32x4.mul(q_yminusx_4567, v19) + const q_yminusx_89xx_19: v128 = i32x4.mul(q_yminusx_89xx, v19) + + // f[0] + fi = v128.extract_lane(r_Y_0123, 0) + fv = i32x4.splat(fi) + + h01 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123) + h23 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123) + h45 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567) + h67 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567) + h89 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx) + + // f[1] + fi = v128.extract_lane(r_Y_0123, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + h12 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123) + h34 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123) + h56 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567) + h78 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567) + h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yminusx_89xx, q_yminusx_89xx_19, m)) + + // f[2] + fi = v128.extract_lane(r_Y_0123, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19) + + h23 = i64x2.add(h23, t0) + h45 = i64x2.add(h45, t1) + h67 = i64x2.add(h67, t2) + h89 = i64x2.add(h89, t3) + h01 = i64x2.add(h01, t4) + + // f[3] + fi = v128.extract_lane(r_Y_0123, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_yminusx_4567, q_yminusx_4567_19, m)) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19) + + h34 = i64x2.add(h34, t0) + h56 = i64x2.add(h56, t1) + h78 = i64x2.add(h78, t2) + h90 = i64x2.add(h90, t3) + h12 = i64x2.add(h12, t4) + + // f[4] + fi = v128.extract_lane(r_Y_4567, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19) + + h45 = i64x2.add(h45, t0) + h67 = i64x2.add(h67, t1) + h89 = i64x2.add(h89, t2) + h01 = i64x2.add(h01, t3) + h23 = i64x2.add(h23, t4) + + // f[5] + fi = v128.extract_lane(r_Y_4567, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yminusx_4567, q_yminusx_4567_19, m)) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19) + + h56 = i64x2.add(h56, t0) + h78 = i64x2.add(h78, t1) + h90 = i64x2.add(h90, t2) + h12 = i64x2.add(h12, t3) + h34 = i64x2.add(h34, t4) + + // f[6] + fi = v128.extract_lane(r_Y_4567, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19) + + h67 = i64x2.add(h67, t0) + h89 = i64x2.add(h89, t1) + h01 = i64x2.add(h01, t2) + h23 = i64x2.add(h23, t3) + h45 = i64x2.add(h45, t4) + + // f[7] + fi = v128.extract_lane(r_Y_4567, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_yminusx_0123, q_yminusx_0123_19, m)) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19) + + h78 = i64x2.add(h78, t0) + h90 = i64x2.add(h90, t1) + h12 = i64x2.add(h12, t2) + h34 = i64x2.add(h34, t3) + h56 = i64x2.add(h56, t4) + + // f[8] + fi = v128.extract_lane(r_Y_89xx, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19) + + h89 = i64x2.add(h89, t0) + h01 = i64x2.add(h01, t1) + h23 = i64x2.add(h23, t2) + h45 = i64x2.add(h45, t3) + h67 = i64x2.add(h67, t4) + + // f[9] + fi = v128.extract_lane(r_Y_89xx, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yminusx_0123, q_yminusx_0123_19, m)) + t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19) + + h90 = i64x2.add(h90, t0) + h12 = i64x2.add(h12, t1) + h34 = i64x2.add(h34, t2) + h56 = i64x2.add(h56, t3) + h78 = i64x2.add(h78, t4) + + // extract scalars from vectors + h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0) + h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0) + h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0) + h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0) + h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0) + h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0) + h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0) + h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0) + h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0) + h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0) + + // carry + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + (1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + (1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + (1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + (1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + (1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + (1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + (1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + (1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + r_Y_0123 = i32x4(h0, h1, h2, h3) + r_Y_4567 = i32x4(h4, h5, h6, h7) + r_Y_89xx = i32x4(h8, h9, 0, 0) + + // r.T = q.xy2d * p.T + const p_T_0123_19: v128 = i32x4.mul(p_T_0123, v19) + const p_T_4567_19: v128 = i32x4.mul(p_T_4567, v19) + const p_T_89xx_19: v128 = i32x4.mul(p_T_89xx, v19) + + // f[0] + fi = v128.extract_lane(q_xy2d_0123, 0) + fv = i32x4.splat(fi) + + h01 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + h23 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + h45 = i64x2.extmul_low_i32x4_s(fv, p_T_4567) + h67 = i64x2.extmul_high_i32x4_s(fv, p_T_4567) + h89 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx) + + // f[1] + fi = v128.extract_lane(q_xy2d_0123, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + h12 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + h34 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + h56 = i64x2.extmul_low_i32x4_s(fv, p_T_4567) + h78 = i64x2.extmul_high_i32x4_s(fv, p_T_4567) + h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_89xx, p_T_89xx_19, m)) + + // f[2] + fi = v128.extract_lane(q_xy2d_0123, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h23 = i64x2.add(h23, t0) + h45 = i64x2.add(h45, t1) + h67 = i64x2.add(h67, t2) + h89 = i64x2.add(h89, t3) + h01 = i64x2.add(h01, t4) + + // f[3] + fi = v128.extract_lane(q_xy2d_0123, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(p_T_4567, p_T_4567_19, m)) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h34 = i64x2.add(h34, t0) + h56 = i64x2.add(h56, t1) + h78 = i64x2.add(h78, t2) + h90 = i64x2.add(h90, t3) + h12 = i64x2.add(h12, t4) + + // f[4] + fi = v128.extract_lane(q_xy2d_4567, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h45 = i64x2.add(h45, t0) + h67 = i64x2.add(h67, t1) + h89 = i64x2.add(h89, t2) + h01 = i64x2.add(h01, t3) + h23 = i64x2.add(h23, t4) + + // f[5] + fi = v128.extract_lane(q_xy2d_4567, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_4567, p_T_4567_19, m)) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h56 = i64x2.add(h56, t0) + h78 = i64x2.add(h78, t1) + h90 = i64x2.add(h90, t2) + h12 = i64x2.add(h12, t3) + h34 = i64x2.add(h34, t4) + + // f[6] + fi = v128.extract_lane(q_xy2d_4567, 2) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h67 = i64x2.add(h67, t0) + h89 = i64x2.add(h89, t1) + h01 = i64x2.add(h01, t2) + h23 = i64x2.add(h23, t3) + h45 = i64x2.add(h45, t4) + + // f[7] + fi = v128.extract_lane(q_xy2d_4567, 3) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(p_T_0123, p_T_0123_19, m)) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h78 = i64x2.add(h78, t0) + h90 = i64x2.add(h90, t1) + h12 = i64x2.add(h12, t2) + h34 = i64x2.add(h34, t3) + h56 = i64x2.add(h56, t4) + + // f[8] + fi = v128.extract_lane(q_xy2d_89xx, 0) + fv = i32x4.splat(fi) + + t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h89 = i64x2.add(h89, t0) + h01 = i64x2.add(h01, t1) + h23 = i64x2.add(h23, t2) + h45 = i64x2.add(h45, t3) + h67 = i64x2.add(h67, t4) + + // f[9] + fi = v128.extract_lane(q_xy2d_89xx, 1) + fv = i32x4.splat(fi) + fv = i32x4.add(fv, v128.and(fv, odds)) + + t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_0123, p_T_0123_19, m)) + t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123_19) + t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19) + t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19) + t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19) + + h90 = i64x2.add(h90, t0) + h12 = i64x2.add(h12, t1) + h34 = i64x2.add(h34, t2) + h56 = i64x2.add(h56, t3) + h78 = i64x2.add(h78, t4) + + // extract scalars from vectors + h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0) + h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0) + h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0) + h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0) + h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0) + h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0) + h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0) + h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0) + h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0) + h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0) + + // carry + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + + c1 = (h1 + (1 << 24)) >> 25 + h2 += c1 + h1 -= c1 << 25 + c5 = (h5 + (1 << 24)) >> 25 + h6 += c5 + h5 -= c5 << 25 + + c2 = (h2 + (1 << 25)) >> 26 + h3 += c2 + h2 -= c2 << 26 + c6 = (h6 + (1 << 25)) >> 26 + h7 += c6 + h6 -= c6 << 26 + + c3 = (h3 + (1 << 24)) >> 25 + h4 += c3 + h3 -= c3 << 25 + c7 = (h7 + (1 << 24)) >> 25 + h8 += c7 + h7 -= c7 << 25 + + c4 = (h4 + (1 << 25)) >> 26 + h5 += c4 + h4 -= c4 << 26 + c8 = (h8 + (1 << 25)) >> 26 + h9 += c8 + h8 -= c8 << 26 + + c9 = (h9 + (1 << 24)) >> 25 + h0 += c9 * 19 + h9 -= c9 << 25 + + c0 = (h0 + (1 << 25)) >> 26 + h1 += c0 + h0 -= c0 << 26 + + // assign reduced results to output + r_T_0123 = i32x4(h0, h1, h2, h3) + r_T_4567 = i32x4(h4, h5, h6, h7) + r_T_89xx = i32x4(h8, h9, 0, 0) + + // p.Z = 2*p.Z + p_Z_0123 = v128.shl(p_Z_0123, 1) + p_Z_4567 = v128.shl(p_Z_4567, 1) + p_Z_89xx = v128.shl(p_Z_89xx, 1) + + // r.X = r.Z - r.Y + r_X_0123 = v128.sub(r_Z_0123, r_Y_0123) + r_X_4567 = v128.sub(r_Z_4567, r_Y_4567) + r_X_89xx = v128.sub(r_Z_89xx, r_Y_89xx) + + // r.Y = r.Z + r.Y + r_Y_0123 = v128.add(r_Z_0123, r_Y_0123) + r_Y_4567 = v128.add(r_Z_4567, r_Y_4567) + r_Y_89xx = v128.add(r_Z_89xx, r_Y_89xx) + + // r.Z = p.Z + r.T + r_Z_0123 = v128.add(p_Z_0123, r_T_0123) + r_Z_4567 = v128.add(p_Z_4567, r_T_4567) + r_Z_89xx = v128.add(p_Z_89xx, r_T_89xx) + + // r.T = p.Z - r.T + r_T_0123 = v128.sub(p_Z_0123, r_T_0123) + r_T_4567 = v128.sub(p_Z_4567, r_T_4567) + r_T_89xx = v128.sub(p_Z_89xx, r_T_89xx) + + // r.X = [0..3, 4..7, 8..11] + v128.store(changetype(r.X), r_X_0123, 0) + v128.store(changetype(r.X), r_X_4567, 16) + v128.store_lane(changetype(r.X), r_X_89xx, 0, 32) + + // r.Y = [0..3, 4..7, 8..11] + v128.store(changetype(r.Y), r_Y_0123, 0) + v128.store(changetype(r.Y), r_Y_4567, 16) + v128.store_lane(changetype(r.Y), r_Y_89xx, 0, 32) + + // r.Z = [0..3, 4..7, 8..11] + v128.store(changetype(r.Z), r_Z_0123, 0) + v128.store(changetype(r.Z), r_Z_4567, 16) + v128.store_lane(changetype(r.Z), r_Z_89xx, 0, 32) + + // r.T = [0..3, 4..7, 8..11] + v128.store(changetype(r.T), r_T_0123, 0) + v128.store(changetype(r.T), r_T_4567, 16) + v128.store_lane(changetype(r.T), r_T_89xx, 0, 32) + } const ge_sub_p3_q_cached = new ge_cached() -- 2.52.0