]> git.codecow.com Git - nano25519.git/commitdiff
Start converting to generated arithmetic.
authorChris Duncan <chris@codecow.com>
Wed, 30 Sep 2026 21:12:42 +0000 (14:12 -0700)
committerChris Duncan <chris@codecow.com>
Wed, 30 Sep 2026 21:12:42 +0000 (14:12 -0700)
scripts/fe.gen.mjs
scripts/p.gen.mjs
src/assembly/ed25519/fe.ts
src/assembly/ed25519/p.ts

index 5bd688b24d62d23f76d438bb38a3c817e03419bc..6a36597ad36b6f64eb0a30ae8e39e0117e0f4638 100644 (file)
@@ -9,6 +9,117 @@
  * AssemblyScript compiler and type checker.
  */
 
+/**
+ * Initialize common algorithm variables once per group element operation.
+ *
+ * @param {string[]} fe List of FieldElement variable names to initialize
+ * @returns {string} Variable initializing statements
+ */
+export function fe_init_v128 (...fe) {
+       return `
+       let ${fe.map(fe => `
+       ${fe.replaceAll('.', '_')}_0123: v128,
+       ${fe.replaceAll('.', '_')}_4567: v128,
+       ${fe.replaceAll('.', '_')}_89xx: v128
+`)}
+`
+}
+
+/**
+ * Initialize common algorithm variables once per group element operation.
+ *
+ * @param {string[]} fe List of FieldElement variable names to initialize
+ * @returns {string} Variable initializing statements
+ */
+export function fe_init (...fe) {
+       return `
+       ${fe_init_v128(...fe)}
+
+       // mask to select even lane from first vector and odd lane from second
+       const m: v128 = i32x4(-1, 0, -1, 0)
+
+       // mask to double odd-on-odd indexes
+       const odds: v128 = i32x4(0, -1, 0, -1)
+
+       // factor of 19 for wrap from h9 to h0
+       const v19: v128 = i32x4.splat(19)
+
+       let c0: i64
+       let c1: i64
+       let c2: i64
+       let c3: i64
+       let c4: i64
+       let c5: i64
+       let c6: i64
+       let c7: i64
+       let c8: i64
+       let c9: i64
+
+       let f0: i64
+       let f1: i64
+       let f2: i64
+       let f3: i64
+       let f4: i64
+       let f5: i64
+       let f6: i64
+       let f7: i64
+       let f8: i64
+       let f9: i64
+
+       let f0_2: i64
+       let f1_2: i64
+       let f2_2: i64
+       let f3_2: i64
+       let f4_2: i64
+       let f5_2: i64
+       let f6_2: i64
+       let f7_2: i64
+       let f8_2: i64
+       let f9_2: i64
+
+       let f5_19: i64
+       let f6_19: i64
+       let f7_19: i64
+       let f8_19: i64
+       let f9_19: i64
+
+       let f5_38: i64
+       let f7_38: i64
+       let f9_38: i64
+
+       let fi: i32
+       let fv: v128
+
+       let h0: i64
+       let h1: i64
+       let h2: i64
+       let h3: i64
+       let h4: i64
+       let h5: i64
+       let h6: i64
+       let h7: i64
+       let h8: i64
+       let h9: i64
+
+       let h01: v128
+       let h12: v128
+       let h23: v128
+       let h34: v128
+       let h45: v128
+       let h56: v128
+       let h67: v128
+       let h78: v128
+       let h89: v128
+       let h90: v128
+
+       let t0: v128
+       let t1: v128
+       let t2: v128
+       let t3: v128
+       let t4: v128
+`
+}
+
 /**
  * Add two FieldElements and store the sum.
  *
  * @returns {string} Three vector addition commands
  */
 export function fe_add (h, f, g) {
+       const h_name = h.replaceAll('.', '_')
+       const f_name = f.replaceAll('.', '_')
+       const g_name = g.replaceAll('.', '_')
        return `
        // ${h} = ${f} + ${g}
-       ${h}0 = v128.add<i32>(${f}0, ${g}0)
-       ${h}1 = v128.add<i32>(${f}1, ${g}1)
-       ${h}2 = v128.add<i32>(${f}2, ${g}2)
+       ${h_name}_0123 = v128.add<i32>(${f_name}_0123, ${g_name}_0123)
+       ${h_name}_4567 = v128.add<i32>(${f_name}_4567, ${g_name}_4567)
+       ${h_name}_89xx = v128.add<i32>(${f_name}_89xx, ${g_name}_89xx)
 `
 }
 
 /**
- * Subtract a FieldElement another and store the result.
+ * Add a FieldElement to itself and store the sum.
  *
- * @param {string} h Variable name of FieldElement difference destination
- * @param {string} f Variable name of FieldElement minuend source
- * @param {string} g Variable name of FieldElement subtrahend source
- * @returns {string} Three vector subtraction commands
+ * @param {string} h Variable name of FieldElement sum destination
+ * @param {string} f Variable name of FieldElement summand source
+ * @returns {string} Three vector doubling commands
  */
-export function fe_sub (h, f, g) {
+export function fe_dbl (h, f) {
+       const h_name = h.replaceAll('.', '_')
+       const f_name = f.replaceAll('.', '_')
        return `
-       // ${h} = ${f} - ${g}
-       ${h}0 = v128.sub<i32>(${f}0, ${g}0)
-       ${h}1 = v128.sub<i32>(${f}1, ${g}1)
-       ${h}2 = v128.sub<i32>(${f}2, ${g}2)
+       // ${h} = 2*${f}
+       ${h_name}_0123 = v128.shl<i32>(${f_name}_0123, 1)
+       ${h_name}_4567 = v128.shl<i32>(${f_name}_4567, 1)
+       ${h_name}_89xx = v128.shl<i32>(${f_name}_89xx, 1)
 `
 }
 
 /**
- * Add a FieldElement to itself and store the sum.
+ * Loads a FieldElement from memory into named vectors.
  *
- * @param {string} h Variable name of FieldElement sum destination
- * @param {string} f Variable name of FieldElement summand source
- * @returns {string} Three vector doubling commands
+ * @param {string} f Variable name of FieldElement source
+ * @returns {string} Three vectors
  */
-export function fe_dbl (h, f) {
+export function fe_load (f) {
+       const f_name = f.replaceAll('.', '_')
+       const f_ptr = `changetype<usize>(${f})`
        return `
-       // ${h} = 2*${f}
-       ${h}0 = v128.shl<i32>(${f}0, 1)
-       ${h}1 = v128.shl<i32>(${f}1, 1)
-       ${h}2 = v128.shl<i32>(${f}2, 1)
+       // [0..3, 4..7, 8..11] = ${f}
+       ${f_name}_0123 = v128.load(${f_ptr}, 0)
+       ${f_name}_4567 = v128.load(${f_ptr}, 16)
+       ${f_name}_89xx = v128.load(${f_ptr}, 32)
+`
+}
+
+/**
+ * Multiply two FieldElements and store the product.
+ *
+ * @param {string} h Variable name of FieldElement product destination
+ * @param {string} f Variable name of FieldElement multiplicand source
+ * @param {string} g Variable name of FieldElement multiplicand source
+ * @returns {string} Field multiplication algorithm
+ */
+export function fe_mul (h, f, g) {
+       const h_name = h.replaceAll('.', '_')
+       const f_name = f.replaceAll('.', '_')
+       const g_name = g.replaceAll('.', '_')
+       return `
+       // ${h} = ${f} * ${g}
+       const ${g_name}_0123_19: v128 = i32x4.mul(${g_name}_0123, v19)
+       const ${g_name}_4567_19: v128 = i32x4.mul(${g_name}_4567, v19)
+       const ${g_name}_89xx_19: v128 = i32x4.mul(${g_name}_89xx, v19)
+
+       // f[0]
+       fi = v128.extract_lane<i32>(${f_name}_0123, 0)
+       fv = i32x4.splat(fi)
+
+       h01 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+       h23 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+       h45 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567)
+       h67 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567)
+       h89 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx)
+
+       // f[1]
+       fi = v128.extract_lane<i32>(${f_name}_0123, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       h12 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+       h34 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+       h56 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567)
+       h78 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567)
+       h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(${g_name}_89xx, ${g_name}_89xx_19, m))
+
+       // f[2]
+       fi = v128.extract_lane<i32>(${f_name}_0123, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567)
+       t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+       h23 = i64x2.add(h23, t0)
+       h45 = i64x2.add(h45, t1)
+       h67 = i64x2.add(h67, t2)
+       h89 = i64x2.add(h89, t3)
+       h01 = i64x2.add(h01, t4)
+
+       // f[3]
+       fi = v128.extract_lane<i32>(${f_name}_0123, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(${g_name}_4567, ${g_name}_4567_19, m))
+       t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+       h34 = i64x2.add(h34, t0)
+       h56 = i64x2.add(h56, t1)
+       h78 = i64x2.add(h78, t2)
+       h90 = i64x2.add(h90, t3)
+       h12 = i64x2.add(h12, t4)
+
+       // f[4]
+       fi = v128.extract_lane<i32>(${f_name}_4567, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+       h45 = i64x2.add(h45, t0)
+       h67 = i64x2.add(h67, t1)
+       h89 = i64x2.add(h89, t2)
+       h01 = i64x2.add(h01, t3)
+       h23 = i64x2.add(h23, t4)
+
+       // f[5]
+       fi = v128.extract_lane<i32>(${f_name}_4567, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(${g_name}_4567, ${g_name}_4567_19, m))
+       t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+       h56 = i64x2.add(h56, t0)
+       h78 = i64x2.add(h78, t1)
+       h90 = i64x2.add(h90, t2)
+       h12 = i64x2.add(h12, t3)
+       h34 = i64x2.add(h34, t4)
+
+       // f[6]
+       fi = v128.extract_lane<i32>(${f_name}_4567, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+       h67 = i64x2.add(h67, t0)
+       h89 = i64x2.add(h89, t1)
+       h01 = i64x2.add(h01, t2)
+       h23 = i64x2.add(h23, t3)
+       h45 = i64x2.add(h45, t4)
+
+       // f[7]
+       fi = v128.extract_lane<i32>(${f_name}_4567, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(${g_name}_0123, ${g_name}_0123_19, m))
+       t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+       h78 = i64x2.add(h78, t0)
+       h90 = i64x2.add(h90, t1)
+       h12 = i64x2.add(h12, t2)
+       h34 = i64x2.add(h34, t3)
+       h56 = i64x2.add(h56, t4)
+
+       // f[8]
+       fi = v128.extract_lane<i32>(${f_name}_89xx, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+       h89 = i64x2.add(h89, t0)
+       h01 = i64x2.add(h01, t1)
+       h23 = i64x2.add(h23, t2)
+       h45 = i64x2.add(h45, t3)
+       h67 = i64x2.add(h67, t4)
+
+       // f[9]
+       fi = v128.extract_lane<i32>(${f_name}_89xx, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(${g_name}_0123, ${g_name}_0123_19, m))
+       t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+       h90 = i64x2.add(h90, t0)
+       h12 = i64x2.add(h12, t1)
+       h34 = i64x2.add(h34, t2)
+       h56 = i64x2.add(h56, t3)
+       h78 = i64x2.add(h78, t4)
+
+       // extract scalars from vectors
+       h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+       h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+       h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+       h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+       h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+       h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+       h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+       h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+       h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+       h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+       // carry
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + (1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + (1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + (1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + (1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + (1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + (1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + (1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + (1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       ${h_name}_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       ${h_name}_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       ${h_name}_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+`
+}
+
+/**
+ * Negate the values of a FieldElement and store the result.
+ *
+ * @param {string} h Variable name of FieldElement result destination
+ * @param {string} f Variable name of FieldElement operand source
+ * @returns {string} Three vector negation commands
+ */
+export function fe_neg (h, f) {
+       const h_name = h.replaceAll('.', '_')
+       const f_name = f.replaceAll('.', '_')
+       return `
+       // ${h} = -${f}
+       ${h_name}_0123 = v128.neg<i32>(${f_name}_0123)
+       ${h_name}_4567 = v128.neg<i32>(${f_name}_4567)
+       ${h_name}_89xx = v128.neg<i32>(${f_name}_89xx)
+`
+}
+
+/**
+ * Square a FieldElement and store the result.
+ *
+ * Some iterations of the inner loop can be skipped since
+ * \`f[x]*f[y] + f[y]*f[x] = 2*f[x]*f[y]\`
+ *
+ * \`\`\`
+ * // non-constant-time example
+ * for (let i = 0; i < 10; i++) {
+ *     for (let j = i; j < 10; j++) {
+ *             h[i + j] += f[i] * g[j]
+ *             if (i < j) {
+ *                     h[i + j] *= 2
+ *             }
+ *     }
+ * }
+ * \`\`\`
+ *
+ * h = f * f
+ * Can overlap h with f.
+ *
+ * Preconditions:
+ * |f| bounded by 1.65*2²⁶,1.65*2²⁵,1.65*2²⁶,1.65*2²⁵,etc.
+ *
+ * Postconditions:
+ * |h| bounded by 1.01*2²⁵,1.01*2²⁴4,1.01*2²⁵,1.01*2²⁴,etc.
+ *
+ * @param {string} h Variable name of FieldElement power destination
+ * @param {string} f Variable name of FieldElement base source
+ * @returns {string} Field squaring algorithm
+ */
+export function fe_sq (h, f) {
+       const h_name = h.replaceAll('.', '_')
+       const f_name = f.replaceAll('.', '_')
+       return `
+       // ${h} = ${f}²
+       f0 = i64(v128.extract_lane<i32>(${f_name}_0123, 0))
+       f1 = i64(v128.extract_lane<i32>(${f_name}_0123, 1))
+       f2 = i64(v128.extract_lane<i32>(${f_name}_0123, 2))
+       f3 = i64(v128.extract_lane<i32>(${f_name}_0123, 3))
+       f4 = i64(v128.extract_lane<i32>(${f_name}_4567, 0))
+       f5 = i64(v128.extract_lane<i32>(${f_name}_4567, 1))
+       f6 = i64(v128.extract_lane<i32>(${f_name}_4567, 2))
+       f7 = i64(v128.extract_lane<i32>(${f_name}_4567, 3))
+       f8 = i64(v128.extract_lane<i32>(${f_name}_89xx, 0))
+       f9 = i64(v128.extract_lane<i32>(${f_name}_89xx, 1))
+
+       f0_2 = f0 * 2
+       f1_2 = f1 * 2
+       f2_2 = f2 * 2
+       f3_2 = f3 * 2
+       f4_2 = f4 * 2
+       f5_2 = f5 * 2
+       f6_2 = f6 * 2
+       f7_2 = f7 * 2
+
+       f5_19 = f5 * 19 /* 1.959375*2²⁹ */
+       f6_19 = f6 * 19 /* 1.959375*2³⁰ */
+       f7_19 = f7 * 19 /* 1.959375*2²⁹ */
+       f8_19 = f8 * 19 /* 1.959375*2³⁰ */
+       f9_19 = f9 * 19 /* 1.959375*2²⁹ */
+
+       f7_38 = f7 * 38 /* 1.959375*2³⁰ */
+       f9_38 = f9 * 38 /* 1.959375*2³⁰ */
+
+       h0 = f0 * f0
+       h1 = f0_2 * f1
+       h2 = f0_2 * f2
+       h3 = f0_2 * f3
+       h4 = f0_2 * f4
+       h5 = f0_2 * f5
+       h6 = f0_2 * f6
+       h7 = f0_2 * f7
+       h8 = f0_2 * f8
+       h9 = f0_2 * f9
+
+       h2 += f1_2 * f1
+       h3 += f1_2 * f2
+       h4 += f1_2 * f3_2
+       h5 += f1_2 * f4
+       h6 += f1_2 * f5_2
+       h7 += f1_2 * f6
+       h8 += f1_2 * f7_2
+       h9 += f1_2 * f8
+       h0 += f1_2 * f9_38
+
+       h4 += f2 * f2
+       h5 += f2_2 * f3
+       h6 += f2_2 * f4
+       h7 += f2_2 * f5
+       h8 += f2_2 * f6
+       h9 += f2_2 * f7
+       h0 += f2_2 * f8_19
+       h1 += f2_2 * f9_19
+
+       h6 += f3_2 * f3
+       h7 += f3_2 * f4
+       h8 += f3_2 * f5_2
+       h9 += f3_2 * f6
+       h0 += f3_2 * f7_38
+       h1 += f3_2 * f8_19
+       h2 += f3_2 * f9_38
+
+       h8 += f4 * f4
+       h9 += f4_2 * f5
+       h0 += f4_2 * f6_19
+       h1 += f4_2 * f7_19
+       h2 += f4_2 * f8_19
+       h3 += f4_2 * f9_19
+
+       h0 += f5_2 * f5_19
+       h1 += f5_2 * f6_19
+       h2 += f5_2 * f7_38
+       h3 += f5_2 * f8_19
+       h4 += f5_2 * f9_38
+
+       h2 += f6 * f6_19
+       h3 += f6_2 * f7_19
+       h4 += f6_2 * f8_19
+       h5 += f6_2 * f9_19
+
+       h4 += f7_2 * f7_19
+       h5 += f7_2 * f8_19
+       h6 += f7_2 * f9_38
+
+       h6 += f8 * f8_19
+       h7 += f8 * f9_38
+
+       h8 += f9 * f9_38
+
+       c0 = (h0 + i64(1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + i64(1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + i64(1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + i64(1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + i64(1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + i64(1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + i64(1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + i64(1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + i64(1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + i64(1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + i64(1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + i64(1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       ${h_name}_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       ${h_name}_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       ${h_name}_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+`
+}
+
+/**
+ * Square a FieldElement, double it, and store the result.
+ *
+ * h = 2 * f * f
+ * Can overlap h with f.
+ *
+ * Preconditions:
+ * |f| bounded by 1.65*2^26,1.65*2^25,1.65*2^26,1.65*2^25,etc.
+ *
+ * Postconditions:
+ * |h| bounded by 1.01*2^25,1.01*2^24,1.01*2^25,1.01*2^24,etc.
+ *
+ * @param {string} h Variable name of FieldElement power destination
+ * @param {string} f Variable name of FieldElement base source
+ * @returns {string} Field squaring algorithm
+ */
+export function fe_sq2 (h, f) {
+       const h_name = h.replaceAll('.', '_')
+       const f_name = f.replaceAll('.', '_')
+       return `
+       // ${h} = 2${f}²
+       f0 = i64(v128.extract_lane<i32>(${f_name}_0123, 0))
+       f1 = i64(v128.extract_lane<i32>(${f_name}_0123, 1))
+       f2 = i64(v128.extract_lane<i32>(${f_name}_0123, 2))
+       f3 = i64(v128.extract_lane<i32>(${f_name}_0123, 3))
+       f4 = i64(v128.extract_lane<i32>(${f_name}_4567, 0))
+       f5 = i64(v128.extract_lane<i32>(${f_name}_4567, 1))
+       f6 = i64(v128.extract_lane<i32>(${f_name}_4567, 2))
+       f7 = i64(v128.extract_lane<i32>(${f_name}_4567, 3))
+       f8 = i64(v128.extract_lane<i32>(${f_name}_89xx, 0))
+       f9 = i64(v128.extract_lane<i32>(${f_name}_89xx, 1))
+
+       f0_2 = f0 * 2
+       f1_2 = f1 * 2
+       f2_2 = f2 * 2
+       f3_2 = f3 * 2
+       f4_2 = f4 * 2
+       f5_2 = f5 * 2
+       f6_2 = f6 * 2
+       f7_2 = f7 * 2
+
+       f5_38 = f5 * 38 /* 1.959375*2^30 */
+       f6_19 = f6 * 19 /* 1.959375*2^30 */
+       f7_38 = f7 * 38/* 1.959375*2^30 */
+       f8_19 = f8 * 19/* 1.959375*2^30 */
+       f9_38 = f9 * 38/* 1.959375*2^30 */
+
+       const f0f0: i64 = f0 * f0
+       const f0f1_2: i64 = f0_2 * f1
+       const f0f2_2: i64 = f0_2 * f2
+       const f0f3_2: i64 = f0_2 * f3
+       const f0f4_2: i64 = f0_2 * f4
+       const f0f5_2: i64 = f0_2 * f5
+       const f0f6_2: i64 = f0_2 * f6
+       const f0f7_2: i64 = f0_2 * f7
+       const f0f8_2: i64 = f0_2 * f8
+       const f0f9_2: i64 = f0_2 * f9
+
+       const f1f1_2: i64 = f1_2 * f1
+       const f1f2_2: i64 = f1_2 * f2
+       const f1f3_4: i64 = f1_2 * f3_2
+       const f1f4_2: i64 = f1_2 * f4
+       const f1f5_4: i64 = f1_2 * f5_2
+       const f1f6_2: i64 = f1_2 * f6
+       const f1f7_4: i64 = f1_2 * f7_2
+       const f1f8_2: i64 = f1_2 * f8
+       const f1f9_76: i64 = f1_2 * f9_38
+
+       const f2f2: i64 = f2 * f2
+       const f2f3_2: i64 = f2_2 * f3
+       const f2f4_2: i64 = f2_2 * f4
+       const f2f5_2: i64 = f2_2 * f5
+       const f2f6_2: i64 = f2_2 * f6
+       const f2f7_2: i64 = f2_2 * f7
+       const f2f8_38: i64 = f2_2 * f8_19
+       const f2f9_38: i64 = f2 * f9_38
+
+       const f3f3_2: i64 = f3_2 * f3
+       const f3f4_2: i64 = f3_2 * f4
+       const f3f5_4: i64 = f3_2 * f5_2
+       const f3f6_2: i64 = f3_2 * f6
+       const f3f7_76: i64 = f3_2 * f7_38
+       const f3f8_38: i64 = f3_2 * f8_19
+       const f3f9_76: i64 = f3_2 * f9_38
+
+       const f4f4: i64 = f4 * f4
+       const f4f5_2: i64 = f4_2 * f5
+       const f4f6_38: i64 = f4_2 * f6_19
+       const f4f7_38: i64 = f4 * f7_38
+       const f4f8_38: i64 = f4_2 * f8_19
+       const f4f9_38: i64 = f4 * f9_38
+
+       const f5f5_38: i64 = f5 * f5_38
+       const f5f6_38: i64 = f5_2 * f6_19
+       const f5f7_76: i64 = f5_2 * f7_38
+       const f5f8_38: i64 = f5_2 * f8_19
+       const f5f9_76: i64 = f5_2 * f9_38
+
+       const f6f6_19: i64 = f6 * f6_19
+       const f6f7_38: i64 = f6 * f7_38
+       const f6f8_38: i64 = f6_2 * f8_19
+       const f6f9_38: i64 = f6 * f9_38
+
+       const f7f7_38: i64 = f7 * f7_38
+       const f7f8_38: i64 = f7_2 * f8_19
+       const f7f9_76: i64 = f7_2 * f9_38
+
+       const f8f8_19: i64 = f8 * f8_19
+       const f8f9_38: i64 = f8 * f9_38
+
+       const f9f9_38: i64 = f9 * f9_38
+
+       h0 = f0f0 + f1f9_76 + f2f8_38 + f3f7_76 + f4f6_38 + f5f5_38
+       h1 = f0f1_2 + f2f9_38 + f3f8_38 + f4f7_38 + f5f6_38
+       h2 = f0f2_2 + f1f1_2 + f3f9_76 + f4f8_38 + f5f7_76 + f6f6_19
+       h3 = f0f3_2 + f1f2_2 + f4f9_38 + f5f8_38 + f6f7_38
+       h4 = f0f4_2 + f1f3_4 + f2f2 + f5f9_76 + f6f8_38 + f7f7_38
+       h5 = f0f5_2 + f1f4_2 + f2f3_2 + f6f9_38 + f7f8_38
+       h6 = f0f6_2 + f1f5_4 + f2f4_2 + f3f3_2 + f7f9_76 + f8f8_19
+       h7 = f0f7_2 + f1f6_2 + f2f5_2 + f3f4_2 + f8f9_38
+       h8 = f0f8_2 + f1f7_4 + f2f6_2 + f3f5_4 + f4f4 + f9f9_38
+       h9 = f0f9_2 + f1f8_2 + f2f7_2 + f3f6_2 + f4f5_2
+
+       h0 *= 2
+       h1 *= 2
+       h2 *= 2
+       h3 *= 2
+       h4 *= 2
+       h5 *= 2
+       h6 *= 2
+       h7 *= 2
+       h8 *= 2
+       h9 *= 2
+
+       c0 = (h0 + i64(1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + i64(1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + i64(1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + i64(1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + i64(1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + i64(1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + i64(1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + i64(1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + i64(1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + i64(1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + i64(1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + i64(1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       ${h_name}_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       ${h_name}_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       ${h_name}_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+`
+}
+
+/**
+ * Stores a FieldElement from named vectors into memory.
+ *
+ * @param {string} f Variable name of FieldElement
+ * @returns {string} StaticArray in memory
+ */
+export function fe_store (f) {
+       const f_name = f.replaceAll('.', '_')
+       const f_ptr = `changetype<usize>(${f})`
+       return `
+       // ${f} = [0..3, 4..7, 8..11]
+       v128.store(${f_ptr}, ${f_name}_0123, 0)
+       v128.store(${f_ptr}, ${f_name}_4567, 16)
+       v128.store_lane<u64>(${f_ptr}, ${f_name}_89xx, 0, 32)
+`
+}
+
+/**
+ * Subtract a FieldElement another and store the result.
+ *
+ * @param {string} h Variable name of FieldElement difference destination
+ * @param {string} f Variable name of FieldElement minuend source
+ * @param {string} g Variable name of FieldElement subtrahend source
+ * @returns {string} Three vector subtraction commands
+ */
+export function fe_sub (h, f, g) {
+       const h_name = h.replaceAll('.', '_')
+       const f_name = f.replaceAll('.', '_')
+       const g_name = g.replaceAll('.', '_')
+       return `
+       // ${h} = ${f} - ${g}
+       ${h_name}_0123 = v128.sub<i32>(${f_name}_0123, ${g_name}_0123)
+       ${h_name}_4567 = v128.sub<i32>(${f_name}_4567, ${g_name}_4567)
+       ${h_name}_89xx = v128.sub<i32>(${f_name}_89xx, ${g_name}_89xx)
 `
 }
 
@@ -126,26 +895,14 @@ export function fe_1 (h: FieldElement): void {
 //@ts-expect-error
 @inline
 export function fe_add (h: FieldElement, f: FieldElement, g: FieldElement): void {
-       const h_ptr = changetype<usize>(h)
-       const f_ptr = changetype<usize>(f)
-       const g_ptr = changetype<usize>(g)
-
-       let h0: v128, h1: v128, h2: v128
-
-       let f0 = v128.load(f_ptr, 0)
-       let f1 = v128.load(f_ptr, 16)
-       let f2 = v128.load(f_ptr, 32)
-
-       let g0 = v128.load(g_ptr, 0)
-       let g1 = v128.load(g_ptr, 16)
-       let g2 = v128.load(g_ptr, 32)
+       ${fe_init_v128('h', 'f', 'g')}
+       ${fe_load('h')}
+       ${fe_load('f')}
+       ${fe_load('g')}
 
        ${fe_add('h', 'f', 'g')}
 
-       // store h
-       v128.store(h_ptr, h0, 0)
-       v128.store(h_ptr, h1, 16)
-       v128.store_lane<u64>(h_ptr, h2, 0, 32)
+       ${fe_store('h')}
 }
 
 /**
@@ -187,21 +944,13 @@ export function fe_copy (h: FieldElement, f: FieldElement): void {
 //@ts-expect-error
 @inline
 export function fe_dbl (h: FieldElement, f: FieldElement): void {
-       const h_ptr = changetype<usize>(h)
-       const f_ptr = changetype<usize>(f)
-
-       let h0: v128, h1: v128, h2: v128
-
-       let f0 = v128.load(f_ptr, 0)
-       let f1 = v128.load(f_ptr, 16)
-       let f2 = v128.load(f_ptr, 32)
+       ${fe_init_v128('h', 'f')}
+       ${fe_load('h')}
+       ${fe_load('f')}
 
        ${fe_dbl('h', 'f')}
 
-       // store h
-       v128.store(h_ptr, h0, 0)
-       v128.store(h_ptr, h1, 16)
-       v128.store_lane<u64>(h_ptr, h2, 0, 32)
+       ${fe_store('h')}
 }
 
 /**
@@ -369,11 +1118,9 @@ export function fe_iszero (f: FieldElement): u8 {
  * @param g FieldElement multiplicand source
  */
 export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void {
+       ${fe_init_v128('h', 'f', 'g')}
        const f_ptr: usize = changetype<usize>(f)
-       const g_ptr: usize = changetype<usize>(g)
-       const g0123: v128 = v128.load(g_ptr, 0)
-       const g4567: v128 = v128.load(g_ptr, 16)
-       const g89xx: v128 = v128.load(g_ptr, 32)
+       ${fe_load('g')}
 
        // mask to select even lane from first vector and odd lane from second
        const m: v128 = i32x4(-1, 0, -1, 0)
@@ -383,40 +1130,40 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
 
        // factor of 19 for wrap from h9 to h0
        const v19: v128 = i32x4.splat(19)
-       const g0123_19: v128 = i32x4.mul(g0123, v19)
-       const g4567_19: v128 = i32x4.mul(g4567, v19)
-       const g89xx_19: v128 = i32x4.mul(g89xx, v19)
+       const g_0123_19: v128 = i32x4.mul(g_0123, v19)
+       const g_4567_19: v128 = i32x4.mul(g_4567, v19)
+       const g_89xx_19: v128 = i32x4.mul(g_89xx, v19)
 
        // f[0]
        let fi: i32 = load<i32>(f_ptr, 0)
        let fv: v128 = i32x4.splat(fi)
 
-       let h01: v128 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       let h23: v128 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       let h45: v128 = i64x2.extmul_low_i32x4_s(fv, g4567)
-       let h67: v128 = i64x2.extmul_high_i32x4_s(fv, g4567)
-       let h89: v128 = i64x2.extmul_low_i32x4_s(fv, g89xx)
+       let h01: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       let h23: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       let h45: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+       let h67: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567)
+       let h89: v128 = i64x2.extmul_low_i32x4_s(fv, g_89xx)
 
        // f[1]
        fi = load<i32>(f_ptr, 4)
        fv = i32x4.splat(fi)
        fv = i32x4.add(fv, v128.and(fv, odds))
 
-       let h12: v128 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       let h34: v128 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       let h56: v128 = i64x2.extmul_low_i32x4_s(fv, g4567)
-       let h78: v128 = i64x2.extmul_high_i32x4_s(fv, g4567)
-       let h90: v128 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g89xx, g89xx_19, m))
+       let h12: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       let h34: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       let h56: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+       let h78: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567)
+       let h90: v128 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_89xx, g_89xx_19, m))
 
        // f[2]
        fi = load<i32>(f_ptr, 8)
        fv = i32x4.splat(fi)
 
-       let t0: v128 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       let t1: v128 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       let t2: v128 = i64x2.extmul_low_i32x4_s(fv, g4567)
-       let t3: v128 = i64x2.extmul_high_i32x4_s(fv, g4567)
-       let t4: v128 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       let t0: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       let t1: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       let t2: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+       let t3: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567)
+       let t4: v128 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h23 = i64x2.add(h23, t0)
        h45 = i64x2.add(h45, t1)
@@ -429,11 +1176,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fv = i32x4.splat(fi)
        fv = i32x4.add(fv, v128.and(fv, odds))
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       t1 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       t2 = i64x2.extmul_low_i32x4_s(fv, g4567)
-       t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g4567, g4567_19, m))
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g_4567, g_4567_19, m))
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h34 = i64x2.add(h34, t0)
        h56 = i64x2.add(h56, t1)
@@ -445,11 +1192,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fi = load<i32>(f_ptr, 16)
        fv = i32x4.splat(fi)
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       t1 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       t2 = i64x2.extmul_low_i32x4_s(fv, g4567)
-       t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h45 = i64x2.add(h45, t0)
        h67 = i64x2.add(h67, t1)
@@ -462,11 +1209,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fv = i32x4.splat(fi)
        fv = i32x4.add(fv, v128.and(fv, odds))
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       t1 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g4567, g4567_19, m))
-       t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_4567, g_4567_19, m))
+       t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h56 = i64x2.add(h56, t0)
        h78 = i64x2.add(h78, t1)
@@ -478,11 +1225,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fi = load<i32>(f_ptr, 24)
        fv = i32x4.splat(fi)
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       t1 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19)
-       t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h67 = i64x2.add(h67, t0)
        h89 = i64x2.add(h89, t1)
@@ -495,11 +1242,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fv = i32x4.splat(fi)
        fv = i32x4.add(fv, v128.and(fv, odds))
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g0123, g0123_19, m))
-       t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19)
-       t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g_0123, g_0123_19, m))
+       t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h78 = i64x2.add(h78, t0)
        h90 = i64x2.add(h90, t1)
@@ -511,11 +1258,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fi = load<i32>(f_ptr, 32)
        fv = i32x4.splat(fi)
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       t1 = i64x2.extmul_high_i32x4_s(fv, g0123_19)
-       t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19)
-       t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, g_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h89 = i64x2.add(h89, t0)
        h01 = i64x2.add(h01, t1)
@@ -528,11 +1275,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fv = i32x4.splat(fi)
        fv = i32x4.add(fv, v128.and(fv, odds))
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g0123, g0123_19, m))
-       t1 = i64x2.extmul_high_i32x4_s(fv, g0123_19)
-       t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19)
-       t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_0123, g_0123_19, m))
+       t1 = i64x2.extmul_high_i32x4_s(fv, g_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h90 = i64x2.add(h90, t0)
        h12 = i64x2.add(h12, t1)
@@ -630,11 +1377,13 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
 //@ts-expect-error
 @inline
 export function fe_neg (h: FieldElement, f: FieldElement): void {
-       const h_ptr = changetype<usize>(h)
-       const f_ptr = changetype<usize>(f)
-       v128.store(h_ptr, v128.neg<i32>(v128.load(f_ptr)))
-       v128.store(h_ptr, v128.neg<i32>(v128.load(f_ptr, 16)), 16)
-       v128.store_lane<i64>(h_ptr, v128.neg<i32>(v128.load(f_ptr, 32)), 0, 32)
+       ${fe_init_v128('h', 'f')}
+       ${fe_load('h')}
+       ${fe_load('f')}
+
+       ${fe_neg('h', 'f')}
+
+       ${fe_store('h')}
 }
 
 const fe_pow22523_t0: FieldElement = fe()
@@ -1477,26 +2226,14 @@ export function fe_sq2 (h: FieldElement, f: FieldElement): void {
 //@ts-expect-error
 @inline
 export function fe_sub (h: FieldElement, f: FieldElement, g: FieldElement): void {
-       const h_ptr = changetype<usize>(h)
-       const f_ptr = changetype<usize>(f)
-       const g_ptr = changetype<usize>(g)
-
-       let h0: v128, h1: v128, h2: v128
-
-       let f0 = v128.load(f_ptr, 0)
-       let f1 = v128.load(f_ptr, 16)
-       let f2 = v128.load(f_ptr, 32)
-
-       let g0 = v128.load(g_ptr, 0)
-       let g1 = v128.load(g_ptr, 16)
-       let g2 = v128.load(g_ptr, 32)
+       ${fe_init_v128('h', 'f', 'g')}
+       ${fe_load('h')}
+       ${fe_load('f')}
+       ${fe_load('g')}
 
        ${fe_sub('h', 'f', 'g')}
 
-       // store h
-       v128.store(h_ptr, h0, 0)
-       v128.store(h_ptr, h1, 16)
-       v128.store_lane<u64>(h_ptr, h2, 0, 32)
+       ${fe_store('h')}
 }
 
 const fe_tobytes_t: FieldElement = fe()
index 25a3e4d9e9d418760f6c3043f3ba2dd6091278b4..a4936e6c855069a864291a55b9108c3be63eb256 100644 (file)
@@ -9,7 +9,7 @@
  * AssemblyScript compiler and type checker.
  */
 
-import { fe_add, fe_dbl, fe_sub } from './fe.gen.mjs'
+import { fe_add, fe_dbl, fe_init, fe_load, fe_mul, fe_sq, fe_sq2, fe_store, fe_sub } from './fe.gen.mjs'
 
 export const P = `//! SPDX-FileCopyrightText: 2026 Chris Duncan <chris@codecow.com>
 //! SPDX-License-Identifier: GPL-3.0-or-later
@@ -60,70 +60,40 @@ export function ge_p2_0 (h: ge_p2): void {
        fe_1(h.Z)
 }
 
-const ge_p2_dbl_t0: FieldElement = fe()
+const ge_p2_dbl_t: FieldElement = fe()
 //@ts-expect-error
 @inline
 export function ge_p2_dbl (r: ge_p1p1, p: ge_p2): void {
-       const rX_ptr = changetype<usize>(r.X)
-       const rY_ptr = changetype<usize>(r.Y)
-       const rZ_ptr = changetype<usize>(r.Z)
-       const rT_ptr = changetype<usize>(r.T)
-
-       const pX_ptr = changetype<usize>(p.X)
-       const pY_ptr = changetype<usize>(p.Y)
-       const pZ_ptr = changetype<usize>(p.Z)
-
-       const rYsq = ge_p2_dbl_t0
-       const rYsq_ptr = changetype<usize>(rYsq)
-
-       fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rY = pY²
-       fe_sq2(r.T, p.Z) // rT = pZ²
-
-       // Start converting to performant inlined ops with locals to avoid load/store
-       let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
-       let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
-       let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32)
-       let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32)
-
-       const pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32)
-       const pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32)
-       const pZ0 = v128.load(pZ_ptr), pZ1 = v128.load(pZ_ptr, 16), pZ2 = v128.load(pZ_ptr, 32)
-
-       // rY = pX + pY
-       v128.store(rY_ptr, v128.add<i32>(pX0, pY0))
-       v128.store(rY_ptr, v128.add<i32>(pX1, pY1), 16)
-       v128.store_lane<u64>(rY_ptr, v128.add<i32>(pX2, pY2), 0, 32)
-
-       // t0 = r.Y²
-       fe_sq(rYsq, r.Y)
-       const rYsq0 = v128.load(rYsq_ptr, 0)
-       const rYsq1 = v128.load(rYsq_ptr, 16)
-       const rYsq2 = v128.load(rYsq_ptr, 32)
-
-       ${fe_add('rY', 'rZ', 'rX')}
-       ${fe_sub('rZ', 'rZ', 'rX')}
-       ${fe_sub('rX', 'rYsq', 'rY')}
-       ${fe_sub('rT', 'rT', 'rZ')}
-
-       // store r.X
-       v128.store(rX_ptr, rX0)
-       v128.store(rX_ptr, rX1, 16)
-       v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
-       // store r.Y
-       v128.store(rY_ptr, rY0)
-       v128.store(rY_ptr, rY1, 16)
-       v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
-       // store r.Z
-       v128.store(rZ_ptr, rZ0)
-       v128.store(rZ_ptr, rZ1, 16)
-       v128.store_lane<u64>(rZ_ptr, rZ2, 0, 32)
-
-       // store r.T
-       v128.store(rT_ptr, rT0)
-       v128.store(rT_ptr, rT1, 16)
-       v128.store_lane<u64>(rT_ptr, rT2, 0, 32)
+       ${fe_init('p.X', 'p.Y', 'p.Z', 'r.X', 'r.Y', 'r.Z', 'r.T', 'rY2')}
+       ${fe_load('p.X')}
+       ${fe_load('p.Y')}
+       ${fe_load('p.Z')}
+
+       ${fe_load('r.X')}
+       ${fe_load('r.Y')}
+       ${fe_load('r.Z')}
+       ${fe_load('r.T')}
+
+       const rY2 = ge_p2_dbl_t
+       ${fe_load('rY2')}
+
+       // could swap below with fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rZ = pY²
+       ${fe_sq('r.X', 'p.X')}
+       ${fe_sq('r.Z', 'p.Y')}
+       // could swap above with fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rZ = pY²
+
+       ${fe_sq2('r.T', 'p.Z')}
+       ${fe_add('r.Y', 'p.X', 'p.Y')}
+       ${fe_sq('rY2', 'r.Y')}
+       ${fe_add('r.Y', 'r.Z', 'r.X')}
+       ${fe_sub('r.Z', 'r.Z', 'r.X')}
+       ${fe_sub('r.X', 'rY2', 'r.Y')}
+       ${fe_sub('r.T', 'r.T', 'r.Z')}
+
+       ${fe_store('r.X')}
+       ${fe_store('r.Y')}
+       ${fe_store('r.Z')}
+       ${fe_store('r.T')}
 }
 
 /**
@@ -222,70 +192,42 @@ export function ge_precomp_0 (h: ge_precomp): void {
  * r = p + q
  */
 export function ge_add_cached (r: ge_p1p1, p: ge_p3, q: ge_cached): void {
-       const rX_ptr = changetype<usize>(r.X)
-       const rY_ptr = changetype<usize>(r.Y)
-       const rZ_ptr = changetype<usize>(r.Z)
-       const rT_ptr = changetype<usize>(r.T)
-
-       const pX_ptr = changetype<usize>(p.X)
-       const pY_ptr = changetype<usize>(p.Y)
-
-       let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
-       let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
-       let pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32)
-       let pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32)
-
-       ${fe_add('rX', 'pY', 'pX')}
-       ${fe_sub('rY', 'pY', 'pX')}
-
-       // store r.X
-       v128.store(rX_ptr, rX0)
-       v128.store(rX_ptr, rX1, 16)
-       v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
-       // store r.Y
-       v128.store(rY_ptr, rY0)
-       v128.store(rY_ptr, rY1, 16)
-       v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
-       // not yet converted to inline for size and complexity
-       fe_mul(r.Z, r.X, q.YplusX)
-       fe_mul(r.Y, r.Y, q.YminusX)
-       fe_mul(r.T, q.T2d, p.T)
-       fe_mul(r.X, p.Z, q.Z)
-
-       // remaining arithmetic done inline to avoid unnecessary load/store
-       rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
-       rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
-       let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32)
-       let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32)
-       let t0: v128, t1: v128, t2: v128
-
-       ${fe_dbl('t', 'rX')}
-       ${fe_sub('rX', 'rZ', 'rY')}
-       ${fe_add('rY', 'rZ', 'rY')}
-       ${fe_add('rZ', 't', 'rT')}
-       ${fe_sub('rT', 't', 'rT')}
-
-       // store r.X
-       v128.store(rX_ptr, rX0)
-       v128.store(rX_ptr, rX1, 16)
-       v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
-       // store r.Y
-       v128.store(rY_ptr, rY0)
-       v128.store(rY_ptr, rY1, 16)
-       v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
-       // store r.Z
-       v128.store(rZ_ptr, rZ0)
-       v128.store(rZ_ptr, rZ1, 16)
-       v128.store_lane<u64>(rZ_ptr, rZ2, 0, 32)
-
-       // store r.T
-       v128.store(rT_ptr, rT0)
-       v128.store(rT_ptr, rT1, 16)
-       v128.store_lane<u64>(rT_ptr, rT2, 0, 32)
+       ${fe_init('r.X', 'r.Y', 'r.Z', 'r.T', 'p.X', 'p.Y', 'p.Z', 'p.T', 'q.YplusX', 'q.YminusX', 'q.Z', 'q.T2d')}
+       ${fe_load('r.X')}
+       ${fe_load('r.Y')}
+       ${fe_load('r.Z')}
+       ${fe_load('r.T')}
+
+       ${fe_load('p.X')}
+       ${fe_load('p.Y')}
+       ${fe_load('p.Z')}
+       ${fe_load('p.T')}
+
+       ${fe_load('q.YplusX')}
+       ${fe_load('q.YminusX')}
+       ${fe_load('q.Z')}
+       ${fe_load('q.T2d')}
+
+       ${fe_add('r.X', 'p.Y', 'p.X')}
+       ${fe_sub('r.Y', 'p.Y', 'p.X')}
+
+       ${fe_mul('r.Z', 'r.X', 'q.YplusX')}
+       ${fe_mul('r.Y', 'r.Y', 'q.YminusX')}
+       ${fe_mul('r.T', 'q.T2d', 'p.T')}
+       ${fe_mul('r.X', 'p.Z', 'q.Z')}
+
+       let t_0123: v128, t_4567: v128, t_89xx: v128
+
+       ${fe_dbl('t', 'r.X')}
+       ${fe_sub('r.X', 'r.Z', 'r.Y')}
+       ${fe_add('r.Y', 'r.Z', 'r.Y')}
+       ${fe_add('r.Z', 't', 'r.T')}
+       ${fe_sub('r.T', 't', 'r.T')}
+
+       ${fe_store('r.X')}
+       ${fe_store('r.Y')}
+       ${fe_store('r.Z')}
+       ${fe_store('r.T')}
 }
 
 /**
@@ -294,73 +236,36 @@ export function ge_add_cached (r: ge_p1p1, p: ge_p3, q: ge_cached): void {
 //@ts-expect-error
 @inline
 export function ge_add_precomp (r: ge_p1p1, p: ge_p3, q: ge_precomp): void {
-       const rX_ptr = changetype<usize>(r.X)
-       const rY_ptr = changetype<usize>(r.Y)
-       const rZ_ptr = changetype<usize>(r.Z)
-       const rT_ptr = changetype<usize>(r.T)
-
-       const pX_ptr = changetype<usize>(p.X)
-       const pY_ptr = changetype<usize>(p.Y)
-       const pZ_ptr = changetype<usize>(p.Z)
-
-       let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
-       let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
-       const pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32)
-       const pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32)
-
-       ${fe_add('rX', 'pY', 'pX')}
-       ${fe_sub('rY', 'pY', 'pX')}
-
-       // store r.X
-       v128.store(rX_ptr, rX0)
-       v128.store(rX_ptr, rX1, 16)
-       v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
-       // store r.Y
-       v128.store(rY_ptr, rY0)
-       v128.store(rY_ptr, rY1, 16)
-       v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
-       // not yet converted to inline for size and complexity
-       fe_mul(r.Z, r.X, q.yplusx)
-       fe_mul(r.Y, r.Y, q.yminusx)
-       fe_mul(r.T, q.xy2d, p.T)
-
-       // remaining arithmetic done inline to avoid unnecessary load/store
-       rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
-       rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
-       let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32)
-       let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32)
-
-       let pZ0 = v128.load(pZ_ptr, 0)
-       let pZ1 = v128.load(pZ_ptr, 16)
-       let pZ2 = v128.load(pZ_ptr, 32)
-
-       ${fe_dbl('pZ', 'pZ')}
-       ${fe_sub('rX', 'rZ', 'rY')}
-       ${fe_add('rY', 'rZ', 'rY')}
-       ${fe_add('rZ', 'pZ', 'rT')}
-       ${fe_sub('rT', 'pZ', 'rT')}
-
-       // store r.X
-       v128.store(rX_ptr, rX0)
-       v128.store(rX_ptr, rX1, 16)
-       v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
-       // store r.Y
-       v128.store(rY_ptr, rY0)
-       v128.store(rY_ptr, rY1, 16)
-       v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
-       // store r.Z
-       v128.store(rZ_ptr, rZ0)
-       v128.store(rZ_ptr, rZ1, 16)
-       v128.store_lane<u64>(rZ_ptr, rZ2, 0, 32)
-
-       // store r.T
-       v128.store(rT_ptr, rT0)
-       v128.store(rT_ptr, rT1, 16)
-       v128.store_lane<u64>(rT_ptr, rT2, 0, 32)
+       ${fe_init('r.X', 'r.Y', 'r.Z', 'r.T', 'p.X', 'p.Y', 'p.Z', 'p.T', 'q.yplusx', 'q.yminusx', 'q.xy2d')}
+       ${fe_load('r.X')}
+       ${fe_load('r.Y')}
+       ${fe_load('r.Z')}
+       ${fe_load('r.T')}
+
+       ${fe_load('p.X')}
+       ${fe_load('p.Y')}
+       ${fe_load('p.Z')}
+       ${fe_load('p.T')}
+
+       ${fe_load('q.yplusx')}
+       ${fe_load('q.yminusx')}
+       ${fe_load('q.xy2d')}
+
+       ${fe_add('r.X', 'p.Y', 'p.X')}
+       ${fe_sub('r.Y', 'p.Y', 'p.X')}
+       ${fe_mul('r.Z', 'r.X', 'q.yplusx')}
+       ${fe_mul('r.Y', 'r.Y', 'q.yminusx')}
+       ${fe_mul('r.T', 'q.xy2d', 'p.T')}
+       ${fe_dbl('p.Z', 'p.Z')}
+       ${fe_sub('r.X', 'r.Z', 'r.Y')}
+       ${fe_add('r.Y', 'r.Z', 'r.Y')}
+       ${fe_add('r.Z', 'p.Z', 'r.T')}
+       ${fe_sub('r.T', 'p.Z', 'r.T')}
+
+       ${fe_store('r.X')}
+       ${fe_store('r.Y')}
+       ${fe_store('r.Z')}
+       ${fe_store('r.T')}
 }
 
 const ge_sub_p3_q_cached = new ge_cached()
index 93e443a0908443bcb80471b941948c5b4cbae654..f4ab1c3e7095e0a7b5bbef1e75fb521af5225b94 100644 (file)
@@ -65,29 +65,45 @@ export function fe_1 (h: FieldElement): void {
 //@ts-expect-error
 @inline
 export function fe_add (h: FieldElement, f: FieldElement, g: FieldElement): void {
-       const h_ptr = changetype<usize>(h)
-       const f_ptr = changetype<usize>(f)
-       const g_ptr = changetype<usize>(g)
+       
+       let 
+       h_0123: v128,
+       h_4567: v128,
+       h_89xx: v128
+,
+       f_0123: v128,
+       f_4567: v128,
+       f_89xx: v128
+,
+       g_0123: v128,
+       g_4567: v128,
+       g_89xx: v128
+
+       // [0..3, 4..7, 8..11] = h
+       h_0123 = v128.load(changetype<usize>(h), 0)
+       h_4567 = v128.load(changetype<usize>(h), 16)
+       h_89xx = v128.load(changetype<usize>(h), 32)
+
+       // [0..3, 4..7, 8..11] = f
+       f_0123 = v128.load(changetype<usize>(f), 0)
+       f_4567 = v128.load(changetype<usize>(f), 16)
+       f_89xx = v128.load(changetype<usize>(f), 32)
+
+       // [0..3, 4..7, 8..11] = g
+       g_0123 = v128.load(changetype<usize>(g), 0)
+       g_4567 = v128.load(changetype<usize>(g), 16)
+       g_89xx = v128.load(changetype<usize>(g), 32)
 
-       let h0: v128, h1: v128, h2: v128
-
-       let f0 = v128.load(f_ptr, 0)
-       let f1 = v128.load(f_ptr, 16)
-       let f2 = v128.load(f_ptr, 32)
+       // h = f + g
+       h_0123 = v128.add<i32>(f_0123, g_0123)
+       h_4567 = v128.add<i32>(f_4567, g_4567)
+       h_89xx = v128.add<i32>(f_89xx, g_89xx)
 
-       let g0 = v128.load(g_ptr, 0)
-       let g1 = v128.load(g_ptr, 16)
-       let g2 = v128.load(g_ptr, 32)
+       // h = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(h), h_0123, 0)
+       v128.store(changetype<usize>(h), h_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(h), h_89xx, 0, 32)
 
-       // h = f + g
-       h0 = v128.add<i32>(f0, g0)
-       h1 = v128.add<i32>(f1, g1)
-       h2 = v128.add<i32>(f2, g2)
-
-       // store h
-       v128.store(h_ptr, h0, 0)
-       v128.store(h_ptr, h1, 16)
-       v128.store_lane<u64>(h_ptr, h2, 0, 32)
 }
 
 /**
@@ -129,24 +145,36 @@ export function fe_copy (h: FieldElement, f: FieldElement): void {
 //@ts-expect-error
 @inline
 export function fe_dbl (h: FieldElement, f: FieldElement): void {
-       const h_ptr = changetype<usize>(h)
-       const f_ptr = changetype<usize>(f)
+       
+       let 
+       h_0123: v128,
+       h_4567: v128,
+       h_89xx: v128
+,
+       f_0123: v128,
+       f_4567: v128,
+       f_89xx: v128
+
+       // [0..3, 4..7, 8..11] = h
+       h_0123 = v128.load(changetype<usize>(h), 0)
+       h_4567 = v128.load(changetype<usize>(h), 16)
+       h_89xx = v128.load(changetype<usize>(h), 32)
+
+       // [0..3, 4..7, 8..11] = f
+       f_0123 = v128.load(changetype<usize>(f), 0)
+       f_4567 = v128.load(changetype<usize>(f), 16)
+       f_89xx = v128.load(changetype<usize>(f), 32)
 
-       let h0: v128, h1: v128, h2: v128
+       // h = 2*f
+       h_0123 = v128.shl<i32>(f_0123, 1)
+       h_4567 = v128.shl<i32>(f_4567, 1)
+       h_89xx = v128.shl<i32>(f_89xx, 1)
 
-       let f0 = v128.load(f_ptr, 0)
-       let f1 = v128.load(f_ptr, 16)
-       let f2 = v128.load(f_ptr, 32)
+       // h = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(h), h_0123, 0)
+       v128.store(changetype<usize>(h), h_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(h), h_89xx, 0, 32)
 
-       // h = 2*f
-       h0 = v128.shl<i32>(f0, 1)
-       h1 = v128.shl<i32>(f1, 1)
-       h2 = v128.shl<i32>(f2, 1)
-
-       // store h
-       v128.store(h_ptr, h0, 0)
-       v128.store(h_ptr, h1, 16)
-       v128.store_lane<u64>(h_ptr, h2, 0, 32)
 }
 
 /**
@@ -314,11 +342,26 @@ export function fe_iszero (f: FieldElement): u8 {
  * @param g FieldElement multiplicand source
  */
 export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void {
+       
+       let 
+       h_0123: v128,
+       h_4567: v128,
+       h_89xx: v128
+,
+       f_0123: v128,
+       f_4567: v128,
+       f_89xx: v128
+,
+       g_0123: v128,
+       g_4567: v128,
+       g_89xx: v128
+
        const f_ptr: usize = changetype<usize>(f)
-       const g_ptr: usize = changetype<usize>(g)
-       const g0123: v128 = v128.load(g_ptr, 0)
-       const g4567: v128 = v128.load(g_ptr, 16)
-       const g89xx: v128 = v128.load(g_ptr, 32)
+       
+       // [0..3, 4..7, 8..11] = g
+       g_0123 = v128.load(changetype<usize>(g), 0)
+       g_4567 = v128.load(changetype<usize>(g), 16)
+       g_89xx = v128.load(changetype<usize>(g), 32)
 
        // mask to select even lane from first vector and odd lane from second
        const m: v128 = i32x4(-1, 0, -1, 0)
@@ -328,40 +371,40 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
 
        // factor of 19 for wrap from h9 to h0
        const v19: v128 = i32x4.splat(19)
-       const g0123_19: v128 = i32x4.mul(g0123, v19)
-       const g4567_19: v128 = i32x4.mul(g4567, v19)
-       const g89xx_19: v128 = i32x4.mul(g89xx, v19)
+       const g_0123_19: v128 = i32x4.mul(g_0123, v19)
+       const g_4567_19: v128 = i32x4.mul(g_4567, v19)
+       const g_89xx_19: v128 = i32x4.mul(g_89xx, v19)
 
        // f[0]
        let fi: i32 = load<i32>(f_ptr, 0)
        let fv: v128 = i32x4.splat(fi)
 
-       let h01: v128 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       let h23: v128 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       let h45: v128 = i64x2.extmul_low_i32x4_s(fv, g4567)
-       let h67: v128 = i64x2.extmul_high_i32x4_s(fv, g4567)
-       let h89: v128 = i64x2.extmul_low_i32x4_s(fv, g89xx)
+       let h01: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       let h23: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       let h45: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+       let h67: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567)
+       let h89: v128 = i64x2.extmul_low_i32x4_s(fv, g_89xx)
 
        // f[1]
        fi = load<i32>(f_ptr, 4)
        fv = i32x4.splat(fi)
        fv = i32x4.add(fv, v128.and(fv, odds))
 
-       let h12: v128 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       let h34: v128 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       let h56: v128 = i64x2.extmul_low_i32x4_s(fv, g4567)
-       let h78: v128 = i64x2.extmul_high_i32x4_s(fv, g4567)
-       let h90: v128 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g89xx, g89xx_19, m))
+       let h12: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       let h34: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       let h56: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+       let h78: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567)
+       let h90: v128 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_89xx, g_89xx_19, m))
 
        // f[2]
        fi = load<i32>(f_ptr, 8)
        fv = i32x4.splat(fi)
 
-       let t0: v128 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       let t1: v128 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       let t2: v128 = i64x2.extmul_low_i32x4_s(fv, g4567)
-       let t3: v128 = i64x2.extmul_high_i32x4_s(fv, g4567)
-       let t4: v128 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       let t0: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       let t1: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       let t2: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+       let t3: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567)
+       let t4: v128 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h23 = i64x2.add(h23, t0)
        h45 = i64x2.add(h45, t1)
@@ -374,11 +417,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fv = i32x4.splat(fi)
        fv = i32x4.add(fv, v128.and(fv, odds))
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       t1 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       t2 = i64x2.extmul_low_i32x4_s(fv, g4567)
-       t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g4567, g4567_19, m))
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g_4567, g_4567_19, m))
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h34 = i64x2.add(h34, t0)
        h56 = i64x2.add(h56, t1)
@@ -390,11 +433,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fi = load<i32>(f_ptr, 16)
        fv = i32x4.splat(fi)
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       t1 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       t2 = i64x2.extmul_low_i32x4_s(fv, g4567)
-       t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h45 = i64x2.add(h45, t0)
        h67 = i64x2.add(h67, t1)
@@ -407,11 +450,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fv = i32x4.splat(fi)
        fv = i32x4.add(fv, v128.and(fv, odds))
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       t1 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g4567, g4567_19, m))
-       t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_4567, g_4567_19, m))
+       t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h56 = i64x2.add(h56, t0)
        h78 = i64x2.add(h78, t1)
@@ -423,11 +466,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fi = load<i32>(f_ptr, 24)
        fv = i32x4.splat(fi)
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       t1 = i64x2.extmul_high_i32x4_s(fv, g0123)
-       t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19)
-       t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h67 = i64x2.add(h67, t0)
        h89 = i64x2.add(h89, t1)
@@ -440,11 +483,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fv = i32x4.splat(fi)
        fv = i32x4.add(fv, v128.and(fv, odds))
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g0123, g0123_19, m))
-       t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19)
-       t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g_0123, g_0123_19, m))
+       t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h78 = i64x2.add(h78, t0)
        h90 = i64x2.add(h90, t1)
@@ -456,11 +499,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fi = load<i32>(f_ptr, 32)
        fv = i32x4.splat(fi)
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
-       t1 = i64x2.extmul_high_i32x4_s(fv, g0123_19)
-       t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19)
-       t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, g_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h89 = i64x2.add(h89, t0)
        h01 = i64x2.add(h01, t1)
@@ -473,11 +516,11 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
        fv = i32x4.splat(fi)
        fv = i32x4.add(fv, v128.and(fv, odds))
 
-       t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g0123, g0123_19, m))
-       t1 = i64x2.extmul_high_i32x4_s(fv, g0123_19)
-       t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19)
-       t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
-       t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+       t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_0123, g_0123_19, m))
+       t1 = i64x2.extmul_high_i32x4_s(fv, g_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
 
        h90 = i64x2.add(h90, t0)
        h12 = i64x2.add(h12, t1)
@@ -575,11 +618,36 @@ export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void
 //@ts-expect-error
 @inline
 export function fe_neg (h: FieldElement, f: FieldElement): void {
-       const h_ptr = changetype<usize>(h)
-       const f_ptr = changetype<usize>(f)
-       v128.store(h_ptr, v128.neg<i32>(v128.load(f_ptr)))
-       v128.store(h_ptr, v128.neg<i32>(v128.load(f_ptr, 16)), 16)
-       v128.store_lane<i64>(h_ptr, v128.neg<i32>(v128.load(f_ptr, 32)), 0, 32)
+       
+       let 
+       h_0123: v128,
+       h_4567: v128,
+       h_89xx: v128
+,
+       f_0123: v128,
+       f_4567: v128,
+       f_89xx: v128
+
+       // [0..3, 4..7, 8..11] = h
+       h_0123 = v128.load(changetype<usize>(h), 0)
+       h_4567 = v128.load(changetype<usize>(h), 16)
+       h_89xx = v128.load(changetype<usize>(h), 32)
+
+       // [0..3, 4..7, 8..11] = f
+       f_0123 = v128.load(changetype<usize>(f), 0)
+       f_4567 = v128.load(changetype<usize>(f), 16)
+       f_89xx = v128.load(changetype<usize>(f), 32)
+
+       // h = -f
+       h_0123 = v128.neg<i32>(f_0123)
+       h_4567 = v128.neg<i32>(f_4567)
+       h_89xx = v128.neg<i32>(f_89xx)
+
+       // h = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(h), h_0123, 0)
+       v128.store(changetype<usize>(h), h_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(h), h_89xx, 0, 32)
+
 }
 
 const fe_pow22523_t0: FieldElement = fe()
@@ -1422,29 +1490,45 @@ export function fe_sq2 (h: FieldElement, f: FieldElement): void {
 //@ts-expect-error
 @inline
 export function fe_sub (h: FieldElement, f: FieldElement, g: FieldElement): void {
-       const h_ptr = changetype<usize>(h)
-       const f_ptr = changetype<usize>(f)
-       const g_ptr = changetype<usize>(g)
-
-       let h0: v128, h1: v128, h2: v128
+       
+       let 
+       h_0123: v128,
+       h_4567: v128,
+       h_89xx: v128
+,
+       f_0123: v128,
+       f_4567: v128,
+       f_89xx: v128
+,
+       g_0123: v128,
+       g_4567: v128,
+       g_89xx: v128
+
+       // [0..3, 4..7, 8..11] = h
+       h_0123 = v128.load(changetype<usize>(h), 0)
+       h_4567 = v128.load(changetype<usize>(h), 16)
+       h_89xx = v128.load(changetype<usize>(h), 32)
+
+       // [0..3, 4..7, 8..11] = f
+       f_0123 = v128.load(changetype<usize>(f), 0)
+       f_4567 = v128.load(changetype<usize>(f), 16)
+       f_89xx = v128.load(changetype<usize>(f), 32)
+
+       // [0..3, 4..7, 8..11] = g
+       g_0123 = v128.load(changetype<usize>(g), 0)
+       g_4567 = v128.load(changetype<usize>(g), 16)
+       g_89xx = v128.load(changetype<usize>(g), 32)
 
-       let f0 = v128.load(f_ptr, 0)
-       let f1 = v128.load(f_ptr, 16)
-       let f2 = v128.load(f_ptr, 32)
+       // h = f - g
+       h_0123 = v128.sub<i32>(f_0123, g_0123)
+       h_4567 = v128.sub<i32>(f_4567, g_4567)
+       h_89xx = v128.sub<i32>(f_89xx, g_89xx)
 
-       let g0 = v128.load(g_ptr, 0)
-       let g1 = v128.load(g_ptr, 16)
-       let g2 = v128.load(g_ptr, 32)
+       // h = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(h), h_0123, 0)
+       v128.store(changetype<usize>(h), h_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(h), h_89xx, 0, 32)
 
-       // h = f - g
-       h0 = v128.sub<i32>(f0, g0)
-       h1 = v128.sub<i32>(f1, g1)
-       h2 = v128.sub<i32>(f2, g2)
-
-       // store h
-       v128.store(h_ptr, h0, 0)
-       v128.store(h_ptr, h1, 16)
-       v128.store_lane<u64>(h_ptr, h2, 0, 32)
 }
 
 const fe_tobytes_t: FieldElement = fe()
index 0472a568bb353248fda3655b76354576eb5d086a..9b030f2fd4e3ec59a88978b07fa35c397bd1edf4 100644 (file)
@@ -47,85 +47,809 @@ export function ge_p2_0 (h: ge_p2): void {
        fe_1(h.Z)
 }
 
-const ge_p2_dbl_t0: FieldElement = fe()
+const ge_p2_dbl_t: FieldElement = fe()
 //@ts-expect-error
 @inline
 export function ge_p2_dbl (r: ge_p1p1, p: ge_p2): void {
-       const rX_ptr = changetype<usize>(r.X)
-       const rY_ptr = changetype<usize>(r.Y)
-       const rZ_ptr = changetype<usize>(r.Z)
-       const rT_ptr = changetype<usize>(r.T)
-
-       const pX_ptr = changetype<usize>(p.X)
-       const pY_ptr = changetype<usize>(p.Y)
-       const pZ_ptr = changetype<usize>(p.Z)
-
-       const rYsq = ge_p2_dbl_t0
-       const rYsq_ptr = changetype<usize>(rYsq)
-
-       fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rY = pY²
-       fe_sq2(r.T, p.Z) // rT = pZ²
-
-       // Start converting to performant inlined ops with locals to avoid load/store
-       let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
-       let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
-       let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32)
-       let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32)
-
-       const pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32)
-       const pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32)
-       const pZ0 = v128.load(pZ_ptr), pZ1 = v128.load(pZ_ptr, 16), pZ2 = v128.load(pZ_ptr, 32)
-
-       // rY = pX + pY
-       v128.store(rY_ptr, v128.add<i32>(pX0, pY0))
-       v128.store(rY_ptr, v128.add<i32>(pX1, pY1), 16)
-       v128.store_lane<u64>(rY_ptr, v128.add<i32>(pX2, pY2), 0, 32)
-
-       // t0 = r.Y²
-       fe_sq(rYsq, r.Y)
-       const rYsq0 = v128.load(rYsq_ptr, 0)
-       const rYsq1 = v128.load(rYsq_ptr, 16)
-       const rYsq2 = v128.load(rYsq_ptr, 32)
-
-       // rY = rZ + rX
-       rY0 = v128.add<i32>(rZ0, rX0)
-       rY1 = v128.add<i32>(rZ1, rX1)
-       rY2 = v128.add<i32>(rZ2, rX2)
-
-       // rZ = rZ - rX
-       rZ0 = v128.sub<i32>(rZ0, rX0)
-       rZ1 = v128.sub<i32>(rZ1, rX1)
-       rZ2 = v128.sub<i32>(rZ2, rX2)
-
-       // rX = rYsq - rY
-       rX0 = v128.sub<i32>(rYsq0, rY0)
-       rX1 = v128.sub<i32>(rYsq1, rY1)
-       rX2 = v128.sub<i32>(rYsq2, rY2)
-
-       // rT = rT - rZ
-       rT0 = v128.sub<i32>(rT0, rZ0)
-       rT1 = v128.sub<i32>(rT1, rZ1)
-       rT2 = v128.sub<i32>(rT2, rZ2)
-
-       // store r.X
-       v128.store(rX_ptr, rX0)
-       v128.store(rX_ptr, rX1, 16)
-       v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
-       // store r.Y
-       v128.store(rY_ptr, rY0)
-       v128.store(rY_ptr, rY1, 16)
-       v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
-       // store r.Z
-       v128.store(rZ_ptr, rZ0)
-       v128.store(rZ_ptr, rZ1, 16)
-       v128.store_lane<u64>(rZ_ptr, rZ2, 0, 32)
-
-       // store r.T
-       v128.store(rT_ptr, rT0)
-       v128.store(rT_ptr, rT1, 16)
-       v128.store_lane<u64>(rT_ptr, rT2, 0, 32)
+
+       let 
+       p_X_0123: v128,
+       p_X_4567: v128,
+       p_X_89xx: v128
+,
+       p_Y_0123: v128,
+       p_Y_4567: v128,
+       p_Y_89xx: v128
+,
+       p_Z_0123: v128,
+       p_Z_4567: v128,
+       p_Z_89xx: v128
+,
+       r_X_0123: v128,
+       r_X_4567: v128,
+       r_X_89xx: v128
+,
+       r_Y_0123: v128,
+       r_Y_4567: v128,
+       r_Y_89xx: v128
+,
+       r_Z_0123: v128,
+       r_Z_4567: v128,
+       r_Z_89xx: v128
+,
+       r_T_0123: v128,
+       r_T_4567: v128,
+       r_T_89xx: v128
+,
+       rY2_0123: v128,
+       rY2_4567: v128,
+       rY2_89xx: v128
+
+       // mask to select even lane from first vector and odd lane from second
+       const m: v128 = i32x4(-1, 0, -1, 0)
+
+       // mask to double odd-on-odd indexes
+       const odds: v128 = i32x4(0, -1, 0, -1)
+
+       // factor of 19 for wrap from h9 to h0
+       const v19: v128 = i32x4.splat(19)
+
+       let c0: i64
+       let c1: i64
+       let c2: i64
+       let c3: i64
+       let c4: i64
+       let c5: i64
+       let c6: i64
+       let c7: i64
+       let c8: i64
+       let c9: i64
+
+       let f0: i64
+       let f1: i64
+       let f2: i64
+       let f3: i64
+       let f4: i64
+       let f5: i64
+       let f6: i64
+       let f7: i64
+       let f8: i64
+       let f9: i64
+
+       let f0_2: i64
+       let f1_2: i64
+       let f2_2: i64
+       let f3_2: i64
+       let f4_2: i64
+       let f5_2: i64
+       let f6_2: i64
+       let f7_2: i64
+       let f8_2: i64
+       let f9_2: i64
+
+       let f5_19: i64
+       let f6_19: i64
+       let f7_19: i64
+       let f8_19: i64
+       let f9_19: i64
+
+       let f5_38: i64
+       let f7_38: i64
+       let f9_38: i64
+
+       let fi: i32
+       let fv: v128
+
+       let h0: i64
+       let h1: i64
+       let h2: i64
+       let h3: i64
+       let h4: i64
+       let h5: i64
+       let h6: i64
+       let h7: i64
+       let h8: i64
+       let h9: i64
+
+       let h01: v128
+       let h12: v128
+       let h23: v128
+       let h34: v128
+       let h45: v128
+       let h56: v128
+       let h67: v128
+       let h78: v128
+       let h89: v128
+       let h90: v128
+
+       let t0: v128
+       let t1: v128
+       let t2: v128
+       let t3: v128
+       let t4: v128
+
+       // [0..3, 4..7, 8..11] = p.X
+       p_X_0123 = v128.load(changetype<usize>(p.X), 0)
+       p_X_4567 = v128.load(changetype<usize>(p.X), 16)
+       p_X_89xx = v128.load(changetype<usize>(p.X), 32)
+
+       // [0..3, 4..7, 8..11] = p.Y
+       p_Y_0123 = v128.load(changetype<usize>(p.Y), 0)
+       p_Y_4567 = v128.load(changetype<usize>(p.Y), 16)
+       p_Y_89xx = v128.load(changetype<usize>(p.Y), 32)
+
+       // [0..3, 4..7, 8..11] = p.Z
+       p_Z_0123 = v128.load(changetype<usize>(p.Z), 0)
+       p_Z_4567 = v128.load(changetype<usize>(p.Z), 16)
+       p_Z_89xx = v128.load(changetype<usize>(p.Z), 32)
+
+       // [0..3, 4..7, 8..11] = r.X
+       r_X_0123 = v128.load(changetype<usize>(r.X), 0)
+       r_X_4567 = v128.load(changetype<usize>(r.X), 16)
+       r_X_89xx = v128.load(changetype<usize>(r.X), 32)
+
+       // [0..3, 4..7, 8..11] = r.Y
+       r_Y_0123 = v128.load(changetype<usize>(r.Y), 0)
+       r_Y_4567 = v128.load(changetype<usize>(r.Y), 16)
+       r_Y_89xx = v128.load(changetype<usize>(r.Y), 32)
+
+       // [0..3, 4..7, 8..11] = r.Z
+       r_Z_0123 = v128.load(changetype<usize>(r.Z), 0)
+       r_Z_4567 = v128.load(changetype<usize>(r.Z), 16)
+       r_Z_89xx = v128.load(changetype<usize>(r.Z), 32)
+
+       // [0..3, 4..7, 8..11] = r.T
+       r_T_0123 = v128.load(changetype<usize>(r.T), 0)
+       r_T_4567 = v128.load(changetype<usize>(r.T), 16)
+       r_T_89xx = v128.load(changetype<usize>(r.T), 32)
+
+       const rY2 = ge_p2_dbl_t
+       
+       // [0..3, 4..7, 8..11] = rY2
+       rY2_0123 = v128.load(changetype<usize>(rY2), 0)
+       rY2_4567 = v128.load(changetype<usize>(rY2), 16)
+       rY2_89xx = v128.load(changetype<usize>(rY2), 32)
+
+       // could swap below with fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rZ = pY²
+       
+       // r.X = p.X²
+       f0 = i64(v128.extract_lane<i32>(p_X_0123, 0))
+       f1 = i64(v128.extract_lane<i32>(p_X_0123, 1))
+       f2 = i64(v128.extract_lane<i32>(p_X_0123, 2))
+       f3 = i64(v128.extract_lane<i32>(p_X_0123, 3))
+       f4 = i64(v128.extract_lane<i32>(p_X_4567, 0))
+       f5 = i64(v128.extract_lane<i32>(p_X_4567, 1))
+       f6 = i64(v128.extract_lane<i32>(p_X_4567, 2))
+       f7 = i64(v128.extract_lane<i32>(p_X_4567, 3))
+       f8 = i64(v128.extract_lane<i32>(p_X_89xx, 0))
+       f9 = i64(v128.extract_lane<i32>(p_X_89xx, 1))
+
+       f0_2 = f0 * 2
+       f1_2 = f1 * 2
+       f2_2 = f2 * 2
+       f3_2 = f3 * 2
+       f4_2 = f4 * 2
+       f5_2 = f5 * 2
+       f6_2 = f6 * 2
+       f7_2 = f7 * 2
+
+       f5_19 = f5 * 19 /* 1.959375*2²⁹ */
+       f6_19 = f6 * 19 /* 1.959375*2³⁰ */
+       f7_19 = f7 * 19 /* 1.959375*2²⁹ */
+       f8_19 = f8 * 19 /* 1.959375*2³⁰ */
+       f9_19 = f9 * 19 /* 1.959375*2²⁹ */
+
+       f7_38 = f7 * 38 /* 1.959375*2³⁰ */
+       f9_38 = f9 * 38 /* 1.959375*2³⁰ */
+
+       h0 = f0 * f0
+       h1 = f0_2 * f1
+       h2 = f0_2 * f2
+       h3 = f0_2 * f3
+       h4 = f0_2 * f4
+       h5 = f0_2 * f5
+       h6 = f0_2 * f6
+       h7 = f0_2 * f7
+       h8 = f0_2 * f8
+       h9 = f0_2 * f9
+
+       h2 += f1_2 * f1
+       h3 += f1_2 * f2
+       h4 += f1_2 * f3_2
+       h5 += f1_2 * f4
+       h6 += f1_2 * f5_2
+       h7 += f1_2 * f6
+       h8 += f1_2 * f7_2
+       h9 += f1_2 * f8
+       h0 += f1_2 * f9_38
+
+       h4 += f2 * f2
+       h5 += f2_2 * f3
+       h6 += f2_2 * f4
+       h7 += f2_2 * f5
+       h8 += f2_2 * f6
+       h9 += f2_2 * f7
+       h0 += f2_2 * f8_19
+       h1 += f2_2 * f9_19
+
+       h6 += f3_2 * f3
+       h7 += f3_2 * f4
+       h8 += f3_2 * f5_2
+       h9 += f3_2 * f6
+       h0 += f3_2 * f7_38
+       h1 += f3_2 * f8_19
+       h2 += f3_2 * f9_38
+
+       h8 += f4 * f4
+       h9 += f4_2 * f5
+       h0 += f4_2 * f6_19
+       h1 += f4_2 * f7_19
+       h2 += f4_2 * f8_19
+       h3 += f4_2 * f9_19
+
+       h0 += f5_2 * f5_19
+       h1 += f5_2 * f6_19
+       h2 += f5_2 * f7_38
+       h3 += f5_2 * f8_19
+       h4 += f5_2 * f9_38
+
+       h2 += f6 * f6_19
+       h3 += f6_2 * f7_19
+       h4 += f6_2 * f8_19
+       h5 += f6_2 * f9_19
+
+       h4 += f7_2 * f7_19
+       h5 += f7_2 * f8_19
+       h6 += f7_2 * f9_38
+
+       h6 += f8 * f8_19
+       h7 += f8 * f9_38
+
+       h8 += f9 * f9_38
+
+       c0 = (h0 + i64(1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + i64(1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + i64(1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + i64(1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + i64(1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + i64(1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + i64(1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + i64(1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + i64(1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + i64(1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + i64(1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + i64(1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       r_X_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       r_X_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       r_X_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+       // r.Z = p.Y²
+       f0 = i64(v128.extract_lane<i32>(p_Y_0123, 0))
+       f1 = i64(v128.extract_lane<i32>(p_Y_0123, 1))
+       f2 = i64(v128.extract_lane<i32>(p_Y_0123, 2))
+       f3 = i64(v128.extract_lane<i32>(p_Y_0123, 3))
+       f4 = i64(v128.extract_lane<i32>(p_Y_4567, 0))
+       f5 = i64(v128.extract_lane<i32>(p_Y_4567, 1))
+       f6 = i64(v128.extract_lane<i32>(p_Y_4567, 2))
+       f7 = i64(v128.extract_lane<i32>(p_Y_4567, 3))
+       f8 = i64(v128.extract_lane<i32>(p_Y_89xx, 0))
+       f9 = i64(v128.extract_lane<i32>(p_Y_89xx, 1))
+
+       f0_2 = f0 * 2
+       f1_2 = f1 * 2
+       f2_2 = f2 * 2
+       f3_2 = f3 * 2
+       f4_2 = f4 * 2
+       f5_2 = f5 * 2
+       f6_2 = f6 * 2
+       f7_2 = f7 * 2
+
+       f5_19 = f5 * 19 /* 1.959375*2²⁹ */
+       f6_19 = f6 * 19 /* 1.959375*2³⁰ */
+       f7_19 = f7 * 19 /* 1.959375*2²⁹ */
+       f8_19 = f8 * 19 /* 1.959375*2³⁰ */
+       f9_19 = f9 * 19 /* 1.959375*2²⁹ */
+
+       f7_38 = f7 * 38 /* 1.959375*2³⁰ */
+       f9_38 = f9 * 38 /* 1.959375*2³⁰ */
+
+       h0 = f0 * f0
+       h1 = f0_2 * f1
+       h2 = f0_2 * f2
+       h3 = f0_2 * f3
+       h4 = f0_2 * f4
+       h5 = f0_2 * f5
+       h6 = f0_2 * f6
+       h7 = f0_2 * f7
+       h8 = f0_2 * f8
+       h9 = f0_2 * f9
+
+       h2 += f1_2 * f1
+       h3 += f1_2 * f2
+       h4 += f1_2 * f3_2
+       h5 += f1_2 * f4
+       h6 += f1_2 * f5_2
+       h7 += f1_2 * f6
+       h8 += f1_2 * f7_2
+       h9 += f1_2 * f8
+       h0 += f1_2 * f9_38
+
+       h4 += f2 * f2
+       h5 += f2_2 * f3
+       h6 += f2_2 * f4
+       h7 += f2_2 * f5
+       h8 += f2_2 * f6
+       h9 += f2_2 * f7
+       h0 += f2_2 * f8_19
+       h1 += f2_2 * f9_19
+
+       h6 += f3_2 * f3
+       h7 += f3_2 * f4
+       h8 += f3_2 * f5_2
+       h9 += f3_2 * f6
+       h0 += f3_2 * f7_38
+       h1 += f3_2 * f8_19
+       h2 += f3_2 * f9_38
+
+       h8 += f4 * f4
+       h9 += f4_2 * f5
+       h0 += f4_2 * f6_19
+       h1 += f4_2 * f7_19
+       h2 += f4_2 * f8_19
+       h3 += f4_2 * f9_19
+
+       h0 += f5_2 * f5_19
+       h1 += f5_2 * f6_19
+       h2 += f5_2 * f7_38
+       h3 += f5_2 * f8_19
+       h4 += f5_2 * f9_38
+
+       h2 += f6 * f6_19
+       h3 += f6_2 * f7_19
+       h4 += f6_2 * f8_19
+       h5 += f6_2 * f9_19
+
+       h4 += f7_2 * f7_19
+       h5 += f7_2 * f8_19
+       h6 += f7_2 * f9_38
+
+       h6 += f8 * f8_19
+       h7 += f8 * f9_38
+
+       h8 += f9 * f9_38
+
+       c0 = (h0 + i64(1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + i64(1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + i64(1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + i64(1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + i64(1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + i64(1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + i64(1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + i64(1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + i64(1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + i64(1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + i64(1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + i64(1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       r_Z_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       r_Z_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       r_Z_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+       // could swap above with fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rZ = pY²
+
+       // r.T = 2p.Z²
+       f0 = i64(v128.extract_lane<i32>(p_Z_0123, 0))
+       f1 = i64(v128.extract_lane<i32>(p_Z_0123, 1))
+       f2 = i64(v128.extract_lane<i32>(p_Z_0123, 2))
+       f3 = i64(v128.extract_lane<i32>(p_Z_0123, 3))
+       f4 = i64(v128.extract_lane<i32>(p_Z_4567, 0))
+       f5 = i64(v128.extract_lane<i32>(p_Z_4567, 1))
+       f6 = i64(v128.extract_lane<i32>(p_Z_4567, 2))
+       f7 = i64(v128.extract_lane<i32>(p_Z_4567, 3))
+       f8 = i64(v128.extract_lane<i32>(p_Z_89xx, 0))
+       f9 = i64(v128.extract_lane<i32>(p_Z_89xx, 1))
+
+       f0_2 = f0 * 2
+       f1_2 = f1 * 2
+       f2_2 = f2 * 2
+       f3_2 = f3 * 2
+       f4_2 = f4 * 2
+       f5_2 = f5 * 2
+       f6_2 = f6 * 2
+       f7_2 = f7 * 2
+
+       f5_38 = f5 * 38 /* 1.959375*2^30 */
+       f6_19 = f6 * 19 /* 1.959375*2^30 */
+       f7_38 = f7 * 38/* 1.959375*2^30 */
+       f8_19 = f8 * 19/* 1.959375*2^30 */
+       f9_38 = f9 * 38/* 1.959375*2^30 */
+
+       const f0f0: i64 = f0 * f0
+       const f0f1_2: i64 = f0_2 * f1
+       const f0f2_2: i64 = f0_2 * f2
+       const f0f3_2: i64 = f0_2 * f3
+       const f0f4_2: i64 = f0_2 * f4
+       const f0f5_2: i64 = f0_2 * f5
+       const f0f6_2: i64 = f0_2 * f6
+       const f0f7_2: i64 = f0_2 * f7
+       const f0f8_2: i64 = f0_2 * f8
+       const f0f9_2: i64 = f0_2 * f9
+
+       const f1f1_2: i64 = f1_2 * f1
+       const f1f2_2: i64 = f1_2 * f2
+       const f1f3_4: i64 = f1_2 * f3_2
+       const f1f4_2: i64 = f1_2 * f4
+       const f1f5_4: i64 = f1_2 * f5_2
+       const f1f6_2: i64 = f1_2 * f6
+       const f1f7_4: i64 = f1_2 * f7_2
+       const f1f8_2: i64 = f1_2 * f8
+       const f1f9_76: i64 = f1_2 * f9_38
+
+       const f2f2: i64 = f2 * f2
+       const f2f3_2: i64 = f2_2 * f3
+       const f2f4_2: i64 = f2_2 * f4
+       const f2f5_2: i64 = f2_2 * f5
+       const f2f6_2: i64 = f2_2 * f6
+       const f2f7_2: i64 = f2_2 * f7
+       const f2f8_38: i64 = f2_2 * f8_19
+       const f2f9_38: i64 = f2 * f9_38
+
+       const f3f3_2: i64 = f3_2 * f3
+       const f3f4_2: i64 = f3_2 * f4
+       const f3f5_4: i64 = f3_2 * f5_2
+       const f3f6_2: i64 = f3_2 * f6
+       const f3f7_76: i64 = f3_2 * f7_38
+       const f3f8_38: i64 = f3_2 * f8_19
+       const f3f9_76: i64 = f3_2 * f9_38
+
+       const f4f4: i64 = f4 * f4
+       const f4f5_2: i64 = f4_2 * f5
+       const f4f6_38: i64 = f4_2 * f6_19
+       const f4f7_38: i64 = f4 * f7_38
+       const f4f8_38: i64 = f4_2 * f8_19
+       const f4f9_38: i64 = f4 * f9_38
+
+       const f5f5_38: i64 = f5 * f5_38
+       const f5f6_38: i64 = f5_2 * f6_19
+       const f5f7_76: i64 = f5_2 * f7_38
+       const f5f8_38: i64 = f5_2 * f8_19
+       const f5f9_76: i64 = f5_2 * f9_38
+
+       const f6f6_19: i64 = f6 * f6_19
+       const f6f7_38: i64 = f6 * f7_38
+       const f6f8_38: i64 = f6_2 * f8_19
+       const f6f9_38: i64 = f6 * f9_38
+
+       const f7f7_38: i64 = f7 * f7_38
+       const f7f8_38: i64 = f7_2 * f8_19
+       const f7f9_76: i64 = f7_2 * f9_38
+
+       const f8f8_19: i64 = f8 * f8_19
+       const f8f9_38: i64 = f8 * f9_38
+
+       const f9f9_38: i64 = f9 * f9_38
+
+       h0 = f0f0 + f1f9_76 + f2f8_38 + f3f7_76 + f4f6_38 + f5f5_38
+       h1 = f0f1_2 + f2f9_38 + f3f8_38 + f4f7_38 + f5f6_38
+       h2 = f0f2_2 + f1f1_2 + f3f9_76 + f4f8_38 + f5f7_76 + f6f6_19
+       h3 = f0f3_2 + f1f2_2 + f4f9_38 + f5f8_38 + f6f7_38
+       h4 = f0f4_2 + f1f3_4 + f2f2 + f5f9_76 + f6f8_38 + f7f7_38
+       h5 = f0f5_2 + f1f4_2 + f2f3_2 + f6f9_38 + f7f8_38
+       h6 = f0f6_2 + f1f5_4 + f2f4_2 + f3f3_2 + f7f9_76 + f8f8_19
+       h7 = f0f7_2 + f1f6_2 + f2f5_2 + f3f4_2 + f8f9_38
+       h8 = f0f8_2 + f1f7_4 + f2f6_2 + f3f5_4 + f4f4 + f9f9_38
+       h9 = f0f9_2 + f1f8_2 + f2f7_2 + f3f6_2 + f4f5_2
+
+       h0 *= 2
+       h1 *= 2
+       h2 *= 2
+       h3 *= 2
+       h4 *= 2
+       h5 *= 2
+       h6 *= 2
+       h7 *= 2
+       h8 *= 2
+       h9 *= 2
+
+       c0 = (h0 + i64(1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + i64(1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + i64(1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + i64(1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + i64(1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + i64(1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + i64(1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + i64(1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + i64(1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + i64(1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + i64(1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + i64(1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       r_T_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       r_T_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       r_T_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+       // r.Y = p.X + p.Y
+       r_Y_0123 = v128.add<i32>(p_X_0123, p_Y_0123)
+       r_Y_4567 = v128.add<i32>(p_X_4567, p_Y_4567)
+       r_Y_89xx = v128.add<i32>(p_X_89xx, p_Y_89xx)
+
+       // rY2 = r.Y²
+       f0 = i64(v128.extract_lane<i32>(r_Y_0123, 0))
+       f1 = i64(v128.extract_lane<i32>(r_Y_0123, 1))
+       f2 = i64(v128.extract_lane<i32>(r_Y_0123, 2))
+       f3 = i64(v128.extract_lane<i32>(r_Y_0123, 3))
+       f4 = i64(v128.extract_lane<i32>(r_Y_4567, 0))
+       f5 = i64(v128.extract_lane<i32>(r_Y_4567, 1))
+       f6 = i64(v128.extract_lane<i32>(r_Y_4567, 2))
+       f7 = i64(v128.extract_lane<i32>(r_Y_4567, 3))
+       f8 = i64(v128.extract_lane<i32>(r_Y_89xx, 0))
+       f9 = i64(v128.extract_lane<i32>(r_Y_89xx, 1))
+
+       f0_2 = f0 * 2
+       f1_2 = f1 * 2
+       f2_2 = f2 * 2
+       f3_2 = f3 * 2
+       f4_2 = f4 * 2
+       f5_2 = f5 * 2
+       f6_2 = f6 * 2
+       f7_2 = f7 * 2
+
+       f5_19 = f5 * 19 /* 1.959375*2²⁹ */
+       f6_19 = f6 * 19 /* 1.959375*2³⁰ */
+       f7_19 = f7 * 19 /* 1.959375*2²⁹ */
+       f8_19 = f8 * 19 /* 1.959375*2³⁰ */
+       f9_19 = f9 * 19 /* 1.959375*2²⁹ */
+
+       f7_38 = f7 * 38 /* 1.959375*2³⁰ */
+       f9_38 = f9 * 38 /* 1.959375*2³⁰ */
+
+       h0 = f0 * f0
+       h1 = f0_2 * f1
+       h2 = f0_2 * f2
+       h3 = f0_2 * f3
+       h4 = f0_2 * f4
+       h5 = f0_2 * f5
+       h6 = f0_2 * f6
+       h7 = f0_2 * f7
+       h8 = f0_2 * f8
+       h9 = f0_2 * f9
+
+       h2 += f1_2 * f1
+       h3 += f1_2 * f2
+       h4 += f1_2 * f3_2
+       h5 += f1_2 * f4
+       h6 += f1_2 * f5_2
+       h7 += f1_2 * f6
+       h8 += f1_2 * f7_2
+       h9 += f1_2 * f8
+       h0 += f1_2 * f9_38
+
+       h4 += f2 * f2
+       h5 += f2_2 * f3
+       h6 += f2_2 * f4
+       h7 += f2_2 * f5
+       h8 += f2_2 * f6
+       h9 += f2_2 * f7
+       h0 += f2_2 * f8_19
+       h1 += f2_2 * f9_19
+
+       h6 += f3_2 * f3
+       h7 += f3_2 * f4
+       h8 += f3_2 * f5_2
+       h9 += f3_2 * f6
+       h0 += f3_2 * f7_38
+       h1 += f3_2 * f8_19
+       h2 += f3_2 * f9_38
+
+       h8 += f4 * f4
+       h9 += f4_2 * f5
+       h0 += f4_2 * f6_19
+       h1 += f4_2 * f7_19
+       h2 += f4_2 * f8_19
+       h3 += f4_2 * f9_19
+
+       h0 += f5_2 * f5_19
+       h1 += f5_2 * f6_19
+       h2 += f5_2 * f7_38
+       h3 += f5_2 * f8_19
+       h4 += f5_2 * f9_38
+
+       h2 += f6 * f6_19
+       h3 += f6_2 * f7_19
+       h4 += f6_2 * f8_19
+       h5 += f6_2 * f9_19
+
+       h4 += f7_2 * f7_19
+       h5 += f7_2 * f8_19
+       h6 += f7_2 * f9_38
+
+       h6 += f8 * f8_19
+       h7 += f8 * f9_38
+
+       h8 += f9 * f9_38
+
+       c0 = (h0 + i64(1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + i64(1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + i64(1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + i64(1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + i64(1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + i64(1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + i64(1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + i64(1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + i64(1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + i64(1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + i64(1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + i64(1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       rY2_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       rY2_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       rY2_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+       // r.Y = r.Z + r.X
+       r_Y_0123 = v128.add<i32>(r_Z_0123, r_X_0123)
+       r_Y_4567 = v128.add<i32>(r_Z_4567, r_X_4567)
+       r_Y_89xx = v128.add<i32>(r_Z_89xx, r_X_89xx)
+
+       // r.Z = r.Z - r.X
+       r_Z_0123 = v128.sub<i32>(r_Z_0123, r_X_0123)
+       r_Z_4567 = v128.sub<i32>(r_Z_4567, r_X_4567)
+       r_Z_89xx = v128.sub<i32>(r_Z_89xx, r_X_89xx)
+
+       // r.X = rY2 - r.Y
+       r_X_0123 = v128.sub<i32>(rY2_0123, r_Y_0123)
+       r_X_4567 = v128.sub<i32>(rY2_4567, r_Y_4567)
+       r_X_89xx = v128.sub<i32>(rY2_89xx, r_Y_89xx)
+
+       // r.T = r.T - r.Z
+       r_T_0123 = v128.sub<i32>(r_T_0123, r_Z_0123)
+       r_T_4567 = v128.sub<i32>(r_T_4567, r_Z_4567)
+       r_T_89xx = v128.sub<i32>(r_T_89xx, r_Z_89xx)
+
+       // r.X = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(r.X), r_X_0123, 0)
+       v128.store(changetype<usize>(r.X), r_X_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(r.X), r_X_89xx, 0, 32)
+
+       // r.Y = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(r.Y), r_Y_0123, 0)
+       v128.store(changetype<usize>(r.Y), r_Y_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(r.Y), r_Y_89xx, 0, 32)
+
+       // r.Z = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(r.Z), r_Z_0123, 0)
+       v128.store(changetype<usize>(r.Z), r_Z_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(r.Z), r_Z_89xx, 0, 32)
+
+       // r.T = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(r.T), r_T_0123, 0)
+       v128.store(changetype<usize>(r.T), r_T_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(r.T), r_T_89xx, 0, 32)
+
 }
 
 /**
@@ -224,96 +948,1132 @@ export function ge_precomp_0 (h: ge_precomp): void {
  * r = p + q
  */
 export function ge_add_cached (r: ge_p1p1, p: ge_p3, q: ge_cached): void {
-       const rX_ptr = changetype<usize>(r.X)
-       const rY_ptr = changetype<usize>(r.Y)
-       const rZ_ptr = changetype<usize>(r.Z)
-       const rT_ptr = changetype<usize>(r.T)
-
-       const pX_ptr = changetype<usize>(p.X)
-       const pY_ptr = changetype<usize>(p.Y)
-
-       let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
-       let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
-       let pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32)
-       let pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32)
-
-       // rX = pY + pX
-       rX0 = v128.add<i32>(pY0, pX0)
-       rX1 = v128.add<i32>(pY1, pX1)
-       rX2 = v128.add<i32>(pY2, pX2)
-
-       // rY = pY - pX
-       rY0 = v128.sub<i32>(pY0, pX0)
-       rY1 = v128.sub<i32>(pY1, pX1)
-       rY2 = v128.sub<i32>(pY2, pX2)
-
-       // store r.X
-       v128.store(rX_ptr, rX0)
-       v128.store(rX_ptr, rX1, 16)
-       v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
-       // store r.Y
-       v128.store(rY_ptr, rY0)
-       v128.store(rY_ptr, rY1, 16)
-       v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
-       // not yet converted to inline for size and complexity
-       fe_mul(r.Z, r.X, q.YplusX)
-       fe_mul(r.Y, r.Y, q.YminusX)
-       fe_mul(r.T, q.T2d, p.T)
-       fe_mul(r.X, p.Z, q.Z)
 
-       // remaining arithmetic done inline to avoid unnecessary load/store
-       rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
-       rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
-       let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32)
-       let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32)
-       let t0: v128, t1: v128, t2: v128
-
-       // t = 2*rX
-       t0 = v128.shl<i32>(rX0, 1)
-       t1 = v128.shl<i32>(rX1, 1)
-       t2 = v128.shl<i32>(rX2, 1)
-
-       // rX = rZ - rY
-       rX0 = v128.sub<i32>(rZ0, rY0)
-       rX1 = v128.sub<i32>(rZ1, rY1)
-       rX2 = v128.sub<i32>(rZ2, rY2)
-
-       // rY = rZ + rY
-       rY0 = v128.add<i32>(rZ0, rY0)
-       rY1 = v128.add<i32>(rZ1, rY1)
-       rY2 = v128.add<i32>(rZ2, rY2)
-
-       // rZ = t + rT
-       rZ0 = v128.add<i32>(t0, rT0)
-       rZ1 = v128.add<i32>(t1, rT1)
-       rZ2 = v128.add<i32>(t2, rT2)
-
-       // rT = t - rT
-       rT0 = v128.sub<i32>(t0, rT0)
-       rT1 = v128.sub<i32>(t1, rT1)
-       rT2 = v128.sub<i32>(t2, rT2)
-
-       // store r.X
-       v128.store(rX_ptr, rX0)
-       v128.store(rX_ptr, rX1, 16)
-       v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
-       // store r.Y
-       v128.store(rY_ptr, rY0)
-       v128.store(rY_ptr, rY1, 16)
-       v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
-       // store r.Z
-       v128.store(rZ_ptr, rZ0)
-       v128.store(rZ_ptr, rZ1, 16)
-       v128.store_lane<u64>(rZ_ptr, rZ2, 0, 32)
-
-       // store r.T
-       v128.store(rT_ptr, rT0)
-       v128.store(rT_ptr, rT1, 16)
-       v128.store_lane<u64>(rT_ptr, rT2, 0, 32)
+       let 
+       r_X_0123: v128,
+       r_X_4567: v128,
+       r_X_89xx: v128
+,
+       r_Y_0123: v128,
+       r_Y_4567: v128,
+       r_Y_89xx: v128
+,
+       r_Z_0123: v128,
+       r_Z_4567: v128,
+       r_Z_89xx: v128
+,
+       r_T_0123: v128,
+       r_T_4567: v128,
+       r_T_89xx: v128
+,
+       p_X_0123: v128,
+       p_X_4567: v128,
+       p_X_89xx: v128
+,
+       p_Y_0123: v128,
+       p_Y_4567: v128,
+       p_Y_89xx: v128
+,
+       p_Z_0123: v128,
+       p_Z_4567: v128,
+       p_Z_89xx: v128
+,
+       p_T_0123: v128,
+       p_T_4567: v128,
+       p_T_89xx: v128
+,
+       q_YplusX_0123: v128,
+       q_YplusX_4567: v128,
+       q_YplusX_89xx: v128
+,
+       q_YminusX_0123: v128,
+       q_YminusX_4567: v128,
+       q_YminusX_89xx: v128
+,
+       q_Z_0123: v128,
+       q_Z_4567: v128,
+       q_Z_89xx: v128
+,
+       q_T2d_0123: v128,
+       q_T2d_4567: v128,
+       q_T2d_89xx: v128
+
+       // mask to select even lane from first vector and odd lane from second
+       const m: v128 = i32x4(-1, 0, -1, 0)
+
+       // mask to double odd-on-odd indexes
+       const odds: v128 = i32x4(0, -1, 0, -1)
+
+       // factor of 19 for wrap from h9 to h0
+       const v19: v128 = i32x4.splat(19)
+
+       let c0: i64
+       let c1: i64
+       let c2: i64
+       let c3: i64
+       let c4: i64
+       let c5: i64
+       let c6: i64
+       let c7: i64
+       let c8: i64
+       let c9: i64
+
+       let f0: i64
+       let f1: i64
+       let f2: i64
+       let f3: i64
+       let f4: i64
+       let f5: i64
+       let f6: i64
+       let f7: i64
+       let f8: i64
+       let f9: i64
+
+       let f0_2: i64
+       let f1_2: i64
+       let f2_2: i64
+       let f3_2: i64
+       let f4_2: i64
+       let f5_2: i64
+       let f6_2: i64
+       let f7_2: i64
+       let f8_2: i64
+       let f9_2: i64
+
+       let f5_19: i64
+       let f6_19: i64
+       let f7_19: i64
+       let f8_19: i64
+       let f9_19: i64
+
+       let f5_38: i64
+       let f7_38: i64
+       let f9_38: i64
+
+       let fi: i32
+       let fv: v128
+
+       let h0: i64
+       let h1: i64
+       let h2: i64
+       let h3: i64
+       let h4: i64
+       let h5: i64
+       let h6: i64
+       let h7: i64
+       let h8: i64
+       let h9: i64
+
+       let h01: v128
+       let h12: v128
+       let h23: v128
+       let h34: v128
+       let h45: v128
+       let h56: v128
+       let h67: v128
+       let h78: v128
+       let h89: v128
+       let h90: v128
+
+       let t0: v128
+       let t1: v128
+       let t2: v128
+       let t3: v128
+       let t4: v128
+
+       // [0..3, 4..7, 8..11] = r.X
+       r_X_0123 = v128.load(changetype<usize>(r.X), 0)
+       r_X_4567 = v128.load(changetype<usize>(r.X), 16)
+       r_X_89xx = v128.load(changetype<usize>(r.X), 32)
+
+       // [0..3, 4..7, 8..11] = r.Y
+       r_Y_0123 = v128.load(changetype<usize>(r.Y), 0)
+       r_Y_4567 = v128.load(changetype<usize>(r.Y), 16)
+       r_Y_89xx = v128.load(changetype<usize>(r.Y), 32)
+
+       // [0..3, 4..7, 8..11] = r.Z
+       r_Z_0123 = v128.load(changetype<usize>(r.Z), 0)
+       r_Z_4567 = v128.load(changetype<usize>(r.Z), 16)
+       r_Z_89xx = v128.load(changetype<usize>(r.Z), 32)
+
+       // [0..3, 4..7, 8..11] = r.T
+       r_T_0123 = v128.load(changetype<usize>(r.T), 0)
+       r_T_4567 = v128.load(changetype<usize>(r.T), 16)
+       r_T_89xx = v128.load(changetype<usize>(r.T), 32)
+
+       // [0..3, 4..7, 8..11] = p.X
+       p_X_0123 = v128.load(changetype<usize>(p.X), 0)
+       p_X_4567 = v128.load(changetype<usize>(p.X), 16)
+       p_X_89xx = v128.load(changetype<usize>(p.X), 32)
+
+       // [0..3, 4..7, 8..11] = p.Y
+       p_Y_0123 = v128.load(changetype<usize>(p.Y), 0)
+       p_Y_4567 = v128.load(changetype<usize>(p.Y), 16)
+       p_Y_89xx = v128.load(changetype<usize>(p.Y), 32)
+
+       // [0..3, 4..7, 8..11] = p.Z
+       p_Z_0123 = v128.load(changetype<usize>(p.Z), 0)
+       p_Z_4567 = v128.load(changetype<usize>(p.Z), 16)
+       p_Z_89xx = v128.load(changetype<usize>(p.Z), 32)
+
+       // [0..3, 4..7, 8..11] = p.T
+       p_T_0123 = v128.load(changetype<usize>(p.T), 0)
+       p_T_4567 = v128.load(changetype<usize>(p.T), 16)
+       p_T_89xx = v128.load(changetype<usize>(p.T), 32)
+
+       // [0..3, 4..7, 8..11] = q.YplusX
+       q_YplusX_0123 = v128.load(changetype<usize>(q.YplusX), 0)
+       q_YplusX_4567 = v128.load(changetype<usize>(q.YplusX), 16)
+       q_YplusX_89xx = v128.load(changetype<usize>(q.YplusX), 32)
+
+       // [0..3, 4..7, 8..11] = q.YminusX
+       q_YminusX_0123 = v128.load(changetype<usize>(q.YminusX), 0)
+       q_YminusX_4567 = v128.load(changetype<usize>(q.YminusX), 16)
+       q_YminusX_89xx = v128.load(changetype<usize>(q.YminusX), 32)
+
+       // [0..3, 4..7, 8..11] = q.Z
+       q_Z_0123 = v128.load(changetype<usize>(q.Z), 0)
+       q_Z_4567 = v128.load(changetype<usize>(q.Z), 16)
+       q_Z_89xx = v128.load(changetype<usize>(q.Z), 32)
+
+       // [0..3, 4..7, 8..11] = q.T2d
+       q_T2d_0123 = v128.load(changetype<usize>(q.T2d), 0)
+       q_T2d_4567 = v128.load(changetype<usize>(q.T2d), 16)
+       q_T2d_89xx = v128.load(changetype<usize>(q.T2d), 32)
+
+       // r.X = p.Y + p.X
+       r_X_0123 = v128.add<i32>(p_Y_0123, p_X_0123)
+       r_X_4567 = v128.add<i32>(p_Y_4567, p_X_4567)
+       r_X_89xx = v128.add<i32>(p_Y_89xx, p_X_89xx)
+
+       // r.Y = p.Y - p.X
+       r_Y_0123 = v128.sub<i32>(p_Y_0123, p_X_0123)
+       r_Y_4567 = v128.sub<i32>(p_Y_4567, p_X_4567)
+       r_Y_89xx = v128.sub<i32>(p_Y_89xx, p_X_89xx)
+
+       // r.Z = r.X * q.YplusX
+       const q_YplusX_0123_19: v128 = i32x4.mul(q_YplusX_0123, v19)
+       const q_YplusX_4567_19: v128 = i32x4.mul(q_YplusX_4567, v19)
+       const q_YplusX_89xx_19: v128 = i32x4.mul(q_YplusX_89xx, v19)
+
+       // f[0]
+       fi = v128.extract_lane<i32>(r_X_0123, 0)
+       fv = i32x4.splat(fi)
+
+       h01 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+       h23 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+       h45 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567)
+       h67 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567)
+       h89 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx)
+
+       // f[1]
+       fi = v128.extract_lane<i32>(r_X_0123, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       h12 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+       h34 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+       h56 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567)
+       h78 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567)
+       h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YplusX_89xx, q_YplusX_89xx_19, m))
+
+       // f[2]
+       fi = v128.extract_lane<i32>(r_X_0123, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+       h23 = i64x2.add(h23, t0)
+       h45 = i64x2.add(h45, t1)
+       h67 = i64x2.add(h67, t2)
+       h89 = i64x2.add(h89, t3)
+       h01 = i64x2.add(h01, t4)
+
+       // f[3]
+       fi = v128.extract_lane<i32>(r_X_0123, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_YplusX_4567, q_YplusX_4567_19, m))
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+       h34 = i64x2.add(h34, t0)
+       h56 = i64x2.add(h56, t1)
+       h78 = i64x2.add(h78, t2)
+       h90 = i64x2.add(h90, t3)
+       h12 = i64x2.add(h12, t4)
+
+       // f[4]
+       fi = v128.extract_lane<i32>(r_X_4567, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+       h45 = i64x2.add(h45, t0)
+       h67 = i64x2.add(h67, t1)
+       h89 = i64x2.add(h89, t2)
+       h01 = i64x2.add(h01, t3)
+       h23 = i64x2.add(h23, t4)
+
+       // f[5]
+       fi = v128.extract_lane<i32>(r_X_4567, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YplusX_4567, q_YplusX_4567_19, m))
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+       h56 = i64x2.add(h56, t0)
+       h78 = i64x2.add(h78, t1)
+       h90 = i64x2.add(h90, t2)
+       h12 = i64x2.add(h12, t3)
+       h34 = i64x2.add(h34, t4)
+
+       // f[6]
+       fi = v128.extract_lane<i32>(r_X_4567, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+       h67 = i64x2.add(h67, t0)
+       h89 = i64x2.add(h89, t1)
+       h01 = i64x2.add(h01, t2)
+       h23 = i64x2.add(h23, t3)
+       h45 = i64x2.add(h45, t4)
+
+       // f[7]
+       fi = v128.extract_lane<i32>(r_X_4567, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_YplusX_0123, q_YplusX_0123_19, m))
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+       h78 = i64x2.add(h78, t0)
+       h90 = i64x2.add(h90, t1)
+       h12 = i64x2.add(h12, t2)
+       h34 = i64x2.add(h34, t3)
+       h56 = i64x2.add(h56, t4)
+
+       // f[8]
+       fi = v128.extract_lane<i32>(r_X_89xx, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+       h89 = i64x2.add(h89, t0)
+       h01 = i64x2.add(h01, t1)
+       h23 = i64x2.add(h23, t2)
+       h45 = i64x2.add(h45, t3)
+       h67 = i64x2.add(h67, t4)
+
+       // f[9]
+       fi = v128.extract_lane<i32>(r_X_89xx, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YplusX_0123, q_YplusX_0123_19, m))
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+       h90 = i64x2.add(h90, t0)
+       h12 = i64x2.add(h12, t1)
+       h34 = i64x2.add(h34, t2)
+       h56 = i64x2.add(h56, t3)
+       h78 = i64x2.add(h78, t4)
+
+       // extract scalars from vectors
+       h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+       h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+       h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+       h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+       h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+       h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+       h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+       h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+       h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+       h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+       // carry
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + (1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + (1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + (1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + (1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + (1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + (1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + (1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + (1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       r_Z_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       r_Z_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       r_Z_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+       // r.Y = r.Y * q.YminusX
+       const q_YminusX_0123_19: v128 = i32x4.mul(q_YminusX_0123, v19)
+       const q_YminusX_4567_19: v128 = i32x4.mul(q_YminusX_4567, v19)
+       const q_YminusX_89xx_19: v128 = i32x4.mul(q_YminusX_89xx, v19)
+
+       // f[0]
+       fi = v128.extract_lane<i32>(r_Y_0123, 0)
+       fv = i32x4.splat(fi)
+
+       h01 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+       h23 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+       h45 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567)
+       h67 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567)
+       h89 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx)
+
+       // f[1]
+       fi = v128.extract_lane<i32>(r_Y_0123, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       h12 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+       h34 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+       h56 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567)
+       h78 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567)
+       h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YminusX_89xx, q_YminusX_89xx_19, m))
+
+       // f[2]
+       fi = v128.extract_lane<i32>(r_Y_0123, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+       h23 = i64x2.add(h23, t0)
+       h45 = i64x2.add(h45, t1)
+       h67 = i64x2.add(h67, t2)
+       h89 = i64x2.add(h89, t3)
+       h01 = i64x2.add(h01, t4)
+
+       // f[3]
+       fi = v128.extract_lane<i32>(r_Y_0123, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_YminusX_4567, q_YminusX_4567_19, m))
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+       h34 = i64x2.add(h34, t0)
+       h56 = i64x2.add(h56, t1)
+       h78 = i64x2.add(h78, t2)
+       h90 = i64x2.add(h90, t3)
+       h12 = i64x2.add(h12, t4)
+
+       // f[4]
+       fi = v128.extract_lane<i32>(r_Y_4567, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+       h45 = i64x2.add(h45, t0)
+       h67 = i64x2.add(h67, t1)
+       h89 = i64x2.add(h89, t2)
+       h01 = i64x2.add(h01, t3)
+       h23 = i64x2.add(h23, t4)
+
+       // f[5]
+       fi = v128.extract_lane<i32>(r_Y_4567, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YminusX_4567, q_YminusX_4567_19, m))
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+       h56 = i64x2.add(h56, t0)
+       h78 = i64x2.add(h78, t1)
+       h90 = i64x2.add(h90, t2)
+       h12 = i64x2.add(h12, t3)
+       h34 = i64x2.add(h34, t4)
+
+       // f[6]
+       fi = v128.extract_lane<i32>(r_Y_4567, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+       h67 = i64x2.add(h67, t0)
+       h89 = i64x2.add(h89, t1)
+       h01 = i64x2.add(h01, t2)
+       h23 = i64x2.add(h23, t3)
+       h45 = i64x2.add(h45, t4)
+
+       // f[7]
+       fi = v128.extract_lane<i32>(r_Y_4567, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_YminusX_0123, q_YminusX_0123_19, m))
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+       h78 = i64x2.add(h78, t0)
+       h90 = i64x2.add(h90, t1)
+       h12 = i64x2.add(h12, t2)
+       h34 = i64x2.add(h34, t3)
+       h56 = i64x2.add(h56, t4)
+
+       // f[8]
+       fi = v128.extract_lane<i32>(r_Y_89xx, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+       h89 = i64x2.add(h89, t0)
+       h01 = i64x2.add(h01, t1)
+       h23 = i64x2.add(h23, t2)
+       h45 = i64x2.add(h45, t3)
+       h67 = i64x2.add(h67, t4)
+
+       // f[9]
+       fi = v128.extract_lane<i32>(r_Y_89xx, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YminusX_0123, q_YminusX_0123_19, m))
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+       h90 = i64x2.add(h90, t0)
+       h12 = i64x2.add(h12, t1)
+       h34 = i64x2.add(h34, t2)
+       h56 = i64x2.add(h56, t3)
+       h78 = i64x2.add(h78, t4)
+
+       // extract scalars from vectors
+       h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+       h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+       h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+       h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+       h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+       h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+       h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+       h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+       h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+       h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+       // carry
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + (1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + (1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + (1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + (1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + (1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + (1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + (1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + (1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       r_Y_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       r_Y_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       r_Y_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+       // r.T = q.T2d * p.T
+       const p_T_0123_19: v128 = i32x4.mul(p_T_0123, v19)
+       const p_T_4567_19: v128 = i32x4.mul(p_T_4567, v19)
+       const p_T_89xx_19: v128 = i32x4.mul(p_T_89xx, v19)
+
+       // f[0]
+       fi = v128.extract_lane<i32>(q_T2d_0123, 0)
+       fv = i32x4.splat(fi)
+
+       h01 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       h23 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       h45 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+       h67 = i64x2.extmul_high_i32x4_s(fv, p_T_4567)
+       h89 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx)
+
+       // f[1]
+       fi = v128.extract_lane<i32>(q_T2d_0123, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       h12 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       h34 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       h56 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+       h78 = i64x2.extmul_high_i32x4_s(fv, p_T_4567)
+       h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_89xx, p_T_89xx_19, m))
+
+       // f[2]
+       fi = v128.extract_lane<i32>(q_T2d_0123, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h23 = i64x2.add(h23, t0)
+       h45 = i64x2.add(h45, t1)
+       h67 = i64x2.add(h67, t2)
+       h89 = i64x2.add(h89, t3)
+       h01 = i64x2.add(h01, t4)
+
+       // f[3]
+       fi = v128.extract_lane<i32>(q_T2d_0123, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(p_T_4567, p_T_4567_19, m))
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h34 = i64x2.add(h34, t0)
+       h56 = i64x2.add(h56, t1)
+       h78 = i64x2.add(h78, t2)
+       h90 = i64x2.add(h90, t3)
+       h12 = i64x2.add(h12, t4)
+
+       // f[4]
+       fi = v128.extract_lane<i32>(q_T2d_4567, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h45 = i64x2.add(h45, t0)
+       h67 = i64x2.add(h67, t1)
+       h89 = i64x2.add(h89, t2)
+       h01 = i64x2.add(h01, t3)
+       h23 = i64x2.add(h23, t4)
+
+       // f[5]
+       fi = v128.extract_lane<i32>(q_T2d_4567, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_4567, p_T_4567_19, m))
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h56 = i64x2.add(h56, t0)
+       h78 = i64x2.add(h78, t1)
+       h90 = i64x2.add(h90, t2)
+       h12 = i64x2.add(h12, t3)
+       h34 = i64x2.add(h34, t4)
+
+       // f[6]
+       fi = v128.extract_lane<i32>(q_T2d_4567, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h67 = i64x2.add(h67, t0)
+       h89 = i64x2.add(h89, t1)
+       h01 = i64x2.add(h01, t2)
+       h23 = i64x2.add(h23, t3)
+       h45 = i64x2.add(h45, t4)
+
+       // f[7]
+       fi = v128.extract_lane<i32>(q_T2d_4567, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(p_T_0123, p_T_0123_19, m))
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h78 = i64x2.add(h78, t0)
+       h90 = i64x2.add(h90, t1)
+       h12 = i64x2.add(h12, t2)
+       h34 = i64x2.add(h34, t3)
+       h56 = i64x2.add(h56, t4)
+
+       // f[8]
+       fi = v128.extract_lane<i32>(q_T2d_89xx, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h89 = i64x2.add(h89, t0)
+       h01 = i64x2.add(h01, t1)
+       h23 = i64x2.add(h23, t2)
+       h45 = i64x2.add(h45, t3)
+       h67 = i64x2.add(h67, t4)
+
+       // f[9]
+       fi = v128.extract_lane<i32>(q_T2d_89xx, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_0123, p_T_0123_19, m))
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h90 = i64x2.add(h90, t0)
+       h12 = i64x2.add(h12, t1)
+       h34 = i64x2.add(h34, t2)
+       h56 = i64x2.add(h56, t3)
+       h78 = i64x2.add(h78, t4)
+
+       // extract scalars from vectors
+       h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+       h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+       h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+       h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+       h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+       h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+       h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+       h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+       h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+       h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+       // carry
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + (1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + (1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + (1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + (1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + (1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + (1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + (1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + (1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       r_T_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       r_T_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       r_T_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+       // r.X = p.Z * q.Z
+       const q_Z_0123_19: v128 = i32x4.mul(q_Z_0123, v19)
+       const q_Z_4567_19: v128 = i32x4.mul(q_Z_4567, v19)
+       const q_Z_89xx_19: v128 = i32x4.mul(q_Z_89xx, v19)
+
+       // f[0]
+       fi = v128.extract_lane<i32>(p_Z_0123, 0)
+       fv = i32x4.splat(fi)
+
+       h01 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+       h23 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+       h45 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567)
+       h67 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567)
+       h89 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx)
+
+       // f[1]
+       fi = v128.extract_lane<i32>(p_Z_0123, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       h12 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+       h34 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+       h56 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567)
+       h78 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567)
+       h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_Z_89xx, q_Z_89xx_19, m))
+
+       // f[2]
+       fi = v128.extract_lane<i32>(p_Z_0123, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+       h23 = i64x2.add(h23, t0)
+       h45 = i64x2.add(h45, t1)
+       h67 = i64x2.add(h67, t2)
+       h89 = i64x2.add(h89, t3)
+       h01 = i64x2.add(h01, t4)
+
+       // f[3]
+       fi = v128.extract_lane<i32>(p_Z_0123, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_Z_4567, q_Z_4567_19, m))
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+       h34 = i64x2.add(h34, t0)
+       h56 = i64x2.add(h56, t1)
+       h78 = i64x2.add(h78, t2)
+       h90 = i64x2.add(h90, t3)
+       h12 = i64x2.add(h12, t4)
+
+       // f[4]
+       fi = v128.extract_lane<i32>(p_Z_4567, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+       h45 = i64x2.add(h45, t0)
+       h67 = i64x2.add(h67, t1)
+       h89 = i64x2.add(h89, t2)
+       h01 = i64x2.add(h01, t3)
+       h23 = i64x2.add(h23, t4)
+
+       // f[5]
+       fi = v128.extract_lane<i32>(p_Z_4567, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_Z_4567, q_Z_4567_19, m))
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+       h56 = i64x2.add(h56, t0)
+       h78 = i64x2.add(h78, t1)
+       h90 = i64x2.add(h90, t2)
+       h12 = i64x2.add(h12, t3)
+       h34 = i64x2.add(h34, t4)
+
+       // f[6]
+       fi = v128.extract_lane<i32>(p_Z_4567, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+       h67 = i64x2.add(h67, t0)
+       h89 = i64x2.add(h89, t1)
+       h01 = i64x2.add(h01, t2)
+       h23 = i64x2.add(h23, t3)
+       h45 = i64x2.add(h45, t4)
+
+       // f[7]
+       fi = v128.extract_lane<i32>(p_Z_4567, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_Z_0123, q_Z_0123_19, m))
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+       h78 = i64x2.add(h78, t0)
+       h90 = i64x2.add(h90, t1)
+       h12 = i64x2.add(h12, t2)
+       h34 = i64x2.add(h34, t3)
+       h56 = i64x2.add(h56, t4)
+
+       // f[8]
+       fi = v128.extract_lane<i32>(p_Z_89xx, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+       h89 = i64x2.add(h89, t0)
+       h01 = i64x2.add(h01, t1)
+       h23 = i64x2.add(h23, t2)
+       h45 = i64x2.add(h45, t3)
+       h67 = i64x2.add(h67, t4)
+
+       // f[9]
+       fi = v128.extract_lane<i32>(p_Z_89xx, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_Z_0123, q_Z_0123_19, m))
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+       h90 = i64x2.add(h90, t0)
+       h12 = i64x2.add(h12, t1)
+       h34 = i64x2.add(h34, t2)
+       h56 = i64x2.add(h56, t3)
+       h78 = i64x2.add(h78, t4)
+
+       // extract scalars from vectors
+       h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+       h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+       h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+       h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+       h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+       h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+       h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+       h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+       h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+       h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+       // carry
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + (1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + (1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + (1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + (1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + (1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + (1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + (1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + (1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       r_X_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       r_X_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       r_X_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+       let t_0123: v128, t_4567: v128, t_89xx: v128
+
+       // t = 2*r.X
+       t_0123 = v128.shl<i32>(r_X_0123, 1)
+       t_4567 = v128.shl<i32>(r_X_4567, 1)
+       t_89xx = v128.shl<i32>(r_X_89xx, 1)
+
+       // r.X = r.Z - r.Y
+       r_X_0123 = v128.sub<i32>(r_Z_0123, r_Y_0123)
+       r_X_4567 = v128.sub<i32>(r_Z_4567, r_Y_4567)
+       r_X_89xx = v128.sub<i32>(r_Z_89xx, r_Y_89xx)
+
+       // r.Y = r.Z + r.Y
+       r_Y_0123 = v128.add<i32>(r_Z_0123, r_Y_0123)
+       r_Y_4567 = v128.add<i32>(r_Z_4567, r_Y_4567)
+       r_Y_89xx = v128.add<i32>(r_Z_89xx, r_Y_89xx)
+
+       // r.Z = t + r.T
+       r_Z_0123 = v128.add<i32>(t_0123, r_T_0123)
+       r_Z_4567 = v128.add<i32>(t_4567, r_T_4567)
+       r_Z_89xx = v128.add<i32>(t_89xx, r_T_89xx)
+
+       // r.T = t - r.T
+       r_T_0123 = v128.sub<i32>(t_0123, r_T_0123)
+       r_T_4567 = v128.sub<i32>(t_4567, r_T_4567)
+       r_T_89xx = v128.sub<i32>(t_89xx, r_T_89xx)
+
+       // r.X = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(r.X), r_X_0123, 0)
+       v128.store(changetype<usize>(r.X), r_X_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(r.X), r_X_89xx, 0, 32)
+
+       // r.Y = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(r.Y), r_Y_0123, 0)
+       v128.store(changetype<usize>(r.Y), r_Y_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(r.Y), r_Y_89xx, 0, 32)
+
+       // r.Z = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(r.Z), r_Z_0123, 0)
+       v128.store(changetype<usize>(r.Z), r_Z_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(r.Z), r_Z_89xx, 0, 32)
+
+       // r.T = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(r.T), r_T_0123, 0)
+       v128.store(changetype<usize>(r.T), r_T_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(r.T), r_T_89xx, 0, 32)
+
 }
 
 /**
@@ -322,99 +2082,902 @@ export function ge_add_cached (r: ge_p1p1, p: ge_p3, q: ge_cached): void {
 //@ts-expect-error
 @inline
 export function ge_add_precomp (r: ge_p1p1, p: ge_p3, q: ge_precomp): void {
-       const rX_ptr = changetype<usize>(r.X)
-       const rY_ptr = changetype<usize>(r.Y)
-       const rZ_ptr = changetype<usize>(r.Z)
-       const rT_ptr = changetype<usize>(r.T)
-
-       const pX_ptr = changetype<usize>(p.X)
-       const pY_ptr = changetype<usize>(p.Y)
-       const pZ_ptr = changetype<usize>(p.Z)
-
-       let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
-       let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
-       const pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32)
-       const pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32)
-
-       // rX = pY + pX
-       rX0 = v128.add<i32>(pY0, pX0)
-       rX1 = v128.add<i32>(pY1, pX1)
-       rX2 = v128.add<i32>(pY2, pX2)
-
-       // rY = pY - pX
-       rY0 = v128.sub<i32>(pY0, pX0)
-       rY1 = v128.sub<i32>(pY1, pX1)
-       rY2 = v128.sub<i32>(pY2, pX2)
-
-       // store r.X
-       v128.store(rX_ptr, rX0)
-       v128.store(rX_ptr, rX1, 16)
-       v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
-       // store r.Y
-       v128.store(rY_ptr, rY0)
-       v128.store(rY_ptr, rY1, 16)
-       v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
-       // not yet converted to inline for size and complexity
-       fe_mul(r.Z, r.X, q.yplusx)
-       fe_mul(r.Y, r.Y, q.yminusx)
-       fe_mul(r.T, q.xy2d, p.T)
 
-       // remaining arithmetic done inline to avoid unnecessary load/store
-       rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
-       rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
-       let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32)
-       let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32)
-
-       let pZ0 = v128.load(pZ_ptr, 0)
-       let pZ1 = v128.load(pZ_ptr, 16)
-       let pZ2 = v128.load(pZ_ptr, 32)
-
-       // pZ = 2*pZ
-       pZ0 = v128.shl<i32>(pZ0, 1)
-       pZ1 = v128.shl<i32>(pZ1, 1)
-       pZ2 = v128.shl<i32>(pZ2, 1)
-
-       // rX = rZ - rY
-       rX0 = v128.sub<i32>(rZ0, rY0)
-       rX1 = v128.sub<i32>(rZ1, rY1)
-       rX2 = v128.sub<i32>(rZ2, rY2)
-
-       // rY = rZ + rY
-       rY0 = v128.add<i32>(rZ0, rY0)
-       rY1 = v128.add<i32>(rZ1, rY1)
-       rY2 = v128.add<i32>(rZ2, rY2)
-
-       // rZ = pZ + rT
-       rZ0 = v128.add<i32>(pZ0, rT0)
-       rZ1 = v128.add<i32>(pZ1, rT1)
-       rZ2 = v128.add<i32>(pZ2, rT2)
-
-       // rT = pZ - rT
-       rT0 = v128.sub<i32>(pZ0, rT0)
-       rT1 = v128.sub<i32>(pZ1, rT1)
-       rT2 = v128.sub<i32>(pZ2, rT2)
-
-       // store r.X
-       v128.store(rX_ptr, rX0)
-       v128.store(rX_ptr, rX1, 16)
-       v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
-       // store r.Y
-       v128.store(rY_ptr, rY0)
-       v128.store(rY_ptr, rY1, 16)
-       v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
-       // store r.Z
-       v128.store(rZ_ptr, rZ0)
-       v128.store(rZ_ptr, rZ1, 16)
-       v128.store_lane<u64>(rZ_ptr, rZ2, 0, 32)
-
-       // store r.T
-       v128.store(rT_ptr, rT0)
-       v128.store(rT_ptr, rT1, 16)
-       v128.store_lane<u64>(rT_ptr, rT2, 0, 32)
+       let 
+       r_X_0123: v128,
+       r_X_4567: v128,
+       r_X_89xx: v128
+,
+       r_Y_0123: v128,
+       r_Y_4567: v128,
+       r_Y_89xx: v128
+,
+       r_Z_0123: v128,
+       r_Z_4567: v128,
+       r_Z_89xx: v128
+,
+       r_T_0123: v128,
+       r_T_4567: v128,
+       r_T_89xx: v128
+,
+       p_X_0123: v128,
+       p_X_4567: v128,
+       p_X_89xx: v128
+,
+       p_Y_0123: v128,
+       p_Y_4567: v128,
+       p_Y_89xx: v128
+,
+       p_Z_0123: v128,
+       p_Z_4567: v128,
+       p_Z_89xx: v128
+,
+       p_T_0123: v128,
+       p_T_4567: v128,
+       p_T_89xx: v128
+,
+       q_yplusx_0123: v128,
+       q_yplusx_4567: v128,
+       q_yplusx_89xx: v128
+,
+       q_yminusx_0123: v128,
+       q_yminusx_4567: v128,
+       q_yminusx_89xx: v128
+,
+       q_xy2d_0123: v128,
+       q_xy2d_4567: v128,
+       q_xy2d_89xx: v128
+
+       // mask to select even lane from first vector and odd lane from second
+       const m: v128 = i32x4(-1, 0, -1, 0)
+
+       // mask to double odd-on-odd indexes
+       const odds: v128 = i32x4(0, -1, 0, -1)
+
+       // factor of 19 for wrap from h9 to h0
+       const v19: v128 = i32x4.splat(19)
+
+       let c0: i64
+       let c1: i64
+       let c2: i64
+       let c3: i64
+       let c4: i64
+       let c5: i64
+       let c6: i64
+       let c7: i64
+       let c8: i64
+       let c9: i64
+
+       let f0: i64
+       let f1: i64
+       let f2: i64
+       let f3: i64
+       let f4: i64
+       let f5: i64
+       let f6: i64
+       let f7: i64
+       let f8: i64
+       let f9: i64
+
+       let f0_2: i64
+       let f1_2: i64
+       let f2_2: i64
+       let f3_2: i64
+       let f4_2: i64
+       let f5_2: i64
+       let f6_2: i64
+       let f7_2: i64
+       let f8_2: i64
+       let f9_2: i64
+
+       let f5_19: i64
+       let f6_19: i64
+       let f7_19: i64
+       let f8_19: i64
+       let f9_19: i64
+
+       let f5_38: i64
+       let f7_38: i64
+       let f9_38: i64
+
+       let fi: i32
+       let fv: v128
+
+       let h0: i64
+       let h1: i64
+       let h2: i64
+       let h3: i64
+       let h4: i64
+       let h5: i64
+       let h6: i64
+       let h7: i64
+       let h8: i64
+       let h9: i64
+
+       let h01: v128
+       let h12: v128
+       let h23: v128
+       let h34: v128
+       let h45: v128
+       let h56: v128
+       let h67: v128
+       let h78: v128
+       let h89: v128
+       let h90: v128
+
+       let t0: v128
+       let t1: v128
+       let t2: v128
+       let t3: v128
+       let t4: v128
+
+       // [0..3, 4..7, 8..11] = r.X
+       r_X_0123 = v128.load(changetype<usize>(r.X), 0)
+       r_X_4567 = v128.load(changetype<usize>(r.X), 16)
+       r_X_89xx = v128.load(changetype<usize>(r.X), 32)
+
+       // [0..3, 4..7, 8..11] = r.Y
+       r_Y_0123 = v128.load(changetype<usize>(r.Y), 0)
+       r_Y_4567 = v128.load(changetype<usize>(r.Y), 16)
+       r_Y_89xx = v128.load(changetype<usize>(r.Y), 32)
+
+       // [0..3, 4..7, 8..11] = r.Z
+       r_Z_0123 = v128.load(changetype<usize>(r.Z), 0)
+       r_Z_4567 = v128.load(changetype<usize>(r.Z), 16)
+       r_Z_89xx = v128.load(changetype<usize>(r.Z), 32)
+
+       // [0..3, 4..7, 8..11] = r.T
+       r_T_0123 = v128.load(changetype<usize>(r.T), 0)
+       r_T_4567 = v128.load(changetype<usize>(r.T), 16)
+       r_T_89xx = v128.load(changetype<usize>(r.T), 32)
+
+       // [0..3, 4..7, 8..11] = p.X
+       p_X_0123 = v128.load(changetype<usize>(p.X), 0)
+       p_X_4567 = v128.load(changetype<usize>(p.X), 16)
+       p_X_89xx = v128.load(changetype<usize>(p.X), 32)
+
+       // [0..3, 4..7, 8..11] = p.Y
+       p_Y_0123 = v128.load(changetype<usize>(p.Y), 0)
+       p_Y_4567 = v128.load(changetype<usize>(p.Y), 16)
+       p_Y_89xx = v128.load(changetype<usize>(p.Y), 32)
+
+       // [0..3, 4..7, 8..11] = p.Z
+       p_Z_0123 = v128.load(changetype<usize>(p.Z), 0)
+       p_Z_4567 = v128.load(changetype<usize>(p.Z), 16)
+       p_Z_89xx = v128.load(changetype<usize>(p.Z), 32)
+
+       // [0..3, 4..7, 8..11] = p.T
+       p_T_0123 = v128.load(changetype<usize>(p.T), 0)
+       p_T_4567 = v128.load(changetype<usize>(p.T), 16)
+       p_T_89xx = v128.load(changetype<usize>(p.T), 32)
+
+       // [0..3, 4..7, 8..11] = q.yplusx
+       q_yplusx_0123 = v128.load(changetype<usize>(q.yplusx), 0)
+       q_yplusx_4567 = v128.load(changetype<usize>(q.yplusx), 16)
+       q_yplusx_89xx = v128.load(changetype<usize>(q.yplusx), 32)
+
+       // [0..3, 4..7, 8..11] = q.yminusx
+       q_yminusx_0123 = v128.load(changetype<usize>(q.yminusx), 0)
+       q_yminusx_4567 = v128.load(changetype<usize>(q.yminusx), 16)
+       q_yminusx_89xx = v128.load(changetype<usize>(q.yminusx), 32)
+
+       // [0..3, 4..7, 8..11] = q.xy2d
+       q_xy2d_0123 = v128.load(changetype<usize>(q.xy2d), 0)
+       q_xy2d_4567 = v128.load(changetype<usize>(q.xy2d), 16)
+       q_xy2d_89xx = v128.load(changetype<usize>(q.xy2d), 32)
+
+       // r.X = p.Y + p.X
+       r_X_0123 = v128.add<i32>(p_Y_0123, p_X_0123)
+       r_X_4567 = v128.add<i32>(p_Y_4567, p_X_4567)
+       r_X_89xx = v128.add<i32>(p_Y_89xx, p_X_89xx)
+
+       // r.Y = p.Y - p.X
+       r_Y_0123 = v128.sub<i32>(p_Y_0123, p_X_0123)
+       r_Y_4567 = v128.sub<i32>(p_Y_4567, p_X_4567)
+       r_Y_89xx = v128.sub<i32>(p_Y_89xx, p_X_89xx)
+
+       // r.Z = r.X * q.yplusx
+       const q_yplusx_0123_19: v128 = i32x4.mul(q_yplusx_0123, v19)
+       const q_yplusx_4567_19: v128 = i32x4.mul(q_yplusx_4567, v19)
+       const q_yplusx_89xx_19: v128 = i32x4.mul(q_yplusx_89xx, v19)
+
+       // f[0]
+       fi = v128.extract_lane<i32>(r_X_0123, 0)
+       fv = i32x4.splat(fi)
+
+       h01 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+       h23 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+       h45 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567)
+       h67 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567)
+       h89 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx)
+
+       // f[1]
+       fi = v128.extract_lane<i32>(r_X_0123, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       h12 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+       h34 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+       h56 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567)
+       h78 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567)
+       h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yplusx_89xx, q_yplusx_89xx_19, m))
+
+       // f[2]
+       fi = v128.extract_lane<i32>(r_X_0123, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+       h23 = i64x2.add(h23, t0)
+       h45 = i64x2.add(h45, t1)
+       h67 = i64x2.add(h67, t2)
+       h89 = i64x2.add(h89, t3)
+       h01 = i64x2.add(h01, t4)
+
+       // f[3]
+       fi = v128.extract_lane<i32>(r_X_0123, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_yplusx_4567, q_yplusx_4567_19, m))
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+       h34 = i64x2.add(h34, t0)
+       h56 = i64x2.add(h56, t1)
+       h78 = i64x2.add(h78, t2)
+       h90 = i64x2.add(h90, t3)
+       h12 = i64x2.add(h12, t4)
+
+       // f[4]
+       fi = v128.extract_lane<i32>(r_X_4567, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+       h45 = i64x2.add(h45, t0)
+       h67 = i64x2.add(h67, t1)
+       h89 = i64x2.add(h89, t2)
+       h01 = i64x2.add(h01, t3)
+       h23 = i64x2.add(h23, t4)
+
+       // f[5]
+       fi = v128.extract_lane<i32>(r_X_4567, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yplusx_4567, q_yplusx_4567_19, m))
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+       h56 = i64x2.add(h56, t0)
+       h78 = i64x2.add(h78, t1)
+       h90 = i64x2.add(h90, t2)
+       h12 = i64x2.add(h12, t3)
+       h34 = i64x2.add(h34, t4)
+
+       // f[6]
+       fi = v128.extract_lane<i32>(r_X_4567, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+       h67 = i64x2.add(h67, t0)
+       h89 = i64x2.add(h89, t1)
+       h01 = i64x2.add(h01, t2)
+       h23 = i64x2.add(h23, t3)
+       h45 = i64x2.add(h45, t4)
+
+       // f[7]
+       fi = v128.extract_lane<i32>(r_X_4567, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_yplusx_0123, q_yplusx_0123_19, m))
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+       h78 = i64x2.add(h78, t0)
+       h90 = i64x2.add(h90, t1)
+       h12 = i64x2.add(h12, t2)
+       h34 = i64x2.add(h34, t3)
+       h56 = i64x2.add(h56, t4)
+
+       // f[8]
+       fi = v128.extract_lane<i32>(r_X_89xx, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+       h89 = i64x2.add(h89, t0)
+       h01 = i64x2.add(h01, t1)
+       h23 = i64x2.add(h23, t2)
+       h45 = i64x2.add(h45, t3)
+       h67 = i64x2.add(h67, t4)
+
+       // f[9]
+       fi = v128.extract_lane<i32>(r_X_89xx, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yplusx_0123, q_yplusx_0123_19, m))
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+       h90 = i64x2.add(h90, t0)
+       h12 = i64x2.add(h12, t1)
+       h34 = i64x2.add(h34, t2)
+       h56 = i64x2.add(h56, t3)
+       h78 = i64x2.add(h78, t4)
+
+       // extract scalars from vectors
+       h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+       h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+       h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+       h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+       h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+       h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+       h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+       h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+       h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+       h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+       // carry
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + (1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + (1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + (1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + (1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + (1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + (1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + (1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + (1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       r_Z_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       r_Z_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       r_Z_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+       // r.Y = r.Y * q.yminusx
+       const q_yminusx_0123_19: v128 = i32x4.mul(q_yminusx_0123, v19)
+       const q_yminusx_4567_19: v128 = i32x4.mul(q_yminusx_4567, v19)
+       const q_yminusx_89xx_19: v128 = i32x4.mul(q_yminusx_89xx, v19)
+
+       // f[0]
+       fi = v128.extract_lane<i32>(r_Y_0123, 0)
+       fv = i32x4.splat(fi)
+
+       h01 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+       h23 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+       h45 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567)
+       h67 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567)
+       h89 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx)
+
+       // f[1]
+       fi = v128.extract_lane<i32>(r_Y_0123, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       h12 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+       h34 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+       h56 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567)
+       h78 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567)
+       h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yminusx_89xx, q_yminusx_89xx_19, m))
+
+       // f[2]
+       fi = v128.extract_lane<i32>(r_Y_0123, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+       h23 = i64x2.add(h23, t0)
+       h45 = i64x2.add(h45, t1)
+       h67 = i64x2.add(h67, t2)
+       h89 = i64x2.add(h89, t3)
+       h01 = i64x2.add(h01, t4)
+
+       // f[3]
+       fi = v128.extract_lane<i32>(r_Y_0123, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_yminusx_4567, q_yminusx_4567_19, m))
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+       h34 = i64x2.add(h34, t0)
+       h56 = i64x2.add(h56, t1)
+       h78 = i64x2.add(h78, t2)
+       h90 = i64x2.add(h90, t3)
+       h12 = i64x2.add(h12, t4)
+
+       // f[4]
+       fi = v128.extract_lane<i32>(r_Y_4567, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+       h45 = i64x2.add(h45, t0)
+       h67 = i64x2.add(h67, t1)
+       h89 = i64x2.add(h89, t2)
+       h01 = i64x2.add(h01, t3)
+       h23 = i64x2.add(h23, t4)
+
+       // f[5]
+       fi = v128.extract_lane<i32>(r_Y_4567, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yminusx_4567, q_yminusx_4567_19, m))
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+       h56 = i64x2.add(h56, t0)
+       h78 = i64x2.add(h78, t1)
+       h90 = i64x2.add(h90, t2)
+       h12 = i64x2.add(h12, t3)
+       h34 = i64x2.add(h34, t4)
+
+       // f[6]
+       fi = v128.extract_lane<i32>(r_Y_4567, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+       h67 = i64x2.add(h67, t0)
+       h89 = i64x2.add(h89, t1)
+       h01 = i64x2.add(h01, t2)
+       h23 = i64x2.add(h23, t3)
+       h45 = i64x2.add(h45, t4)
+
+       // f[7]
+       fi = v128.extract_lane<i32>(r_Y_4567, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_yminusx_0123, q_yminusx_0123_19, m))
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+       h78 = i64x2.add(h78, t0)
+       h90 = i64x2.add(h90, t1)
+       h12 = i64x2.add(h12, t2)
+       h34 = i64x2.add(h34, t3)
+       h56 = i64x2.add(h56, t4)
+
+       // f[8]
+       fi = v128.extract_lane<i32>(r_Y_89xx, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+       h89 = i64x2.add(h89, t0)
+       h01 = i64x2.add(h01, t1)
+       h23 = i64x2.add(h23, t2)
+       h45 = i64x2.add(h45, t3)
+       h67 = i64x2.add(h67, t4)
+
+       // f[9]
+       fi = v128.extract_lane<i32>(r_Y_89xx, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yminusx_0123, q_yminusx_0123_19, m))
+       t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+       h90 = i64x2.add(h90, t0)
+       h12 = i64x2.add(h12, t1)
+       h34 = i64x2.add(h34, t2)
+       h56 = i64x2.add(h56, t3)
+       h78 = i64x2.add(h78, t4)
+
+       // extract scalars from vectors
+       h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+       h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+       h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+       h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+       h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+       h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+       h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+       h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+       h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+       h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+       // carry
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + (1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + (1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + (1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + (1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + (1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + (1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + (1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + (1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       r_Y_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       r_Y_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       r_Y_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+       // r.T = q.xy2d * p.T
+       const p_T_0123_19: v128 = i32x4.mul(p_T_0123, v19)
+       const p_T_4567_19: v128 = i32x4.mul(p_T_4567, v19)
+       const p_T_89xx_19: v128 = i32x4.mul(p_T_89xx, v19)
+
+       // f[0]
+       fi = v128.extract_lane<i32>(q_xy2d_0123, 0)
+       fv = i32x4.splat(fi)
+
+       h01 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       h23 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       h45 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+       h67 = i64x2.extmul_high_i32x4_s(fv, p_T_4567)
+       h89 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx)
+
+       // f[1]
+       fi = v128.extract_lane<i32>(q_xy2d_0123, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       h12 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       h34 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       h56 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+       h78 = i64x2.extmul_high_i32x4_s(fv, p_T_4567)
+       h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_89xx, p_T_89xx_19, m))
+
+       // f[2]
+       fi = v128.extract_lane<i32>(q_xy2d_0123, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h23 = i64x2.add(h23, t0)
+       h45 = i64x2.add(h45, t1)
+       h67 = i64x2.add(h67, t2)
+       h89 = i64x2.add(h89, t3)
+       h01 = i64x2.add(h01, t4)
+
+       // f[3]
+       fi = v128.extract_lane<i32>(q_xy2d_0123, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(p_T_4567, p_T_4567_19, m))
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h34 = i64x2.add(h34, t0)
+       h56 = i64x2.add(h56, t1)
+       h78 = i64x2.add(h78, t2)
+       h90 = i64x2.add(h90, t3)
+       h12 = i64x2.add(h12, t4)
+
+       // f[4]
+       fi = v128.extract_lane<i32>(q_xy2d_4567, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h45 = i64x2.add(h45, t0)
+       h67 = i64x2.add(h67, t1)
+       h89 = i64x2.add(h89, t2)
+       h01 = i64x2.add(h01, t3)
+       h23 = i64x2.add(h23, t4)
+
+       // f[5]
+       fi = v128.extract_lane<i32>(q_xy2d_4567, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_4567, p_T_4567_19, m))
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h56 = i64x2.add(h56, t0)
+       h78 = i64x2.add(h78, t1)
+       h90 = i64x2.add(h90, t2)
+       h12 = i64x2.add(h12, t3)
+       h34 = i64x2.add(h34, t4)
+
+       // f[6]
+       fi = v128.extract_lane<i32>(q_xy2d_4567, 2)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h67 = i64x2.add(h67, t0)
+       h89 = i64x2.add(h89, t1)
+       h01 = i64x2.add(h01, t2)
+       h23 = i64x2.add(h23, t3)
+       h45 = i64x2.add(h45, t4)
+
+       // f[7]
+       fi = v128.extract_lane<i32>(q_xy2d_4567, 3)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(p_T_0123, p_T_0123_19, m))
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h78 = i64x2.add(h78, t0)
+       h90 = i64x2.add(h90, t1)
+       h12 = i64x2.add(h12, t2)
+       h34 = i64x2.add(h34, t3)
+       h56 = i64x2.add(h56, t4)
+
+       // f[8]
+       fi = v128.extract_lane<i32>(q_xy2d_89xx, 0)
+       fv = i32x4.splat(fi)
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h89 = i64x2.add(h89, t0)
+       h01 = i64x2.add(h01, t1)
+       h23 = i64x2.add(h23, t2)
+       h45 = i64x2.add(h45, t3)
+       h67 = i64x2.add(h67, t4)
+
+       // f[9]
+       fi = v128.extract_lane<i32>(q_xy2d_89xx, 1)
+       fv = i32x4.splat(fi)
+       fv = i32x4.add(fv, v128.and(fv, odds))
+
+       t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_0123, p_T_0123_19, m))
+       t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123_19)
+       t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+       t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+       t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+       h90 = i64x2.add(h90, t0)
+       h12 = i64x2.add(h12, t1)
+       h34 = i64x2.add(h34, t2)
+       h56 = i64x2.add(h56, t3)
+       h78 = i64x2.add(h78, t4)
+
+       // extract scalars from vectors
+       h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+       h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+       h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+       h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+       h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+       h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+       h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+       h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+       h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+       h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+       // carry
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+
+       c1 = (h1 + (1 << 24)) >> 25
+       h2 += c1
+       h1 -= c1 << 25
+       c5 = (h5 + (1 << 24)) >> 25
+       h6 += c5
+       h5 -= c5 << 25
+
+       c2 = (h2 + (1 << 25)) >> 26
+       h3 += c2
+       h2 -= c2 << 26
+       c6 = (h6 + (1 << 25)) >> 26
+       h7 += c6
+       h6 -= c6 << 26
+
+       c3 = (h3 + (1 << 24)) >> 25
+       h4 += c3
+       h3 -= c3 << 25
+       c7 = (h7 + (1 << 24)) >> 25
+       h8 += c7
+       h7 -= c7 << 25
+
+       c4 = (h4 + (1 << 25)) >> 26
+       h5 += c4
+       h4 -= c4 << 26
+       c8 = (h8 + (1 << 25)) >> 26
+       h9 += c8
+       h8 -= c8 << 26
+
+       c9 = (h9 + (1 << 24)) >> 25
+       h0 += c9 * 19
+       h9 -= c9 << 25
+
+       c0 = (h0 + (1 << 25)) >> 26
+       h1 += c0
+       h0 -= c0 << 26
+
+       // assign reduced results to output
+       r_T_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+       r_T_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+       r_T_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+       // p.Z = 2*p.Z
+       p_Z_0123 = v128.shl<i32>(p_Z_0123, 1)
+       p_Z_4567 = v128.shl<i32>(p_Z_4567, 1)
+       p_Z_89xx = v128.shl<i32>(p_Z_89xx, 1)
+
+       // r.X = r.Z - r.Y
+       r_X_0123 = v128.sub<i32>(r_Z_0123, r_Y_0123)
+       r_X_4567 = v128.sub<i32>(r_Z_4567, r_Y_4567)
+       r_X_89xx = v128.sub<i32>(r_Z_89xx, r_Y_89xx)
+
+       // r.Y = r.Z + r.Y
+       r_Y_0123 = v128.add<i32>(r_Z_0123, r_Y_0123)
+       r_Y_4567 = v128.add<i32>(r_Z_4567, r_Y_4567)
+       r_Y_89xx = v128.add<i32>(r_Z_89xx, r_Y_89xx)
+
+       // r.Z = p.Z + r.T
+       r_Z_0123 = v128.add<i32>(p_Z_0123, r_T_0123)
+       r_Z_4567 = v128.add<i32>(p_Z_4567, r_T_4567)
+       r_Z_89xx = v128.add<i32>(p_Z_89xx, r_T_89xx)
+
+       // r.T = p.Z - r.T
+       r_T_0123 = v128.sub<i32>(p_Z_0123, r_T_0123)
+       r_T_4567 = v128.sub<i32>(p_Z_4567, r_T_4567)
+       r_T_89xx = v128.sub<i32>(p_Z_89xx, r_T_89xx)
+
+       // r.X = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(r.X), r_X_0123, 0)
+       v128.store(changetype<usize>(r.X), r_X_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(r.X), r_X_89xx, 0, 32)
+
+       // r.Y = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(r.Y), r_Y_0123, 0)
+       v128.store(changetype<usize>(r.Y), r_Y_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(r.Y), r_Y_89xx, 0, 32)
+
+       // r.Z = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(r.Z), r_Z_0123, 0)
+       v128.store(changetype<usize>(r.Z), r_Z_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(r.Z), r_Z_89xx, 0, 32)
+
+       // r.T = [0..3, 4..7, 8..11]
+       v128.store(changetype<usize>(r.T), r_T_0123, 0)
+       v128.store(changetype<usize>(r.T), r_T_4567, 16)
+       v128.store_lane<u64>(changetype<usize>(r.T), r_T_89xx, 0, 32)
+
 }
 
 const ge_sub_p3_q_cached = new ge_cached()