* AssemblyScript compiler and type checker.
*/
+/**
+ * Initialize common algorithm variables once per group element operation.
+ *
+ * @param {string[]} fe List of FieldElement variable names to initialize
+ * @returns {string} Variable initializing statements
+ */
+export function fe_init_v128 (...fe) {
+ return `
+ let ${fe.map(fe => `
+ ${fe.replaceAll('.', '_')}_0123: v128,
+ ${fe.replaceAll('.', '_')}_4567: v128,
+ ${fe.replaceAll('.', '_')}_89xx: v128
+`)}
+`
+}
+
+/**
+ * Initialize common algorithm variables once per group element operation.
+ *
+ * @param {string[]} fe List of FieldElement variable names to initialize
+ * @returns {string} Variable initializing statements
+ */
+export function fe_init (...fe) {
+ return `
+ ${fe_init_v128(...fe)}
+
+ // mask to select even lane from first vector and odd lane from second
+ const m: v128 = i32x4(-1, 0, -1, 0)
+
+ // mask to double odd-on-odd indexes
+ const odds: v128 = i32x4(0, -1, 0, -1)
+
+ // factor of 19 for wrap from h9 to h0
+ const v19: v128 = i32x4.splat(19)
+
+ let c0: i64
+ let c1: i64
+ let c2: i64
+ let c3: i64
+ let c4: i64
+ let c5: i64
+ let c6: i64
+ let c7: i64
+ let c8: i64
+ let c9: i64
+
+ let f0: i64
+ let f1: i64
+ let f2: i64
+ let f3: i64
+ let f4: i64
+ let f5: i64
+ let f6: i64
+ let f7: i64
+ let f8: i64
+ let f9: i64
+
+ let f0_2: i64
+ let f1_2: i64
+ let f2_2: i64
+ let f3_2: i64
+ let f4_2: i64
+ let f5_2: i64
+ let f6_2: i64
+ let f7_2: i64
+ let f8_2: i64
+ let f9_2: i64
+
+ let f5_19: i64
+ let f6_19: i64
+ let f7_19: i64
+ let f8_19: i64
+ let f9_19: i64
+
+ let f5_38: i64
+ let f7_38: i64
+ let f9_38: i64
+
+ let fi: i32
+ let fv: v128
+
+ let h0: i64
+ let h1: i64
+ let h2: i64
+ let h3: i64
+ let h4: i64
+ let h5: i64
+ let h6: i64
+ let h7: i64
+ let h8: i64
+ let h9: i64
+
+ let h01: v128
+ let h12: v128
+ let h23: v128
+ let h34: v128
+ let h45: v128
+ let h56: v128
+ let h67: v128
+ let h78: v128
+ let h89: v128
+ let h90: v128
+
+ let t0: v128
+ let t1: v128
+ let t2: v128
+ let t3: v128
+ let t4: v128
+`
+}
+
/**
* Add two FieldElements and store the sum.
*
* @returns {string} Three vector addition commands
*/
export function fe_add (h, f, g) {
+ const h_name = h.replaceAll('.', '_')
+ const f_name = f.replaceAll('.', '_')
+ const g_name = g.replaceAll('.', '_')
return `
// ${h} = ${f} + ${g}
- ${h}0 = v128.add<i32>(${f}0, ${g}0)
- ${h}1 = v128.add<i32>(${f}1, ${g}1)
- ${h}2 = v128.add<i32>(${f}2, ${g}2)
+ ${h_name}_0123 = v128.add<i32>(${f_name}_0123, ${g_name}_0123)
+ ${h_name}_4567 = v128.add<i32>(${f_name}_4567, ${g_name}_4567)
+ ${h_name}_89xx = v128.add<i32>(${f_name}_89xx, ${g_name}_89xx)
`
}
/**
- * Subtract a FieldElement another and store the result.
+ * Add a FieldElement to itself and store the sum.
*
- * @param {string} h Variable name of FieldElement difference destination
- * @param {string} f Variable name of FieldElement minuend source
- * @param {string} g Variable name of FieldElement subtrahend source
- * @returns {string} Three vector subtraction commands
+ * @param {string} h Variable name of FieldElement sum destination
+ * @param {string} f Variable name of FieldElement summand source
+ * @returns {string} Three vector doubling commands
*/
-export function fe_sub (h, f, g) {
+export function fe_dbl (h, f) {
+ const h_name = h.replaceAll('.', '_')
+ const f_name = f.replaceAll('.', '_')
return `
- // ${h} = ${f} - ${g}
- ${h}0 = v128.sub<i32>(${f}0, ${g}0)
- ${h}1 = v128.sub<i32>(${f}1, ${g}1)
- ${h}2 = v128.sub<i32>(${f}2, ${g}2)
+ // ${h} = 2*${f}
+ ${h_name}_0123 = v128.shl<i32>(${f_name}_0123, 1)
+ ${h_name}_4567 = v128.shl<i32>(${f_name}_4567, 1)
+ ${h_name}_89xx = v128.shl<i32>(${f_name}_89xx, 1)
`
}
/**
- * Add a FieldElement to itself and store the sum.
+ * Loads a FieldElement from memory into named vectors.
*
- * @param {string} h Variable name of FieldElement sum destination
- * @param {string} f Variable name of FieldElement summand source
- * @returns {string} Three vector doubling commands
+ * @param {string} f Variable name of FieldElement source
+ * @returns {string} Three vectors
*/
-export function fe_dbl (h, f) {
+export function fe_load (f) {
+ const f_name = f.replaceAll('.', '_')
+ const f_ptr = `changetype<usize>(${f})`
return `
- // ${h} = 2*${f}
- ${h}0 = v128.shl<i32>(${f}0, 1)
- ${h}1 = v128.shl<i32>(${f}1, 1)
- ${h}2 = v128.shl<i32>(${f}2, 1)
+ // [0..3, 4..7, 8..11] = ${f}
+ ${f_name}_0123 = v128.load(${f_ptr}, 0)
+ ${f_name}_4567 = v128.load(${f_ptr}, 16)
+ ${f_name}_89xx = v128.load(${f_ptr}, 32)
+`
+}
+
+/**
+ * Multiply two FieldElements and store the product.
+ *
+ * @param {string} h Variable name of FieldElement product destination
+ * @param {string} f Variable name of FieldElement multiplicand source
+ * @param {string} g Variable name of FieldElement multiplicand source
+ * @returns {string} Field multiplication algorithm
+ */
+export function fe_mul (h, f, g) {
+ const h_name = h.replaceAll('.', '_')
+ const f_name = f.replaceAll('.', '_')
+ const g_name = g.replaceAll('.', '_')
+ return `
+ // ${h} = ${f} * ${g}
+ const ${g_name}_0123_19: v128 = i32x4.mul(${g_name}_0123, v19)
+ const ${g_name}_4567_19: v128 = i32x4.mul(${g_name}_4567, v19)
+ const ${g_name}_89xx_19: v128 = i32x4.mul(${g_name}_89xx, v19)
+
+ // f[0]
+ fi = v128.extract_lane<i32>(${f_name}_0123, 0)
+ fv = i32x4.splat(fi)
+
+ h01 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+ h23 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+ h45 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567)
+ h67 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567)
+ h89 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx)
+
+ // f[1]
+ fi = v128.extract_lane<i32>(${f_name}_0123, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ h12 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+ h34 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+ h56 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567)
+ h78 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567)
+ h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(${g_name}_89xx, ${g_name}_89xx_19, m))
+
+ // f[2]
+ fi = v128.extract_lane<i32>(${f_name}_0123, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567)
+ t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+ h23 = i64x2.add(h23, t0)
+ h45 = i64x2.add(h45, t1)
+ h67 = i64x2.add(h67, t2)
+ h89 = i64x2.add(h89, t3)
+ h01 = i64x2.add(h01, t4)
+
+ // f[3]
+ fi = v128.extract_lane<i32>(${f_name}_0123, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(${g_name}_4567, ${g_name}_4567_19, m))
+ t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+ h34 = i64x2.add(h34, t0)
+ h56 = i64x2.add(h56, t1)
+ h78 = i64x2.add(h78, t2)
+ h90 = i64x2.add(h90, t3)
+ h12 = i64x2.add(h12, t4)
+
+ // f[4]
+ fi = v128.extract_lane<i32>(${f_name}_4567, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+ h45 = i64x2.add(h45, t0)
+ h67 = i64x2.add(h67, t1)
+ h89 = i64x2.add(h89, t2)
+ h01 = i64x2.add(h01, t3)
+ h23 = i64x2.add(h23, t4)
+
+ // f[5]
+ fi = v128.extract_lane<i32>(${f_name}_4567, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(${g_name}_4567, ${g_name}_4567_19, m))
+ t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+ h56 = i64x2.add(h56, t0)
+ h78 = i64x2.add(h78, t1)
+ h90 = i64x2.add(h90, t2)
+ h12 = i64x2.add(h12, t3)
+ h34 = i64x2.add(h34, t4)
+
+ // f[6]
+ fi = v128.extract_lane<i32>(${f_name}_4567, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+ h67 = i64x2.add(h67, t0)
+ h89 = i64x2.add(h89, t1)
+ h01 = i64x2.add(h01, t2)
+ h23 = i64x2.add(h23, t3)
+ h45 = i64x2.add(h45, t4)
+
+ // f[7]
+ fi = v128.extract_lane<i32>(${f_name}_4567, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(${g_name}_0123, ${g_name}_0123_19, m))
+ t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+ h78 = i64x2.add(h78, t0)
+ h90 = i64x2.add(h90, t1)
+ h12 = i64x2.add(h12, t2)
+ h34 = i64x2.add(h34, t3)
+ h56 = i64x2.add(h56, t4)
+
+ // f[8]
+ fi = v128.extract_lane<i32>(${f_name}_89xx, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+ h89 = i64x2.add(h89, t0)
+ h01 = i64x2.add(h01, t1)
+ h23 = i64x2.add(h23, t2)
+ h45 = i64x2.add(h45, t3)
+ h67 = i64x2.add(h67, t4)
+
+ // f[9]
+ fi = v128.extract_lane<i32>(${f_name}_89xx, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(${g_name}_0123, ${g_name}_0123_19, m))
+ t1 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, ${g_name}_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, ${g_name}_89xx_19)
+
+ h90 = i64x2.add(h90, t0)
+ h12 = i64x2.add(h12, t1)
+ h34 = i64x2.add(h34, t2)
+ h56 = i64x2.add(h56, t3)
+ h78 = i64x2.add(h78, t4)
+
+ // extract scalars from vectors
+ h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+ h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+ h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+ h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+ h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+ h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+ h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+ h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+ h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+ h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+ // carry
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + (1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + (1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + (1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + (1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + (1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + (1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + (1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + (1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ ${h_name}_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ ${h_name}_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ ${h_name}_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+`
+}
+
+/**
+ * Negate the values of a FieldElement and store the result.
+ *
+ * @param {string} h Variable name of FieldElement result destination
+ * @param {string} f Variable name of FieldElement operand source
+ * @returns {string} Three vector negation commands
+ */
+export function fe_neg (h, f) {
+ const h_name = h.replaceAll('.', '_')
+ const f_name = f.replaceAll('.', '_')
+ return `
+ // ${h} = -${f}
+ ${h_name}_0123 = v128.neg<i32>(${f_name}_0123)
+ ${h_name}_4567 = v128.neg<i32>(${f_name}_4567)
+ ${h_name}_89xx = v128.neg<i32>(${f_name}_89xx)
+`
+}
+
+/**
+ * Square a FieldElement and store the result.
+ *
+ * Some iterations of the inner loop can be skipped since
+ * \`f[x]*f[y] + f[y]*f[x] = 2*f[x]*f[y]\`
+ *
+ * \`\`\`
+ * // non-constant-time example
+ * for (let i = 0; i < 10; i++) {
+ * for (let j = i; j < 10; j++) {
+ * h[i + j] += f[i] * g[j]
+ * if (i < j) {
+ * h[i + j] *= 2
+ * }
+ * }
+ * }
+ * \`\`\`
+ *
+ * h = f * f
+ * Can overlap h with f.
+ *
+ * Preconditions:
+ * |f| bounded by 1.65*2²⁶,1.65*2²⁵,1.65*2²⁶,1.65*2²⁵,etc.
+ *
+ * Postconditions:
+ * |h| bounded by 1.01*2²⁵,1.01*2²⁴4,1.01*2²⁵,1.01*2²⁴,etc.
+ *
+ * @param {string} h Variable name of FieldElement power destination
+ * @param {string} f Variable name of FieldElement base source
+ * @returns {string} Field squaring algorithm
+ */
+export function fe_sq (h, f) {
+ const h_name = h.replaceAll('.', '_')
+ const f_name = f.replaceAll('.', '_')
+ return `
+ // ${h} = ${f}²
+ f0 = i64(v128.extract_lane<i32>(${f_name}_0123, 0))
+ f1 = i64(v128.extract_lane<i32>(${f_name}_0123, 1))
+ f2 = i64(v128.extract_lane<i32>(${f_name}_0123, 2))
+ f3 = i64(v128.extract_lane<i32>(${f_name}_0123, 3))
+ f4 = i64(v128.extract_lane<i32>(${f_name}_4567, 0))
+ f5 = i64(v128.extract_lane<i32>(${f_name}_4567, 1))
+ f6 = i64(v128.extract_lane<i32>(${f_name}_4567, 2))
+ f7 = i64(v128.extract_lane<i32>(${f_name}_4567, 3))
+ f8 = i64(v128.extract_lane<i32>(${f_name}_89xx, 0))
+ f9 = i64(v128.extract_lane<i32>(${f_name}_89xx, 1))
+
+ f0_2 = f0 * 2
+ f1_2 = f1 * 2
+ f2_2 = f2 * 2
+ f3_2 = f3 * 2
+ f4_2 = f4 * 2
+ f5_2 = f5 * 2
+ f6_2 = f6 * 2
+ f7_2 = f7 * 2
+
+ f5_19 = f5 * 19 /* 1.959375*2²⁹ */
+ f6_19 = f6 * 19 /* 1.959375*2³⁰ */
+ f7_19 = f7 * 19 /* 1.959375*2²⁹ */
+ f8_19 = f8 * 19 /* 1.959375*2³⁰ */
+ f9_19 = f9 * 19 /* 1.959375*2²⁹ */
+
+ f7_38 = f7 * 38 /* 1.959375*2³⁰ */
+ f9_38 = f9 * 38 /* 1.959375*2³⁰ */
+
+ h0 = f0 * f0
+ h1 = f0_2 * f1
+ h2 = f0_2 * f2
+ h3 = f0_2 * f3
+ h4 = f0_2 * f4
+ h5 = f0_2 * f5
+ h6 = f0_2 * f6
+ h7 = f0_2 * f7
+ h8 = f0_2 * f8
+ h9 = f0_2 * f9
+
+ h2 += f1_2 * f1
+ h3 += f1_2 * f2
+ h4 += f1_2 * f3_2
+ h5 += f1_2 * f4
+ h6 += f1_2 * f5_2
+ h7 += f1_2 * f6
+ h8 += f1_2 * f7_2
+ h9 += f1_2 * f8
+ h0 += f1_2 * f9_38
+
+ h4 += f2 * f2
+ h5 += f2_2 * f3
+ h6 += f2_2 * f4
+ h7 += f2_2 * f5
+ h8 += f2_2 * f6
+ h9 += f2_2 * f7
+ h0 += f2_2 * f8_19
+ h1 += f2_2 * f9_19
+
+ h6 += f3_2 * f3
+ h7 += f3_2 * f4
+ h8 += f3_2 * f5_2
+ h9 += f3_2 * f6
+ h0 += f3_2 * f7_38
+ h1 += f3_2 * f8_19
+ h2 += f3_2 * f9_38
+
+ h8 += f4 * f4
+ h9 += f4_2 * f5
+ h0 += f4_2 * f6_19
+ h1 += f4_2 * f7_19
+ h2 += f4_2 * f8_19
+ h3 += f4_2 * f9_19
+
+ h0 += f5_2 * f5_19
+ h1 += f5_2 * f6_19
+ h2 += f5_2 * f7_38
+ h3 += f5_2 * f8_19
+ h4 += f5_2 * f9_38
+
+ h2 += f6 * f6_19
+ h3 += f6_2 * f7_19
+ h4 += f6_2 * f8_19
+ h5 += f6_2 * f9_19
+
+ h4 += f7_2 * f7_19
+ h5 += f7_2 * f8_19
+ h6 += f7_2 * f9_38
+
+ h6 += f8 * f8_19
+ h7 += f8 * f9_38
+
+ h8 += f9 * f9_38
+
+ c0 = (h0 + i64(1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + i64(1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + i64(1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + i64(1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + i64(1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + i64(1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + i64(1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + i64(1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + i64(1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + i64(1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + i64(1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + i64(1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ ${h_name}_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ ${h_name}_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ ${h_name}_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+`
+}
+
+/**
+ * Square a FieldElement, double it, and store the result.
+ *
+ * h = 2 * f * f
+ * Can overlap h with f.
+ *
+ * Preconditions:
+ * |f| bounded by 1.65*2^26,1.65*2^25,1.65*2^26,1.65*2^25,etc.
+ *
+ * Postconditions:
+ * |h| bounded by 1.01*2^25,1.01*2^24,1.01*2^25,1.01*2^24,etc.
+ *
+ * @param {string} h Variable name of FieldElement power destination
+ * @param {string} f Variable name of FieldElement base source
+ * @returns {string} Field squaring algorithm
+ */
+export function fe_sq2 (h, f) {
+ const h_name = h.replaceAll('.', '_')
+ const f_name = f.replaceAll('.', '_')
+ return `
+ // ${h} = 2${f}²
+ f0 = i64(v128.extract_lane<i32>(${f_name}_0123, 0))
+ f1 = i64(v128.extract_lane<i32>(${f_name}_0123, 1))
+ f2 = i64(v128.extract_lane<i32>(${f_name}_0123, 2))
+ f3 = i64(v128.extract_lane<i32>(${f_name}_0123, 3))
+ f4 = i64(v128.extract_lane<i32>(${f_name}_4567, 0))
+ f5 = i64(v128.extract_lane<i32>(${f_name}_4567, 1))
+ f6 = i64(v128.extract_lane<i32>(${f_name}_4567, 2))
+ f7 = i64(v128.extract_lane<i32>(${f_name}_4567, 3))
+ f8 = i64(v128.extract_lane<i32>(${f_name}_89xx, 0))
+ f9 = i64(v128.extract_lane<i32>(${f_name}_89xx, 1))
+
+ f0_2 = f0 * 2
+ f1_2 = f1 * 2
+ f2_2 = f2 * 2
+ f3_2 = f3 * 2
+ f4_2 = f4 * 2
+ f5_2 = f5 * 2
+ f6_2 = f6 * 2
+ f7_2 = f7 * 2
+
+ f5_38 = f5 * 38 /* 1.959375*2^30 */
+ f6_19 = f6 * 19 /* 1.959375*2^30 */
+ f7_38 = f7 * 38/* 1.959375*2^30 */
+ f8_19 = f8 * 19/* 1.959375*2^30 */
+ f9_38 = f9 * 38/* 1.959375*2^30 */
+
+ const f0f0: i64 = f0 * f0
+ const f0f1_2: i64 = f0_2 * f1
+ const f0f2_2: i64 = f0_2 * f2
+ const f0f3_2: i64 = f0_2 * f3
+ const f0f4_2: i64 = f0_2 * f4
+ const f0f5_2: i64 = f0_2 * f5
+ const f0f6_2: i64 = f0_2 * f6
+ const f0f7_2: i64 = f0_2 * f7
+ const f0f8_2: i64 = f0_2 * f8
+ const f0f9_2: i64 = f0_2 * f9
+
+ const f1f1_2: i64 = f1_2 * f1
+ const f1f2_2: i64 = f1_2 * f2
+ const f1f3_4: i64 = f1_2 * f3_2
+ const f1f4_2: i64 = f1_2 * f4
+ const f1f5_4: i64 = f1_2 * f5_2
+ const f1f6_2: i64 = f1_2 * f6
+ const f1f7_4: i64 = f1_2 * f7_2
+ const f1f8_2: i64 = f1_2 * f8
+ const f1f9_76: i64 = f1_2 * f9_38
+
+ const f2f2: i64 = f2 * f2
+ const f2f3_2: i64 = f2_2 * f3
+ const f2f4_2: i64 = f2_2 * f4
+ const f2f5_2: i64 = f2_2 * f5
+ const f2f6_2: i64 = f2_2 * f6
+ const f2f7_2: i64 = f2_2 * f7
+ const f2f8_38: i64 = f2_2 * f8_19
+ const f2f9_38: i64 = f2 * f9_38
+
+ const f3f3_2: i64 = f3_2 * f3
+ const f3f4_2: i64 = f3_2 * f4
+ const f3f5_4: i64 = f3_2 * f5_2
+ const f3f6_2: i64 = f3_2 * f6
+ const f3f7_76: i64 = f3_2 * f7_38
+ const f3f8_38: i64 = f3_2 * f8_19
+ const f3f9_76: i64 = f3_2 * f9_38
+
+ const f4f4: i64 = f4 * f4
+ const f4f5_2: i64 = f4_2 * f5
+ const f4f6_38: i64 = f4_2 * f6_19
+ const f4f7_38: i64 = f4 * f7_38
+ const f4f8_38: i64 = f4_2 * f8_19
+ const f4f9_38: i64 = f4 * f9_38
+
+ const f5f5_38: i64 = f5 * f5_38
+ const f5f6_38: i64 = f5_2 * f6_19
+ const f5f7_76: i64 = f5_2 * f7_38
+ const f5f8_38: i64 = f5_2 * f8_19
+ const f5f9_76: i64 = f5_2 * f9_38
+
+ const f6f6_19: i64 = f6 * f6_19
+ const f6f7_38: i64 = f6 * f7_38
+ const f6f8_38: i64 = f6_2 * f8_19
+ const f6f9_38: i64 = f6 * f9_38
+
+ const f7f7_38: i64 = f7 * f7_38
+ const f7f8_38: i64 = f7_2 * f8_19
+ const f7f9_76: i64 = f7_2 * f9_38
+
+ const f8f8_19: i64 = f8 * f8_19
+ const f8f9_38: i64 = f8 * f9_38
+
+ const f9f9_38: i64 = f9 * f9_38
+
+ h0 = f0f0 + f1f9_76 + f2f8_38 + f3f7_76 + f4f6_38 + f5f5_38
+ h1 = f0f1_2 + f2f9_38 + f3f8_38 + f4f7_38 + f5f6_38
+ h2 = f0f2_2 + f1f1_2 + f3f9_76 + f4f8_38 + f5f7_76 + f6f6_19
+ h3 = f0f3_2 + f1f2_2 + f4f9_38 + f5f8_38 + f6f7_38
+ h4 = f0f4_2 + f1f3_4 + f2f2 + f5f9_76 + f6f8_38 + f7f7_38
+ h5 = f0f5_2 + f1f4_2 + f2f3_2 + f6f9_38 + f7f8_38
+ h6 = f0f6_2 + f1f5_4 + f2f4_2 + f3f3_2 + f7f9_76 + f8f8_19
+ h7 = f0f7_2 + f1f6_2 + f2f5_2 + f3f4_2 + f8f9_38
+ h8 = f0f8_2 + f1f7_4 + f2f6_2 + f3f5_4 + f4f4 + f9f9_38
+ h9 = f0f9_2 + f1f8_2 + f2f7_2 + f3f6_2 + f4f5_2
+
+ h0 *= 2
+ h1 *= 2
+ h2 *= 2
+ h3 *= 2
+ h4 *= 2
+ h5 *= 2
+ h6 *= 2
+ h7 *= 2
+ h8 *= 2
+ h9 *= 2
+
+ c0 = (h0 + i64(1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + i64(1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + i64(1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + i64(1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + i64(1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + i64(1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + i64(1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + i64(1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + i64(1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + i64(1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + i64(1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + i64(1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ ${h_name}_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ ${h_name}_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ ${h_name}_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+`
+}
+
+/**
+ * Stores a FieldElement from named vectors into memory.
+ *
+ * @param {string} f Variable name of FieldElement
+ * @returns {string} StaticArray in memory
+ */
+export function fe_store (f) {
+ const f_name = f.replaceAll('.', '_')
+ const f_ptr = `changetype<usize>(${f})`
+ return `
+ // ${f} = [0..3, 4..7, 8..11]
+ v128.store(${f_ptr}, ${f_name}_0123, 0)
+ v128.store(${f_ptr}, ${f_name}_4567, 16)
+ v128.store_lane<u64>(${f_ptr}, ${f_name}_89xx, 0, 32)
+`
+}
+
+/**
+ * Subtract a FieldElement another and store the result.
+ *
+ * @param {string} h Variable name of FieldElement difference destination
+ * @param {string} f Variable name of FieldElement minuend source
+ * @param {string} g Variable name of FieldElement subtrahend source
+ * @returns {string} Three vector subtraction commands
+ */
+export function fe_sub (h, f, g) {
+ const h_name = h.replaceAll('.', '_')
+ const f_name = f.replaceAll('.', '_')
+ const g_name = g.replaceAll('.', '_')
+ return `
+ // ${h} = ${f} - ${g}
+ ${h_name}_0123 = v128.sub<i32>(${f_name}_0123, ${g_name}_0123)
+ ${h_name}_4567 = v128.sub<i32>(${f_name}_4567, ${g_name}_4567)
+ ${h_name}_89xx = v128.sub<i32>(${f_name}_89xx, ${g_name}_89xx)
`
}
//@ts-expect-error
@inline
export function fe_add (h: FieldElement, f: FieldElement, g: FieldElement): void {
- const h_ptr = changetype<usize>(h)
- const f_ptr = changetype<usize>(f)
- const g_ptr = changetype<usize>(g)
-
- let h0: v128, h1: v128, h2: v128
-
- let f0 = v128.load(f_ptr, 0)
- let f1 = v128.load(f_ptr, 16)
- let f2 = v128.load(f_ptr, 32)
-
- let g0 = v128.load(g_ptr, 0)
- let g1 = v128.load(g_ptr, 16)
- let g2 = v128.load(g_ptr, 32)
+ ${fe_init_v128('h', 'f', 'g')}
+ ${fe_load('h')}
+ ${fe_load('f')}
+ ${fe_load('g')}
${fe_add('h', 'f', 'g')}
- // store h
- v128.store(h_ptr, h0, 0)
- v128.store(h_ptr, h1, 16)
- v128.store_lane<u64>(h_ptr, h2, 0, 32)
+ ${fe_store('h')}
}
/**
//@ts-expect-error
@inline
export function fe_dbl (h: FieldElement, f: FieldElement): void {
- const h_ptr = changetype<usize>(h)
- const f_ptr = changetype<usize>(f)
-
- let h0: v128, h1: v128, h2: v128
-
- let f0 = v128.load(f_ptr, 0)
- let f1 = v128.load(f_ptr, 16)
- let f2 = v128.load(f_ptr, 32)
+ ${fe_init_v128('h', 'f')}
+ ${fe_load('h')}
+ ${fe_load('f')}
${fe_dbl('h', 'f')}
- // store h
- v128.store(h_ptr, h0, 0)
- v128.store(h_ptr, h1, 16)
- v128.store_lane<u64>(h_ptr, h2, 0, 32)
+ ${fe_store('h')}
}
/**
* @param g FieldElement multiplicand source
*/
export function fe_mul (h: FieldElement, f: FieldElement, g: FieldElement): void {
+ ${fe_init_v128('h', 'f', 'g')}
const f_ptr: usize = changetype<usize>(f)
- const g_ptr: usize = changetype<usize>(g)
- const g0123: v128 = v128.load(g_ptr, 0)
- const g4567: v128 = v128.load(g_ptr, 16)
- const g89xx: v128 = v128.load(g_ptr, 32)
+ ${fe_load('g')}
// mask to select even lane from first vector and odd lane from second
const m: v128 = i32x4(-1, 0, -1, 0)
// factor of 19 for wrap from h9 to h0
const v19: v128 = i32x4.splat(19)
- const g0123_19: v128 = i32x4.mul(g0123, v19)
- const g4567_19: v128 = i32x4.mul(g4567, v19)
- const g89xx_19: v128 = i32x4.mul(g89xx, v19)
+ const g_0123_19: v128 = i32x4.mul(g_0123, v19)
+ const g_4567_19: v128 = i32x4.mul(g_4567, v19)
+ const g_89xx_19: v128 = i32x4.mul(g_89xx, v19)
// f[0]
let fi: i32 = load<i32>(f_ptr, 0)
let fv: v128 = i32x4.splat(fi)
- let h01: v128 = i64x2.extmul_low_i32x4_s(fv, g0123)
- let h23: v128 = i64x2.extmul_high_i32x4_s(fv, g0123)
- let h45: v128 = i64x2.extmul_low_i32x4_s(fv, g4567)
- let h67: v128 = i64x2.extmul_high_i32x4_s(fv, g4567)
- let h89: v128 = i64x2.extmul_low_i32x4_s(fv, g89xx)
+ let h01: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+ let h23: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+ let h45: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+ let h67: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567)
+ let h89: v128 = i64x2.extmul_low_i32x4_s(fv, g_89xx)
// f[1]
fi = load<i32>(f_ptr, 4)
fv = i32x4.splat(fi)
fv = i32x4.add(fv, v128.and(fv, odds))
- let h12: v128 = i64x2.extmul_low_i32x4_s(fv, g0123)
- let h34: v128 = i64x2.extmul_high_i32x4_s(fv, g0123)
- let h56: v128 = i64x2.extmul_low_i32x4_s(fv, g4567)
- let h78: v128 = i64x2.extmul_high_i32x4_s(fv, g4567)
- let h90: v128 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g89xx, g89xx_19, m))
+ let h12: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+ let h34: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+ let h56: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+ let h78: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567)
+ let h90: v128 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_89xx, g_89xx_19, m))
// f[2]
fi = load<i32>(f_ptr, 8)
fv = i32x4.splat(fi)
- let t0: v128 = i64x2.extmul_low_i32x4_s(fv, g0123)
- let t1: v128 = i64x2.extmul_high_i32x4_s(fv, g0123)
- let t2: v128 = i64x2.extmul_low_i32x4_s(fv, g4567)
- let t3: v128 = i64x2.extmul_high_i32x4_s(fv, g4567)
- let t4: v128 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+ let t0: v128 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+ let t1: v128 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+ let t2: v128 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+ let t3: v128 = i64x2.extmul_high_i32x4_s(fv, g_4567)
+ let t4: v128 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
h23 = i64x2.add(h23, t0)
h45 = i64x2.add(h45, t1)
fv = i32x4.splat(fi)
fv = i32x4.add(fv, v128.and(fv, odds))
- t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
- t1 = i64x2.extmul_high_i32x4_s(fv, g0123)
- t2 = i64x2.extmul_low_i32x4_s(fv, g4567)
- t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g4567, g4567_19, m))
- t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+ t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g_4567, g_4567_19, m))
+ t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
h34 = i64x2.add(h34, t0)
h56 = i64x2.add(h56, t1)
fi = load<i32>(f_ptr, 16)
fv = i32x4.splat(fi)
- t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
- t1 = i64x2.extmul_high_i32x4_s(fv, g0123)
- t2 = i64x2.extmul_low_i32x4_s(fv, g4567)
- t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
- t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+ t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, g_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
h45 = i64x2.add(h45, t0)
h67 = i64x2.add(h67, t1)
fv = i32x4.splat(fi)
fv = i32x4.add(fv, v128.and(fv, odds))
- t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
- t1 = i64x2.extmul_high_i32x4_s(fv, g0123)
- t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g4567, g4567_19, m))
- t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
- t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+ t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_4567, g_4567_19, m))
+ t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
h56 = i64x2.add(h56, t0)
h78 = i64x2.add(h78, t1)
fi = load<i32>(f_ptr, 24)
fv = i32x4.splat(fi)
- t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
- t1 = i64x2.extmul_high_i32x4_s(fv, g0123)
- t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19)
- t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
- t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+ t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, g_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
h67 = i64x2.add(h67, t0)
h89 = i64x2.add(h89, t1)
fv = i32x4.splat(fi)
fv = i32x4.add(fv, v128.and(fv, odds))
- t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
- t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g0123, g0123_19, m))
- t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19)
- t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
- t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+ t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(g_0123, g_0123_19, m))
+ t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
h78 = i64x2.add(h78, t0)
h90 = i64x2.add(h90, t1)
fi = load<i32>(f_ptr, 32)
fv = i32x4.splat(fi)
- t0 = i64x2.extmul_low_i32x4_s(fv, g0123)
- t1 = i64x2.extmul_high_i32x4_s(fv, g0123_19)
- t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19)
- t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
- t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+ t0 = i64x2.extmul_low_i32x4_s(fv, g_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, g_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
h89 = i64x2.add(h89, t0)
h01 = i64x2.add(h01, t1)
fv = i32x4.splat(fi)
fv = i32x4.add(fv, v128.and(fv, odds))
- t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g0123, g0123_19, m))
- t1 = i64x2.extmul_high_i32x4_s(fv, g0123_19)
- t2 = i64x2.extmul_low_i32x4_s(fv, g4567_19)
- t3 = i64x2.extmul_high_i32x4_s(fv, g4567_19)
- t4 = i64x2.extmul_low_i32x4_s(fv, g89xx_19)
+ t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(g_0123, g_0123_19, m))
+ t1 = i64x2.extmul_high_i32x4_s(fv, g_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, g_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, g_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, g_89xx_19)
h90 = i64x2.add(h90, t0)
h12 = i64x2.add(h12, t1)
//@ts-expect-error
@inline
export function fe_neg (h: FieldElement, f: FieldElement): void {
- const h_ptr = changetype<usize>(h)
- const f_ptr = changetype<usize>(f)
- v128.store(h_ptr, v128.neg<i32>(v128.load(f_ptr)))
- v128.store(h_ptr, v128.neg<i32>(v128.load(f_ptr, 16)), 16)
- v128.store_lane<i64>(h_ptr, v128.neg<i32>(v128.load(f_ptr, 32)), 0, 32)
+ ${fe_init_v128('h', 'f')}
+ ${fe_load('h')}
+ ${fe_load('f')}
+
+ ${fe_neg('h', 'f')}
+
+ ${fe_store('h')}
}
const fe_pow22523_t0: FieldElement = fe()
//@ts-expect-error
@inline
export function fe_sub (h: FieldElement, f: FieldElement, g: FieldElement): void {
- const h_ptr = changetype<usize>(h)
- const f_ptr = changetype<usize>(f)
- const g_ptr = changetype<usize>(g)
-
- let h0: v128, h1: v128, h2: v128
-
- let f0 = v128.load(f_ptr, 0)
- let f1 = v128.load(f_ptr, 16)
- let f2 = v128.load(f_ptr, 32)
-
- let g0 = v128.load(g_ptr, 0)
- let g1 = v128.load(g_ptr, 16)
- let g2 = v128.load(g_ptr, 32)
+ ${fe_init_v128('h', 'f', 'g')}
+ ${fe_load('h')}
+ ${fe_load('f')}
+ ${fe_load('g')}
${fe_sub('h', 'f', 'g')}
- // store h
- v128.store(h_ptr, h0, 0)
- v128.store(h_ptr, h1, 16)
- v128.store_lane<u64>(h_ptr, h2, 0, 32)
+ ${fe_store('h')}
}
const fe_tobytes_t: FieldElement = fe()
fe_1(h.Z)
}
-const ge_p2_dbl_t0: FieldElement = fe()
+const ge_p2_dbl_t: FieldElement = fe()
//@ts-expect-error
@inline
export function ge_p2_dbl (r: ge_p1p1, p: ge_p2): void {
- const rX_ptr = changetype<usize>(r.X)
- const rY_ptr = changetype<usize>(r.Y)
- const rZ_ptr = changetype<usize>(r.Z)
- const rT_ptr = changetype<usize>(r.T)
-
- const pX_ptr = changetype<usize>(p.X)
- const pY_ptr = changetype<usize>(p.Y)
- const pZ_ptr = changetype<usize>(p.Z)
-
- const rYsq = ge_p2_dbl_t0
- const rYsq_ptr = changetype<usize>(rYsq)
-
- fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rY = pY²
- fe_sq2(r.T, p.Z) // rT = pZ²
-
- // Start converting to performant inlined ops with locals to avoid load/store
- let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
- let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
- let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32)
- let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32)
-
- const pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32)
- const pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32)
- const pZ0 = v128.load(pZ_ptr), pZ1 = v128.load(pZ_ptr, 16), pZ2 = v128.load(pZ_ptr, 32)
-
- // rY = pX + pY
- v128.store(rY_ptr, v128.add<i32>(pX0, pY0))
- v128.store(rY_ptr, v128.add<i32>(pX1, pY1), 16)
- v128.store_lane<u64>(rY_ptr, v128.add<i32>(pX2, pY2), 0, 32)
-
- // t0 = r.Y²
- fe_sq(rYsq, r.Y)
- const rYsq0 = v128.load(rYsq_ptr, 0)
- const rYsq1 = v128.load(rYsq_ptr, 16)
- const rYsq2 = v128.load(rYsq_ptr, 32)
-
- // rY = rZ + rX
- rY0 = v128.add<i32>(rZ0, rX0)
- rY1 = v128.add<i32>(rZ1, rX1)
- rY2 = v128.add<i32>(rZ2, rX2)
-
- // rZ = rZ - rX
- rZ0 = v128.sub<i32>(rZ0, rX0)
- rZ1 = v128.sub<i32>(rZ1, rX1)
- rZ2 = v128.sub<i32>(rZ2, rX2)
-
- // rX = rYsq - rY
- rX0 = v128.sub<i32>(rYsq0, rY0)
- rX1 = v128.sub<i32>(rYsq1, rY1)
- rX2 = v128.sub<i32>(rYsq2, rY2)
-
- // rT = rT - rZ
- rT0 = v128.sub<i32>(rT0, rZ0)
- rT1 = v128.sub<i32>(rT1, rZ1)
- rT2 = v128.sub<i32>(rT2, rZ2)
-
- // store r.X
- v128.store(rX_ptr, rX0)
- v128.store(rX_ptr, rX1, 16)
- v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
- // store r.Y
- v128.store(rY_ptr, rY0)
- v128.store(rY_ptr, rY1, 16)
- v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
- // store r.Z
- v128.store(rZ_ptr, rZ0)
- v128.store(rZ_ptr, rZ1, 16)
- v128.store_lane<u64>(rZ_ptr, rZ2, 0, 32)
-
- // store r.T
- v128.store(rT_ptr, rT0)
- v128.store(rT_ptr, rT1, 16)
- v128.store_lane<u64>(rT_ptr, rT2, 0, 32)
+
+ let
+ p_X_0123: v128,
+ p_X_4567: v128,
+ p_X_89xx: v128
+,
+ p_Y_0123: v128,
+ p_Y_4567: v128,
+ p_Y_89xx: v128
+,
+ p_Z_0123: v128,
+ p_Z_4567: v128,
+ p_Z_89xx: v128
+,
+ r_X_0123: v128,
+ r_X_4567: v128,
+ r_X_89xx: v128
+,
+ r_Y_0123: v128,
+ r_Y_4567: v128,
+ r_Y_89xx: v128
+,
+ r_Z_0123: v128,
+ r_Z_4567: v128,
+ r_Z_89xx: v128
+,
+ r_T_0123: v128,
+ r_T_4567: v128,
+ r_T_89xx: v128
+,
+ rY2_0123: v128,
+ rY2_4567: v128,
+ rY2_89xx: v128
+
+ // mask to select even lane from first vector and odd lane from second
+ const m: v128 = i32x4(-1, 0, -1, 0)
+
+ // mask to double odd-on-odd indexes
+ const odds: v128 = i32x4(0, -1, 0, -1)
+
+ // factor of 19 for wrap from h9 to h0
+ const v19: v128 = i32x4.splat(19)
+
+ let c0: i64
+ let c1: i64
+ let c2: i64
+ let c3: i64
+ let c4: i64
+ let c5: i64
+ let c6: i64
+ let c7: i64
+ let c8: i64
+ let c9: i64
+
+ let f0: i64
+ let f1: i64
+ let f2: i64
+ let f3: i64
+ let f4: i64
+ let f5: i64
+ let f6: i64
+ let f7: i64
+ let f8: i64
+ let f9: i64
+
+ let f0_2: i64
+ let f1_2: i64
+ let f2_2: i64
+ let f3_2: i64
+ let f4_2: i64
+ let f5_2: i64
+ let f6_2: i64
+ let f7_2: i64
+ let f8_2: i64
+ let f9_2: i64
+
+ let f5_19: i64
+ let f6_19: i64
+ let f7_19: i64
+ let f8_19: i64
+ let f9_19: i64
+
+ let f5_38: i64
+ let f7_38: i64
+ let f9_38: i64
+
+ let fi: i32
+ let fv: v128
+
+ let h0: i64
+ let h1: i64
+ let h2: i64
+ let h3: i64
+ let h4: i64
+ let h5: i64
+ let h6: i64
+ let h7: i64
+ let h8: i64
+ let h9: i64
+
+ let h01: v128
+ let h12: v128
+ let h23: v128
+ let h34: v128
+ let h45: v128
+ let h56: v128
+ let h67: v128
+ let h78: v128
+ let h89: v128
+ let h90: v128
+
+ let t0: v128
+ let t1: v128
+ let t2: v128
+ let t3: v128
+ let t4: v128
+
+ // [0..3, 4..7, 8..11] = p.X
+ p_X_0123 = v128.load(changetype<usize>(p.X), 0)
+ p_X_4567 = v128.load(changetype<usize>(p.X), 16)
+ p_X_89xx = v128.load(changetype<usize>(p.X), 32)
+
+ // [0..3, 4..7, 8..11] = p.Y
+ p_Y_0123 = v128.load(changetype<usize>(p.Y), 0)
+ p_Y_4567 = v128.load(changetype<usize>(p.Y), 16)
+ p_Y_89xx = v128.load(changetype<usize>(p.Y), 32)
+
+ // [0..3, 4..7, 8..11] = p.Z
+ p_Z_0123 = v128.load(changetype<usize>(p.Z), 0)
+ p_Z_4567 = v128.load(changetype<usize>(p.Z), 16)
+ p_Z_89xx = v128.load(changetype<usize>(p.Z), 32)
+
+ // [0..3, 4..7, 8..11] = r.X
+ r_X_0123 = v128.load(changetype<usize>(r.X), 0)
+ r_X_4567 = v128.load(changetype<usize>(r.X), 16)
+ r_X_89xx = v128.load(changetype<usize>(r.X), 32)
+
+ // [0..3, 4..7, 8..11] = r.Y
+ r_Y_0123 = v128.load(changetype<usize>(r.Y), 0)
+ r_Y_4567 = v128.load(changetype<usize>(r.Y), 16)
+ r_Y_89xx = v128.load(changetype<usize>(r.Y), 32)
+
+ // [0..3, 4..7, 8..11] = r.Z
+ r_Z_0123 = v128.load(changetype<usize>(r.Z), 0)
+ r_Z_4567 = v128.load(changetype<usize>(r.Z), 16)
+ r_Z_89xx = v128.load(changetype<usize>(r.Z), 32)
+
+ // [0..3, 4..7, 8..11] = r.T
+ r_T_0123 = v128.load(changetype<usize>(r.T), 0)
+ r_T_4567 = v128.load(changetype<usize>(r.T), 16)
+ r_T_89xx = v128.load(changetype<usize>(r.T), 32)
+
+ const rY2 = ge_p2_dbl_t
+
+ // [0..3, 4..7, 8..11] = rY2
+ rY2_0123 = v128.load(changetype<usize>(rY2), 0)
+ rY2_4567 = v128.load(changetype<usize>(rY2), 16)
+ rY2_89xx = v128.load(changetype<usize>(rY2), 32)
+
+ // could swap below with fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rZ = pY²
+
+ // r.X = p.X²
+ f0 = i64(v128.extract_lane<i32>(p_X_0123, 0))
+ f1 = i64(v128.extract_lane<i32>(p_X_0123, 1))
+ f2 = i64(v128.extract_lane<i32>(p_X_0123, 2))
+ f3 = i64(v128.extract_lane<i32>(p_X_0123, 3))
+ f4 = i64(v128.extract_lane<i32>(p_X_4567, 0))
+ f5 = i64(v128.extract_lane<i32>(p_X_4567, 1))
+ f6 = i64(v128.extract_lane<i32>(p_X_4567, 2))
+ f7 = i64(v128.extract_lane<i32>(p_X_4567, 3))
+ f8 = i64(v128.extract_lane<i32>(p_X_89xx, 0))
+ f9 = i64(v128.extract_lane<i32>(p_X_89xx, 1))
+
+ f0_2 = f0 * 2
+ f1_2 = f1 * 2
+ f2_2 = f2 * 2
+ f3_2 = f3 * 2
+ f4_2 = f4 * 2
+ f5_2 = f5 * 2
+ f6_2 = f6 * 2
+ f7_2 = f7 * 2
+
+ f5_19 = f5 * 19 /* 1.959375*2²⁹ */
+ f6_19 = f6 * 19 /* 1.959375*2³⁰ */
+ f7_19 = f7 * 19 /* 1.959375*2²⁹ */
+ f8_19 = f8 * 19 /* 1.959375*2³⁰ */
+ f9_19 = f9 * 19 /* 1.959375*2²⁹ */
+
+ f7_38 = f7 * 38 /* 1.959375*2³⁰ */
+ f9_38 = f9 * 38 /* 1.959375*2³⁰ */
+
+ h0 = f0 * f0
+ h1 = f0_2 * f1
+ h2 = f0_2 * f2
+ h3 = f0_2 * f3
+ h4 = f0_2 * f4
+ h5 = f0_2 * f5
+ h6 = f0_2 * f6
+ h7 = f0_2 * f7
+ h8 = f0_2 * f8
+ h9 = f0_2 * f9
+
+ h2 += f1_2 * f1
+ h3 += f1_2 * f2
+ h4 += f1_2 * f3_2
+ h5 += f1_2 * f4
+ h6 += f1_2 * f5_2
+ h7 += f1_2 * f6
+ h8 += f1_2 * f7_2
+ h9 += f1_2 * f8
+ h0 += f1_2 * f9_38
+
+ h4 += f2 * f2
+ h5 += f2_2 * f3
+ h6 += f2_2 * f4
+ h7 += f2_2 * f5
+ h8 += f2_2 * f6
+ h9 += f2_2 * f7
+ h0 += f2_2 * f8_19
+ h1 += f2_2 * f9_19
+
+ h6 += f3_2 * f3
+ h7 += f3_2 * f4
+ h8 += f3_2 * f5_2
+ h9 += f3_2 * f6
+ h0 += f3_2 * f7_38
+ h1 += f3_2 * f8_19
+ h2 += f3_2 * f9_38
+
+ h8 += f4 * f4
+ h9 += f4_2 * f5
+ h0 += f4_2 * f6_19
+ h1 += f4_2 * f7_19
+ h2 += f4_2 * f8_19
+ h3 += f4_2 * f9_19
+
+ h0 += f5_2 * f5_19
+ h1 += f5_2 * f6_19
+ h2 += f5_2 * f7_38
+ h3 += f5_2 * f8_19
+ h4 += f5_2 * f9_38
+
+ h2 += f6 * f6_19
+ h3 += f6_2 * f7_19
+ h4 += f6_2 * f8_19
+ h5 += f6_2 * f9_19
+
+ h4 += f7_2 * f7_19
+ h5 += f7_2 * f8_19
+ h6 += f7_2 * f9_38
+
+ h6 += f8 * f8_19
+ h7 += f8 * f9_38
+
+ h8 += f9 * f9_38
+
+ c0 = (h0 + i64(1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + i64(1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + i64(1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + i64(1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + i64(1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + i64(1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + i64(1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + i64(1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + i64(1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + i64(1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + i64(1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + i64(1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ r_X_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ r_X_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ r_X_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+ // r.Z = p.Y²
+ f0 = i64(v128.extract_lane<i32>(p_Y_0123, 0))
+ f1 = i64(v128.extract_lane<i32>(p_Y_0123, 1))
+ f2 = i64(v128.extract_lane<i32>(p_Y_0123, 2))
+ f3 = i64(v128.extract_lane<i32>(p_Y_0123, 3))
+ f4 = i64(v128.extract_lane<i32>(p_Y_4567, 0))
+ f5 = i64(v128.extract_lane<i32>(p_Y_4567, 1))
+ f6 = i64(v128.extract_lane<i32>(p_Y_4567, 2))
+ f7 = i64(v128.extract_lane<i32>(p_Y_4567, 3))
+ f8 = i64(v128.extract_lane<i32>(p_Y_89xx, 0))
+ f9 = i64(v128.extract_lane<i32>(p_Y_89xx, 1))
+
+ f0_2 = f0 * 2
+ f1_2 = f1 * 2
+ f2_2 = f2 * 2
+ f3_2 = f3 * 2
+ f4_2 = f4 * 2
+ f5_2 = f5 * 2
+ f6_2 = f6 * 2
+ f7_2 = f7 * 2
+
+ f5_19 = f5 * 19 /* 1.959375*2²⁹ */
+ f6_19 = f6 * 19 /* 1.959375*2³⁰ */
+ f7_19 = f7 * 19 /* 1.959375*2²⁹ */
+ f8_19 = f8 * 19 /* 1.959375*2³⁰ */
+ f9_19 = f9 * 19 /* 1.959375*2²⁹ */
+
+ f7_38 = f7 * 38 /* 1.959375*2³⁰ */
+ f9_38 = f9 * 38 /* 1.959375*2³⁰ */
+
+ h0 = f0 * f0
+ h1 = f0_2 * f1
+ h2 = f0_2 * f2
+ h3 = f0_2 * f3
+ h4 = f0_2 * f4
+ h5 = f0_2 * f5
+ h6 = f0_2 * f6
+ h7 = f0_2 * f7
+ h8 = f0_2 * f8
+ h9 = f0_2 * f9
+
+ h2 += f1_2 * f1
+ h3 += f1_2 * f2
+ h4 += f1_2 * f3_2
+ h5 += f1_2 * f4
+ h6 += f1_2 * f5_2
+ h7 += f1_2 * f6
+ h8 += f1_2 * f7_2
+ h9 += f1_2 * f8
+ h0 += f1_2 * f9_38
+
+ h4 += f2 * f2
+ h5 += f2_2 * f3
+ h6 += f2_2 * f4
+ h7 += f2_2 * f5
+ h8 += f2_2 * f6
+ h9 += f2_2 * f7
+ h0 += f2_2 * f8_19
+ h1 += f2_2 * f9_19
+
+ h6 += f3_2 * f3
+ h7 += f3_2 * f4
+ h8 += f3_2 * f5_2
+ h9 += f3_2 * f6
+ h0 += f3_2 * f7_38
+ h1 += f3_2 * f8_19
+ h2 += f3_2 * f9_38
+
+ h8 += f4 * f4
+ h9 += f4_2 * f5
+ h0 += f4_2 * f6_19
+ h1 += f4_2 * f7_19
+ h2 += f4_2 * f8_19
+ h3 += f4_2 * f9_19
+
+ h0 += f5_2 * f5_19
+ h1 += f5_2 * f6_19
+ h2 += f5_2 * f7_38
+ h3 += f5_2 * f8_19
+ h4 += f5_2 * f9_38
+
+ h2 += f6 * f6_19
+ h3 += f6_2 * f7_19
+ h4 += f6_2 * f8_19
+ h5 += f6_2 * f9_19
+
+ h4 += f7_2 * f7_19
+ h5 += f7_2 * f8_19
+ h6 += f7_2 * f9_38
+
+ h6 += f8 * f8_19
+ h7 += f8 * f9_38
+
+ h8 += f9 * f9_38
+
+ c0 = (h0 + i64(1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + i64(1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + i64(1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + i64(1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + i64(1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + i64(1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + i64(1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + i64(1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + i64(1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + i64(1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + i64(1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + i64(1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ r_Z_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ r_Z_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ r_Z_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+ // could swap above with fe_sq_vec(r.X, r.Z, p.X, p.Y) // rX = pX²; rZ = pY²
+
+ // r.T = 2p.Z²
+ f0 = i64(v128.extract_lane<i32>(p_Z_0123, 0))
+ f1 = i64(v128.extract_lane<i32>(p_Z_0123, 1))
+ f2 = i64(v128.extract_lane<i32>(p_Z_0123, 2))
+ f3 = i64(v128.extract_lane<i32>(p_Z_0123, 3))
+ f4 = i64(v128.extract_lane<i32>(p_Z_4567, 0))
+ f5 = i64(v128.extract_lane<i32>(p_Z_4567, 1))
+ f6 = i64(v128.extract_lane<i32>(p_Z_4567, 2))
+ f7 = i64(v128.extract_lane<i32>(p_Z_4567, 3))
+ f8 = i64(v128.extract_lane<i32>(p_Z_89xx, 0))
+ f9 = i64(v128.extract_lane<i32>(p_Z_89xx, 1))
+
+ f0_2 = f0 * 2
+ f1_2 = f1 * 2
+ f2_2 = f2 * 2
+ f3_2 = f3 * 2
+ f4_2 = f4 * 2
+ f5_2 = f5 * 2
+ f6_2 = f6 * 2
+ f7_2 = f7 * 2
+
+ f5_38 = f5 * 38 /* 1.959375*2^30 */
+ f6_19 = f6 * 19 /* 1.959375*2^30 */
+ f7_38 = f7 * 38/* 1.959375*2^30 */
+ f8_19 = f8 * 19/* 1.959375*2^30 */
+ f9_38 = f9 * 38/* 1.959375*2^30 */
+
+ const f0f0: i64 = f0 * f0
+ const f0f1_2: i64 = f0_2 * f1
+ const f0f2_2: i64 = f0_2 * f2
+ const f0f3_2: i64 = f0_2 * f3
+ const f0f4_2: i64 = f0_2 * f4
+ const f0f5_2: i64 = f0_2 * f5
+ const f0f6_2: i64 = f0_2 * f6
+ const f0f7_2: i64 = f0_2 * f7
+ const f0f8_2: i64 = f0_2 * f8
+ const f0f9_2: i64 = f0_2 * f9
+
+ const f1f1_2: i64 = f1_2 * f1
+ const f1f2_2: i64 = f1_2 * f2
+ const f1f3_4: i64 = f1_2 * f3_2
+ const f1f4_2: i64 = f1_2 * f4
+ const f1f5_4: i64 = f1_2 * f5_2
+ const f1f6_2: i64 = f1_2 * f6
+ const f1f7_4: i64 = f1_2 * f7_2
+ const f1f8_2: i64 = f1_2 * f8
+ const f1f9_76: i64 = f1_2 * f9_38
+
+ const f2f2: i64 = f2 * f2
+ const f2f3_2: i64 = f2_2 * f3
+ const f2f4_2: i64 = f2_2 * f4
+ const f2f5_2: i64 = f2_2 * f5
+ const f2f6_2: i64 = f2_2 * f6
+ const f2f7_2: i64 = f2_2 * f7
+ const f2f8_38: i64 = f2_2 * f8_19
+ const f2f9_38: i64 = f2 * f9_38
+
+ const f3f3_2: i64 = f3_2 * f3
+ const f3f4_2: i64 = f3_2 * f4
+ const f3f5_4: i64 = f3_2 * f5_2
+ const f3f6_2: i64 = f3_2 * f6
+ const f3f7_76: i64 = f3_2 * f7_38
+ const f3f8_38: i64 = f3_2 * f8_19
+ const f3f9_76: i64 = f3_2 * f9_38
+
+ const f4f4: i64 = f4 * f4
+ const f4f5_2: i64 = f4_2 * f5
+ const f4f6_38: i64 = f4_2 * f6_19
+ const f4f7_38: i64 = f4 * f7_38
+ const f4f8_38: i64 = f4_2 * f8_19
+ const f4f9_38: i64 = f4 * f9_38
+
+ const f5f5_38: i64 = f5 * f5_38
+ const f5f6_38: i64 = f5_2 * f6_19
+ const f5f7_76: i64 = f5_2 * f7_38
+ const f5f8_38: i64 = f5_2 * f8_19
+ const f5f9_76: i64 = f5_2 * f9_38
+
+ const f6f6_19: i64 = f6 * f6_19
+ const f6f7_38: i64 = f6 * f7_38
+ const f6f8_38: i64 = f6_2 * f8_19
+ const f6f9_38: i64 = f6 * f9_38
+
+ const f7f7_38: i64 = f7 * f7_38
+ const f7f8_38: i64 = f7_2 * f8_19
+ const f7f9_76: i64 = f7_2 * f9_38
+
+ const f8f8_19: i64 = f8 * f8_19
+ const f8f9_38: i64 = f8 * f9_38
+
+ const f9f9_38: i64 = f9 * f9_38
+
+ h0 = f0f0 + f1f9_76 + f2f8_38 + f3f7_76 + f4f6_38 + f5f5_38
+ h1 = f0f1_2 + f2f9_38 + f3f8_38 + f4f7_38 + f5f6_38
+ h2 = f0f2_2 + f1f1_2 + f3f9_76 + f4f8_38 + f5f7_76 + f6f6_19
+ h3 = f0f3_2 + f1f2_2 + f4f9_38 + f5f8_38 + f6f7_38
+ h4 = f0f4_2 + f1f3_4 + f2f2 + f5f9_76 + f6f8_38 + f7f7_38
+ h5 = f0f5_2 + f1f4_2 + f2f3_2 + f6f9_38 + f7f8_38
+ h6 = f0f6_2 + f1f5_4 + f2f4_2 + f3f3_2 + f7f9_76 + f8f8_19
+ h7 = f0f7_2 + f1f6_2 + f2f5_2 + f3f4_2 + f8f9_38
+ h8 = f0f8_2 + f1f7_4 + f2f6_2 + f3f5_4 + f4f4 + f9f9_38
+ h9 = f0f9_2 + f1f8_2 + f2f7_2 + f3f6_2 + f4f5_2
+
+ h0 *= 2
+ h1 *= 2
+ h2 *= 2
+ h3 *= 2
+ h4 *= 2
+ h5 *= 2
+ h6 *= 2
+ h7 *= 2
+ h8 *= 2
+ h9 *= 2
+
+ c0 = (h0 + i64(1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + i64(1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + i64(1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + i64(1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + i64(1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + i64(1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + i64(1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + i64(1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + i64(1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + i64(1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + i64(1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + i64(1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ r_T_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ r_T_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ r_T_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+ // r.Y = p.X + p.Y
+ r_Y_0123 = v128.add<i32>(p_X_0123, p_Y_0123)
+ r_Y_4567 = v128.add<i32>(p_X_4567, p_Y_4567)
+ r_Y_89xx = v128.add<i32>(p_X_89xx, p_Y_89xx)
+
+ // rY2 = r.Y²
+ f0 = i64(v128.extract_lane<i32>(r_Y_0123, 0))
+ f1 = i64(v128.extract_lane<i32>(r_Y_0123, 1))
+ f2 = i64(v128.extract_lane<i32>(r_Y_0123, 2))
+ f3 = i64(v128.extract_lane<i32>(r_Y_0123, 3))
+ f4 = i64(v128.extract_lane<i32>(r_Y_4567, 0))
+ f5 = i64(v128.extract_lane<i32>(r_Y_4567, 1))
+ f6 = i64(v128.extract_lane<i32>(r_Y_4567, 2))
+ f7 = i64(v128.extract_lane<i32>(r_Y_4567, 3))
+ f8 = i64(v128.extract_lane<i32>(r_Y_89xx, 0))
+ f9 = i64(v128.extract_lane<i32>(r_Y_89xx, 1))
+
+ f0_2 = f0 * 2
+ f1_2 = f1 * 2
+ f2_2 = f2 * 2
+ f3_2 = f3 * 2
+ f4_2 = f4 * 2
+ f5_2 = f5 * 2
+ f6_2 = f6 * 2
+ f7_2 = f7 * 2
+
+ f5_19 = f5 * 19 /* 1.959375*2²⁹ */
+ f6_19 = f6 * 19 /* 1.959375*2³⁰ */
+ f7_19 = f7 * 19 /* 1.959375*2²⁹ */
+ f8_19 = f8 * 19 /* 1.959375*2³⁰ */
+ f9_19 = f9 * 19 /* 1.959375*2²⁹ */
+
+ f7_38 = f7 * 38 /* 1.959375*2³⁰ */
+ f9_38 = f9 * 38 /* 1.959375*2³⁰ */
+
+ h0 = f0 * f0
+ h1 = f0_2 * f1
+ h2 = f0_2 * f2
+ h3 = f0_2 * f3
+ h4 = f0_2 * f4
+ h5 = f0_2 * f5
+ h6 = f0_2 * f6
+ h7 = f0_2 * f7
+ h8 = f0_2 * f8
+ h9 = f0_2 * f9
+
+ h2 += f1_2 * f1
+ h3 += f1_2 * f2
+ h4 += f1_2 * f3_2
+ h5 += f1_2 * f4
+ h6 += f1_2 * f5_2
+ h7 += f1_2 * f6
+ h8 += f1_2 * f7_2
+ h9 += f1_2 * f8
+ h0 += f1_2 * f9_38
+
+ h4 += f2 * f2
+ h5 += f2_2 * f3
+ h6 += f2_2 * f4
+ h7 += f2_2 * f5
+ h8 += f2_2 * f6
+ h9 += f2_2 * f7
+ h0 += f2_2 * f8_19
+ h1 += f2_2 * f9_19
+
+ h6 += f3_2 * f3
+ h7 += f3_2 * f4
+ h8 += f3_2 * f5_2
+ h9 += f3_2 * f6
+ h0 += f3_2 * f7_38
+ h1 += f3_2 * f8_19
+ h2 += f3_2 * f9_38
+
+ h8 += f4 * f4
+ h9 += f4_2 * f5
+ h0 += f4_2 * f6_19
+ h1 += f4_2 * f7_19
+ h2 += f4_2 * f8_19
+ h3 += f4_2 * f9_19
+
+ h0 += f5_2 * f5_19
+ h1 += f5_2 * f6_19
+ h2 += f5_2 * f7_38
+ h3 += f5_2 * f8_19
+ h4 += f5_2 * f9_38
+
+ h2 += f6 * f6_19
+ h3 += f6_2 * f7_19
+ h4 += f6_2 * f8_19
+ h5 += f6_2 * f9_19
+
+ h4 += f7_2 * f7_19
+ h5 += f7_2 * f8_19
+ h6 += f7_2 * f9_38
+
+ h6 += f8 * f8_19
+ h7 += f8 * f9_38
+
+ h8 += f9 * f9_38
+
+ c0 = (h0 + i64(1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + i64(1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + i64(1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + i64(1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + i64(1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + i64(1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + i64(1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + i64(1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + i64(1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + i64(1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + i64(1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + i64(1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ rY2_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ rY2_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ rY2_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+ // r.Y = r.Z + r.X
+ r_Y_0123 = v128.add<i32>(r_Z_0123, r_X_0123)
+ r_Y_4567 = v128.add<i32>(r_Z_4567, r_X_4567)
+ r_Y_89xx = v128.add<i32>(r_Z_89xx, r_X_89xx)
+
+ // r.Z = r.Z - r.X
+ r_Z_0123 = v128.sub<i32>(r_Z_0123, r_X_0123)
+ r_Z_4567 = v128.sub<i32>(r_Z_4567, r_X_4567)
+ r_Z_89xx = v128.sub<i32>(r_Z_89xx, r_X_89xx)
+
+ // r.X = rY2 - r.Y
+ r_X_0123 = v128.sub<i32>(rY2_0123, r_Y_0123)
+ r_X_4567 = v128.sub<i32>(rY2_4567, r_Y_4567)
+ r_X_89xx = v128.sub<i32>(rY2_89xx, r_Y_89xx)
+
+ // r.T = r.T - r.Z
+ r_T_0123 = v128.sub<i32>(r_T_0123, r_Z_0123)
+ r_T_4567 = v128.sub<i32>(r_T_4567, r_Z_4567)
+ r_T_89xx = v128.sub<i32>(r_T_89xx, r_Z_89xx)
+
+ // r.X = [0..3, 4..7, 8..11]
+ v128.store(changetype<usize>(r.X), r_X_0123, 0)
+ v128.store(changetype<usize>(r.X), r_X_4567, 16)
+ v128.store_lane<u64>(changetype<usize>(r.X), r_X_89xx, 0, 32)
+
+ // r.Y = [0..3, 4..7, 8..11]
+ v128.store(changetype<usize>(r.Y), r_Y_0123, 0)
+ v128.store(changetype<usize>(r.Y), r_Y_4567, 16)
+ v128.store_lane<u64>(changetype<usize>(r.Y), r_Y_89xx, 0, 32)
+
+ // r.Z = [0..3, 4..7, 8..11]
+ v128.store(changetype<usize>(r.Z), r_Z_0123, 0)
+ v128.store(changetype<usize>(r.Z), r_Z_4567, 16)
+ v128.store_lane<u64>(changetype<usize>(r.Z), r_Z_89xx, 0, 32)
+
+ // r.T = [0..3, 4..7, 8..11]
+ v128.store(changetype<usize>(r.T), r_T_0123, 0)
+ v128.store(changetype<usize>(r.T), r_T_4567, 16)
+ v128.store_lane<u64>(changetype<usize>(r.T), r_T_89xx, 0, 32)
+
}
/**
* r = p + q
*/
export function ge_add_cached (r: ge_p1p1, p: ge_p3, q: ge_cached): void {
- const rX_ptr = changetype<usize>(r.X)
- const rY_ptr = changetype<usize>(r.Y)
- const rZ_ptr = changetype<usize>(r.Z)
- const rT_ptr = changetype<usize>(r.T)
-
- const pX_ptr = changetype<usize>(p.X)
- const pY_ptr = changetype<usize>(p.Y)
-
- let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
- let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
- let pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32)
- let pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32)
-
- // rX = pY + pX
- rX0 = v128.add<i32>(pY0, pX0)
- rX1 = v128.add<i32>(pY1, pX1)
- rX2 = v128.add<i32>(pY2, pX2)
-
- // rY = pY - pX
- rY0 = v128.sub<i32>(pY0, pX0)
- rY1 = v128.sub<i32>(pY1, pX1)
- rY2 = v128.sub<i32>(pY2, pX2)
-
- // store r.X
- v128.store(rX_ptr, rX0)
- v128.store(rX_ptr, rX1, 16)
- v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
- // store r.Y
- v128.store(rY_ptr, rY0)
- v128.store(rY_ptr, rY1, 16)
- v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
- // not yet converted to inline for size and complexity
- fe_mul(r.Z, r.X, q.YplusX)
- fe_mul(r.Y, r.Y, q.YminusX)
- fe_mul(r.T, q.T2d, p.T)
- fe_mul(r.X, p.Z, q.Z)
- // remaining arithmetic done inline to avoid unnecessary load/store
- rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
- rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
- let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32)
- let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32)
- let t0: v128, t1: v128, t2: v128
-
- // t = 2*rX
- t0 = v128.shl<i32>(rX0, 1)
- t1 = v128.shl<i32>(rX1, 1)
- t2 = v128.shl<i32>(rX2, 1)
-
- // rX = rZ - rY
- rX0 = v128.sub<i32>(rZ0, rY0)
- rX1 = v128.sub<i32>(rZ1, rY1)
- rX2 = v128.sub<i32>(rZ2, rY2)
-
- // rY = rZ + rY
- rY0 = v128.add<i32>(rZ0, rY0)
- rY1 = v128.add<i32>(rZ1, rY1)
- rY2 = v128.add<i32>(rZ2, rY2)
-
- // rZ = t + rT
- rZ0 = v128.add<i32>(t0, rT0)
- rZ1 = v128.add<i32>(t1, rT1)
- rZ2 = v128.add<i32>(t2, rT2)
-
- // rT = t - rT
- rT0 = v128.sub<i32>(t0, rT0)
- rT1 = v128.sub<i32>(t1, rT1)
- rT2 = v128.sub<i32>(t2, rT2)
-
- // store r.X
- v128.store(rX_ptr, rX0)
- v128.store(rX_ptr, rX1, 16)
- v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
- // store r.Y
- v128.store(rY_ptr, rY0)
- v128.store(rY_ptr, rY1, 16)
- v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
- // store r.Z
- v128.store(rZ_ptr, rZ0)
- v128.store(rZ_ptr, rZ1, 16)
- v128.store_lane<u64>(rZ_ptr, rZ2, 0, 32)
-
- // store r.T
- v128.store(rT_ptr, rT0)
- v128.store(rT_ptr, rT1, 16)
- v128.store_lane<u64>(rT_ptr, rT2, 0, 32)
+ let
+ r_X_0123: v128,
+ r_X_4567: v128,
+ r_X_89xx: v128
+,
+ r_Y_0123: v128,
+ r_Y_4567: v128,
+ r_Y_89xx: v128
+,
+ r_Z_0123: v128,
+ r_Z_4567: v128,
+ r_Z_89xx: v128
+,
+ r_T_0123: v128,
+ r_T_4567: v128,
+ r_T_89xx: v128
+,
+ p_X_0123: v128,
+ p_X_4567: v128,
+ p_X_89xx: v128
+,
+ p_Y_0123: v128,
+ p_Y_4567: v128,
+ p_Y_89xx: v128
+,
+ p_Z_0123: v128,
+ p_Z_4567: v128,
+ p_Z_89xx: v128
+,
+ p_T_0123: v128,
+ p_T_4567: v128,
+ p_T_89xx: v128
+,
+ q_YplusX_0123: v128,
+ q_YplusX_4567: v128,
+ q_YplusX_89xx: v128
+,
+ q_YminusX_0123: v128,
+ q_YminusX_4567: v128,
+ q_YminusX_89xx: v128
+,
+ q_Z_0123: v128,
+ q_Z_4567: v128,
+ q_Z_89xx: v128
+,
+ q_T2d_0123: v128,
+ q_T2d_4567: v128,
+ q_T2d_89xx: v128
+
+ // mask to select even lane from first vector and odd lane from second
+ const m: v128 = i32x4(-1, 0, -1, 0)
+
+ // mask to double odd-on-odd indexes
+ const odds: v128 = i32x4(0, -1, 0, -1)
+
+ // factor of 19 for wrap from h9 to h0
+ const v19: v128 = i32x4.splat(19)
+
+ let c0: i64
+ let c1: i64
+ let c2: i64
+ let c3: i64
+ let c4: i64
+ let c5: i64
+ let c6: i64
+ let c7: i64
+ let c8: i64
+ let c9: i64
+
+ let f0: i64
+ let f1: i64
+ let f2: i64
+ let f3: i64
+ let f4: i64
+ let f5: i64
+ let f6: i64
+ let f7: i64
+ let f8: i64
+ let f9: i64
+
+ let f0_2: i64
+ let f1_2: i64
+ let f2_2: i64
+ let f3_2: i64
+ let f4_2: i64
+ let f5_2: i64
+ let f6_2: i64
+ let f7_2: i64
+ let f8_2: i64
+ let f9_2: i64
+
+ let f5_19: i64
+ let f6_19: i64
+ let f7_19: i64
+ let f8_19: i64
+ let f9_19: i64
+
+ let f5_38: i64
+ let f7_38: i64
+ let f9_38: i64
+
+ let fi: i32
+ let fv: v128
+
+ let h0: i64
+ let h1: i64
+ let h2: i64
+ let h3: i64
+ let h4: i64
+ let h5: i64
+ let h6: i64
+ let h7: i64
+ let h8: i64
+ let h9: i64
+
+ let h01: v128
+ let h12: v128
+ let h23: v128
+ let h34: v128
+ let h45: v128
+ let h56: v128
+ let h67: v128
+ let h78: v128
+ let h89: v128
+ let h90: v128
+
+ let t0: v128
+ let t1: v128
+ let t2: v128
+ let t3: v128
+ let t4: v128
+
+ // [0..3, 4..7, 8..11] = r.X
+ r_X_0123 = v128.load(changetype<usize>(r.X), 0)
+ r_X_4567 = v128.load(changetype<usize>(r.X), 16)
+ r_X_89xx = v128.load(changetype<usize>(r.X), 32)
+
+ // [0..3, 4..7, 8..11] = r.Y
+ r_Y_0123 = v128.load(changetype<usize>(r.Y), 0)
+ r_Y_4567 = v128.load(changetype<usize>(r.Y), 16)
+ r_Y_89xx = v128.load(changetype<usize>(r.Y), 32)
+
+ // [0..3, 4..7, 8..11] = r.Z
+ r_Z_0123 = v128.load(changetype<usize>(r.Z), 0)
+ r_Z_4567 = v128.load(changetype<usize>(r.Z), 16)
+ r_Z_89xx = v128.load(changetype<usize>(r.Z), 32)
+
+ // [0..3, 4..7, 8..11] = r.T
+ r_T_0123 = v128.load(changetype<usize>(r.T), 0)
+ r_T_4567 = v128.load(changetype<usize>(r.T), 16)
+ r_T_89xx = v128.load(changetype<usize>(r.T), 32)
+
+ // [0..3, 4..7, 8..11] = p.X
+ p_X_0123 = v128.load(changetype<usize>(p.X), 0)
+ p_X_4567 = v128.load(changetype<usize>(p.X), 16)
+ p_X_89xx = v128.load(changetype<usize>(p.X), 32)
+
+ // [0..3, 4..7, 8..11] = p.Y
+ p_Y_0123 = v128.load(changetype<usize>(p.Y), 0)
+ p_Y_4567 = v128.load(changetype<usize>(p.Y), 16)
+ p_Y_89xx = v128.load(changetype<usize>(p.Y), 32)
+
+ // [0..3, 4..7, 8..11] = p.Z
+ p_Z_0123 = v128.load(changetype<usize>(p.Z), 0)
+ p_Z_4567 = v128.load(changetype<usize>(p.Z), 16)
+ p_Z_89xx = v128.load(changetype<usize>(p.Z), 32)
+
+ // [0..3, 4..7, 8..11] = p.T
+ p_T_0123 = v128.load(changetype<usize>(p.T), 0)
+ p_T_4567 = v128.load(changetype<usize>(p.T), 16)
+ p_T_89xx = v128.load(changetype<usize>(p.T), 32)
+
+ // [0..3, 4..7, 8..11] = q.YplusX
+ q_YplusX_0123 = v128.load(changetype<usize>(q.YplusX), 0)
+ q_YplusX_4567 = v128.load(changetype<usize>(q.YplusX), 16)
+ q_YplusX_89xx = v128.load(changetype<usize>(q.YplusX), 32)
+
+ // [0..3, 4..7, 8..11] = q.YminusX
+ q_YminusX_0123 = v128.load(changetype<usize>(q.YminusX), 0)
+ q_YminusX_4567 = v128.load(changetype<usize>(q.YminusX), 16)
+ q_YminusX_89xx = v128.load(changetype<usize>(q.YminusX), 32)
+
+ // [0..3, 4..7, 8..11] = q.Z
+ q_Z_0123 = v128.load(changetype<usize>(q.Z), 0)
+ q_Z_4567 = v128.load(changetype<usize>(q.Z), 16)
+ q_Z_89xx = v128.load(changetype<usize>(q.Z), 32)
+
+ // [0..3, 4..7, 8..11] = q.T2d
+ q_T2d_0123 = v128.load(changetype<usize>(q.T2d), 0)
+ q_T2d_4567 = v128.load(changetype<usize>(q.T2d), 16)
+ q_T2d_89xx = v128.load(changetype<usize>(q.T2d), 32)
+
+ // r.X = p.Y + p.X
+ r_X_0123 = v128.add<i32>(p_Y_0123, p_X_0123)
+ r_X_4567 = v128.add<i32>(p_Y_4567, p_X_4567)
+ r_X_89xx = v128.add<i32>(p_Y_89xx, p_X_89xx)
+
+ // r.Y = p.Y - p.X
+ r_Y_0123 = v128.sub<i32>(p_Y_0123, p_X_0123)
+ r_Y_4567 = v128.sub<i32>(p_Y_4567, p_X_4567)
+ r_Y_89xx = v128.sub<i32>(p_Y_89xx, p_X_89xx)
+
+ // r.Z = r.X * q.YplusX
+ const q_YplusX_0123_19: v128 = i32x4.mul(q_YplusX_0123, v19)
+ const q_YplusX_4567_19: v128 = i32x4.mul(q_YplusX_4567, v19)
+ const q_YplusX_89xx_19: v128 = i32x4.mul(q_YplusX_89xx, v19)
+
+ // f[0]
+ fi = v128.extract_lane<i32>(r_X_0123, 0)
+ fv = i32x4.splat(fi)
+
+ h01 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+ h23 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+ h45 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567)
+ h67 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567)
+ h89 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx)
+
+ // f[1]
+ fi = v128.extract_lane<i32>(r_X_0123, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ h12 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+ h34 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+ h56 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567)
+ h78 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567)
+ h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YplusX_89xx, q_YplusX_89xx_19, m))
+
+ // f[2]
+ fi = v128.extract_lane<i32>(r_X_0123, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+ h23 = i64x2.add(h23, t0)
+ h45 = i64x2.add(h45, t1)
+ h67 = i64x2.add(h67, t2)
+ h89 = i64x2.add(h89, t3)
+ h01 = i64x2.add(h01, t4)
+
+ // f[3]
+ fi = v128.extract_lane<i32>(r_X_0123, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_YplusX_4567, q_YplusX_4567_19, m))
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+ h34 = i64x2.add(h34, t0)
+ h56 = i64x2.add(h56, t1)
+ h78 = i64x2.add(h78, t2)
+ h90 = i64x2.add(h90, t3)
+ h12 = i64x2.add(h12, t4)
+
+ // f[4]
+ fi = v128.extract_lane<i32>(r_X_4567, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+ h45 = i64x2.add(h45, t0)
+ h67 = i64x2.add(h67, t1)
+ h89 = i64x2.add(h89, t2)
+ h01 = i64x2.add(h01, t3)
+ h23 = i64x2.add(h23, t4)
+
+ // f[5]
+ fi = v128.extract_lane<i32>(r_X_4567, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YplusX_4567, q_YplusX_4567_19, m))
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+ h56 = i64x2.add(h56, t0)
+ h78 = i64x2.add(h78, t1)
+ h90 = i64x2.add(h90, t2)
+ h12 = i64x2.add(h12, t3)
+ h34 = i64x2.add(h34, t4)
+
+ // f[6]
+ fi = v128.extract_lane<i32>(r_X_4567, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+ h67 = i64x2.add(h67, t0)
+ h89 = i64x2.add(h89, t1)
+ h01 = i64x2.add(h01, t2)
+ h23 = i64x2.add(h23, t3)
+ h45 = i64x2.add(h45, t4)
+
+ // f[7]
+ fi = v128.extract_lane<i32>(r_X_4567, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_YplusX_0123, q_YplusX_0123_19, m))
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+ h78 = i64x2.add(h78, t0)
+ h90 = i64x2.add(h90, t1)
+ h12 = i64x2.add(h12, t2)
+ h34 = i64x2.add(h34, t3)
+ h56 = i64x2.add(h56, t4)
+
+ // f[8]
+ fi = v128.extract_lane<i32>(r_X_89xx, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+ h89 = i64x2.add(h89, t0)
+ h01 = i64x2.add(h01, t1)
+ h23 = i64x2.add(h23, t2)
+ h45 = i64x2.add(h45, t3)
+ h67 = i64x2.add(h67, t4)
+
+ // f[9]
+ fi = v128.extract_lane<i32>(r_X_89xx, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YplusX_0123, q_YplusX_0123_19, m))
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YplusX_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YplusX_89xx_19)
+
+ h90 = i64x2.add(h90, t0)
+ h12 = i64x2.add(h12, t1)
+ h34 = i64x2.add(h34, t2)
+ h56 = i64x2.add(h56, t3)
+ h78 = i64x2.add(h78, t4)
+
+ // extract scalars from vectors
+ h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+ h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+ h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+ h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+ h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+ h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+ h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+ h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+ h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+ h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+ // carry
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + (1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + (1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + (1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + (1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + (1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + (1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + (1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + (1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ r_Z_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ r_Z_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ r_Z_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+ // r.Y = r.Y * q.YminusX
+ const q_YminusX_0123_19: v128 = i32x4.mul(q_YminusX_0123, v19)
+ const q_YminusX_4567_19: v128 = i32x4.mul(q_YminusX_4567, v19)
+ const q_YminusX_89xx_19: v128 = i32x4.mul(q_YminusX_89xx, v19)
+
+ // f[0]
+ fi = v128.extract_lane<i32>(r_Y_0123, 0)
+ fv = i32x4.splat(fi)
+
+ h01 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+ h23 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+ h45 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567)
+ h67 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567)
+ h89 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx)
+
+ // f[1]
+ fi = v128.extract_lane<i32>(r_Y_0123, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ h12 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+ h34 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+ h56 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567)
+ h78 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567)
+ h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YminusX_89xx, q_YminusX_89xx_19, m))
+
+ // f[2]
+ fi = v128.extract_lane<i32>(r_Y_0123, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+ h23 = i64x2.add(h23, t0)
+ h45 = i64x2.add(h45, t1)
+ h67 = i64x2.add(h67, t2)
+ h89 = i64x2.add(h89, t3)
+ h01 = i64x2.add(h01, t4)
+
+ // f[3]
+ fi = v128.extract_lane<i32>(r_Y_0123, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_YminusX_4567, q_YminusX_4567_19, m))
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+ h34 = i64x2.add(h34, t0)
+ h56 = i64x2.add(h56, t1)
+ h78 = i64x2.add(h78, t2)
+ h90 = i64x2.add(h90, t3)
+ h12 = i64x2.add(h12, t4)
+
+ // f[4]
+ fi = v128.extract_lane<i32>(r_Y_4567, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+ h45 = i64x2.add(h45, t0)
+ h67 = i64x2.add(h67, t1)
+ h89 = i64x2.add(h89, t2)
+ h01 = i64x2.add(h01, t3)
+ h23 = i64x2.add(h23, t4)
+
+ // f[5]
+ fi = v128.extract_lane<i32>(r_Y_4567, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YminusX_4567, q_YminusX_4567_19, m))
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+ h56 = i64x2.add(h56, t0)
+ h78 = i64x2.add(h78, t1)
+ h90 = i64x2.add(h90, t2)
+ h12 = i64x2.add(h12, t3)
+ h34 = i64x2.add(h34, t4)
+
+ // f[6]
+ fi = v128.extract_lane<i32>(r_Y_4567, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+ h67 = i64x2.add(h67, t0)
+ h89 = i64x2.add(h89, t1)
+ h01 = i64x2.add(h01, t2)
+ h23 = i64x2.add(h23, t3)
+ h45 = i64x2.add(h45, t4)
+
+ // f[7]
+ fi = v128.extract_lane<i32>(r_Y_4567, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_YminusX_0123, q_YminusX_0123_19, m))
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+ h78 = i64x2.add(h78, t0)
+ h90 = i64x2.add(h90, t1)
+ h12 = i64x2.add(h12, t2)
+ h34 = i64x2.add(h34, t3)
+ h56 = i64x2.add(h56, t4)
+
+ // f[8]
+ fi = v128.extract_lane<i32>(r_Y_89xx, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+ h89 = i64x2.add(h89, t0)
+ h01 = i64x2.add(h01, t1)
+ h23 = i64x2.add(h23, t2)
+ h45 = i64x2.add(h45, t3)
+ h67 = i64x2.add(h67, t4)
+
+ // f[9]
+ fi = v128.extract_lane<i32>(r_Y_89xx, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_YminusX_0123, q_YminusX_0123_19, m))
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_YminusX_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_YminusX_89xx_19)
+
+ h90 = i64x2.add(h90, t0)
+ h12 = i64x2.add(h12, t1)
+ h34 = i64x2.add(h34, t2)
+ h56 = i64x2.add(h56, t3)
+ h78 = i64x2.add(h78, t4)
+
+ // extract scalars from vectors
+ h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+ h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+ h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+ h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+ h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+ h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+ h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+ h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+ h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+ h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+ // carry
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + (1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + (1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + (1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + (1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + (1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + (1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + (1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + (1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ r_Y_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ r_Y_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ r_Y_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+ // r.T = q.T2d * p.T
+ const p_T_0123_19: v128 = i32x4.mul(p_T_0123, v19)
+ const p_T_4567_19: v128 = i32x4.mul(p_T_4567, v19)
+ const p_T_89xx_19: v128 = i32x4.mul(p_T_89xx, v19)
+
+ // f[0]
+ fi = v128.extract_lane<i32>(q_T2d_0123, 0)
+ fv = i32x4.splat(fi)
+
+ h01 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ h23 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ h45 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+ h67 = i64x2.extmul_high_i32x4_s(fv, p_T_4567)
+ h89 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx)
+
+ // f[1]
+ fi = v128.extract_lane<i32>(q_T2d_0123, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ h12 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ h34 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ h56 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+ h78 = i64x2.extmul_high_i32x4_s(fv, p_T_4567)
+ h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_89xx, p_T_89xx_19, m))
+
+ // f[2]
+ fi = v128.extract_lane<i32>(q_T2d_0123, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h23 = i64x2.add(h23, t0)
+ h45 = i64x2.add(h45, t1)
+ h67 = i64x2.add(h67, t2)
+ h89 = i64x2.add(h89, t3)
+ h01 = i64x2.add(h01, t4)
+
+ // f[3]
+ fi = v128.extract_lane<i32>(q_T2d_0123, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(p_T_4567, p_T_4567_19, m))
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h34 = i64x2.add(h34, t0)
+ h56 = i64x2.add(h56, t1)
+ h78 = i64x2.add(h78, t2)
+ h90 = i64x2.add(h90, t3)
+ h12 = i64x2.add(h12, t4)
+
+ // f[4]
+ fi = v128.extract_lane<i32>(q_T2d_4567, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h45 = i64x2.add(h45, t0)
+ h67 = i64x2.add(h67, t1)
+ h89 = i64x2.add(h89, t2)
+ h01 = i64x2.add(h01, t3)
+ h23 = i64x2.add(h23, t4)
+
+ // f[5]
+ fi = v128.extract_lane<i32>(q_T2d_4567, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_4567, p_T_4567_19, m))
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h56 = i64x2.add(h56, t0)
+ h78 = i64x2.add(h78, t1)
+ h90 = i64x2.add(h90, t2)
+ h12 = i64x2.add(h12, t3)
+ h34 = i64x2.add(h34, t4)
+
+ // f[6]
+ fi = v128.extract_lane<i32>(q_T2d_4567, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h67 = i64x2.add(h67, t0)
+ h89 = i64x2.add(h89, t1)
+ h01 = i64x2.add(h01, t2)
+ h23 = i64x2.add(h23, t3)
+ h45 = i64x2.add(h45, t4)
+
+ // f[7]
+ fi = v128.extract_lane<i32>(q_T2d_4567, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(p_T_0123, p_T_0123_19, m))
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h78 = i64x2.add(h78, t0)
+ h90 = i64x2.add(h90, t1)
+ h12 = i64x2.add(h12, t2)
+ h34 = i64x2.add(h34, t3)
+ h56 = i64x2.add(h56, t4)
+
+ // f[8]
+ fi = v128.extract_lane<i32>(q_T2d_89xx, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h89 = i64x2.add(h89, t0)
+ h01 = i64x2.add(h01, t1)
+ h23 = i64x2.add(h23, t2)
+ h45 = i64x2.add(h45, t3)
+ h67 = i64x2.add(h67, t4)
+
+ // f[9]
+ fi = v128.extract_lane<i32>(q_T2d_89xx, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_0123, p_T_0123_19, m))
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h90 = i64x2.add(h90, t0)
+ h12 = i64x2.add(h12, t1)
+ h34 = i64x2.add(h34, t2)
+ h56 = i64x2.add(h56, t3)
+ h78 = i64x2.add(h78, t4)
+
+ // extract scalars from vectors
+ h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+ h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+ h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+ h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+ h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+ h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+ h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+ h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+ h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+ h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+ // carry
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + (1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + (1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + (1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + (1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + (1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + (1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + (1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + (1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ r_T_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ r_T_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ r_T_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+ // r.X = p.Z * q.Z
+ const q_Z_0123_19: v128 = i32x4.mul(q_Z_0123, v19)
+ const q_Z_4567_19: v128 = i32x4.mul(q_Z_4567, v19)
+ const q_Z_89xx_19: v128 = i32x4.mul(q_Z_89xx, v19)
+
+ // f[0]
+ fi = v128.extract_lane<i32>(p_Z_0123, 0)
+ fv = i32x4.splat(fi)
+
+ h01 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+ h23 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+ h45 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567)
+ h67 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567)
+ h89 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx)
+
+ // f[1]
+ fi = v128.extract_lane<i32>(p_Z_0123, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ h12 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+ h34 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+ h56 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567)
+ h78 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567)
+ h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_Z_89xx, q_Z_89xx_19, m))
+
+ // f[2]
+ fi = v128.extract_lane<i32>(p_Z_0123, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+ h23 = i64x2.add(h23, t0)
+ h45 = i64x2.add(h45, t1)
+ h67 = i64x2.add(h67, t2)
+ h89 = i64x2.add(h89, t3)
+ h01 = i64x2.add(h01, t4)
+
+ // f[3]
+ fi = v128.extract_lane<i32>(p_Z_0123, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_Z_4567, q_Z_4567_19, m))
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+ h34 = i64x2.add(h34, t0)
+ h56 = i64x2.add(h56, t1)
+ h78 = i64x2.add(h78, t2)
+ h90 = i64x2.add(h90, t3)
+ h12 = i64x2.add(h12, t4)
+
+ // f[4]
+ fi = v128.extract_lane<i32>(p_Z_4567, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+ h45 = i64x2.add(h45, t0)
+ h67 = i64x2.add(h67, t1)
+ h89 = i64x2.add(h89, t2)
+ h01 = i64x2.add(h01, t3)
+ h23 = i64x2.add(h23, t4)
+
+ // f[5]
+ fi = v128.extract_lane<i32>(p_Z_4567, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_Z_4567, q_Z_4567_19, m))
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+ h56 = i64x2.add(h56, t0)
+ h78 = i64x2.add(h78, t1)
+ h90 = i64x2.add(h90, t2)
+ h12 = i64x2.add(h12, t3)
+ h34 = i64x2.add(h34, t4)
+
+ // f[6]
+ fi = v128.extract_lane<i32>(p_Z_4567, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+ h67 = i64x2.add(h67, t0)
+ h89 = i64x2.add(h89, t1)
+ h01 = i64x2.add(h01, t2)
+ h23 = i64x2.add(h23, t3)
+ h45 = i64x2.add(h45, t4)
+
+ // f[7]
+ fi = v128.extract_lane<i32>(p_Z_4567, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_Z_0123, q_Z_0123_19, m))
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+ h78 = i64x2.add(h78, t0)
+ h90 = i64x2.add(h90, t1)
+ h12 = i64x2.add(h12, t2)
+ h34 = i64x2.add(h34, t3)
+ h56 = i64x2.add(h56, t4)
+
+ // f[8]
+ fi = v128.extract_lane<i32>(p_Z_89xx, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_Z_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+ h89 = i64x2.add(h89, t0)
+ h01 = i64x2.add(h01, t1)
+ h23 = i64x2.add(h23, t2)
+ h45 = i64x2.add(h45, t3)
+ h67 = i64x2.add(h67, t4)
+
+ // f[9]
+ fi = v128.extract_lane<i32>(p_Z_89xx, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_Z_0123, q_Z_0123_19, m))
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_Z_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_Z_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_Z_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_Z_89xx_19)
+
+ h90 = i64x2.add(h90, t0)
+ h12 = i64x2.add(h12, t1)
+ h34 = i64x2.add(h34, t2)
+ h56 = i64x2.add(h56, t3)
+ h78 = i64x2.add(h78, t4)
+
+ // extract scalars from vectors
+ h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+ h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+ h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+ h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+ h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+ h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+ h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+ h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+ h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+ h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+ // carry
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + (1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + (1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + (1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + (1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + (1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + (1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + (1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + (1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ r_X_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ r_X_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ r_X_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+ let t_0123: v128, t_4567: v128, t_89xx: v128
+
+ // t = 2*r.X
+ t_0123 = v128.shl<i32>(r_X_0123, 1)
+ t_4567 = v128.shl<i32>(r_X_4567, 1)
+ t_89xx = v128.shl<i32>(r_X_89xx, 1)
+
+ // r.X = r.Z - r.Y
+ r_X_0123 = v128.sub<i32>(r_Z_0123, r_Y_0123)
+ r_X_4567 = v128.sub<i32>(r_Z_4567, r_Y_4567)
+ r_X_89xx = v128.sub<i32>(r_Z_89xx, r_Y_89xx)
+
+ // r.Y = r.Z + r.Y
+ r_Y_0123 = v128.add<i32>(r_Z_0123, r_Y_0123)
+ r_Y_4567 = v128.add<i32>(r_Z_4567, r_Y_4567)
+ r_Y_89xx = v128.add<i32>(r_Z_89xx, r_Y_89xx)
+
+ // r.Z = t + r.T
+ r_Z_0123 = v128.add<i32>(t_0123, r_T_0123)
+ r_Z_4567 = v128.add<i32>(t_4567, r_T_4567)
+ r_Z_89xx = v128.add<i32>(t_89xx, r_T_89xx)
+
+ // r.T = t - r.T
+ r_T_0123 = v128.sub<i32>(t_0123, r_T_0123)
+ r_T_4567 = v128.sub<i32>(t_4567, r_T_4567)
+ r_T_89xx = v128.sub<i32>(t_89xx, r_T_89xx)
+
+ // r.X = [0..3, 4..7, 8..11]
+ v128.store(changetype<usize>(r.X), r_X_0123, 0)
+ v128.store(changetype<usize>(r.X), r_X_4567, 16)
+ v128.store_lane<u64>(changetype<usize>(r.X), r_X_89xx, 0, 32)
+
+ // r.Y = [0..3, 4..7, 8..11]
+ v128.store(changetype<usize>(r.Y), r_Y_0123, 0)
+ v128.store(changetype<usize>(r.Y), r_Y_4567, 16)
+ v128.store_lane<u64>(changetype<usize>(r.Y), r_Y_89xx, 0, 32)
+
+ // r.Z = [0..3, 4..7, 8..11]
+ v128.store(changetype<usize>(r.Z), r_Z_0123, 0)
+ v128.store(changetype<usize>(r.Z), r_Z_4567, 16)
+ v128.store_lane<u64>(changetype<usize>(r.Z), r_Z_89xx, 0, 32)
+
+ // r.T = [0..3, 4..7, 8..11]
+ v128.store(changetype<usize>(r.T), r_T_0123, 0)
+ v128.store(changetype<usize>(r.T), r_T_4567, 16)
+ v128.store_lane<u64>(changetype<usize>(r.T), r_T_89xx, 0, 32)
+
}
/**
//@ts-expect-error
@inline
export function ge_add_precomp (r: ge_p1p1, p: ge_p3, q: ge_precomp): void {
- const rX_ptr = changetype<usize>(r.X)
- const rY_ptr = changetype<usize>(r.Y)
- const rZ_ptr = changetype<usize>(r.Z)
- const rT_ptr = changetype<usize>(r.T)
-
- const pX_ptr = changetype<usize>(p.X)
- const pY_ptr = changetype<usize>(p.Y)
- const pZ_ptr = changetype<usize>(p.Z)
-
- let rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
- let rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
- const pX0 = v128.load(pX_ptr), pX1 = v128.load(pX_ptr, 16), pX2 = v128.load(pX_ptr, 32)
- const pY0 = v128.load(pY_ptr), pY1 = v128.load(pY_ptr, 16), pY2 = v128.load(pY_ptr, 32)
-
- // rX = pY + pX
- rX0 = v128.add<i32>(pY0, pX0)
- rX1 = v128.add<i32>(pY1, pX1)
- rX2 = v128.add<i32>(pY2, pX2)
-
- // rY = pY - pX
- rY0 = v128.sub<i32>(pY0, pX0)
- rY1 = v128.sub<i32>(pY1, pX1)
- rY2 = v128.sub<i32>(pY2, pX2)
-
- // store r.X
- v128.store(rX_ptr, rX0)
- v128.store(rX_ptr, rX1, 16)
- v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
- // store r.Y
- v128.store(rY_ptr, rY0)
- v128.store(rY_ptr, rY1, 16)
- v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
- // not yet converted to inline for size and complexity
- fe_mul(r.Z, r.X, q.yplusx)
- fe_mul(r.Y, r.Y, q.yminusx)
- fe_mul(r.T, q.xy2d, p.T)
- // remaining arithmetic done inline to avoid unnecessary load/store
- rX0 = v128.load(rX_ptr), rX1 = v128.load(rX_ptr, 16), rX2 = v128.load(rX_ptr, 32)
- rY0 = v128.load(rY_ptr), rY1 = v128.load(rY_ptr, 16), rY2 = v128.load(rY_ptr, 32)
- let rZ0 = v128.load(rZ_ptr), rZ1 = v128.load(rZ_ptr, 16), rZ2 = v128.load(rZ_ptr, 32)
- let rT0 = v128.load(rT_ptr), rT1 = v128.load(rT_ptr, 16), rT2 = v128.load(rT_ptr, 32)
-
- let pZ0 = v128.load(pZ_ptr, 0)
- let pZ1 = v128.load(pZ_ptr, 16)
- let pZ2 = v128.load(pZ_ptr, 32)
-
- // pZ = 2*pZ
- pZ0 = v128.shl<i32>(pZ0, 1)
- pZ1 = v128.shl<i32>(pZ1, 1)
- pZ2 = v128.shl<i32>(pZ2, 1)
-
- // rX = rZ - rY
- rX0 = v128.sub<i32>(rZ0, rY0)
- rX1 = v128.sub<i32>(rZ1, rY1)
- rX2 = v128.sub<i32>(rZ2, rY2)
-
- // rY = rZ + rY
- rY0 = v128.add<i32>(rZ0, rY0)
- rY1 = v128.add<i32>(rZ1, rY1)
- rY2 = v128.add<i32>(rZ2, rY2)
-
- // rZ = pZ + rT
- rZ0 = v128.add<i32>(pZ0, rT0)
- rZ1 = v128.add<i32>(pZ1, rT1)
- rZ2 = v128.add<i32>(pZ2, rT2)
-
- // rT = pZ - rT
- rT0 = v128.sub<i32>(pZ0, rT0)
- rT1 = v128.sub<i32>(pZ1, rT1)
- rT2 = v128.sub<i32>(pZ2, rT2)
-
- // store r.X
- v128.store(rX_ptr, rX0)
- v128.store(rX_ptr, rX1, 16)
- v128.store_lane<u64>(rX_ptr, rX2, 0, 32)
-
- // store r.Y
- v128.store(rY_ptr, rY0)
- v128.store(rY_ptr, rY1, 16)
- v128.store_lane<u64>(rY_ptr, rY2, 0, 32)
-
- // store r.Z
- v128.store(rZ_ptr, rZ0)
- v128.store(rZ_ptr, rZ1, 16)
- v128.store_lane<u64>(rZ_ptr, rZ2, 0, 32)
-
- // store r.T
- v128.store(rT_ptr, rT0)
- v128.store(rT_ptr, rT1, 16)
- v128.store_lane<u64>(rT_ptr, rT2, 0, 32)
+ let
+ r_X_0123: v128,
+ r_X_4567: v128,
+ r_X_89xx: v128
+,
+ r_Y_0123: v128,
+ r_Y_4567: v128,
+ r_Y_89xx: v128
+,
+ r_Z_0123: v128,
+ r_Z_4567: v128,
+ r_Z_89xx: v128
+,
+ r_T_0123: v128,
+ r_T_4567: v128,
+ r_T_89xx: v128
+,
+ p_X_0123: v128,
+ p_X_4567: v128,
+ p_X_89xx: v128
+,
+ p_Y_0123: v128,
+ p_Y_4567: v128,
+ p_Y_89xx: v128
+,
+ p_Z_0123: v128,
+ p_Z_4567: v128,
+ p_Z_89xx: v128
+,
+ p_T_0123: v128,
+ p_T_4567: v128,
+ p_T_89xx: v128
+,
+ q_yplusx_0123: v128,
+ q_yplusx_4567: v128,
+ q_yplusx_89xx: v128
+,
+ q_yminusx_0123: v128,
+ q_yminusx_4567: v128,
+ q_yminusx_89xx: v128
+,
+ q_xy2d_0123: v128,
+ q_xy2d_4567: v128,
+ q_xy2d_89xx: v128
+
+ // mask to select even lane from first vector and odd lane from second
+ const m: v128 = i32x4(-1, 0, -1, 0)
+
+ // mask to double odd-on-odd indexes
+ const odds: v128 = i32x4(0, -1, 0, -1)
+
+ // factor of 19 for wrap from h9 to h0
+ const v19: v128 = i32x4.splat(19)
+
+ let c0: i64
+ let c1: i64
+ let c2: i64
+ let c3: i64
+ let c4: i64
+ let c5: i64
+ let c6: i64
+ let c7: i64
+ let c8: i64
+ let c9: i64
+
+ let f0: i64
+ let f1: i64
+ let f2: i64
+ let f3: i64
+ let f4: i64
+ let f5: i64
+ let f6: i64
+ let f7: i64
+ let f8: i64
+ let f9: i64
+
+ let f0_2: i64
+ let f1_2: i64
+ let f2_2: i64
+ let f3_2: i64
+ let f4_2: i64
+ let f5_2: i64
+ let f6_2: i64
+ let f7_2: i64
+ let f8_2: i64
+ let f9_2: i64
+
+ let f5_19: i64
+ let f6_19: i64
+ let f7_19: i64
+ let f8_19: i64
+ let f9_19: i64
+
+ let f5_38: i64
+ let f7_38: i64
+ let f9_38: i64
+
+ let fi: i32
+ let fv: v128
+
+ let h0: i64
+ let h1: i64
+ let h2: i64
+ let h3: i64
+ let h4: i64
+ let h5: i64
+ let h6: i64
+ let h7: i64
+ let h8: i64
+ let h9: i64
+
+ let h01: v128
+ let h12: v128
+ let h23: v128
+ let h34: v128
+ let h45: v128
+ let h56: v128
+ let h67: v128
+ let h78: v128
+ let h89: v128
+ let h90: v128
+
+ let t0: v128
+ let t1: v128
+ let t2: v128
+ let t3: v128
+ let t4: v128
+
+ // [0..3, 4..7, 8..11] = r.X
+ r_X_0123 = v128.load(changetype<usize>(r.X), 0)
+ r_X_4567 = v128.load(changetype<usize>(r.X), 16)
+ r_X_89xx = v128.load(changetype<usize>(r.X), 32)
+
+ // [0..3, 4..7, 8..11] = r.Y
+ r_Y_0123 = v128.load(changetype<usize>(r.Y), 0)
+ r_Y_4567 = v128.load(changetype<usize>(r.Y), 16)
+ r_Y_89xx = v128.load(changetype<usize>(r.Y), 32)
+
+ // [0..3, 4..7, 8..11] = r.Z
+ r_Z_0123 = v128.load(changetype<usize>(r.Z), 0)
+ r_Z_4567 = v128.load(changetype<usize>(r.Z), 16)
+ r_Z_89xx = v128.load(changetype<usize>(r.Z), 32)
+
+ // [0..3, 4..7, 8..11] = r.T
+ r_T_0123 = v128.load(changetype<usize>(r.T), 0)
+ r_T_4567 = v128.load(changetype<usize>(r.T), 16)
+ r_T_89xx = v128.load(changetype<usize>(r.T), 32)
+
+ // [0..3, 4..7, 8..11] = p.X
+ p_X_0123 = v128.load(changetype<usize>(p.X), 0)
+ p_X_4567 = v128.load(changetype<usize>(p.X), 16)
+ p_X_89xx = v128.load(changetype<usize>(p.X), 32)
+
+ // [0..3, 4..7, 8..11] = p.Y
+ p_Y_0123 = v128.load(changetype<usize>(p.Y), 0)
+ p_Y_4567 = v128.load(changetype<usize>(p.Y), 16)
+ p_Y_89xx = v128.load(changetype<usize>(p.Y), 32)
+
+ // [0..3, 4..7, 8..11] = p.Z
+ p_Z_0123 = v128.load(changetype<usize>(p.Z), 0)
+ p_Z_4567 = v128.load(changetype<usize>(p.Z), 16)
+ p_Z_89xx = v128.load(changetype<usize>(p.Z), 32)
+
+ // [0..3, 4..7, 8..11] = p.T
+ p_T_0123 = v128.load(changetype<usize>(p.T), 0)
+ p_T_4567 = v128.load(changetype<usize>(p.T), 16)
+ p_T_89xx = v128.load(changetype<usize>(p.T), 32)
+
+ // [0..3, 4..7, 8..11] = q.yplusx
+ q_yplusx_0123 = v128.load(changetype<usize>(q.yplusx), 0)
+ q_yplusx_4567 = v128.load(changetype<usize>(q.yplusx), 16)
+ q_yplusx_89xx = v128.load(changetype<usize>(q.yplusx), 32)
+
+ // [0..3, 4..7, 8..11] = q.yminusx
+ q_yminusx_0123 = v128.load(changetype<usize>(q.yminusx), 0)
+ q_yminusx_4567 = v128.load(changetype<usize>(q.yminusx), 16)
+ q_yminusx_89xx = v128.load(changetype<usize>(q.yminusx), 32)
+
+ // [0..3, 4..7, 8..11] = q.xy2d
+ q_xy2d_0123 = v128.load(changetype<usize>(q.xy2d), 0)
+ q_xy2d_4567 = v128.load(changetype<usize>(q.xy2d), 16)
+ q_xy2d_89xx = v128.load(changetype<usize>(q.xy2d), 32)
+
+ // r.X = p.Y + p.X
+ r_X_0123 = v128.add<i32>(p_Y_0123, p_X_0123)
+ r_X_4567 = v128.add<i32>(p_Y_4567, p_X_4567)
+ r_X_89xx = v128.add<i32>(p_Y_89xx, p_X_89xx)
+
+ // r.Y = p.Y - p.X
+ r_Y_0123 = v128.sub<i32>(p_Y_0123, p_X_0123)
+ r_Y_4567 = v128.sub<i32>(p_Y_4567, p_X_4567)
+ r_Y_89xx = v128.sub<i32>(p_Y_89xx, p_X_89xx)
+
+ // r.Z = r.X * q.yplusx
+ const q_yplusx_0123_19: v128 = i32x4.mul(q_yplusx_0123, v19)
+ const q_yplusx_4567_19: v128 = i32x4.mul(q_yplusx_4567, v19)
+ const q_yplusx_89xx_19: v128 = i32x4.mul(q_yplusx_89xx, v19)
+
+ // f[0]
+ fi = v128.extract_lane<i32>(r_X_0123, 0)
+ fv = i32x4.splat(fi)
+
+ h01 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+ h23 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+ h45 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567)
+ h67 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567)
+ h89 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx)
+
+ // f[1]
+ fi = v128.extract_lane<i32>(r_X_0123, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ h12 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+ h34 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+ h56 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567)
+ h78 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567)
+ h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yplusx_89xx, q_yplusx_89xx_19, m))
+
+ // f[2]
+ fi = v128.extract_lane<i32>(r_X_0123, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+ h23 = i64x2.add(h23, t0)
+ h45 = i64x2.add(h45, t1)
+ h67 = i64x2.add(h67, t2)
+ h89 = i64x2.add(h89, t3)
+ h01 = i64x2.add(h01, t4)
+
+ // f[3]
+ fi = v128.extract_lane<i32>(r_X_0123, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_yplusx_4567, q_yplusx_4567_19, m))
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+ h34 = i64x2.add(h34, t0)
+ h56 = i64x2.add(h56, t1)
+ h78 = i64x2.add(h78, t2)
+ h90 = i64x2.add(h90, t3)
+ h12 = i64x2.add(h12, t4)
+
+ // f[4]
+ fi = v128.extract_lane<i32>(r_X_4567, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+ h45 = i64x2.add(h45, t0)
+ h67 = i64x2.add(h67, t1)
+ h89 = i64x2.add(h89, t2)
+ h01 = i64x2.add(h01, t3)
+ h23 = i64x2.add(h23, t4)
+
+ // f[5]
+ fi = v128.extract_lane<i32>(r_X_4567, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yplusx_4567, q_yplusx_4567_19, m))
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+ h56 = i64x2.add(h56, t0)
+ h78 = i64x2.add(h78, t1)
+ h90 = i64x2.add(h90, t2)
+ h12 = i64x2.add(h12, t3)
+ h34 = i64x2.add(h34, t4)
+
+ // f[6]
+ fi = v128.extract_lane<i32>(r_X_4567, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+ h67 = i64x2.add(h67, t0)
+ h89 = i64x2.add(h89, t1)
+ h01 = i64x2.add(h01, t2)
+ h23 = i64x2.add(h23, t3)
+ h45 = i64x2.add(h45, t4)
+
+ // f[7]
+ fi = v128.extract_lane<i32>(r_X_4567, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_yplusx_0123, q_yplusx_0123_19, m))
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+ h78 = i64x2.add(h78, t0)
+ h90 = i64x2.add(h90, t1)
+ h12 = i64x2.add(h12, t2)
+ h34 = i64x2.add(h34, t3)
+ h56 = i64x2.add(h56, t4)
+
+ // f[8]
+ fi = v128.extract_lane<i32>(r_X_89xx, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+ h89 = i64x2.add(h89, t0)
+ h01 = i64x2.add(h01, t1)
+ h23 = i64x2.add(h23, t2)
+ h45 = i64x2.add(h45, t3)
+ h67 = i64x2.add(h67, t4)
+
+ // f[9]
+ fi = v128.extract_lane<i32>(r_X_89xx, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yplusx_0123, q_yplusx_0123_19, m))
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yplusx_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yplusx_89xx_19)
+
+ h90 = i64x2.add(h90, t0)
+ h12 = i64x2.add(h12, t1)
+ h34 = i64x2.add(h34, t2)
+ h56 = i64x2.add(h56, t3)
+ h78 = i64x2.add(h78, t4)
+
+ // extract scalars from vectors
+ h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+ h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+ h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+ h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+ h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+ h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+ h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+ h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+ h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+ h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+ // carry
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + (1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + (1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + (1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + (1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + (1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + (1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + (1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + (1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ r_Z_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ r_Z_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ r_Z_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+ // r.Y = r.Y * q.yminusx
+ const q_yminusx_0123_19: v128 = i32x4.mul(q_yminusx_0123, v19)
+ const q_yminusx_4567_19: v128 = i32x4.mul(q_yminusx_4567, v19)
+ const q_yminusx_89xx_19: v128 = i32x4.mul(q_yminusx_89xx, v19)
+
+ // f[0]
+ fi = v128.extract_lane<i32>(r_Y_0123, 0)
+ fv = i32x4.splat(fi)
+
+ h01 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+ h23 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+ h45 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567)
+ h67 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567)
+ h89 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx)
+
+ // f[1]
+ fi = v128.extract_lane<i32>(r_Y_0123, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ h12 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+ h34 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+ h56 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567)
+ h78 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567)
+ h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yminusx_89xx, q_yminusx_89xx_19, m))
+
+ // f[2]
+ fi = v128.extract_lane<i32>(r_Y_0123, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+ h23 = i64x2.add(h23, t0)
+ h45 = i64x2.add(h45, t1)
+ h67 = i64x2.add(h67, t2)
+ h89 = i64x2.add(h89, t3)
+ h01 = i64x2.add(h01, t4)
+
+ // f[3]
+ fi = v128.extract_lane<i32>(r_Y_0123, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_yminusx_4567, q_yminusx_4567_19, m))
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+ h34 = i64x2.add(h34, t0)
+ h56 = i64x2.add(h56, t1)
+ h78 = i64x2.add(h78, t2)
+ h90 = i64x2.add(h90, t3)
+ h12 = i64x2.add(h12, t4)
+
+ // f[4]
+ fi = v128.extract_lane<i32>(r_Y_4567, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+ h45 = i64x2.add(h45, t0)
+ h67 = i64x2.add(h67, t1)
+ h89 = i64x2.add(h89, t2)
+ h01 = i64x2.add(h01, t3)
+ h23 = i64x2.add(h23, t4)
+
+ // f[5]
+ fi = v128.extract_lane<i32>(r_Y_4567, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yminusx_4567, q_yminusx_4567_19, m))
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+ h56 = i64x2.add(h56, t0)
+ h78 = i64x2.add(h78, t1)
+ h90 = i64x2.add(h90, t2)
+ h12 = i64x2.add(h12, t3)
+ h34 = i64x2.add(h34, t4)
+
+ // f[6]
+ fi = v128.extract_lane<i32>(r_Y_4567, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+ h67 = i64x2.add(h67, t0)
+ h89 = i64x2.add(h89, t1)
+ h01 = i64x2.add(h01, t2)
+ h23 = i64x2.add(h23, t3)
+ h45 = i64x2.add(h45, t4)
+
+ // f[7]
+ fi = v128.extract_lane<i32>(r_Y_4567, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(q_yminusx_0123, q_yminusx_0123_19, m))
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+ h78 = i64x2.add(h78, t0)
+ h90 = i64x2.add(h90, t1)
+ h12 = i64x2.add(h12, t2)
+ h34 = i64x2.add(h34, t3)
+ h56 = i64x2.add(h56, t4)
+
+ // f[8]
+ fi = v128.extract_lane<i32>(r_Y_89xx, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+ h89 = i64x2.add(h89, t0)
+ h01 = i64x2.add(h01, t1)
+ h23 = i64x2.add(h23, t2)
+ h45 = i64x2.add(h45, t3)
+ h67 = i64x2.add(h67, t4)
+
+ // f[9]
+ fi = v128.extract_lane<i32>(r_Y_89xx, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(q_yminusx_0123, q_yminusx_0123_19, m))
+ t1 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, q_yminusx_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, q_yminusx_89xx_19)
+
+ h90 = i64x2.add(h90, t0)
+ h12 = i64x2.add(h12, t1)
+ h34 = i64x2.add(h34, t2)
+ h56 = i64x2.add(h56, t3)
+ h78 = i64x2.add(h78, t4)
+
+ // extract scalars from vectors
+ h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+ h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+ h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+ h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+ h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+ h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+ h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+ h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+ h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+ h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+ // carry
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + (1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + (1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + (1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + (1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + (1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + (1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + (1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + (1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ r_Y_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ r_Y_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ r_Y_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+ // r.T = q.xy2d * p.T
+ const p_T_0123_19: v128 = i32x4.mul(p_T_0123, v19)
+ const p_T_4567_19: v128 = i32x4.mul(p_T_4567, v19)
+ const p_T_89xx_19: v128 = i32x4.mul(p_T_89xx, v19)
+
+ // f[0]
+ fi = v128.extract_lane<i32>(q_xy2d_0123, 0)
+ fv = i32x4.splat(fi)
+
+ h01 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ h23 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ h45 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+ h67 = i64x2.extmul_high_i32x4_s(fv, p_T_4567)
+ h89 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx)
+
+ // f[1]
+ fi = v128.extract_lane<i32>(q_xy2d_0123, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ h12 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ h34 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ h56 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+ h78 = i64x2.extmul_high_i32x4_s(fv, p_T_4567)
+ h90 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_89xx, p_T_89xx_19, m))
+
+ // f[2]
+ fi = v128.extract_lane<i32>(q_xy2d_0123, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h23 = i64x2.add(h23, t0)
+ h45 = i64x2.add(h45, t1)
+ h67 = i64x2.add(h67, t2)
+ h89 = i64x2.add(h89, t3)
+ h01 = i64x2.add(h01, t4)
+
+ // f[3]
+ fi = v128.extract_lane<i32>(q_xy2d_0123, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(p_T_4567, p_T_4567_19, m))
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h34 = i64x2.add(h34, t0)
+ h56 = i64x2.add(h56, t1)
+ h78 = i64x2.add(h78, t2)
+ h90 = i64x2.add(h90, t3)
+ h12 = i64x2.add(h12, t4)
+
+ // f[4]
+ fi = v128.extract_lane<i32>(q_xy2d_4567, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567)
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h45 = i64x2.add(h45, t0)
+ h67 = i64x2.add(h67, t1)
+ h89 = i64x2.add(h89, t2)
+ h01 = i64x2.add(h01, t3)
+ h23 = i64x2.add(h23, t4)
+
+ // f[5]
+ fi = v128.extract_lane<i32>(q_xy2d_4567, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_4567, p_T_4567_19, m))
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h56 = i64x2.add(h56, t0)
+ h78 = i64x2.add(h78, t1)
+ h90 = i64x2.add(h90, t2)
+ h12 = i64x2.add(h12, t3)
+ h34 = i64x2.add(h34, t4)
+
+ // f[6]
+ fi = v128.extract_lane<i32>(q_xy2d_4567, 2)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123)
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h67 = i64x2.add(h67, t0)
+ h89 = i64x2.add(h89, t1)
+ h01 = i64x2.add(h01, t2)
+ h23 = i64x2.add(h23, t3)
+ h45 = i64x2.add(h45, t4)
+
+ // f[7]
+ fi = v128.extract_lane<i32>(q_xy2d_4567, 3)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, v128.bitselect(p_T_0123, p_T_0123_19, m))
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h78 = i64x2.add(h78, t0)
+ h90 = i64x2.add(h90, t1)
+ h12 = i64x2.add(h12, t2)
+ h34 = i64x2.add(h34, t3)
+ h56 = i64x2.add(h56, t4)
+
+ // f[8]
+ fi = v128.extract_lane<i32>(q_xy2d_89xx, 0)
+ fv = i32x4.splat(fi)
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, p_T_0123)
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h89 = i64x2.add(h89, t0)
+ h01 = i64x2.add(h01, t1)
+ h23 = i64x2.add(h23, t2)
+ h45 = i64x2.add(h45, t3)
+ h67 = i64x2.add(h67, t4)
+
+ // f[9]
+ fi = v128.extract_lane<i32>(q_xy2d_89xx, 1)
+ fv = i32x4.splat(fi)
+ fv = i32x4.add(fv, v128.and(fv, odds))
+
+ t0 = i64x2.extmul_low_i32x4_s(fv, v128.bitselect(p_T_0123, p_T_0123_19, m))
+ t1 = i64x2.extmul_high_i32x4_s(fv, p_T_0123_19)
+ t2 = i64x2.extmul_low_i32x4_s(fv, p_T_4567_19)
+ t3 = i64x2.extmul_high_i32x4_s(fv, p_T_4567_19)
+ t4 = i64x2.extmul_low_i32x4_s(fv, p_T_89xx_19)
+
+ h90 = i64x2.add(h90, t0)
+ h12 = i64x2.add(h12, t1)
+ h34 = i64x2.add(h34, t2)
+ h56 = i64x2.add(h56, t3)
+ h78 = i64x2.add(h78, t4)
+
+ // extract scalars from vectors
+ h0 = i64x2.extract_lane(h90, 1) + i64x2.extract_lane(h01, 0)
+ h1 = i64x2.extract_lane(h01, 1) + i64x2.extract_lane(h12, 0)
+ h2 = i64x2.extract_lane(h12, 1) + i64x2.extract_lane(h23, 0)
+ h3 = i64x2.extract_lane(h23, 1) + i64x2.extract_lane(h34, 0)
+ h4 = i64x2.extract_lane(h34, 1) + i64x2.extract_lane(h45, 0)
+ h5 = i64x2.extract_lane(h45, 1) + i64x2.extract_lane(h56, 0)
+ h6 = i64x2.extract_lane(h56, 1) + i64x2.extract_lane(h67, 0)
+ h7 = i64x2.extract_lane(h67, 1) + i64x2.extract_lane(h78, 0)
+ h8 = i64x2.extract_lane(h78, 1) + i64x2.extract_lane(h89, 0)
+ h9 = i64x2.extract_lane(h89, 1) + i64x2.extract_lane(h90, 0)
+
+ // carry
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+
+ c1 = (h1 + (1 << 24)) >> 25
+ h2 += c1
+ h1 -= c1 << 25
+ c5 = (h5 + (1 << 24)) >> 25
+ h6 += c5
+ h5 -= c5 << 25
+
+ c2 = (h2 + (1 << 25)) >> 26
+ h3 += c2
+ h2 -= c2 << 26
+ c6 = (h6 + (1 << 25)) >> 26
+ h7 += c6
+ h6 -= c6 << 26
+
+ c3 = (h3 + (1 << 24)) >> 25
+ h4 += c3
+ h3 -= c3 << 25
+ c7 = (h7 + (1 << 24)) >> 25
+ h8 += c7
+ h7 -= c7 << 25
+
+ c4 = (h4 + (1 << 25)) >> 26
+ h5 += c4
+ h4 -= c4 << 26
+ c8 = (h8 + (1 << 25)) >> 26
+ h9 += c8
+ h8 -= c8 << 26
+
+ c9 = (h9 + (1 << 24)) >> 25
+ h0 += c9 * 19
+ h9 -= c9 << 25
+
+ c0 = (h0 + (1 << 25)) >> 26
+ h1 += c0
+ h0 -= c0 << 26
+
+ // assign reduced results to output
+ r_T_0123 = i32x4(<i32>h0, <i32>h1, <i32>h2, <i32>h3)
+ r_T_4567 = i32x4(<i32>h4, <i32>h5, <i32>h6, <i32>h7)
+ r_T_89xx = i32x4(<i32>h8, <i32>h9, 0, 0)
+
+ // p.Z = 2*p.Z
+ p_Z_0123 = v128.shl<i32>(p_Z_0123, 1)
+ p_Z_4567 = v128.shl<i32>(p_Z_4567, 1)
+ p_Z_89xx = v128.shl<i32>(p_Z_89xx, 1)
+
+ // r.X = r.Z - r.Y
+ r_X_0123 = v128.sub<i32>(r_Z_0123, r_Y_0123)
+ r_X_4567 = v128.sub<i32>(r_Z_4567, r_Y_4567)
+ r_X_89xx = v128.sub<i32>(r_Z_89xx, r_Y_89xx)
+
+ // r.Y = r.Z + r.Y
+ r_Y_0123 = v128.add<i32>(r_Z_0123, r_Y_0123)
+ r_Y_4567 = v128.add<i32>(r_Z_4567, r_Y_4567)
+ r_Y_89xx = v128.add<i32>(r_Z_89xx, r_Y_89xx)
+
+ // r.Z = p.Z + r.T
+ r_Z_0123 = v128.add<i32>(p_Z_0123, r_T_0123)
+ r_Z_4567 = v128.add<i32>(p_Z_4567, r_T_4567)
+ r_Z_89xx = v128.add<i32>(p_Z_89xx, r_T_89xx)
+
+ // r.T = p.Z - r.T
+ r_T_0123 = v128.sub<i32>(p_Z_0123, r_T_0123)
+ r_T_4567 = v128.sub<i32>(p_Z_4567, r_T_4567)
+ r_T_89xx = v128.sub<i32>(p_Z_89xx, r_T_89xx)
+
+ // r.X = [0..3, 4..7, 8..11]
+ v128.store(changetype<usize>(r.X), r_X_0123, 0)
+ v128.store(changetype<usize>(r.X), r_X_4567, 16)
+ v128.store_lane<u64>(changetype<usize>(r.X), r_X_89xx, 0, 32)
+
+ // r.Y = [0..3, 4..7, 8..11]
+ v128.store(changetype<usize>(r.Y), r_Y_0123, 0)
+ v128.store(changetype<usize>(r.Y), r_Y_4567, 16)
+ v128.store_lane<u64>(changetype<usize>(r.Y), r_Y_89xx, 0, 32)
+
+ // r.Z = [0..3, 4..7, 8..11]
+ v128.store(changetype<usize>(r.Z), r_Z_0123, 0)
+ v128.store(changetype<usize>(r.Z), r_Z_4567, 16)
+ v128.store_lane<u64>(changetype<usize>(r.Z), r_Z_89xx, 0, 32)
+
+ // r.T = [0..3, 4..7, 8..11]
+ v128.store(changetype<usize>(r.T), r_T_0123, 0)
+ v128.store(changetype<usize>(r.T), r_T_4567, 16)
+ v128.store_lane<u64>(changetype<usize>(r.T), r_T_89xx, 0, 32)
+
}
const ge_sub_p3_q_cached = new ge_cached()