Stage 3: one reduction per coefficient in Fp6 and in the cyclotomic square

The reduction modulo p of a BigInt costs several times a product, and the
formulas of kilic reduce every product. The field layer gains FpWide, a sum
of products not yet reduced, and the product and the square of Fp6, its
product by the sparse element of a line and the square in Fp4 of the
cyclotomic square add their products in that form and reduce each
coefficient once. The values are the same, which the vectors of Go check;
a pairing takes about 15 % less on the VM and 13 % less on Node.js.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
v0.11
dev 2 days ago
parent d497c68408
commit 7d303bb339

@ -165,5 +165,35 @@ extension type const Fp._(BigInt _v) {
Fp._((a._v * b._v + c._v * d._v) % _p); Fp._((a._v * b._v + c._v * d._v) % _p);
} }
/// A sum of products of elements of Fp, not yet reduced modulo p: the tower
/// adds and subtracts products in this form and reduces each coefficient
/// once (lazy reduction), as the reduction costs several times a product.
/// Any integer, negative ones included. A layer of fixed limbs would make it
/// a value of double width.
extension type const FpWide._(BigInt _w) {
/// a·b.
static FpWide mul(Fp a, Fp b) => FpWide._(a._v * b._v);
/// a·b + c·d.
static FpWide mulAdd(Fp a, Fp b, Fp c, Fp d) =>
FpWide._(a._v * b._v + c._v * d._v);
/// a·b − c·d.
static FpWide mulSub(Fp a, Fp b, Fp c, Fp d) =>
FpWide._(a._v * b._v - c._v * d._v);
/// The sum.
FpWide operator +(FpWide o) => FpWide._(_w + o._w);
/// The difference.
FpWide operator -(FpWide o) => FpWide._(_w - o._w);
/// Twice this.
FpWide double() => FpWide._(_w + _w);
/// This modulo p.
Fp reduce() => Fp._(_w % _p);
}
BigInt _bigFromBytes(List<int> b) => BigInt _bigFromBytes(List<int> b) =>
b.isEmpty ? BigInt.zero : BigInt.parse(toHex(b), radix: 16); b.isEmpty ? BigInt.zero : BigInt.parse(toHex(b), radix: 16);

@ -8,7 +8,11 @@
/// The formulas are those of kilic/bls12-381 v0.1.0, which drand/kyber /// The formulas are those of kilic/bls12-381 v0.1.0, which drand/kyber
/// pairs with: an element of GT is the value its pairing computes, and H2 of /// pairs with: an element of GT is the value its pairing computes, and H2 of
/// the tlock IBE hashes it in the order of [Fp12.toBytes], c1 before c0 at /// the tlock IBE hashes it in the order of [Fp12.toBytes], c1 before c0 at
/// every level. Everything is written over the operations of the field layer /// every level. The product and the square of Fp6, its product by a sparse
/// element and the square in Fp4 add their products unreduced and reduce
/// each coefficient once, where kilic reduces every product: the values are
/// the same, with a third of the reductions, the costliest operation of the
/// field layer. Everything is written over the operations of the field layer
/// (bls12381_fp.dart), and like it, is not constant time. /// (bls12381_fp.dart), and like it, is not constant time.
library; library;
@ -167,46 +171,37 @@ final class Fp6 {
/// The negation. /// The negation.
Fp6 operator -() => Fp6(-c0, -c1, -c2); Fp6 operator -() => Fp6(-c0, -c1, -c2);
/// The product, Algorithm 5.21 of the Guide to Pairing-Based /// The product: c0 = a0b0 + ξ(a1b2 + a2b1), c1 = a0b1 + a1b0 + ξa2b2,
/// Cryptography, as kilic. /// c2 = a0b2 + a1b1 + a2b0, each coefficient reduced once.
Fp6 operator *(Fp6 b) { Fp6 operator *(Fp6 b) {
final v0 = c0 * b.c0; _Fp2Wide m(Fp2 x, Fp2 y) => _Fp2Wide.mul(x, y);
final v1 = c1 * b.c1;
final v2 = c2 * b.c2;
return Fp6( return Fp6(
((c1 + c2) * (b.c1 + b.c2) - v1 - v2).mulByNonResidue() + v0, (m(c0, b.c0) + (m(c1, b.c2) + m(c2, b.c1)).mulByNonResidue()).reduce(),
(c0 + c1) * (b.c0 + b.c1) - v0 - v1 + v2.mulByNonResidue(), (m(c0, b.c1) + m(c1, b.c0) + m(c2, b.c2).mulByNonResidue()).reduce(),
(c0 + c2) * (b.c0 + b.c2) - v0 - v2 + v1, (m(c0, b.c2) + m(c1, b.c1) + m(c2, b.c0)).reduce(),
); );
} }
/// The square, CH-SQR2 of eprint 2006/471, as kilic. /// The square: c0 = a0² + 2ξa1a2, c1 = 2a0a1 + ξa2², c2 = a1² + 2a0a2,
Fp6 square() { /// each coefficient reduced once.
final s0 = c0.square(); Fp6 square() => Fp6(
final s1 = (c0 * c1).double(); (_Fp2Wide.square(c0) + _Fp2Wide.mul(c1, c2).double().mulByNonResidue())
final s2 = (c0 - c1 + c2).square(); .reduce(),
final s3 = (c1 * c2).double(); (_Fp2Wide.mul(c0, c1).double() + _Fp2Wide.square(c2).mulByNonResidue())
final s4 = c2.square(); .reduce(),
return Fp6( (_Fp2Wide.square(c1) + _Fp2Wide.mul(c0, c2).double()).reduce(),
s0 + s3.mulByNonResidue(), );
s1 + s4.mulByNonResidue(),
s1 + s2 + s3 - s0 - s4,
);
}
/// The product by v, the non-residue of Fp12: (ξ·c2, c0, c1). /// The product by v, the non-residue of Fp12: (ξ·c2, c0, c1).
Fp6 mulByNonResidue() => Fp6(c2.mulByNonResidue(), c0, c1); Fp6 mulByNonResidue() => Fp6(c2.mulByNonResidue(), c0, c1);
/// The product by b0 + b1·v, as mul01 of kilic. /// The product by b0 + b1·v, mul01 of kilic: c0 = a0b0 + ξa2b1, c1 =
Fp6 mul01(Fp2 b0, Fp2 b1) { /// a0b1 + a1b0, c2 = a1b1 + a2b0, each coefficient reduced once.
final v0 = c0 * b0; Fp6 mul01(Fp2 b0, Fp2 b1) => Fp6(
final v1 = c1 * b1; (_Fp2Wide.mul(c0, b0) + _Fp2Wide.mul(c2, b1).mulByNonResidue()).reduce(),
return Fp6( (_Fp2Wide.mul(c0, b1) + _Fp2Wide.mul(c1, b0)).reduce(),
((c1 + c2) * b1 - v1).mulByNonResidue() + v0, (_Fp2Wide.mul(c1, b1) + _Fp2Wide.mul(c2, b0)).reduce(),
(c0 + c1) * (b0 + b1) - v0 - v1, );
(c0 + c2) * b0 - v0 + v1,
);
}
/// The product by b1·v, as mul1 of kilic. /// The product by b1·v, as mul1 of kilic.
Fp6 mul1(Fp2 b1) => Fp6((c2 * b1).mulByNonResidue(), c0 * b1, c1 * b1); Fp6 mul1(Fp2 b1) => Fp6((c2 * b1).mulByNonResidue(), c0 * b1, c1 * b1);
@ -315,11 +310,41 @@ final class Fp12 {
Uint8List toBytes() => concatBytes([c1.toBytes(), c0.toBytes()]); Uint8List toBytes() => concatBytes([c1.toBytes(), c0.toBytes()]);
} }
// The square of a0 + a1·t in Fp4 = Fp2[t]/(t² − ξ), as fp4Square of kilic. // The square of a0 + a1·t in Fp4 = Fp2[t]/(t² − ξ), fp4Square of kilic:
(Fp2, Fp2) _fp4Square(Fp2 a0, Fp2 a1) { // (a0² + ξa1², 2a0a1), each coefficient reduced once.
final t0 = a0.square(); (Fp2, Fp2) _fp4Square(Fp2 a0, Fp2 a1) => (
final t1 = a1.square(); (_Fp2Wide.square(a1).mulByNonResidue() + _Fp2Wide.square(a0)).reduce(),
return (t1.mulByNonResidue() + t0, (a0 + a1).square() - t0 - t1); _Fp2Wide.mul(a0, a1).double().reduce(),
);
// An element of Fp2 whose coefficients are sums of products not reduced yet
// (FpWide): Fp6 and Fp12 add their products in this form and reduce each
// coefficient once, which takes a third of the reductions of the formulas
// of kilic over reduced products. The values are the same.
final class _Fp2Wide {
const _Fp2Wide(this.c0, this.c1);
// a·b = (a0b0 − a1b1) + (a0b1 + a1b0)·u.
_Fp2Wide.mul(Fp2 a, Fp2 b)
: c0 = FpWide.mulSub(a.c0, b.c0, a.c1, b.c1),
c1 = FpWide.mulAdd(a.c0, b.c1, a.c1, b.c0);
// a² = (a0² − a1²) + 2a0a1·u.
_Fp2Wide.square(Fp2 a)
: c0 = FpWide.mulSub(a.c0, a.c0, a.c1, a.c1),
c1 = FpWide.mul(a.c0, a.c1).double();
final FpWide c0;
final FpWide c1;
_Fp2Wide operator +(_Fp2Wide o) => _Fp2Wide(c0 + o.c0, c1 + o.c1);
_Fp2Wide double() => _Fp2Wide(c0.double(), c1.double());
// The product by ξ = u + 1.
_Fp2Wide mulByNonResidue() => _Fp2Wide(c0 - c1, c0 + c1);
Fp2 reduce() => Fp2(c0.reduce(), c1.reduce());
} }
/// γ_k = ξ^((p^k − 1)/6) for k = 1, 2, 3, the coefficients of the Frobenius /// γ_k = ξ^((p^k − 1)/6) for k = 1, 2, 3, the coefficients of the Frobenius

Loading…
Cancel
Save

Powered by TurnKey Linux.