Skip to content
Merged
Show file tree
Hide file tree
Changes from 6 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 6 additions & 6 deletions internal/stats/latest_stats.csv
Original file line number Diff line number Diff line change
Expand Up @@ -223,42 +223,42 @@ pairing_bw6761,bls24_315,plonk,0,0
pairing_bw6761,bls24_317,plonk,0,0
pairing_bw6761,bw6_761,plonk,0,0
pairing_bw6761,bw6_633,plonk,0,0
scalar_mul_G1_bn254,bn254,groth16,59255,91375
scalar_mul_G1_bn254,bn254,groth16,55279,87825
scalar_mul_G1_bn254,bls12_377,groth16,0,0
scalar_mul_G1_bn254,bls12_381,groth16,0,0
scalar_mul_G1_bn254,bls24_315,groth16,0,0
scalar_mul_G1_bn254,bls24_317,groth16,0,0
scalar_mul_G1_bn254,bw6_761,groth16,0,0
scalar_mul_G1_bn254,bw6_633,groth16,0,0
scalar_mul_G1_bn254,bn254,plonk,209299,202001
scalar_mul_G1_bn254,bn254,plonk,201599,194727
scalar_mul_G1_bn254,bls12_377,plonk,0,0
scalar_mul_G1_bn254,bls12_381,plonk,0,0
scalar_mul_G1_bn254,bls24_315,plonk,0,0
scalar_mul_G1_bn254,bls24_317,plonk,0,0
scalar_mul_G1_bn254,bw6_761,plonk,0,0
scalar_mul_G1_bn254,bw6_633,plonk,0,0
scalar_mul_P256,bn254,groth16,78854,124732
scalar_mul_P256,bn254,groth16,75326,121582
scalar_mul_P256,bls12_377,groth16,0,0
scalar_mul_P256,bls12_381,groth16,0,0
scalar_mul_P256,bls24_315,groth16,0,0
scalar_mul_P256,bls24_317,groth16,0,0
scalar_mul_P256,bw6_761,groth16,0,0
scalar_mul_P256,bw6_633,groth16,0,0
scalar_mul_P256,bn254,plonk,277196,267528
scalar_mul_P256,bn254,plonk,270364,261074
scalar_mul_P256,bls12_377,plonk,0,0
scalar_mul_P256,bls12_381,plonk,0,0
scalar_mul_P256,bls24_315,plonk,0,0
scalar_mul_P256,bls24_317,plonk,0,0
scalar_mul_P256,bw6_761,plonk,0,0
scalar_mul_P256,bw6_633,plonk,0,0
scalar_mul_secp256k1,bn254,groth16,59993,92505
scalar_mul_secp256k1,bn254,groth16,55961,88905
scalar_mul_secp256k1,bls12_377,groth16,0,0
scalar_mul_secp256k1,bls12_381,groth16,0,0
scalar_mul_secp256k1,bls24_315,groth16,0,0
scalar_mul_secp256k1,bls24_317,groth16,0,0
scalar_mul_secp256k1,bw6_761,groth16,0,0
scalar_mul_secp256k1,bw6_633,groth16,0,0
scalar_mul_secp256k1,bn254,plonk,211916,204521
scalar_mul_secp256k1,bn254,plonk,204108,197145
scalar_mul_secp256k1,bls12_377,plonk,0,0
scalar_mul_secp256k1,bls12_381,plonk,0,0
scalar_mul_secp256k1,bls24_315,plonk,0,0
Expand Down
44 changes: 20 additions & 24 deletions std/algebra/emulated/sw_bls12381/g2.go
Original file line number Diff line number Diff line change
Expand Up @@ -270,6 +270,20 @@ func (g2 G2) neg(p *G2Affine) *G2Affine {
}
}

// muxE2Y8Signed selects from 8 E2 Y-values using selector (0-7) and conditionally
// negates based on signBit. This optimizes the common GLV pattern where Y[i] =
// -Y[15-i], reducing a 16-to-1 Mux to an 8-to-1 Mux plus conditional negation.
func (g2 *G2) muxE2Y8Signed(signBit frontend.Variable, selector frontend.Variable, yA0, yA1 [8]*emulated.Element[BaseField]) *fields_bls12381.E2 {
baseA0 := g2.fp.Mux(selector, yA0[0], yA0[1], yA0[2], yA0[3], yA0[4], yA0[5], yA0[6], yA0[7])
Comment thread
yelhousni marked this conversation as resolved.
Outdated
baseA1 := g2.fp.Mux(selector, yA1[0], yA1[1], yA1[2], yA1[3], yA1[4], yA1[5], yA1[6], yA1[7])
negA0 := g2.fp.Neg(baseA0)
negA1 := g2.fp.Neg(baseA1)
return &fields_bls12381.E2{
A0: *g2.fp.Select(signBit, negA0, baseA0),
A1: *g2.fp.Select(signBit, negA1, baseA1),
}
}

func (g2 G2) sub(p, q *G2Affine) *G2Affine {
qNeg := g2.neg(q)
return g2.add(p, qNeg)
Expand Down Expand Up @@ -692,18 +706,12 @@ func (g2 *G2) scalarMulGLV(Q *G2Affine, s *Scalar, opts ...algopts.AlgebraOption
// T = Q + [3]Φ(Q)
// P = B1 and P' = B4
T4 := g2.add(tableQ[1], tablePhiQ[2])
Comment thread
yelhousni marked this conversation as resolved.
Outdated
// T = -Q - Φ(Q)
// P = B2 and P' = B1
T5 := g2.neg(T2)
// T = -[3](Q + Φ(Q))
// P = B2 and P' = B2
T6 := g2.neg(T1)
// T = -Q - [3]Φ(Q)
// P = B2 and P' = B3
T7 := g2.neg(T4)
// T = -[3]Q - Φ(Q)
// P = B2 and P' = B4
T8 := g2.neg(T3)
// T = [3]Q - Φ(Q)
// P = B3 and P' = B1
T9 := g2.add(tableQ[2], tablePhiQ[0])
Expand All @@ -717,18 +725,12 @@ func (g2 *G2) scalarMulGLV(Q *G2Affine, s *Scalar, opts ...algopts.AlgebraOption
// T = -Φ(Q) + Q
// P = B3 and P' = B4
T12 := g2.add(tablePhiQ[0], tableQ[1])
// T = [3]Φ(Q) - Q
// P = B4 and P' = B1
T13 := g2.neg(T10)
// T = Φ(Q) - [3]Q
// P = B4 and P' = B2
T14 := g2.neg(T9)
// T = Φ(Q) - Q
// P = B4 and P' = B3
T15 := g2.neg(T12)
// T = [3](Φ(Q) - Q)
// P = B4 and P' = B4
T16 := g2.neg(T11)
// note that half the points are negatives of the other half,
// hence have the same X coordinates.

Expand All @@ -748,24 +750,18 @@ func (g2 *G2) scalarMulGLV(Q *G2Affine, s *Scalar, opts ...algopts.AlgebraOption
g2.api.Mul(selectorY, g2.api.Sub(1, g2.api.Mul(s2bits[i-1], 2))),
g2.api.Mul(s2bits[i-1], 15),
)
// Bi.Y are distincts so we need a 16-to-1 multiplexer,
// but only half of the Bi.X are distinct so we need a 8-to-1.
// Half of the Bi.X are distinct (8-to-1) and Y[i] = -Y[15-i],
// so we use 8-to-1 Mux for both X and Y, with conditional negation for Y.
T := &G2Affine{
P: g2AffP{
X: fields_bls12381.E2{
A0: *g2.fp.Mux(selectorX, &T6.P.X.A0, &T10.P.X.A0, &T14.P.X.A0, &T2.P.X.A0, &T7.P.X.A0, &T11.P.X.A0, &T15.P.X.A0, &T3.P.X.A0),
A1: *g2.fp.Mux(selectorX, &T6.P.X.A1, &T10.P.X.A1, &T14.P.X.A1, &T2.P.X.A1, &T7.P.X.A1, &T11.P.X.A1, &T15.P.X.A1, &T3.P.X.A1),
},
Y: fields_bls12381.E2{
A0: *g2.fp.Mux(selectorY,
&T6.P.Y.A0, &T10.P.Y.A0, &T14.P.Y.A0, &T2.P.Y.A0, &T7.P.Y.A0, &T11.P.Y.A0, &T15.P.Y.A0, &T3.P.Y.A0,
&T8.P.Y.A0, &T12.P.Y.A0, &T16.P.Y.A0, &T4.P.Y.A0, &T5.P.Y.A0, &T9.P.Y.A0, &T13.P.Y.A0, &T1.P.Y.A0,
),
A1: *g2.fp.Mux(selectorY,
&T6.P.Y.A1, &T10.P.Y.A1, &T14.P.Y.A1, &T2.P.Y.A1, &T7.P.Y.A1, &T11.P.Y.A1, &T15.P.Y.A1, &T3.P.Y.A1,
&T8.P.Y.A1, &T12.P.Y.A1, &T16.P.Y.A1, &T4.P.Y.A1, &T5.P.Y.A1, &T9.P.Y.A1, &T13.P.Y.A1, &T1.P.Y.A1,
),
},
Y: *g2.muxE2Y8Signed(s2bits[i-1], selectorX,
[8]*emulated.Element[BaseField]{&T6.P.Y.A0, &T10.P.Y.A0, &T14.P.Y.A0, &T2.P.Y.A0, &T7.P.Y.A0, &T11.P.Y.A0, &T15.P.Y.A0, &T3.P.Y.A0},
[8]*emulated.Element[BaseField]{&T6.P.Y.A1, &T10.P.Y.A1, &T14.P.Y.A1, &T2.P.Y.A1, &T7.P.Y.A1, &T11.P.Y.A1, &T15.P.Y.A1, &T3.P.Y.A1},
),
},
}
// Acc = [4]Acc + T
Expand Down
79 changes: 26 additions & 53 deletions std/algebra/emulated/sw_emulated/point.go
Original file line number Diff line number Diff line change
Expand Up @@ -541,6 +541,17 @@ func (c *Curve[B, S]) Mux(sel frontend.Variable, inputs ...*AffinePoint[B]) *Aff
}
}

// muxY8Signed selects from 8 Y values using selector (0-7) and conditionally
// negates based on signBit. This optimizes the common GLV pattern where Y[i] =
// -Y[15-i], reducing a 16-to-1 Mux to an 8-to-1 Mux plus conditional negation.
func (c *Curve[B, S]) muxY8Signed(signBit frontend.Variable, selector frontend.Variable, yValues ...*emulated.Element[B]) *emulated.Element[B] {
if len(yValues) != 8 {
panic("muxY8Signed requires exactly 8 Y values")
}
baseY := c.baseApi.Mux(selector, yValues...)
return c.baseApi.Select(signBit, c.baseApi.Neg(baseY), baseY)
}

// ScalarMul computes [s]p and returns it. It doesn't modify p nor s.
// This function doesn't check that the p is on the curve. See AssertIsOnCurve.
//
Expand Down Expand Up @@ -669,9 +680,6 @@ func (c *Curve[B, S]) scalarMulGLV(Q *AffinePoint[B], s *emulated.Element[S], op
// T = -Q - [3]Φ(Q)
// P = B2 and P' = B3
T7 := c.Neg(T4)
// T = -[3]Q - Φ(Q)
// P = B2 and P' = B4
T8 := c.Neg(T3)
// T = [3]Q - Φ(Q)
// P = B3 and P' = B1
T9 := c.Add(tableQ[2], tablePhiQ[0])
Expand All @@ -685,18 +693,12 @@ func (c *Curve[B, S]) scalarMulGLV(Q *AffinePoint[B], s *emulated.Element[S], op
// T = -Φ(Q) + Q
// P = B3 and P' = B4
T12 := c.Add(tablePhiQ[0], tableQ[1])
// T = [3]Φ(Q) - Q
// P = B4 and P' = B1
T13 := c.Neg(T10)
// T = Φ(Q) - [3]Q
// P = B4 and P' = B2
T14 := c.Neg(T9)
// T = Φ(Q) - Q
// P = B4 and P' = B3
T15 := c.Neg(T12)
// T = [3](Φ(Q) - Q)
// P = B4 and P' = B4
T16 := c.Neg(T11)
// note that half the points are negatives of the other half,
// hence have the same X coordinates.

Expand Down Expand Up @@ -730,15 +732,14 @@ func (c *Curve[B, S]) scalarMulGLV(Q *AffinePoint[B], s *emulated.Element[S], op
c.api.Mul(selectorY, c.api.Sub(1, c.api.Mul(s2bits[i-1], 2))),
c.api.Mul(s2bits[i-1], 15),
)
// Bi.Y are distincts so we need a 16-to-1 multiplexer,
// but only half of the Bi.X are distinct so we need a 8-to-1.
// Half of the Bi.X are distinct (8-to-1) and Y[i] = -Y[15-i],
// so we use 8-to-1 Mux for both X and Y, with conditional negation for Y.
T := &AffinePoint[B]{
X: *c.baseApi.Mux(selectorX,
&T6.X, &T10.X, &T14.X, &T2.X, &T7.X, &T11.X, &T15.X, &T3.X,
),
Y: *c.baseApi.Mux(selectorY,
Y: *c.muxY8Signed(s2bits[i-1], selectorX,
&T6.Y, &T10.Y, &T14.Y, &T2.Y, &T7.Y, &T11.Y, &T15.Y, &T3.Y,
&T8.Y, &T12.Y, &T16.Y, &T4.Y, &T5.Y, &T9.Y, &T13.Y, &T1.Y,
),
}
// Acc = [4]Acc + T
Expand Down Expand Up @@ -1058,20 +1059,12 @@ func (c *Curve[B, S]) jointScalarMulGLVUnsafe(Q, R *AffinePoint[B], s, t *emulat
B7 := c.Add(tableS[2], tablePhiS[3])
// B8 = +Q - R - Φ(Q) - Φ(R)
B8 := c.Add(tableS[2], tablePhiS[0])
// B9 = -Q + R + Φ(Q) + Φ(R)
B9 := c.Neg(B8)
// B10 = -Q + R + Φ(Q) - Φ(R)
B10 := c.Neg(B7)
// B11 = -Q + R - Φ(Q) + Φ(R)
B11 := c.Neg(B6)
// B12 = -Q + R - Φ(Q) - Φ(R)
B12 := c.Neg(B5)
// B13 = -Q - R + Φ(Q) + Φ(R)
B13 := c.Neg(B4)
// B14 = -Q - R + Φ(Q) - Φ(R)
B14 := c.Neg(B3)
// B15 = -Q - R - Φ(Q) + Φ(R)
B15 := c.Neg(B2)
// B16 = -Q - R - Φ(Q) - Φ(R)
B16 := c.Neg(B1)
// note that half the points are negatives of the other half,
Expand All @@ -1093,15 +1086,14 @@ func (c *Curve[B, S]) jointScalarMulGLVUnsafe(Q, R *AffinePoint[B], s, t *emulat
c.api.Mul(selectorY, c.api.Sub(1, c.api.Mul(t2bits[i], 2))),
c.api.Mul(t2bits[i], 15),
)
// Bi.Y are distincts so we need a 16-to-1 multiplexer,
// but only half of the Bi.X are distinct so we need a 8-to-1.
// Half of the Bi.X are distinct (8-to-1) and Y[i] = -Y[15-i],
// so we use 8-to-1 Mux for both X and Y, with conditional negation for Y.
Bi = &AffinePoint[B]{
X: *c.baseApi.Mux(selectorX,
&B16.X, &B8.X, &B14.X, &B6.X, &B12.X, &B4.X, &B10.X, &B2.X,
),
Y: *c.baseApi.Mux(selectorY,
Y: *c.muxY8Signed(t2bits[i], selectorX,
&B16.Y, &B8.Y, &B14.Y, &B6.Y, &B12.Y, &B4.Y, &B10.Y, &B2.Y,
&B15.Y, &B7.Y, &B13.Y, &B5.Y, &B11.Y, &B3.Y, &B9.Y, &B1.Y,
),
}
// Acc = [2]Acc + Bi
Expand Down Expand Up @@ -1360,8 +1352,6 @@ func (c *Curve[B, S]) scalarMulFakeGLV(Q *AffinePoint[B], s *emulated.Element[S]
// P = B2 and P' = B3
T7 := c.Neg(T4)
// T = -[3]Q - R
Comment thread
yelhousni marked this conversation as resolved.
// P = B2 and P' = B4
Comment thread
yelhousni marked this conversation as resolved.
T8 := c.Neg(T3)
// T = [3]Q - R
// P = B3 and P' = B1
T9 := addFn(tableQ[2], tableR[0])
Expand All @@ -1375,18 +1365,12 @@ func (c *Curve[B, S]) scalarMulFakeGLV(Q *AffinePoint[B], s *emulated.Element[S]
// T = -R + Q
// P = B3 and P' = B4
T12 := addFn(tableR[0], tableQ[1])
// T = [3]R - Q
// P = B4 and P' = B1
T13 := c.Neg(T10)
// T = R - [3]Q
// P = B4 and P' = B2
T14 := c.Neg(T9)
// T = R - Q
// P = B4 and P' = B3
T15 := c.Neg(T12)
// T = [3](R - Q)
// P = B4 and P' = B4
T16 := c.Neg(T11)
// note that half of these points are negatives of the other half,
// hence have the same X coordinates.

Expand Down Expand Up @@ -1420,15 +1404,14 @@ func (c *Curve[B, S]) scalarMulFakeGLV(Q *AffinePoint[B], s *emulated.Element[S]
c.api.Mul(selectorY, c.api.Sub(1, c.api.Mul(s2bits[i-1], 2))),
c.api.Mul(s2bits[i-1], 15),
)
// Bi.Y are distincts so we need a 16-to-1 multiplexer,
// but only half of the Bi.X are distinct so we need a 8-to-1.
// Half of the Bi.X are distinct (8-to-1) and Y[i] = -Y[15-i],
// so we use 8-to-1 Mux for both X and Y, with conditional negation for Y.
T := &AffinePoint[B]{
X: *c.baseApi.Mux(selectorX,
&T6.X, &T10.X, &T14.X, &T2.X, &T7.X, &T11.X, &T15.X, &T3.X,
),
Y: *c.baseApi.Mux(selectorY,
Y: *c.muxY8Signed(s2bits[i-1], selectorX,
&T6.Y, &T10.Y, &T14.Y, &T2.Y, &T7.Y, &T11.Y, &T15.Y, &T3.Y,
&T8.Y, &T12.Y, &T16.Y, &T4.Y, &T5.Y, &T9.Y, &T13.Y, &T1.Y,
),
}
// Acc = [4]Acc + T
Expand All @@ -1453,15 +1436,14 @@ func (c *Curve[B, S]) scalarMulFakeGLV(Q *AffinePoint[B], s *emulated.Element[S]
c.api.Mul(selectorY, c.api.Sub(1, c.api.Mul(s2bits[1], 2))),
c.api.Mul(s2bits[1], 15),
)
// Bi.Y are distincts so we need a 16-to-1 multiplexer,
// but only half of the Bi.X are distinct so we need a 8-to-1.
// Half of the Bi.X are distinct (8-to-1) and Y[i] = -Y[15-i],
// so we use 8-to-1 Mux for both X and Y, with conditional negation for Y.
T := &AffinePoint[B]{
X: *c.baseApi.Mux(selectorX,
&T6.X, &T10.X, &T14.X, &T2.X, &T7.X, &T11.X, &T15.X, &T3.X,
),
Y: *c.baseApi.Mux(selectorY,
Y: *c.muxY8Signed(s2bits[1], selectorX,
&T6.Y, &T10.Y, &T14.Y, &T2.Y, &T7.Y, &T11.Y, &T15.Y, &T3.Y,
&T8.Y, &T12.Y, &T16.Y, &T4.Y, &T5.Y, &T9.Y, &T13.Y, &T1.Y,
),
}
// to avoid incomplete additions we add [3]R to the precomputed T before computing [4]Acc+T
Expand Down Expand Up @@ -1693,20 +1675,12 @@ func (c *Curve[B, S]) scalarMulGLVAndFakeGLV(P *AffinePoint[B], s *emulated.Elem
B7 := c.Add(tableS[2], tablePhiS[3])
// B8 = +P - Q - Φ(P) - Φ(Q)
B8 := c.Add(tableS[2], tablePhiS[0])
// B9 = -P + Q + Φ(P) + Φ(Q)
B9 := c.Neg(B8)
// B10 = -P + Q + Φ(P) - Φ(Q)
B10 := c.Neg(B7)
// B11 = -P + Q - Φ(P) + Φ(Q)
B11 := c.Neg(B6)
// B12 = -P + Q - Φ(P) - Φ(Q)
B12 := c.Neg(B5)
// B13 = -P - Q + Φ(P) + Φ(Q)
B13 := c.Neg(B4)
// B14 = -P - Q + Φ(P) - Φ(Q)
B14 := c.Neg(B3)
// B15 = -P - Q - Φ(P) + Φ(Q)
B15 := c.Neg(B2)
// B16 = -P - Q - Φ(P) - Φ(Q)
B16 := c.Neg(B1)
// note that half the points are negatives of the other half,
Expand All @@ -1728,15 +1702,14 @@ func (c *Curve[B, S]) scalarMulGLVAndFakeGLV(P *AffinePoint[B], s *emulated.Elem
c.api.Mul(selectorY, c.api.Sub(1, c.api.Mul(v2bits[i], 2))),
c.api.Mul(v2bits[i], 15),
)
// Bi.Y are distincts so we need a 16-to-1 multiplexer,
// but only half of the Bi.X are distinct so we need a 8-to-1.
// Half of the Bi.X are distinct (8-to-1) and Y[i] = -Y[15-i],
// so we use 8-to-1 Mux for both X and Y, with conditional negation for Y.
Bi = &AffinePoint[B]{
X: *c.baseApi.Mux(selectorX,
&B16.X, &B8.X, &B14.X, &B6.X, &B12.X, &B4.X, &B10.X, &B2.X,
),
Y: *c.baseApi.Mux(selectorY,
Y: *c.muxY8Signed(v2bits[i], selectorX,
&B16.Y, &B8.Y, &B14.Y, &B6.Y, &B12.Y, &B4.Y, &B10.Y, &B2.Y,
&B15.Y, &B7.Y, &B13.Y, &B5.Y, &B11.Y, &B3.Y, &B9.Y, &B1.Y,
),
}
// Acc = [2]Acc + Bi
Expand Down
27 changes: 25 additions & 2 deletions std/math/emulated/field_ops.go
Original file line number Diff line number Diff line change
Expand Up @@ -3,10 +3,13 @@ package emulated
import (
"errors"
"fmt"
"math/big"
"math/bits"

"github.com/consensys/gnark/frontend"
"github.com/consensys/gnark/profile"
mathbits "github.com/consensys/gnark/std/math/bits"
"github.com/consensys/gnark/std/math/cmp"
"github.com/consensys/gnark/std/selector"
)

Expand Down Expand Up @@ -341,9 +344,29 @@ func (f *Field[T]) Mux(sel frontend.Variable, inputs ...*Element[T]) *Element[T]
normLimbsTransposed[i][j] = normLimbs[j][i]
}
}

e := f.newInternalElement(make([]frontend.Variable, nbLimbs), overflow)
for i := range nbLimbs {
e.Limbs[i] = selector.Mux(f.api, sel, normLimbsTransposed[i]...)

// Optimization: decompose sel into bits once and reuse for all limbs
// instead of decomposing inside each selector.Mux call.
// ToBinary already constrains bits to be boolean, so we use the Unchecked
// variants to avoid redundant boolean assertions.
n := uint(nbInputs)
nbBits := bits.Len(n - 1) // we use n-1 as sel is 0-indexed
selBits := mathbits.ToBinary(f.api, sel, mathbits.WithNbDigits(nbBits))
Comment thread
yelhousni marked this conversation as resolved.

if bits.OnesCount(n) == 1 {
// Power of 2: binary decomposition guarantees sel < 2^nbBits = n
for i := range nbLimbs {
e.Limbs[i] = selector.BinaryMuxUnchecked(f.api, selBits, normLimbsTransposed[i])
}
} else {
// Non-power of 2: need additional bound check sel <= n-1
bcmp := cmp.NewBoundedComparator(f.api, big.NewInt((1<<nbBits)-1), false)
bcmp.AssertIsLessEq(sel, n-1)
Comment thread
yelhousni marked this conversation as resolved.
Outdated
for i := range nbLimbs {
e.Limbs[i] = selector.GeneralMuxUnchecked(f.api, selBits, normLimbsTransposed[i])
}
}

// Record operation for profiling
Expand Down
Loading
Loading