SHA-256 rounds unrolled by eight, without masks on the rotations

The working variables rotate by name instead of by assignment, and the
left halves of the rotations are no longer masked: on the VM their bits
above 32 only reach sums that are masked before any other use, and on the
web `<<` keeps 32 bits itself. PBKDF2 at 600 000 iterations goes from 1.27
to 1.11 s on the VM; HMAC of package:crypto takes about 3.5 s for the same
work.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
v0.11
dev 2 days ago
parent 7a19b3fa5d
commit a2e2bbfcd6

@ -51,18 +51,18 @@ const _iv = [
/// Compresses the 16 words of [w] (big-endian words of one block, w[0..15];
/// w[16..63] are scratch) into the eight words of [state].
///
/// The left halves of the rotations are not masked: on the VM their bits
/// above 32 only reach sums that are masked before any other use, and an
/// addition carries upward only; on the web, `<<` keeps 32 bits itself. The
/// rounds are unrolled by eight, so that the working variables rotate by
/// name instead of by assignment.
void _compress(Uint32List state, Uint32List w) {
for (var t = 16; t < 64; t++) {
final x = w[t - 15];
final y = w[t - 2];
final s0 =
((x >>> 7) | (x << 25) & _mask32) ^
((x >>> 18) | (x << 14) & _mask32) ^
(x >>> 3);
final s1 =
((y >>> 17) | (y << 15) & _mask32) ^
((y >>> 19) | (y << 13) & _mask32) ^
(y >>> 10);
final s0 = (x >>> 7 | x << 25) ^ (x >>> 18 | x << 14) ^ x >>> 3;
final s1 = (y >>> 17 | y << 15) ^ (y >>> 19 | y << 13) ^ y >>> 10;
w[t] = (w[t - 16] + s0 + w[t - 7] + s1) & _mask32;
}
var a = state[0];
@ -73,28 +73,135 @@ void _compress(Uint32List state, Uint32List w) {
var f = state[5];
var g = state[6];
var h = state[7];
for (var t = 0; t < 64; t++) {
final s1 =
((e >>> 6) | (e << 26) & _mask32) ^
((e >>> 11) | (e << 21) & _mask32) ^
((e >>> 25) | (e << 7) & _mask32);
// Ch(e, f, g) = (e & f) ^ (~e & g), with ~e as 32 bits on the VM too.
final ch = (e & f) ^ ((e ^ _mask32) & g);
final t1 = (h + s1 + ch + _k[t] + w[t]) & _mask32;
final s0 =
((a >>> 2) | (a << 30) & _mask32) ^
((a >>> 13) | (a << 19) & _mask32) ^
((a >>> 22) | (a << 10) & _mask32);
final maj = (a & b) ^ (a & c) ^ (b & c);
final t2 = (s0 + maj) & _mask32;
h = g;
g = f;
f = e;
e = (d + t1) & _mask32;
d = c;
c = b;
b = a;
a = (t1 + t2) & _mask32;
for (var t = 0; t < 64; t += 8) {
// Round t + 0.
final t0 =
(h +
((e >>> 6 | e << 26) ^ (e >>> 11 | e << 21) ^ (e >>> 25 | e << 7)) +
((e & f) ^ ((e ^ _mask32) & g)) +
_k[t + 0] +
w[t + 0]) &
_mask32;
d = (d + t0) & _mask32;
h =
(t0 +
((a >>> 2 | a << 30) ^
(a >>> 13 | a << 19) ^
(a >>> 22 | a << 10)) +
((a & b) ^ (a & c) ^ (b & c))) &
_mask32;
// Round t + 1.
final t1 =
(g +
((d >>> 6 | d << 26) ^ (d >>> 11 | d << 21) ^ (d >>> 25 | d << 7)) +
((d & e) ^ ((d ^ _mask32) & f)) +
_k[t + 1] +
w[t + 1]) &
_mask32;
c = (c + t1) & _mask32;
g =
(t1 +
((h >>> 2 | h << 30) ^
(h >>> 13 | h << 19) ^
(h >>> 22 | h << 10)) +
((h & a) ^ (h & b) ^ (a & b))) &
_mask32;
// Round t + 2.
final t2 =
(f +
((c >>> 6 | c << 26) ^ (c >>> 11 | c << 21) ^ (c >>> 25 | c << 7)) +
((c & d) ^ ((c ^ _mask32) & e)) +
_k[t + 2] +
w[t + 2]) &
_mask32;
b = (b + t2) & _mask32;
f =
(t2 +
((g >>> 2 | g << 30) ^
(g >>> 13 | g << 19) ^
(g >>> 22 | g << 10)) +
((g & h) ^ (g & a) ^ (h & a))) &
_mask32;
// Round t + 3.
final t3 =
(e +
((b >>> 6 | b << 26) ^ (b >>> 11 | b << 21) ^ (b >>> 25 | b << 7)) +
((b & c) ^ ((b ^ _mask32) & d)) +
_k[t + 3] +
w[t + 3]) &
_mask32;
a = (a + t3) & _mask32;
e =
(t3 +
((f >>> 2 | f << 30) ^
(f >>> 13 | f << 19) ^
(f >>> 22 | f << 10)) +
((f & g) ^ (f & h) ^ (g & h))) &
_mask32;
// Round t + 4.
final t4 =
(d +
((a >>> 6 | a << 26) ^ (a >>> 11 | a << 21) ^ (a >>> 25 | a << 7)) +
((a & b) ^ ((a ^ _mask32) & c)) +
_k[t + 4] +
w[t + 4]) &
_mask32;
h = (h + t4) & _mask32;
d =
(t4 +
((e >>> 2 | e << 30) ^
(e >>> 13 | e << 19) ^
(e >>> 22 | e << 10)) +
((e & f) ^ (e & g) ^ (f & g))) &
_mask32;
// Round t + 5.
final t5 =
(c +
((h >>> 6 | h << 26) ^ (h >>> 11 | h << 21) ^ (h >>> 25 | h << 7)) +
((h & a) ^ ((h ^ _mask32) & b)) +
_k[t + 5] +
w[t + 5]) &
_mask32;
g = (g + t5) & _mask32;
c =
(t5 +
((d >>> 2 | d << 30) ^
(d >>> 13 | d << 19) ^
(d >>> 22 | d << 10)) +
((d & e) ^ (d & f) ^ (e & f))) &
_mask32;
// Round t + 6.
final t6 =
(b +
((g >>> 6 | g << 26) ^ (g >>> 11 | g << 21) ^ (g >>> 25 | g << 7)) +
((g & h) ^ ((g ^ _mask32) & a)) +
_k[t + 6] +
w[t + 6]) &
_mask32;
f = (f + t6) & _mask32;
b =
(t6 +
((c >>> 2 | c << 30) ^
(c >>> 13 | c << 19) ^
(c >>> 22 | c << 10)) +
((c & d) ^ (c & e) ^ (d & e))) &
_mask32;
// Round t + 7.
final t7 =
(a +
((f >>> 6 | f << 26) ^ (f >>> 11 | f << 21) ^ (f >>> 25 | f << 7)) +
((f & g) ^ ((f ^ _mask32) & h)) +
_k[t + 7] +
w[t + 7]) &
_mask32;
e = (e + t7) & _mask32;
a =
(t7 +
((b >>> 2 | b << 30) ^
(b >>> 13 | b << 19) ^
(b >>> 22 | b << 10)) +
((b & c) ^ (b & d) ^ (c & d))) &
_mask32;
}
state[0] = (state[0] + a) & _mask32;
state[1] = (state[1] + b) & _mask32;

Loading…
Cancel
Save

Powered by TurnKey Linux.