From a2e2bbfcd667ef478f94aeef81d48651e5bff8d9 Mon Sep 17 00:00:00 2001 From: dev Date: Mon, 5 Oct 2026 16:51:17 +0200 Subject: [PATCH] SHA-256 rounds unrolled by eight, without masks on the rotations The working variables rotate by name instead of by assignment, and the left halves of the rotations are no longer masked: on the VM their bits above 32 only reach sums that are masked before any other use, and on the web `<<` keeps 32 bits itself. PBKDF2 at 600 000 iterations goes from 1.27 to 1.11 s on the VM; HMAC of package:crypto takes about 3.5 s for the same work. Co-Authored-By: Claude Opus 5.5 --- lib/src/sha256.dart | 167 ++++++++++++++++++++++++++++++++++++-------- 1 file changed, 137 insertions(+), 30 deletions(-) diff --git a/lib/src/sha256.dart b/lib/src/sha256.dart index 92c304b..7002aba 100644 --- a/lib/src/sha256.dart +++ b/lib/src/sha256.dart @@ -51,18 +51,18 @@ const _iv = [ /// Compresses the 16 words of [w] (big-endian words of one block, w[0..15]; /// w[16..63] are scratch) into the eight words of [state]. +/// +/// The left halves of the rotations are not masked: on the VM their bits +/// above 32 only reach sums that are masked before any other use, and an +/// addition carries upward only; on the web, `<<` keeps 32 bits itself. The +/// rounds are unrolled by eight, so that the working variables rotate by +/// name instead of by assignment. void _compress(Uint32List state, Uint32List w) { for (var t = 16; t < 64; t++) { final x = w[t - 15]; final y = w[t - 2]; - final s0 = - ((x >>> 7) | (x << 25) & _mask32) ^ - ((x >>> 18) | (x << 14) & _mask32) ^ - (x >>> 3); - final s1 = - ((y >>> 17) | (y << 15) & _mask32) ^ - ((y >>> 19) | (y << 13) & _mask32) ^ - (y >>> 10); + final s0 = (x >>> 7 | x << 25) ^ (x >>> 18 | x << 14) ^ x >>> 3; + final s1 = (y >>> 17 | y << 15) ^ (y >>> 19 | y << 13) ^ y >>> 10; w[t] = (w[t - 16] + s0 + w[t - 7] + s1) & _mask32; } var a = state[0]; @@ -73,28 +73,135 @@ void _compress(Uint32List state, Uint32List w) { var f = state[5]; var g = state[6]; var h = state[7]; - for (var t = 0; t < 64; t++) { - final s1 = - ((e >>> 6) | (e << 26) & _mask32) ^ - ((e >>> 11) | (e << 21) & _mask32) ^ - ((e >>> 25) | (e << 7) & _mask32); - // Ch(e, f, g) = (e & f) ^ (~e & g), with ~e as 32 bits on the VM too. - final ch = (e & f) ^ ((e ^ _mask32) & g); - final t1 = (h + s1 + ch + _k[t] + w[t]) & _mask32; - final s0 = - ((a >>> 2) | (a << 30) & _mask32) ^ - ((a >>> 13) | (a << 19) & _mask32) ^ - ((a >>> 22) | (a << 10) & _mask32); - final maj = (a & b) ^ (a & c) ^ (b & c); - final t2 = (s0 + maj) & _mask32; - h = g; - g = f; - f = e; - e = (d + t1) & _mask32; - d = c; - c = b; - b = a; - a = (t1 + t2) & _mask32; + for (var t = 0; t < 64; t += 8) { + // Round t + 0. + final t0 = + (h + + ((e >>> 6 | e << 26) ^ (e >>> 11 | e << 21) ^ (e >>> 25 | e << 7)) + + ((e & f) ^ ((e ^ _mask32) & g)) + + _k[t + 0] + + w[t + 0]) & + _mask32; + d = (d + t0) & _mask32; + h = + (t0 + + ((a >>> 2 | a << 30) ^ + (a >>> 13 | a << 19) ^ + (a >>> 22 | a << 10)) + + ((a & b) ^ (a & c) ^ (b & c))) & + _mask32; + // Round t + 1. + final t1 = + (g + + ((d >>> 6 | d << 26) ^ (d >>> 11 | d << 21) ^ (d >>> 25 | d << 7)) + + ((d & e) ^ ((d ^ _mask32) & f)) + + _k[t + 1] + + w[t + 1]) & + _mask32; + c = (c + t1) & _mask32; + g = + (t1 + + ((h >>> 2 | h << 30) ^ + (h >>> 13 | h << 19) ^ + (h >>> 22 | h << 10)) + + ((h & a) ^ (h & b) ^ (a & b))) & + _mask32; + // Round t + 2. + final t2 = + (f + + ((c >>> 6 | c << 26) ^ (c >>> 11 | c << 21) ^ (c >>> 25 | c << 7)) + + ((c & d) ^ ((c ^ _mask32) & e)) + + _k[t + 2] + + w[t + 2]) & + _mask32; + b = (b + t2) & _mask32; + f = + (t2 + + ((g >>> 2 | g << 30) ^ + (g >>> 13 | g << 19) ^ + (g >>> 22 | g << 10)) + + ((g & h) ^ (g & a) ^ (h & a))) & + _mask32; + // Round t + 3. + final t3 = + (e + + ((b >>> 6 | b << 26) ^ (b >>> 11 | b << 21) ^ (b >>> 25 | b << 7)) + + ((b & c) ^ ((b ^ _mask32) & d)) + + _k[t + 3] + + w[t + 3]) & + _mask32; + a = (a + t3) & _mask32; + e = + (t3 + + ((f >>> 2 | f << 30) ^ + (f >>> 13 | f << 19) ^ + (f >>> 22 | f << 10)) + + ((f & g) ^ (f & h) ^ (g & h))) & + _mask32; + // Round t + 4. + final t4 = + (d + + ((a >>> 6 | a << 26) ^ (a >>> 11 | a << 21) ^ (a >>> 25 | a << 7)) + + ((a & b) ^ ((a ^ _mask32) & c)) + + _k[t + 4] + + w[t + 4]) & + _mask32; + h = (h + t4) & _mask32; + d = + (t4 + + ((e >>> 2 | e << 30) ^ + (e >>> 13 | e << 19) ^ + (e >>> 22 | e << 10)) + + ((e & f) ^ (e & g) ^ (f & g))) & + _mask32; + // Round t + 5. + final t5 = + (c + + ((h >>> 6 | h << 26) ^ (h >>> 11 | h << 21) ^ (h >>> 25 | h << 7)) + + ((h & a) ^ ((h ^ _mask32) & b)) + + _k[t + 5] + + w[t + 5]) & + _mask32; + g = (g + t5) & _mask32; + c = + (t5 + + ((d >>> 2 | d << 30) ^ + (d >>> 13 | d << 19) ^ + (d >>> 22 | d << 10)) + + ((d & e) ^ (d & f) ^ (e & f))) & + _mask32; + // Round t + 6. + final t6 = + (b + + ((g >>> 6 | g << 26) ^ (g >>> 11 | g << 21) ^ (g >>> 25 | g << 7)) + + ((g & h) ^ ((g ^ _mask32) & a)) + + _k[t + 6] + + w[t + 6]) & + _mask32; + f = (f + t6) & _mask32; + b = + (t6 + + ((c >>> 2 | c << 30) ^ + (c >>> 13 | c << 19) ^ + (c >>> 22 | c << 10)) + + ((c & d) ^ (c & e) ^ (d & e))) & + _mask32; + // Round t + 7. + final t7 = + (a + + ((f >>> 6 | f << 26) ^ (f >>> 11 | f << 21) ^ (f >>> 25 | f << 7)) + + ((f & g) ^ ((f ^ _mask32) & h)) + + _k[t + 7] + + w[t + 7]) & + _mask32; + e = (e + t7) & _mask32; + a = + (t7 + + ((b >>> 2 | b << 30) ^ + (b >>> 13 | b << 19) ^ + (b >>> 22 | b << 10)) + + ((b & c) ^ (b & d) ^ (c & d))) & + _mask32; } state[0] = (state[0] + a) & _mask32; state[1] = (state[1] + b) & _mask32;