diff --git a/src/lib/inspector/wordkey.test.ts b/src/lib/inspector/wordkey.test.ts index 48f1bb2..ad80f00 100644 --- a/src/lib/inspector/wordkey.test.ts +++ b/src/lib/inspector/wordkey.test.ts @@ -8,10 +8,14 @@ import { normalizeWords, WORD_KEY_ROUNDS, wordKey } from './wordkey.ts'; const CHAIN = '52db9ba70e0cc0f6eaf7803dd07447a1f5477735fd3f661792ba94600c84e971'; describe('normalizeWords', () => { - it('ignores case, accents and extra spaces', () => { - const accented = ` ${String.fromCharCode(0xc1)}baco ${String.fromCharCode(0xc1)}RBOL\tni${String.fromCharCode(0xf1)}o `; - expect(normalizeWords(accented)).toEqual(['abaco', 'arbol', 'nino']); + it('ignores case, accents and extra spaces, as Go wordkey.Normalize does', () => { + const nbsp = String.fromCharCode(0xa0); + const tab = String.fromCharCode(9); + // The case of TestNormalize of datekeys-go: Σ in lower case on its own, + // not as a final sigma, and İ without its dot. + expect(normalizeWords(` Ábaco${nbsp}ÁRBOL${tab}niño ΣΑΣ İ `)).toEqual(['abaco', 'arbol', 'nino', 'σασ', 'i']); expect(normalizeWords(' ')).toEqual([]); + expect(normalizeWords(`a${String.fromCharCode(0xfeff)}b`)).toEqual([`a${String.fromCharCode(0xfeff)}b`]); }); }); @@ -21,6 +25,8 @@ describe('wordKey', () => { const key = await wordKey(words, CHAIN, 1000); const want = pbkdf2Sync('perro luna casa verde tren mar', `DateKeys llave de palabras v1|${CHAIN}|1000`, WORD_KEY_ROUNDS, 32, 'sha256'); expect(Buffer.from(key).toString('hex')).toBe(want.toString('hex')); + // The vector of TestKeyMatchesTypeScript of datekeys-go. + expect(Buffer.from(key).toString('hex')).toBe('be74aecd9ea734bfece963597a2269188a4ea8a1247bc381dffbd6a38a2c6376'); expect(Buffer.from(await wordKey(words, CHAIN, 1001)).equals(Buffer.from(key))).toBe(false); }); }); diff --git a/src/lib/inspector/wordkey.ts b/src/lib/inspector/wordkey.ts index e7a62b8..ea8ae4a 100644 --- a/src/lib/inspector/wordkey.ts +++ b/src/lib/inspector/wordkey.ts @@ -13,17 +13,29 @@ export const MIN_WORDS = 6; /** The rounds of PBKDF2-SHA256, OWASP's figure for 2023. */ export const WORD_KEY_ROUNDS = 600_000; -// The combining marks that NFD splits from accented letters. -const MARKS = new RegExp(`[${String.fromCharCode(0x300)}-${String.fromCharCode(0x36f)}]`, 'g'); +// The white space at which the words are split: Go's unicode.IsSpace, so +// that the CLI of the reference reads the same words. +const SPACES = new Set([0x09, 0x0a, 0x0b, 0x0c, 0x0d, 0x20, 0x85, 0xa0, 0x1680, 0x2028, 0x2029, 0x202f, 0x205f, 0x3000]); +for (let r = 0x2000; r <= 0x200a; r++) SPACES.add(r); -/** The words of a text: in lower case, without accents, split by spaces. */ +/** + * The words of a text, as the CLI of the reference reads them (Go's + * wordkey.Normalize): its NFD without the combining marks U+0300 to U+036F, + * each code point in lower case on its own, split at white space. + */ export function normalizeWords(text: string): string[] { - return text - .normalize('NFD') - .replace(MARKS, '') - .toLowerCase() - .split(/\s+/) - .filter((w) => w !== ''); + const words: string[] = []; + let word = ''; + for (const ch of text.normalize('NFD')) { + const r = ch.codePointAt(0)!; + if (r >= 0x300 && r <= 0x36f) continue; + if (SPACES.has(r)) { + if (word !== '') words.push(word); + word = ''; + } else word += ch.toLowerCase(); + } + if (word !== '') words.push(word); + return words; } /**