Words read as the Go CLI reads them

normalizeWords lowers each code point on its own, as Go's simple
mapping does (no final sigma), and splits at the white space of Go's
unicode.IsSpace, so that the page and wordkey.Normalize of datekeys-go
give the same words; both test the same PBKDF2 vector.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
main
dev 7 days ago
parent b617007054
commit 67d626ee9a

@ -8,10 +8,14 @@ import { normalizeWords, WORD_KEY_ROUNDS, wordKey } from './wordkey.ts';
const CHAIN = '52db9ba70e0cc0f6eaf7803dd07447a1f5477735fd3f661792ba94600c84e971';
describe('normalizeWords', () => {
it('ignores case, accents and extra spaces', () => {
const accented = ` ${String.fromCharCode(0xc1)}baco ${String.fromCharCode(0xc1)}RBOL\tni${String.fromCharCode(0xf1)}o `;
expect(normalizeWords(accented)).toEqual(['abaco', 'arbol', 'nino']);
it('ignores case, accents and extra spaces, as Go wordkey.Normalize does', () => {
const nbsp = String.fromCharCode(0xa0);
const tab = String.fromCharCode(9);
// The case of TestNormalize of datekeys-go: Σ in lower case on its own,
// not as a final sigma, and İ without its dot.
expect(normalizeWords(` Ábaco${nbsp}ÁRBOL${tab}niño ΣΑΣ İ `)).toEqual(['abaco', 'arbol', 'nino', 'σασ', 'i']);
expect(normalizeWords(' ')).toEqual([]);
expect(normalizeWords(`a${String.fromCharCode(0xfeff)}b`)).toEqual([`a${String.fromCharCode(0xfeff)}b`]);
});
});
@ -21,6 +25,8 @@ describe('wordKey', () => {
const key = await wordKey(words, CHAIN, 1000);
const want = pbkdf2Sync('perro luna casa verde tren mar', `DateKeys llave de palabras v1|${CHAIN}|1000`, WORD_KEY_ROUNDS, 32, 'sha256');
expect(Buffer.from(key).toString('hex')).toBe(want.toString('hex'));
// The vector of TestKeyMatchesTypeScript of datekeys-go.
expect(Buffer.from(key).toString('hex')).toBe('be74aecd9ea734bfece963597a2269188a4ea8a1247bc381dffbd6a38a2c6376');
expect(Buffer.from(await wordKey(words, CHAIN, 1001)).equals(Buffer.from(key))).toBe(false);
});
});

@ -13,17 +13,29 @@ export const MIN_WORDS = 6;
/** The rounds of PBKDF2-SHA256, OWASP's figure for 2023. */
export const WORD_KEY_ROUNDS = 600_000;
// The combining marks that NFD splits from accented letters.
const MARKS = new RegExp(`[${String.fromCharCode(0x300)}-${String.fromCharCode(0x36f)}]`, 'g');
// The white space at which the words are split: Go's unicode.IsSpace, so
// that the CLI of the reference reads the same words.
const SPACES = new Set([0x09, 0x0a, 0x0b, 0x0c, 0x0d, 0x20, 0x85, 0xa0, 0x1680, 0x2028, 0x2029, 0x202f, 0x205f, 0x3000]);
for (let r = 0x2000; r <= 0x200a; r++) SPACES.add(r);
/** The words of a text: in lower case, without accents, split by spaces. */
/**
* The words of a text, as the CLI of the reference reads them (Go's
* wordkey.Normalize): its NFD without the combining marks U+0300 to U+036F,
* each code point in lower case on its own, split at white space.
*/
export function normalizeWords(text: string): string[] {
return text
.normalize('NFD')
.replace(MARKS, '')
.toLowerCase()
.split(/\s+/)
.filter((w) => w !== '');
const words: string[] = [];
let word = '';
for (const ch of text.normalize('NFD')) {
const r = ch.codePointAt(0)!;
if (r >= 0x300 && r <= 0x36f) continue;
if (SPACES.has(r)) {
if (word !== '') words.push(word);
word = '';
} else word += ch.toLowerCase();
}
if (word !== '') words.push(word);
return words;
}
/**

Loading…
Cancel
Save

Powered by TurnKey Linux.