diff --git a/src/lib/dkc/author.test.ts b/src/lib/dkc/author.test.ts index 02abbf7..08bd303 100644 --- a/src/lib/dkc/author.test.ts +++ b/src/lib/dkc/author.test.ts @@ -77,6 +77,15 @@ describe('what is signed', () => { expect(authorCode(message.subarray(1))).toBe(''); }); + // Go's AuthorCode takes the bytes of the digest as they are: a byte order mark there stays in the code, as + // capsule.AuthorCode gives it at spec-v0.11. + it('takes the code byte by byte, a leading byte order mark included, as Go', () => { + const te = new TextEncoder(); + const message = new Uint8Array([...te.encode('datekeys:dkc3:author-signature:v1\n'), 0xef, 0xbb, 0xbf, ...te.encode(`${'a'.repeat(61)}\n`)]); + expect(message.length).toBe(AUTHOR_MESSAGE_SIZE); + expect(authorCode(message)).toBe('\u{feff}a-aaaa'); + }); + it('binds alg and the list of signers, and leaves L out and I_PAYLOAD hidden', () => { expect(hx(signersDigest(ALG_ED25519))).not.toBe(hx(signersDigest(ALG_CMS))); expect(hx(signersDigest(ALG_CMS, Uint8Array.of(1)))).not.toBe(hx(signersDigest(ALG_CMS, Uint8Array.of(2)))); diff --git a/src/lib/dkc/author.ts b/src/lib/dkc/author.ts index adabe58..e615a8a 100644 --- a/src/lib/dkc/author.ts +++ b/src/lib/dkc/author.ts @@ -85,9 +85,16 @@ export function sealSubject(controlCommitment: Uint8Array, headDigestValue: Uint return domainHash(SEAL_SUBJECT_PREFIX, controlCommitment, headDigestValue, sigPart(signature)); } -/** The code of AUTHOR_MESSAGE that a person compares before signing: the first 8 hexadecimal digits of its digest, in two groups of 4. */ +// The bytes of the code as text, a leading U+FEFF kept, as Go's string(b). +const CODE_TEXT = new TextDecoder('utf-8', { ignoreBOM: true }); + +/** + * The code of AUTHOR_MESSAGE that a person compares before signing: the first + * 8 hexadecimal digits of its digest, in two groups of 4, taken byte by byte + * as Go's AuthorCode takes them. + */ export function authorCode(message: Uint8Array): string { if (message.length !== AUTHOR_MESSAGE_SIZE) return ''; - const d = new TextDecoder().decode(message.subarray(AUTHOR_MESSAGE_PREFIX.length + 1)); - return `${d.slice(0, 4)}-${d.slice(4, 8)}`; + const d = message.subarray(AUTHOR_MESSAGE_PREFIX.length + 1); + return `${CODE_TEXT.decode(d.subarray(0, 4))}-${CODE_TEXT.decode(d.subarray(4, 8))}`; } diff --git a/src/lib/dkc/head.test.ts b/src/lib/dkc/head.test.ts index 28a4cbe..35273f6 100644 --- a/src/lib/dkc/head.test.ts +++ b/src/lib/dkc/head.test.ts @@ -119,6 +119,11 @@ describe('decodeHead', () => { ['a comment with U+202E', head([3, t('a\u202eb')]), 'ERR_HEAD_INVALID', 'comment: text: bidirectional control U+202E'], ['a comment with the tag U+E0041', head([3, t('a\u{e0041}')]), 'ERR_HEAD_INVALID', 'comment: text: invisible U+E0041'], ['an author with LF', head([4, t('a\u000ab')]), 'ERR_HEAD_INVALID', 'declared author: text: control U+000A in the declared author'], + // A text or a path that starts with U+FEFF keeps it, as Go reads the bytes, and its rule refuses it with the text of Go. + ['a comment that starts with U+FEFF', head([3, t('\u{feff}Hola')]), 'ERR_HEAD_INVALID', 'comment: text: byte order mark U+FEFF'], + ['an author that starts with U+FEFF', head([4, t('\u{feff}Ana')]), 'ERR_HEAD_INVALID', 'declared author: text: byte order mark U+FEFF'], + ['R4: a path that starts with U+FEFF', head([5, arr(file('\u{feff}a.txt'))]), 'ERR_HEAD_INVALID', 'file 1: R4: segment 1: invisible U+FEFF'], + ['R4: a segment that starts with U+FEFF', head([5, arr(file('a/\u{feff}b.txt'))]), 'ERR_HEAD_INVALID', 'file 1: R4: segment 2: invisible U+FEFF'], ['R3: ..', head([5, arr(file('..'))]), 'ERR_HEAD_INVALID', 'file 1: R3: segment 1: the segment is two dots'], ['R2: /a', head([5, arr(file('/a'))]), 'ERR_HEAD_INVALID', 'file 1: R2: segment 1 is empty'], ['R4b: a and VS16', head([5, arr(file('a\ufe0f'))]), 'ERR_HEAD_INVALID', 'file 1: R4b: segment 1: U+FE0F is not part of an emoji variation sequence'], diff --git a/src/lib/dkc/note.test.ts b/src/lib/dkc/note.test.ts index 898a622..5218b01 100644 --- a/src/lib/dkc/note.test.ts +++ b/src/lib/dkc/note.test.ts @@ -1,35 +1,60 @@ // Tests of note.ts, the public note of spec v0.11 §24.1: the same cases as the -// tests of extension.CheckNote, NewNote and Note of the Go reference. +// tests of extension.CheckNote, NewNote and Note of the Go reference, with the +// texts that extension.CheckNote gives at spec-v0.11. import { describe, expect, it } from 'vitest'; import { checkNote, MAX_NOTE_LEN, newNote, NOTE_ID, publicNote } from './note.ts'; +const lone = (unit: number): string => String.fromCharCode(unit); +const invalid = (detail: string): string => `a public note that breaks the rules of text: text: ${detail}: ERR_EXTENSION_DATA_INVALID`; + describe('checkNote', () => { it('accepts a line of 1 to 1024 bytes that meets the rules of the declared author', () => { expect(() => checkNote('Cartas del viaje a Lisboa')).not.toThrow(); expect(() => checkNote('x'.repeat(MAX_NOTE_LEN))).not.toThrow(); expect(() => checkNote('Ñandú, 日本')).not.toThrow(); + expect(() => checkNote('\u{1f600}')).not.toThrow(); + }); + + it.each([ + ['empty', '', 'a public note of 0 bytes, not 1 to 1024: ERR_EXTENSION_DATA_INVALID'], + ['a tab', 'a\tb', invalid('control U+0009 in the declared author')], + ['a line feed', 'a\nb', invalid('control U+000A in the declared author')], + ['a carriage return', 'a\rb', invalid('control U+000D')], + ['a space at the start', ' a', invalid('the declared author starts or ends with U+0020')], + ['a space at the end', 'a ', invalid('the declared author starts or ends with U+0020')], + ['too long', 'x'.repeat(MAX_NOTE_LEN + 1), 'a public note of 1025 bytes, not 1 to 1024: ERR_EXTENSION_DATA_INVALID'], + ['a right-to-left override', 'a\u{202e}b', invalid('bidirectional control U+202E')], + ['a zero-width space', 'a\u{200b}b', invalid('invisible U+200B')], + ['a byte order mark at the start', '\u{feff}abc', invalid('byte order mark U+FEFF')], + ])('refuses %s with the text of Go', (_name, text, message) => { + expect(() => checkNote(text)).toThrow(message); }); + // UTF-8 cannot hold a lone surrogate: Go refuses the bytes of one, and a note + // is never written with U+FFFD in its place. The length is checked first, + // with three bytes for each, as Go counts them. it.each([ - ['empty', ''], - ['a tab', 'a\tb'], - ['a line feed', 'a\nb'], - ['a space at the start', ' a'], - ['a space at the end', 'a '], - ['too long', 'x'.repeat(MAX_NOTE_LEN + 1)], - ['a right-to-left override', 'a\u202eb'], - ['a zero-width space', 'a\u200bb'], - ])('refuses %s', (_name, text) => { - expect(() => checkNote(text)).toThrow(/public note/); + ['a lone high surrogate', `a${lone(0xd800)}b`], + ['a high surrogate at the end', `ab${lone(0xdbff)}`], + ['a lone low surrogate', `${lone(0xdc00)}ab`], + ['a pair the wrong way round', `a${lone(0xdc00)}${lone(0xd800)}`], + ['1024 bytes with a lone surrogate', 'x'.repeat(MAX_NOTE_LEN - 3) + lone(0xd800)], + ])('refuses malformed UTF-16, %s, as Go refuses invalid UTF-8', (_name, text) => { + expect(() => checkNote(text)).toThrow('a public note that is not valid UTF-8: ERR_EXTENSION_DATA_INVALID'); + expect(() => newNote(text)).toThrow('a public note that is not valid UTF-8: ERR_EXTENSION_DATA_INVALID'); + }); + + it('measures a lone surrogate as three bytes before it looks at its form', () => { + expect(() => checkNote('x'.repeat(MAX_NOTE_LEN - 2) + lone(0xd800))).toThrow('a public note of 1025 bytes, not 1 to 1024: ERR_EXTENSION_DATA_INVALID'); }); }); describe('newNote and publicNote', () => { - it('makes the extension, and reads it back among the noncritical ones', () => { - const e = newNote('Hola'); - expect([e.id, e.version, new TextDecoder().decode(e.data)]).toEqual([NOTE_ID, 1, 'Hola']); - expect(publicNote([{ id: 'x.other', version: 1, data: undefined }, e])).toBe('Hola'); + it('makes the extension, with the UTF-8 of the text, and reads it back among the noncritical ones', () => { + const e = newNote('Hola, Ñandú \u{1f600}'); + expect([e.id, e.version, e.data]).toEqual([NOTE_ID, 1, new TextEncoder().encode('Hola, Ñandú \u{1f600}')]); + expect(publicNote([{ id: 'x.other', version: 1, data: undefined }, e])).toBe('Hola, Ñandú \u{1f600}'); expect(publicNote([])).toBeUndefined(); expect(() => newNote('a\nb')).toThrow(/public note/); }); @@ -39,5 +64,14 @@ describe('newNote and publicNote', () => { expect(publicNote([{ id: NOTE_ID, version: 1, data: undefined }])).toBeUndefined(); expect(publicNote([{ id: NOTE_ID, version: 1, data: new TextEncoder().encode('a\nb') }])).toBeUndefined(); expect(publicNote([{ id: NOTE_ID, version: 1, data: Uint8Array.of(0xff, 0xfe) }])).toBeUndefined(); + // The UTF-8 of a surrogate, which Go refuses as well. + expect(publicNote([{ id: NOTE_ID, version: 1, data: Uint8Array.of(0x61, 0xed, 0xa0, 0x80) }])).toBeUndefined(); + }); + + // Go reads the bytes: a leading U+FEFF stays, the rules of text refuse it, and + // extension.Note finds the note unusable. A TextDecoder without ignoreBOM + // would drop it and show "abc". + it('shows nothing for a note that starts with a byte order mark, as Go', () => { + expect(publicNote([{ id: NOTE_ID, version: 1, data: Uint8Array.of(0xef, 0xbb, 0xbf, 0x61, 0x62, 0x63) }])).toBeUndefined(); }); }); diff --git a/src/lib/dkc/note.ts b/src/lib/dkc/note.ts index 08eb104..3f68380 100644 --- a/src/lib/dkc/note.ts +++ b/src/lib/dkc/note.ts @@ -6,7 +6,7 @@ // before the date, and nobody can check who wrote it. As Go's extension.CheckNote, // NewNote and Note. -import { utf8Length } from './bytes.ts'; +import { decodeUtf8, utf8Bytes, utf8Length } from './bytes.ts'; import { DateKeysError } from './errors.ts'; import type { Extension } from './extension.ts'; import { checkAuthor } from './pathrule.ts'; @@ -15,10 +15,19 @@ import { checkAuthor } from './pathrule.ts'; export const NOTE_ID = 'datekeys.note'; export const MAX_NOTE_LEN = 1024; -/** The text of a public note is from 1 to 1024 bytes of valid UTF-8 that meets the rules of the declared author (spec §24.1). Throws a DateKeysError when it is not. */ +/** + * The text of a public note is from 1 to 1024 bytes of valid UTF-8 that meets + * the rules of the declared author (spec §24.1), checked in the order and + * with the texts of Go. A string with a lone surrogate, malformed UTF-16, has + * no UTF-8: it is refused, never written with U+FFFD in its place, since the + * note is public and permanent (spec §62.1 rules 15 and 23). Its length + * counts each lone surrogate as the three bytes that Go counts for one + * written in UTF-8. Throws a DateKeysError when it is not. + */ export function checkNote(text: string): void { const n = utf8Length(text); if (n < 1 || n > MAX_NOTE_LEN) throw new DateKeysError('ERR_EXTENSION_DATA_INVALID', `a public note of ${n} bytes, not 1 to ${MAX_NOTE_LEN}`); + if (!text.isWellFormed()) throw new DateKeysError('ERR_EXTENSION_DATA_INVALID', 'a public note that is not valid UTF-8'); try { checkAuthor(text); } catch (err) { @@ -29,20 +38,22 @@ export function checkNote(text: string): void { /** The extension of a public note for `text`, which must pass checkNote. */ export function newNote(text: string): Extension { checkNote(text); - return { id: NOTE_ID, version: 1, data: new TextEncoder().encode(text) }; + return { id: NOTE_ID, version: 1, data: utf8Bytes(text) }; } /** * The text of the public note among the noncritical extensions of a * PUBLIC_HEADER, and undefined when there is none that is usable: a note whose * data breaks the rules of §24.1 is unusable, and shows nothing (spec §54). + * Its bytes are read as Go reads them: a leading U+FEFF stays, and the rules + * of text refuse it. */ export function publicNote(noncritical: readonly Extension[]): string | undefined { const e = noncritical.find((x) => x.id === NOTE_ID && x.version === 1); if (e?.data === undefined) return undefined; - let text: string; + const text = decodeUtf8(e.data); + if (text === undefined) return undefined; try { - text = new TextDecoder('utf-8', { fatal: true }).decode(e.data); checkNote(text); } catch { return undefined; diff --git a/src/lib/inspector/drand.test.ts b/src/lib/inspector/drand.test.ts index cc29a50..db94f2d 100644 --- a/src/lib/inspector/drand.test.ts +++ b/src/lib/inspector/drand.test.ts @@ -71,6 +71,16 @@ describe('fetchRelease', () => { ); }); + // The client of the reference reads the body with json.Unmarshal, which refuses a byte order mark before the JSON. + it('refuses an answer that starts with a byte order mark, as the client of the reference does', async () => { + const body = new TextEncoder().encode(JSON.stringify({ round: ROUND, signature: SIG, randomness: RANDOMNESS })); + const bom = new Response(new Uint8Array([0xef, 0xbb, 0xbf, ...body]), { status: 200 }); + const offline = (): Error => new TypeError('offline'); + await expect(fetchRelease(CHAIN, ROUND, relays({ 'api.drand.sh': bom, 'api2.drand.sh': offline(), 'api3.drand.sh': offline() }))).rejects.toThrow( + 'Ningún relay de drand dio la firma: api.drand.sh respondió algo que no es una firma; api2.drand.sh no se pudo conectar; api3.drand.sh no se pudo conectar.', + ); + }); + it('gives up after its timeout', async () => { await expect(fetchRelease(CHAIN, ROUND, relays({ 'api.drand.sh': 'hang', 'api2.drand.sh': 'hang', 'api3.drand.sh': 'hang' }), 20)).rejects.toThrow( 'Ningún relay de drand dio la firma: api.drand.sh no contestó a tiempo; api2.drand.sh no contestó a tiempo; api3.drand.sh no contestó a tiempo.', diff --git a/src/lib/inspector/drand.ts b/src/lib/inspector/drand.ts index 0ae5d29..1e679de 100644 --- a/src/lib/inspector/drand.ts +++ b/src/lib/inspector/drand.ts @@ -82,6 +82,10 @@ export async function fetchRelease(chainHash: string, round: number, fetcher: ty } } +// The body of an answer as text, a leading U+FEFF kept: JSON.parse refuses it, +// as json.Unmarshal does in the client of the reference. +const BODY_TEXT = new TextDecoder('utf-8', { ignoreBOM: true }); + // The body of an answer, read up to MAX_RESPONSE bytes and not one more. async function bounded(res: Response): Promise { const reader = res.body?.getReader(); @@ -102,7 +106,7 @@ async function bounded(res: Response): Promise { all.set(p, at); at += p.length; } - return new TextDecoder().decode(all); + return BODY_TEXT.decode(all); } async function sha256Hex(hex: string): Promise {