The public note refuses malformed UTF-16, and no decoder drops a leading BOM

Fixes T4 and the rest of T5 of the review of the session of 1 and 2
October:

- note.ts: checkNote refuses a string with a lone surrogate, which UTF-8
  cannot hold, after its length and with the text that extension.CheckNote
  gives for invalid UTF-8. newNote wrote U+FFFD in its place, and the note
  is public and permanent (spec §62.1 rules 15 and 23).
- note.ts: publicNote reads the data with decodeUtf8, which keeps a leading
  U+FEFF: such a note is unusable, as in Go, where it showed without it.
- author.ts: authorCode takes the eight bytes of the code as Go's
  AuthorCode takes them, a leading U+FEFF kept.
- inspector/drand.ts: an answer of a relay that starts with a BOM is
  refused, as json.Unmarshal refuses it in the client of the reference.
- The texts of the head and the paths of format 3 already kept it, since
  cbor.ts decodes them with decodeUtf8: head.test.ts now pins it with the
  texts of Go. The other TextDecoder of src/lib, in age.ts, reads tokens of
  ASCII that are checked byte by byte before.

The texts are those of extension.CheckNote, extension.Note,
capsule.DecodeHead and capsule.AuthorCode at spec-v0.11, taken with an
oracle on the same bytes; the drand client of the reference refuses the
answer with json.Unmarshal. npm run verify passes.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
main
dev 6 days ago
parent 06e992fa74
commit 67a9037839

@ -77,6 +77,15 @@ describe('what is signed', () => {
expect(authorCode(message.subarray(1))).toBe('');
});
// Go's AuthorCode takes the bytes of the digest as they are: a byte order mark there stays in the code, as
// capsule.AuthorCode gives it at spec-v0.11.
it('takes the code byte by byte, a leading byte order mark included, as Go', () => {
const te = new TextEncoder();
const message = new Uint8Array([...te.encode('datekeys:dkc3:author-signature:v1\n'), 0xef, 0xbb, 0xbf, ...te.encode(`${'a'.repeat(61)}\n`)]);
expect(message.length).toBe(AUTHOR_MESSAGE_SIZE);
expect(authorCode(message)).toBe('\u{feff}a-aaaa');
});
it('binds alg and the list of signers, and leaves L out and I_PAYLOAD hidden', () => {
expect(hx(signersDigest(ALG_ED25519))).not.toBe(hx(signersDigest(ALG_CMS)));
expect(hx(signersDigest(ALG_CMS, Uint8Array.of(1)))).not.toBe(hx(signersDigest(ALG_CMS, Uint8Array.of(2))));

@ -85,9 +85,16 @@ export function sealSubject(controlCommitment: Uint8Array, headDigestValue: Uint
return domainHash(SEAL_SUBJECT_PREFIX, controlCommitment, headDigestValue, sigPart(signature));
}
/** The code of AUTHOR_MESSAGE that a person compares before signing: the first 8 hexadecimal digits of its digest, in two groups of 4. */
// The bytes of the code as text, a leading U+FEFF kept, as Go's string(b).
const CODE_TEXT = new TextDecoder('utf-8', { ignoreBOM: true });
/**
* The code of AUTHOR_MESSAGE that a person compares before signing: the first
* 8 hexadecimal digits of its digest, in two groups of 4, taken byte by byte
* as Go's AuthorCode takes them.
*/
export function authorCode(message: Uint8Array): string {
if (message.length !== AUTHOR_MESSAGE_SIZE) return '';
const d = new TextDecoder().decode(message.subarray(AUTHOR_MESSAGE_PREFIX.length + 1));
return `${d.slice(0, 4)}-${d.slice(4, 8)}`;
const d = message.subarray(AUTHOR_MESSAGE_PREFIX.length + 1);
return `${CODE_TEXT.decode(d.subarray(0, 4))}-${CODE_TEXT.decode(d.subarray(4, 8))}`;
}

@ -119,6 +119,11 @@ describe('decodeHead', () => {
['a comment with U+202E', head([3, t('a\u202eb')]), 'ERR_HEAD_INVALID', 'comment: text: bidirectional control U+202E'],
['a comment with the tag U+E0041', head([3, t('a\u{e0041}')]), 'ERR_HEAD_INVALID', 'comment: text: invisible U+E0041'],
['an author with LF', head([4, t('a\u000ab')]), 'ERR_HEAD_INVALID', 'declared author: text: control U+000A in the declared author'],
// A text or a path that starts with U+FEFF keeps it, as Go reads the bytes, and its rule refuses it with the text of Go.
['a comment that starts with U+FEFF', head([3, t('\u{feff}Hola')]), 'ERR_HEAD_INVALID', 'comment: text: byte order mark U+FEFF'],
['an author that starts with U+FEFF', head([4, t('\u{feff}Ana')]), 'ERR_HEAD_INVALID', 'declared author: text: byte order mark U+FEFF'],
['R4: a path that starts with U+FEFF', head([5, arr(file('\u{feff}a.txt'))]), 'ERR_HEAD_INVALID', 'file 1: R4: segment 1: invisible U+FEFF'],
['R4: a segment that starts with U+FEFF', head([5, arr(file('a/\u{feff}b.txt'))]), 'ERR_HEAD_INVALID', 'file 1: R4: segment 2: invisible U+FEFF'],
['R3: ..', head([5, arr(file('..'))]), 'ERR_HEAD_INVALID', 'file 1: R3: segment 1: the segment is two dots'],
['R2: /a', head([5, arr(file('/a'))]), 'ERR_HEAD_INVALID', 'file 1: R2: segment 1 is empty'],
['R4b: a and VS16', head([5, arr(file('a\ufe0f'))]), 'ERR_HEAD_INVALID', 'file 1: R4b: segment 1: U+FE0F is not part of an emoji variation sequence'],

@ -1,35 +1,60 @@
// Tests of note.ts, the public note of spec v0.11 §24.1: the same cases as the
// tests of extension.CheckNote, NewNote and Note of the Go reference.
// tests of extension.CheckNote, NewNote and Note of the Go reference, with the
// texts that extension.CheckNote gives at spec-v0.11.
import { describe, expect, it } from 'vitest';
import { checkNote, MAX_NOTE_LEN, newNote, NOTE_ID, publicNote } from './note.ts';
const lone = (unit: number): string => String.fromCharCode(unit);
const invalid = (detail: string): string => `a public note that breaks the rules of text: text: ${detail}: ERR_EXTENSION_DATA_INVALID`;
describe('checkNote', () => {
it('accepts a line of 1 to 1024 bytes that meets the rules of the declared author', () => {
expect(() => checkNote('Cartas del viaje a Lisboa')).not.toThrow();
expect(() => checkNote('x'.repeat(MAX_NOTE_LEN))).not.toThrow();
expect(() => checkNote('Ñandú, 日本')).not.toThrow();
expect(() => checkNote('\u{1f600}')).not.toThrow();
});
it.each([
['empty', '', 'a public note of 0 bytes, not 1 to 1024: ERR_EXTENSION_DATA_INVALID'],
['a tab', 'a\tb', invalid('control U+0009 in the declared author')],
['a line feed', 'a\nb', invalid('control U+000A in the declared author')],
['a carriage return', 'a\rb', invalid('control U+000D')],
['a space at the start', ' a', invalid('the declared author starts or ends with U+0020')],
['a space at the end', 'a ', invalid('the declared author starts or ends with U+0020')],
['too long', 'x'.repeat(MAX_NOTE_LEN + 1), 'a public note of 1025 bytes, not 1 to 1024: ERR_EXTENSION_DATA_INVALID'],
['a right-to-left override', 'a\u{202e}b', invalid('bidirectional control U+202E')],
['a zero-width space', 'a\u{200b}b', invalid('invisible U+200B')],
['a byte order mark at the start', '\u{feff}abc', invalid('byte order mark U+FEFF')],
])('refuses %s with the text of Go', (_name, text, message) => {
expect(() => checkNote(text)).toThrow(message);
});
// UTF-8 cannot hold a lone surrogate: Go refuses the bytes of one, and a note
// is never written with U+FFFD in its place. The length is checked first,
// with three bytes for each, as Go counts them.
it.each([
['empty', ''],
['a tab', 'a\tb'],
['a line feed', 'a\nb'],
['a space at the start', ' a'],
['a space at the end', 'a '],
['too long', 'x'.repeat(MAX_NOTE_LEN + 1)],
['a right-to-left override', 'a\u202eb'],
['a zero-width space', 'a\u200bb'],
])('refuses %s', (_name, text) => {
expect(() => checkNote(text)).toThrow(/public note/);
['a lone high surrogate', `a${lone(0xd800)}b`],
['a high surrogate at the end', `ab${lone(0xdbff)}`],
['a lone low surrogate', `${lone(0xdc00)}ab`],
['a pair the wrong way round', `a${lone(0xdc00)}${lone(0xd800)}`],
['1024 bytes with a lone surrogate', 'x'.repeat(MAX_NOTE_LEN - 3) + lone(0xd800)],
])('refuses malformed UTF-16, %s, as Go refuses invalid UTF-8', (_name, text) => {
expect(() => checkNote(text)).toThrow('a public note that is not valid UTF-8: ERR_EXTENSION_DATA_INVALID');
expect(() => newNote(text)).toThrow('a public note that is not valid UTF-8: ERR_EXTENSION_DATA_INVALID');
});
it('measures a lone surrogate as three bytes before it looks at its form', () => {
expect(() => checkNote('x'.repeat(MAX_NOTE_LEN - 2) + lone(0xd800))).toThrow('a public note of 1025 bytes, not 1 to 1024: ERR_EXTENSION_DATA_INVALID');
});
});
describe('newNote and publicNote', () => {
it('makes the extension, and reads it back among the noncritical ones', () => {
const e = newNote('Hola');
expect([e.id, e.version, new TextDecoder().decode(e.data)]).toEqual([NOTE_ID, 1, 'Hola']);
expect(publicNote([{ id: 'x.other', version: 1, data: undefined }, e])).toBe('Hola');
it('makes the extension, with the UTF-8 of the text, and reads it back among the noncritical ones', () => {
const e = newNote('Hola, Ñandú \u{1f600}');
expect([e.id, e.version, e.data]).toEqual([NOTE_ID, 1, new TextEncoder().encode('Hola, Ñandú \u{1f600}')]);
expect(publicNote([{ id: 'x.other', version: 1, data: undefined }, e])).toBe('Hola, Ñandú \u{1f600}');
expect(publicNote([])).toBeUndefined();
expect(() => newNote('a\nb')).toThrow(/public note/);
});
@ -39,5 +64,14 @@ describe('newNote and publicNote', () => {
expect(publicNote([{ id: NOTE_ID, version: 1, data: undefined }])).toBeUndefined();
expect(publicNote([{ id: NOTE_ID, version: 1, data: new TextEncoder().encode('a\nb') }])).toBeUndefined();
expect(publicNote([{ id: NOTE_ID, version: 1, data: Uint8Array.of(0xff, 0xfe) }])).toBeUndefined();
// The UTF-8 of a surrogate, which Go refuses as well.
expect(publicNote([{ id: NOTE_ID, version: 1, data: Uint8Array.of(0x61, 0xed, 0xa0, 0x80) }])).toBeUndefined();
});
// Go reads the bytes: a leading U+FEFF stays, the rules of text refuse it, and
// extension.Note finds the note unusable. A TextDecoder without ignoreBOM
// would drop it and show "abc".
it('shows nothing for a note that starts with a byte order mark, as Go', () => {
expect(publicNote([{ id: NOTE_ID, version: 1, data: Uint8Array.of(0xef, 0xbb, 0xbf, 0x61, 0x62, 0x63) }])).toBeUndefined();
});
});

@ -6,7 +6,7 @@
// before the date, and nobody can check who wrote it. As Go's extension.CheckNote,
// NewNote and Note.
import { utf8Length } from './bytes.ts';
import { decodeUtf8, utf8Bytes, utf8Length } from './bytes.ts';
import { DateKeysError } from './errors.ts';
import type { Extension } from './extension.ts';
import { checkAuthor } from './pathrule.ts';
@ -15,10 +15,19 @@ import { checkAuthor } from './pathrule.ts';
export const NOTE_ID = 'datekeys.note';
export const MAX_NOTE_LEN = 1024;
/** The text of a public note is from 1 to 1024 bytes of valid UTF-8 that meets the rules of the declared author (spec §24.1). Throws a DateKeysError when it is not. */
/**
* The text of a public note is from 1 to 1024 bytes of valid UTF-8 that meets
* the rules of the declared author (spec §24.1), checked in the order and
* with the texts of Go. A string with a lone surrogate, malformed UTF-16, has
* no UTF-8: it is refused, never written with U+FFFD in its place, since the
* note is public and permanent (spec §62.1 rules 15 and 23). Its length
* counts each lone surrogate as the three bytes that Go counts for one
* written in UTF-8. Throws a DateKeysError when it is not.
*/
export function checkNote(text: string): void {
const n = utf8Length(text);
if (n < 1 || n > MAX_NOTE_LEN) throw new DateKeysError('ERR_EXTENSION_DATA_INVALID', `a public note of ${n} bytes, not 1 to ${MAX_NOTE_LEN}`);
if (!text.isWellFormed()) throw new DateKeysError('ERR_EXTENSION_DATA_INVALID', 'a public note that is not valid UTF-8');
try {
checkAuthor(text);
} catch (err) {
@ -29,20 +38,22 @@ export function checkNote(text: string): void {
/** The extension of a public note for `text`, which must pass checkNote. */
export function newNote(text: string): Extension {
checkNote(text);
return { id: NOTE_ID, version: 1, data: new TextEncoder().encode(text) };
return { id: NOTE_ID, version: 1, data: utf8Bytes(text) };
}
/**
* The text of the public note among the noncritical extensions of a
* PUBLIC_HEADER, and undefined when there is none that is usable: a note whose
* data breaks the rules of §24.1 is unusable, and shows nothing (spec §54).
* Its bytes are read as Go reads them: a leading U+FEFF stays, and the rules
* of text refuse it.
*/
export function publicNote(noncritical: readonly Extension[]): string | undefined {
const e = noncritical.find((x) => x.id === NOTE_ID && x.version === 1);
if (e?.data === undefined) return undefined;
let text: string;
const text = decodeUtf8(e.data);
if (text === undefined) return undefined;
try {
text = new TextDecoder('utf-8', { fatal: true }).decode(e.data);
checkNote(text);
} catch {
return undefined;

@ -71,6 +71,16 @@ describe('fetchRelease', () => {
);
});
// The client of the reference reads the body with json.Unmarshal, which refuses a byte order mark before the JSON.
it('refuses an answer that starts with a byte order mark, as the client of the reference does', async () => {
const body = new TextEncoder().encode(JSON.stringify({ round: ROUND, signature: SIG, randomness: RANDOMNESS }));
const bom = new Response(new Uint8Array([0xef, 0xbb, 0xbf, ...body]), { status: 200 });
const offline = (): Error => new TypeError('offline');
await expect(fetchRelease(CHAIN, ROUND, relays({ 'api.drand.sh': bom, 'api2.drand.sh': offline(), 'api3.drand.sh': offline() }))).rejects.toThrow(
'Ningún relay de drand dio la firma: api.drand.sh respondió algo que no es una firma; api2.drand.sh no se pudo conectar; api3.drand.sh no se pudo conectar.',
);
});
it('gives up after its timeout', async () => {
await expect(fetchRelease(CHAIN, ROUND, relays({ 'api.drand.sh': 'hang', 'api2.drand.sh': 'hang', 'api3.drand.sh': 'hang' }), 20)).rejects.toThrow(
'Ningún relay de drand dio la firma: api.drand.sh no contestó a tiempo; api2.drand.sh no contestó a tiempo; api3.drand.sh no contestó a tiempo.',

@ -82,6 +82,10 @@ export async function fetchRelease(chainHash: string, round: number, fetcher: ty
}
}
// The body of an answer as text, a leading U+FEFF kept: JSON.parse refuses it,
// as json.Unmarshal does in the client of the reference.
const BODY_TEXT = new TextDecoder('utf-8', { ignoreBOM: true });
// The body of an answer, read up to MAX_RESPONSE bytes and not one more.
async function bounded(res: Response): Promise<string> {
const reader = res.body?.getReader();
@ -102,7 +106,7 @@ async function bounded(res: Response): Promise<string> {
all.set(p, at);
at += p.length;
}
return new TextDecoder().decode(all);
return BODY_TEXT.decode(all);
}
async function sha256Hex(hex: string): Promise<string> {

Loading…
Cancel
Save

Powered by TurnKey Linux.