Escape the invisible and confusable characters of the sources

The tests of create-files.ts, format.ts, opening.ts and create-input.ts
held literal code points that a reader cannot tell apart, or cannot
see: U+212A KELVIN SIGN and U+017F LONG S, fullwidth and dotted I,
ZWJ, a combining acute, U+202E and U+FEFF, and equalFold compared with
two of them in its code. They are now \uXXXX escapes, with the same
values; no source in src or scripts holds a format character.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
main
dev 7 days ago
parent 7dc88ef240
commit 1ff4ca5f5f

@ -38,19 +38,19 @@ describe('systemFile', () => {
['.DS_Store', '.DS_Store'], ['.DS_Store', '.DS_Store'],
['.ds_store', '.DS_Store'], ['.ds_store', '.DS_Store'],
['.DS_STORE', '.DS_Store'], ['.DS_STORE', '.DS_Store'],
['.Dſ_Store', '.DS_Store'], ['.D\u017f_Store', '.DS_Store'],
['Thumbs.db', 'Thumbs.db'], ['Thumbs.db', 'Thumbs.db'],
['Thumbſ.db', 'Thumbs.db'], ['Thumb\u017f.db', 'Thumbs.db'],
['THUMBS.DB', 'Thumbs.db'], ['THUMBS.DB', 'Thumbs.db'],
['desktop.ini', 'desktop.ini'], ['desktop.ini', 'desktop.ini'],
['desKtop.ini', 'desktop.ini'], ['des\u212atop.ini', 'desktop.ini'],
['DESKTOP.INI', 'desktop.ini'], ['DESKTOP.INI', 'desktop.ini'],
['desktop.ini ', undefined], ['desktop.ini ', undefined],
['desktop.ini', undefined], ['\uff44esktop.ini', undefined],
['desktop.İni', undefined], ['desktop.\u0130ni', undefined],
['desktop.ıni', undefined], ['desktop.\u0131ni', undefined],
['desk‍top.ini', undefined], ['desk\u200dtop.ini', undefined],
['Thumbs.db́', undefined], ['Thumbs.db\u0301', undefined],
['._foto.jpg', '._*'], ['._foto.jpg', '._*'],
['._', '._*'], ['._', '._*'],
['_.foto', undefined], ['_.foto', undefined],

@ -41,7 +41,7 @@ function equalFold(s: string, ascii: string): boolean {
if (cps.length !== ascii.length) return false; if (cps.length !== ascii.length) return false;
return cps.every((c, i) => { return cps.every((c, i) => {
const want = lowerASCII(ascii[i]!); const want = lowerASCII(ascii[i]!);
return lowerASCII(c) === want || (want === 'k' && c === 'K') || (want === 's' && c === 'ſ'); return lowerASCII(c) === want || (want === 'k' && c === '\u212a') || (want === 's' && c === '\u017f');
}); });
} }

@ -235,7 +235,7 @@ describe('the names of the files', () => {
expect(downloadName('regalo.DKC', '.dkc', 'x.dkc')).toBe('regalo.DKC'); expect(downloadName('regalo.DKC', '.dkc', 'x.dkc')).toBe('regalo.DKC');
expect(downloadName('regalo', '.dkk', 'x.dkk')).toBe('regalo.dkk'); expect(downloadName('regalo', '.dkk', 'x.dkk')).toBe('regalo.dkk');
expect(downloadName('../a/b\\c', '.dkc', 'x.dkc')).toBe('.._a_b_c.dkc'); expect(downloadName('../a/b\\c', '.dkc', 'x.dkc')).toBe('.._a_b_c.dkc');
expect(downloadName('evil‮cod.dkc', '.dkc', 'x.dkc')).toBe('evil_cod.dkc'); expect(downloadName('evil\u202ecod.dkc', '.dkc', 'x.dkc')).toBe('evil_cod.dkc');
expect(downloadName(' ', '.dkc', 'x.dkc')).toBe('x.dkc'); expect(downloadName(' ', '.dkc', 'x.dkc')).toBe('x.dkc');
}); });
}); });

@ -69,7 +69,7 @@ describe('steps and codes', () => {
describe('text', () => { describe('text', () => {
it('replaces what is not printable in a file name', () => { it('replaces what is not printable in a file name', () => {
expect(safeFileName('informe 2026 — ñ.pdf')).toBe('informe 2026 — ñ.pdf'); expect(safeFileName('informe 2026 — ñ.pdf')).toBe('informe 2026 — ñ.pdf');
expect(safeFileName('a‮b\tc\nd\u{1f600}')).toBe('a_b_c_d\u{1f600}'); expect(safeFileName('a\u202eb\tc\nd\u{1f600}')).toBe('a_b_c_d\u{1f600}');
}); });
it('shows printable UTF-8 as text', () => { it('shows printable UTF-8 as text', () => {

@ -181,16 +181,16 @@ describe('plaintextPreview', () => {
}); });
it('shows a text written on Windows: CR LF as a line feed and no BOM; the download keeps the bytes', () => { it('shows a text written on Windows: CR LF as a line feed and no BOM; the download keeps the bytes', () => {
expect(whole('primera\r\nsegunda\r\n')).toEqual({ text: 'primera\nsegunda\n', cut: false }); expect(whole('\ufeffprimera\r\nsegunda\r\n')).toEqual({ text: 'primera\nsegunda\n', cut: false });
// A lone CR, or a BOM that is not at the start, is not a printable text. // A lone CR, or a BOM that is not at the start, is not a printable text.
expect(whole('a\rb')).toBeUndefined(); expect(whole('a\rb')).toBeUndefined();
expect(whole('ab')).toBeUndefined(); expect(whole('a\ufeffb')).toBeUndefined();
}); });
it('shows nothing of what is not UTF-8 made of printable characters', () => { it('shows nothing of what is not UTF-8 made of printable characters', () => {
expect(plaintextPreview(new Uint8Array([0x25, 0x50, 0x44, 0x46, 0x00]), true)).toBeUndefined(); expect(plaintextPreview(new Uint8Array([0x25, 0x50, 0x44, 0x46, 0x00]), true)).toBeUndefined();
expect(plaintextPreview(new Uint8Array([0xff, 0xfe, 0x41, 0x00]), true)).toBeUndefined(); expect(plaintextPreview(new Uint8Array([0xff, 0xfe, 0x41, 0x00]), true)).toBeUndefined();
expect(whole('a‮b')).toBeUndefined(); expect(whole('a\u202eb')).toBeUndefined();
}); });
it('drops a character of 2, 3 or 4 bytes that the first bytes cut, and a CR whose LF they cut', () => { it('drops a character of 2, 3 or 4 bytes that the first bytes cut, and a CR whose LF they cut', () => {

Loading…
Cancel
Save

Powered by TurnKey Linux.