Paste previously flattened everything to plain text (the signed import doctrine deferred rich paste). Now the editor parses clipboard text/html into the V2 model, the inverse of serialize-html.ts: - parse-html.ts: parseWordsHtml(domRoot) -> WordsBlock[]. Maps p / h1-6 / blockquote / pre / ul-ol-li (incl. checkboxes) / hr / img / figure / table, with inline marks (bold/italic/underline/strike/code, colour + background from style) and links. URLs sanitised; unknown blocks fall back to a paragraph, unknown inline tags are transparent. - insert-blocks.ts: insertBlocks(state, blocks) drops blocks at the caret. Single pasted paragraph inline-merges (a phrase stays in the sentence); multiple/block-level paste splits the host; a nested or non-inline caret degrades to a structure-preserving plain-text insert; a trailing paragraph guarantees a caret home after a terminal table/image. New insertBlocks command + dispatch. - provider onpaste: parse text/html in an INERT document (createHTMLDocument off the active dom — no scripts run, no resources load), then applyCommand insertBlocks. Image-files / single-URL / plain-text paths unchanged; plain text is still the fallback. Tests: insert-blocks 6/6 (merge/split/replace/append/no-op), parse-html 10/10 browser (marks, lists, tables, colour, javascript: URL sanitised). Browser-verified: pasting an h2 + a bold/italic paragraph + a list inserts the heading, the formatted paragraph and the list at the caret. Updated the provider's unsupported-paste test (real HTML now parses; only content-free fragments log unsupported-html). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>active-uix
parent
48695cffad
commit
65883089b3
@ -0,0 +1,94 @@
|
|||||||
|
/**
|
||||||
|
* insertBlocks — the rich-paste sink. Pure (no DOM): blocks come from
|
||||||
|
* factories, the op drops them at the caret.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { describe, expect, it } from 'vitest';
|
||||||
|
import { insertBlocks } from './insert-blocks';
|
||||||
|
import { createHeading, createParagraph, createText } from './factories';
|
||||||
|
import { createState } from './text';
|
||||||
|
import { createCollapsedSelection } from '../selection';
|
||||||
|
import { WORDS_VERSION, type WordsBlock, type WordsDocument } from '../types';
|
||||||
|
|
||||||
|
function doc(...children: WordsBlock[]): WordsDocument {
|
||||||
|
return { version: WORDS_VERSION, children };
|
||||||
|
}
|
||||||
|
const para = (text: string) => createParagraph([createText(text)]);
|
||||||
|
|
||||||
|
describe('insertBlocks (rich paste)', () => {
|
||||||
|
it('inline-merges a single pasted paragraph into the host (marks kept)', () => {
|
||||||
|
const d = doc(para('ab'));
|
||||||
|
// caret between a|b
|
||||||
|
const sel = createCollapsedSelection([0, 0], 1);
|
||||||
|
const pasted = [createParagraph([createText('X', ['bold'])])];
|
||||||
|
const r = insertBlocks(createState(d, sel), pasted);
|
||||||
|
expect(r.changed).toBe(true);
|
||||||
|
// Still ONE block — the phrase stayed in the sentence.
|
||||||
|
expect(r.state.document.children).toHaveLength(1);
|
||||||
|
const b = r.state.document.children[0];
|
||||||
|
if (b.type !== 'paragraph') throw new Error('expected paragraph');
|
||||||
|
expect(b.children.map((c) => (c.type === 'text' ? c.text : ''))).toEqual(['a', 'X', 'b']);
|
||||||
|
// The X carries bold.
|
||||||
|
const x = b.children[1];
|
||||||
|
expect(x.type === 'text' && x.marks).toContain('bold');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('splits the host and inserts multiple blocks between the halves', () => {
|
||||||
|
const d = doc(para('hello world'));
|
||||||
|
// caret after "hello " (offset 6)
|
||||||
|
const sel = createCollapsedSelection([0, 0], 6);
|
||||||
|
const pasted = [createHeading(1, [createText('Title')]), para('Body')];
|
||||||
|
const r = insertBlocks(createState(d, sel), pasted);
|
||||||
|
expect(r.changed).toBe(true);
|
||||||
|
const types = r.state.document.children.map((b) => b.type);
|
||||||
|
// before("hello ") · heading · paragraph("Body") · after("world")
|
||||||
|
expect(types).toEqual(['paragraph', 'heading', 'paragraph', 'paragraph']);
|
||||||
|
const texts = r.state.document.children.map((b) =>
|
||||||
|
b.type === 'paragraph' || b.type === 'heading'
|
||||||
|
? b.children.map((c) => (c.type === 'text' ? c.text : '')).join('')
|
||||||
|
: ''
|
||||||
|
);
|
||||||
|
expect(texts).toEqual(['hello ', 'Title', 'Body', 'world']);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('replaces a non-collapsed selection before inserting', () => {
|
||||||
|
const d = doc(para('abcde'));
|
||||||
|
// select "bcd" (offset 1..4)
|
||||||
|
const sel = {
|
||||||
|
anchor: { path: [0, 0], offset: 1 },
|
||||||
|
focus: { path: [0, 0], offset: 4 }
|
||||||
|
};
|
||||||
|
const r = insertBlocks(createState(d, sel), [createParagraph([createText('X')])]);
|
||||||
|
expect(r.changed).toBe(true);
|
||||||
|
expect(r.state.document.children).toHaveLength(1);
|
||||||
|
const b = r.state.document.children[0];
|
||||||
|
if (b.type !== 'paragraph') throw new Error('expected paragraph');
|
||||||
|
expect(b.children.map((c) => (c.type === 'text' ? c.text : '')).join('')).toBe('aXe');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('replaces an empty host paragraph with the pasted block', () => {
|
||||||
|
const d = doc(createParagraph([createText('')]));
|
||||||
|
const sel = createCollapsedSelection([0, 0], 0);
|
||||||
|
const r = insertBlocks(createState(d, sel), [createHeading(2, [createText('H')])]);
|
||||||
|
expect(r.changed).toBe(true);
|
||||||
|
expect(r.state.document.children).toHaveLength(1);
|
||||||
|
expect(r.state.document.children[0].type).toBe('heading');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('appends at the document end when there is no selection', () => {
|
||||||
|
const d = doc(para('one'));
|
||||||
|
const r = insertBlocks(createState(d, null), [para('two'), para('three')]);
|
||||||
|
expect(r.changed).toBe(true);
|
||||||
|
expect(r.state.document.children.map((b) => b.type)).toEqual([
|
||||||
|
'paragraph',
|
||||||
|
'paragraph',
|
||||||
|
'paragraph'
|
||||||
|
]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('no-ops on empty input', () => {
|
||||||
|
const d = doc(para('x'));
|
||||||
|
const r = insertBlocks(createState(d, createCollapsedSelection([0, 0], 0)), []);
|
||||||
|
expect(r.changed).toBe(false);
|
||||||
|
});
|
||||||
|
});
|
||||||
@ -0,0 +1,232 @@
|
|||||||
|
/**
|
||||||
|
* V2 `insertBlocks` — insert a list of blocks at the selection.
|
||||||
|
*
|
||||||
|
* The paste sink: `parseWordsHtml` turns clipboard HTML into `WordsBlock[]`,
|
||||||
|
* this op drops them into the document at the caret.
|
||||||
|
*
|
||||||
|
* Behaviour:
|
||||||
|
* - A non-empty selection is deleted first (paste replaces it).
|
||||||
|
* - No caret → append at the document end.
|
||||||
|
* - Caret in a TOP-LEVEL inline block (paragraph / heading / quote):
|
||||||
|
* · a single pasted paragraph splices inline (a pasted phrase stays
|
||||||
|
* in the sentence);
|
||||||
|
* · otherwise the host splits at the caret and the blocks land
|
||||||
|
* between the two halves.
|
||||||
|
* - Caret NESTED (inside a column / table cell) or in a non-inline host:
|
||||||
|
* degrade to a structure-preserving plain-text insert (marks/blocks
|
||||||
|
* flatten, nothing breaks). Top-level rich paste is the 99% path.
|
||||||
|
*
|
||||||
|
* The caret lands at the end of the pasted content; when the paste ends in
|
||||||
|
* a non-inline block (table / image) a trailing paragraph is added so the
|
||||||
|
* caret always has a home.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { createCollapsedSelection, normalizeRange } from '../selection';
|
||||||
|
import { createParagraph, createText } from './factories';
|
||||||
|
import { deleteRange } from './delete';
|
||||||
|
import { editingContainerPath, pointFromInlineTextOffset, replaceAt } from './helpers';
|
||||||
|
import { splitContainerAtPoint } from './inline-split';
|
||||||
|
import { insertText } from './text';
|
||||||
|
import { normalizeDocument } from './normalize';
|
||||||
|
import { getActiveMarksForSelection } from './selection-walkers';
|
||||||
|
import { changed, noOp, type WordsEditorState, type WordsOperationResult } from './types';
|
||||||
|
import type { WordsBlock, WordsDocument, WordsInline } from '../types';
|
||||||
|
|
||||||
|
const INLINE_BLOCKS = new Set(['paragraph', 'heading', 'quote']);
|
||||||
|
|
||||||
|
export function insertBlocks(
|
||||||
|
state: WordsEditorState,
|
||||||
|
blocks: readonly WordsBlock[]
|
||||||
|
): WordsOperationResult {
|
||||||
|
if (blocks.length === 0) return noOp(state);
|
||||||
|
|
||||||
|
// 1. Replace a non-empty selection.
|
||||||
|
let cur = state;
|
||||||
|
if (cur.selection && !normalizeRange(cur.selection).collapsed) {
|
||||||
|
cur = deleteRange(cur).state;
|
||||||
|
}
|
||||||
|
|
||||||
|
const doc = cur.document;
|
||||||
|
const sel = cur.selection;
|
||||||
|
|
||||||
|
// 2. No caret → append at the end.
|
||||||
|
if (!sel) {
|
||||||
|
const children = [...doc.children, ...blocks];
|
||||||
|
return commit(doc, children, doc.children.length + blocks.length - 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
const point = normalizeRange(sel).start;
|
||||||
|
const containerPath = editingContainerPath(doc, point.path);
|
||||||
|
|
||||||
|
// 3. Caret on an atomic block (no editable container) → insert after it.
|
||||||
|
if (!containerPath) {
|
||||||
|
const at = (point.path[0] ?? doc.children.length - 1) + 1;
|
||||||
|
const children = [...doc.children.slice(0, at), ...blocks, ...doc.children.slice(at)];
|
||||||
|
return commit(doc, children, at + blocks.length - 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
const topIndex = containerPath[0] ?? 0;
|
||||||
|
const host = doc.children[topIndex];
|
||||||
|
const hostInline = !!host && INLINE_BLOCKS.has(host.type);
|
||||||
|
|
||||||
|
// 4. Nested caret OR non-inline host → degrade to a plain-text insert that
|
||||||
|
// splices safely at any depth (reuses the tested insertText path).
|
||||||
|
if (containerPath.length > 1 || !hostInline) {
|
||||||
|
return insertText(cur, flattenToPlainText(blocks));
|
||||||
|
}
|
||||||
|
|
||||||
|
// 5. Top-level inline host — split its inlines at the caret.
|
||||||
|
const split = splitContainerAtPoint(doc, containerPath, point);
|
||||||
|
|
||||||
|
// 5a. A single pasted paragraph → splice its inlines inline (one block).
|
||||||
|
if (blocks.length === 1 && blocks[0].type === 'paragraph') {
|
||||||
|
const inlines = [...split.before, ...blocks[0].children, ...split.after];
|
||||||
|
const merged = withInlines(host, inlines);
|
||||||
|
const children = replaceAt(doc.children, topIndex, merged);
|
||||||
|
const caretOffset = inlineTextLen(split.before) + inlineTextLen(blocks[0].children);
|
||||||
|
return commitAt(doc, children, topIndex, caretOffset);
|
||||||
|
}
|
||||||
|
|
||||||
|
// 5b. Multiple / block-level paste → split host, blocks between the halves.
|
||||||
|
const beforeBlocks = inlineRunHasContent(split.before) ? [withInlines(host, split.before)] : [];
|
||||||
|
const afterBlocks = inlineRunHasContent(split.after) ? [createParagraph(split.after)] : [];
|
||||||
|
// Guarantee a caret home when the paste ends in a non-inline block.
|
||||||
|
const lastPasted = blocks[blocks.length - 1];
|
||||||
|
const needsTrailing = afterBlocks.length === 0 && !INLINE_BLOCKS.has(lastPasted.type);
|
||||||
|
const trailing = needsTrailing ? [createParagraph()] : [];
|
||||||
|
const mid = [...beforeBlocks, ...blocks, ...afterBlocks, ...trailing];
|
||||||
|
const children = [
|
||||||
|
...doc.children.slice(0, topIndex),
|
||||||
|
...mid,
|
||||||
|
...doc.children.slice(topIndex + 1)
|
||||||
|
];
|
||||||
|
// Caret: end of the last inline-bearing pasted block, else the trailing /
|
||||||
|
// after paragraph.
|
||||||
|
const lastPastedIndex = topIndex + beforeBlocks.length + blocks.length - 1;
|
||||||
|
const caretIndex =
|
||||||
|
afterBlocks.length > 0 || trailing.length > 0 ? lastPastedIndex + 1 : lastPastedIndex;
|
||||||
|
return commit(doc, children, caretIndex);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── commit helpers ─────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
/** Normalize + place the caret at the END of the block at `caretIndex`
|
||||||
|
* (or the nearest inline-bearing block at/after it). */
|
||||||
|
function commit(
|
||||||
|
doc: WordsDocument,
|
||||||
|
children: readonly WordsBlock[],
|
||||||
|
caretIndex: number
|
||||||
|
): WordsOperationResult {
|
||||||
|
const normalized = normalizeDocument({ ...doc, children }).document;
|
||||||
|
const point = safeCaret(normalized, caretIndex);
|
||||||
|
const selection = createCollapsedSelection(point.path, point.offset);
|
||||||
|
return changed({
|
||||||
|
document: normalized,
|
||||||
|
selection,
|
||||||
|
activeMarks: getActiveMarksForSelection(normalized, selection)
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Normalize + place the caret inside the block at `blockIndex` at a known
|
||||||
|
* character offset (used by the single-paragraph inline-merge). */
|
||||||
|
function commitAt(
|
||||||
|
doc: WordsDocument,
|
||||||
|
children: readonly WordsBlock[],
|
||||||
|
blockIndex: number,
|
||||||
|
offset: number
|
||||||
|
): WordsOperationResult {
|
||||||
|
const normalized = normalizeDocument({ ...doc, children }).document;
|
||||||
|
const point = pointFromInlineTextOffset(normalized, [blockIndex], offset);
|
||||||
|
const selection = createCollapsedSelection(point.path, point.offset);
|
||||||
|
return changed({
|
||||||
|
document: normalized,
|
||||||
|
selection,
|
||||||
|
activeMarks: getActiveMarksForSelection(normalized, selection)
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/** A caret at the end of the inline-bearing block at/after `preferred`,
|
||||||
|
* else the nearest one before it, else the document start. */
|
||||||
|
function safeCaret(doc: WordsDocument, preferred: number): { path: number[]; offset: number } {
|
||||||
|
const n = doc.children.length;
|
||||||
|
for (let i = Math.max(0, preferred); i < n; i++) {
|
||||||
|
const b = doc.children[i];
|
||||||
|
if (b && INLINE_BLOCKS.has(b.type)) {
|
||||||
|
const off = i === preferred ? blockTextLen(b) : 0;
|
||||||
|
const p = pointFromInlineTextOffset(doc, [i], off);
|
||||||
|
return { path: [...p.path], offset: p.offset };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (let i = Math.min(preferred, n - 1); i >= 0; i--) {
|
||||||
|
const b = doc.children[i];
|
||||||
|
if (b && INLINE_BLOCKS.has(b.type)) {
|
||||||
|
const p = pointFromInlineTextOffset(doc, [i], blockTextLen(b));
|
||||||
|
return { path: [...p.path], offset: p.offset };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const p = pointFromInlineTextOffset(doc, [0], 0);
|
||||||
|
return { path: [...p.path], offset: p.offset };
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── small helpers ────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
function withInlines(block: WordsBlock, inlines: readonly WordsInline[]): WordsBlock {
|
||||||
|
const safe = inlines.length > 0 ? inlines : [createText('')];
|
||||||
|
return { ...block, children: safe } as WordsBlock;
|
||||||
|
}
|
||||||
|
|
||||||
|
function inlineTextLen(inlines: readonly WordsInline[]): number {
|
||||||
|
let n = 0;
|
||||||
|
for (const node of inlines) {
|
||||||
|
if (node.type === 'text') n += node.text.length;
|
||||||
|
else if (node.type === 'link') for (const c of node.children) n += c.text.length;
|
||||||
|
}
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
|
function blockTextLen(block: WordsBlock): number {
|
||||||
|
if (block.type === 'paragraph' || block.type === 'heading' || block.type === 'quote') {
|
||||||
|
return inlineTextLen(block.children);
|
||||||
|
}
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
function inlineRunHasContent(run: readonly WordsInline[]): boolean {
|
||||||
|
return run.some((n) => (n.type === 'text' ? n.text.trim().length > 0 : true));
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Flatten blocks to plain text (used for the nested-caret degrade path). */
|
||||||
|
function flattenToPlainText(blocks: readonly WordsBlock[]): string {
|
||||||
|
return blocks.map(blockToPlainText).join('\n');
|
||||||
|
}
|
||||||
|
|
||||||
|
function blockToPlainText(block: WordsBlock): string {
|
||||||
|
switch (block.type) {
|
||||||
|
case 'paragraph':
|
||||||
|
case 'heading':
|
||||||
|
case 'quote':
|
||||||
|
return inlinesToPlainText(block.children);
|
||||||
|
case 'code':
|
||||||
|
return block.children.map((c) => c.text).join('');
|
||||||
|
case 'list':
|
||||||
|
return block.items.map((i) => inlinesToPlainText(i.children)).join('\n');
|
||||||
|
case 'callout':
|
||||||
|
return block.children.map(blockToPlainText).join('\n');
|
||||||
|
case 'columns':
|
||||||
|
return block.columns.map((c) => c.children.map(blockToPlainText).join('\n')).join('\n');
|
||||||
|
case 'table':
|
||||||
|
return block.rows
|
||||||
|
.map((r) => r.cells.map((c) => c.children.map(blockToPlainText).join(' ')).join('\t'))
|
||||||
|
.join('\n');
|
||||||
|
default:
|
||||||
|
return '';
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function inlinesToPlainText(inlines: readonly WordsInline[]): string {
|
||||||
|
return inlines
|
||||||
|
.map((n) =>
|
||||||
|
n.type === 'text' ? n.text : n.type === 'link' ? n.children.map((c) => c.text).join('') : ''
|
||||||
|
)
|
||||||
|
.join('');
|
||||||
|
}
|
||||||
@ -0,0 +1,110 @@
|
|||||||
|
/**
|
||||||
|
* parseWordsHtml — clipboard HTML → blocks. Runs as a browser test
|
||||||
|
* (`*.svelte.test.ts`) because it walks a real DOM tree.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { describe, expect, it } from 'vitest';
|
||||||
|
import { parseWordsHtml } from './parse-html';
|
||||||
|
import type { WordsBlock, WordsInline, WordsMark } from '../types';
|
||||||
|
|
||||||
|
function parse(html: string): readonly WordsBlock[] {
|
||||||
|
const div = document.createElement('div');
|
||||||
|
div.innerHTML = html;
|
||||||
|
return parseWordsHtml(div);
|
||||||
|
}
|
||||||
|
function text(n: WordsInline): string {
|
||||||
|
return n.type === 'text' ? n.text : n.type === 'link' ? n.children.map((c) => c.text).join('') : '';
|
||||||
|
}
|
||||||
|
function marksOf(n: WordsInline): readonly WordsMark[] {
|
||||||
|
return n.type === 'text' ? (n.marks ?? []) : [];
|
||||||
|
}
|
||||||
|
|
||||||
|
describe('parseWordsHtml', () => {
|
||||||
|
it('paragraph with a bold run', () => {
|
||||||
|
const blocks = parse('<p>Hello <strong>world</strong></p>');
|
||||||
|
expect(blocks).toHaveLength(1);
|
||||||
|
const p = blocks[0];
|
||||||
|
if (p.type !== 'paragraph') throw new Error('expected paragraph');
|
||||||
|
expect(p.children.map(text).join('')).toBe('Hello world');
|
||||||
|
const bold = p.children.find((c) => text(c) === 'world');
|
||||||
|
expect(bold && marksOf(bold)).toContain('bold');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('headings (h1-h3 direct, h4+ clamp to 3)', () => {
|
||||||
|
const blocks = parse('<h1>One</h1><h2>Two</h2><h5>Five</h5>');
|
||||||
|
expect(blocks.map((b) => b.type)).toEqual(['heading', 'heading', 'heading']);
|
||||||
|
expect(blocks.map((b) => (b.type === 'heading' ? b.level : 0))).toEqual([1, 2, 3]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('nested marks compose (italic + bold)', () => {
|
||||||
|
const blocks = parse('<p><em><strong>bi</strong></em></p>');
|
||||||
|
const p = blocks[0];
|
||||||
|
if (p.type !== 'paragraph') throw new Error('expected paragraph');
|
||||||
|
const node = p.children[0];
|
||||||
|
expect(text(node)).toBe('bi');
|
||||||
|
expect(marksOf(node)).toEqual(expect.arrayContaining(['bold', 'italic']));
|
||||||
|
});
|
||||||
|
|
||||||
|
it('unordered + ordered lists', () => {
|
||||||
|
const ul = parse('<ul><li>a</li><li>b</li></ul>');
|
||||||
|
expect(ul[0].type).toBe('list');
|
||||||
|
if (ul[0].type === 'list') {
|
||||||
|
expect(ul[0].kind).toBe('unordered');
|
||||||
|
expect(ul[0].items.map((i) => i.children.map(text).join(''))).toEqual(['a', 'b']);
|
||||||
|
}
|
||||||
|
const ol = parse('<ol><li>x</li></ol>');
|
||||||
|
expect(ol[0].type === 'list' && ol[0].kind).toBe('ordered');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('links (sanitised href)', () => {
|
||||||
|
const blocks = parse('<p>see <a href="https://example.com">here</a></p>');
|
||||||
|
const p = blocks[0];
|
||||||
|
if (p.type !== 'paragraph') throw new Error('expected paragraph');
|
||||||
|
const link = p.children.find((c) => c.type === 'link');
|
||||||
|
expect(link?.type).toBe('link');
|
||||||
|
expect(link?.type === 'link' && link.href).toBe('https://example.com');
|
||||||
|
// A javascript: URL is dropped (sanitised) → no link, just its text.
|
||||||
|
const evil = parse('<p><a href="javascript:alert(1)">x</a></p>');
|
||||||
|
const ep = evil[0];
|
||||||
|
expect(ep.type === 'paragraph' && ep.children.every((c) => c.type !== 'link')).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('quote + code', () => {
|
||||||
|
expect(parse('<blockquote>q</blockquote>')[0].type).toBe('quote');
|
||||||
|
const code = parse('<pre><code>const x = 1</code></pre>')[0];
|
||||||
|
expect(code.type).toBe('code');
|
||||||
|
if (code.type === 'code') expect(code.children.map((c) => c.text).join('')).toBe('const x = 1');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('inline colour from style', () => {
|
||||||
|
const blocks = parse('<p><span style="color: rgb(255, 0, 0)">red</span></p>');
|
||||||
|
const p = blocks[0];
|
||||||
|
if (p.type !== 'paragraph') throw new Error('expected paragraph');
|
||||||
|
const node = p.children.find((c) => text(c) === 'red');
|
||||||
|
const colour = marksOf(node!).find((m) => typeof m === 'object' && m.type === 'color');
|
||||||
|
expect(colour).toBeTruthy();
|
||||||
|
});
|
||||||
|
|
||||||
|
it('table with a header row', () => {
|
||||||
|
const blocks = parse('<table><tr><th>H1</th><th>H2</th></tr><tr><td>a</td><td>b</td></tr></table>');
|
||||||
|
expect(blocks[0].type).toBe('table');
|
||||||
|
if (blocks[0].type === 'table') {
|
||||||
|
expect(blocks[0].headerRow).toBe(true);
|
||||||
|
expect(blocks[0].rows).toHaveLength(2);
|
||||||
|
expect(blocks[0].rows[0].cells).toHaveLength(2);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
it('bare inline content (no wrapping <p>) becomes a paragraph', () => {
|
||||||
|
const blocks = parse('plain <b>bold</b> text');
|
||||||
|
expect(blocks).toHaveLength(1);
|
||||||
|
expect(blocks[0].type).toBe('paragraph');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('hr → divider; img → image', () => {
|
||||||
|
expect(parse('<hr>')[0].type).toBe('divider');
|
||||||
|
const img = parse('<img src="https://example.com/a.png" alt="cat">')[0];
|
||||||
|
expect(img.type).toBe('image');
|
||||||
|
if (img.type === 'image') expect(img.src).toBe('https://example.com/a.png');
|
||||||
|
});
|
||||||
|
});
|
||||||
@ -0,0 +1,318 @@
|
|||||||
|
/**
|
||||||
|
* V2 HTML paste parser — a pasted DOM tree → `WordsBlock[]`.
|
||||||
|
*
|
||||||
|
* The inverse of `serialize-html.ts`: walks an already-parsed DOM fragment
|
||||||
|
* and maps standard HTML into the V2 block model — paragraph / heading /
|
||||||
|
* quote / code / list / divider / image / table — with inline marks
|
||||||
|
* (bold / italic / underline / strike / code / colour / background) and
|
||||||
|
* links. URLs are sanitised through `sanitizeWordsUrl`.
|
||||||
|
*
|
||||||
|
* Pure DOM-walking: the caller supplies the parsed root (the provider via
|
||||||
|
* `DOMParser`, tests via happy-dom), so this module never reaches for a
|
||||||
|
* global `document`. Output feeds `insertBlocks` / `normalizeDocument`,
|
||||||
|
* which assign ids and coalesce adjacent same-mark text.
|
||||||
|
*
|
||||||
|
* Scope: the common tags browsers and editors emit on copy. Unknown block
|
||||||
|
* tags fall back to a paragraph of their inline content; unknown inline
|
||||||
|
* tags are transparent (their children are parsed with the same marks).
|
||||||
|
*/
|
||||||
|
|
||||||
|
import {
|
||||||
|
createCodeBlock,
|
||||||
|
createDivider,
|
||||||
|
createHeading,
|
||||||
|
createImage,
|
||||||
|
createLink,
|
||||||
|
createList,
|
||||||
|
createListItem,
|
||||||
|
createParagraph,
|
||||||
|
createQuote,
|
||||||
|
createTable,
|
||||||
|
createTableCell,
|
||||||
|
createTableRow,
|
||||||
|
createText
|
||||||
|
} from './factories';
|
||||||
|
import { sanitizeWordsUrl } from '../normalize';
|
||||||
|
import type {
|
||||||
|
ListItem,
|
||||||
|
TableCell,
|
||||||
|
TableRow,
|
||||||
|
WordsBlock,
|
||||||
|
WordsHeadingLevel,
|
||||||
|
WordsInline,
|
||||||
|
WordsListKind,
|
||||||
|
WordsMark,
|
||||||
|
WordsText
|
||||||
|
} from '../types';
|
||||||
|
|
||||||
|
const ELEMENT_NODE = 1;
|
||||||
|
const TEXT_NODE = 3;
|
||||||
|
|
||||||
|
const BLOCK_TAGS = new Set([
|
||||||
|
'p',
|
||||||
|
'h1',
|
||||||
|
'h2',
|
||||||
|
'h3',
|
||||||
|
'h4',
|
||||||
|
'h5',
|
||||||
|
'h6',
|
||||||
|
'blockquote',
|
||||||
|
'pre',
|
||||||
|
'ul',
|
||||||
|
'ol',
|
||||||
|
'hr',
|
||||||
|
'img',
|
||||||
|
'figure',
|
||||||
|
'table',
|
||||||
|
'div',
|
||||||
|
'section',
|
||||||
|
'article',
|
||||||
|
'main',
|
||||||
|
'header',
|
||||||
|
'footer',
|
||||||
|
'aside'
|
||||||
|
]);
|
||||||
|
|
||||||
|
// ── Entry point ──────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
export function parseWordsHtml(root: ParentNode): readonly WordsBlock[] {
|
||||||
|
const blocks: WordsBlock[] = [];
|
||||||
|
collectBlocks(root, blocks);
|
||||||
|
return blocks;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Block level ──────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Walk `parent`'s children, flushing runs of inline content into paragraphs
|
||||||
|
* around the block-level elements. Bare inline content (e.g. pasting
|
||||||
|
* "**a** b" with no wrapping `<p>`) becomes a paragraph.
|
||||||
|
*/
|
||||||
|
function collectBlocks(parent: ParentNode, out: WordsBlock[]): void {
|
||||||
|
let inlineRun: WordsInline[] = [];
|
||||||
|
const flush = () => {
|
||||||
|
if (inlineRunHasContent(inlineRun)) out.push(createParagraph(inlineRun));
|
||||||
|
inlineRun = [];
|
||||||
|
};
|
||||||
|
for (const node of Array.from(parent.childNodes)) {
|
||||||
|
if (node.nodeType === TEXT_NODE) {
|
||||||
|
const text = node.textContent ?? '';
|
||||||
|
// Whitespace-only text BETWEEN block elements is layout, not content.
|
||||||
|
if (text.trim().length > 0) inlineRun.push(createText(collapseWs(text)));
|
||||||
|
} else if (node.nodeType === ELEMENT_NODE) {
|
||||||
|
const el = node as Element;
|
||||||
|
const tag = el.tagName.toLowerCase();
|
||||||
|
if (BLOCK_TAGS.has(tag)) {
|
||||||
|
flush();
|
||||||
|
parseBlockElement(el, tag, out);
|
||||||
|
} else {
|
||||||
|
inlineRun.push(...parseInlines(el, []));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
flush();
|
||||||
|
}
|
||||||
|
|
||||||
|
function parseBlockElement(el: Element, tag: string, out: WordsBlock[]): void {
|
||||||
|
switch (tag) {
|
||||||
|
case 'p':
|
||||||
|
out.push(createParagraph(parseInlines(el, [])));
|
||||||
|
return;
|
||||||
|
case 'h1':
|
||||||
|
case 'h2':
|
||||||
|
case 'h3':
|
||||||
|
case 'h4':
|
||||||
|
case 'h5':
|
||||||
|
case 'h6':
|
||||||
|
out.push(createHeading(headingLevel(tag), parseInlines(el, [])));
|
||||||
|
return;
|
||||||
|
case 'blockquote':
|
||||||
|
out.push(createQuote(parseInlines(el, [])));
|
||||||
|
return;
|
||||||
|
case 'pre':
|
||||||
|
out.push(createCodeBlock([createText(el.textContent ?? '')]));
|
||||||
|
return;
|
||||||
|
case 'ul':
|
||||||
|
out.push(createList('unordered', parseListItems(el, 'unordered')));
|
||||||
|
return;
|
||||||
|
case 'ol':
|
||||||
|
out.push(createList('ordered', parseListItems(el, 'ordered')));
|
||||||
|
return;
|
||||||
|
case 'hr':
|
||||||
|
out.push(createDivider());
|
||||||
|
return;
|
||||||
|
case 'img': {
|
||||||
|
const img = parseImage(el);
|
||||||
|
if (img) out.push(img);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
case 'figure': {
|
||||||
|
const inner = el.querySelector('img');
|
||||||
|
if (inner) {
|
||||||
|
const img = parseImage(inner);
|
||||||
|
if (img) out.push(img);
|
||||||
|
} else {
|
||||||
|
collectBlocks(el, out);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
case 'table': {
|
||||||
|
const table = parseTable(el);
|
||||||
|
if (table) out.push(table);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
// div / section / article / … — unwrap structural containers.
|
||||||
|
collectBlocks(el, out);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function parseListItems(listEl: Element, kind: WordsListKind): readonly ListItem[] {
|
||||||
|
const items: ListItem[] = [];
|
||||||
|
for (const li of Array.from(listEl.children)) {
|
||||||
|
if (li.tagName.toLowerCase() !== 'li') continue;
|
||||||
|
// Checkbox lists: GitHub / editors emit `<li><input type=checkbox>…`.
|
||||||
|
const checkbox = li.querySelector('input[type="checkbox"]');
|
||||||
|
const checked = checkbox ? (checkbox as HTMLInputElement).checked : undefined;
|
||||||
|
const inlines = parseInlines(li, []).filter((n) => !isEmptyText(n));
|
||||||
|
items.push(
|
||||||
|
createListItem(inlines, kind === 'check' || checkbox ? { checked: checked ?? false } : {})
|
||||||
|
);
|
||||||
|
}
|
||||||
|
return items;
|
||||||
|
}
|
||||||
|
|
||||||
|
function parseImage(el: Element): WordsBlock | undefined {
|
||||||
|
const src = sanitizeWordsUrl(el.getAttribute('src') ?? '');
|
||||||
|
if (!src) return undefined;
|
||||||
|
const alt = el.getAttribute('alt');
|
||||||
|
return createImage({ src, ...(alt ? { alt } : {}) });
|
||||||
|
}
|
||||||
|
|
||||||
|
function parseTable(tableEl: Element): WordsBlock | undefined {
|
||||||
|
const rows: TableRow[] = [];
|
||||||
|
let headerRow = false;
|
||||||
|
// Rows from thead/tbody/tfoot or direct <tr>.
|
||||||
|
const trs = Array.from(tableEl.querySelectorAll('tr'));
|
||||||
|
trs.forEach((tr, rowIdx) => {
|
||||||
|
const cells: TableCell[] = [];
|
||||||
|
for (const cell of Array.from(tr.children)) {
|
||||||
|
const ct = cell.tagName.toLowerCase();
|
||||||
|
if (ct !== 'td' && ct !== 'th') continue;
|
||||||
|
if (rowIdx === 0 && ct === 'th') headerRow = true;
|
||||||
|
// Cells hold blocks (since P5m). Parse the cell's content; seed a
|
||||||
|
// paragraph when it has only inline text.
|
||||||
|
const cellBlocks: WordsBlock[] = [];
|
||||||
|
collectBlocks(cell, cellBlocks);
|
||||||
|
cells.push(createTableCell(cellBlocks.length > 0 ? cellBlocks : [createParagraph()]));
|
||||||
|
}
|
||||||
|
if (cells.length > 0) rows.push(createTableRow(cells));
|
||||||
|
});
|
||||||
|
if (rows.length === 0) return undefined;
|
||||||
|
return createTable(rows, headerRow ? { headerRow: true } : {});
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Inline level ─────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
/** Parse `parent`'s inline content, accumulating marks down the tree. */
|
||||||
|
function parseInlines(parent: ParentNode, marks: readonly WordsMark[]): WordsInline[] {
|
||||||
|
const result: WordsInline[] = [];
|
||||||
|
for (const node of Array.from(parent.childNodes)) {
|
||||||
|
if (node.nodeType === TEXT_NODE) {
|
||||||
|
const text = collapseWs(node.textContent ?? '');
|
||||||
|
if (text.length > 0) result.push(createText(text, marks));
|
||||||
|
} else if (node.nodeType === ELEMENT_NODE) {
|
||||||
|
const el = node as Element;
|
||||||
|
const tag = el.tagName.toLowerCase();
|
||||||
|
if (tag === 'br') {
|
||||||
|
// V2 has no inline break node; a soft break reads as a space.
|
||||||
|
result.push(createText(' ', marks));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (tag === 'a') {
|
||||||
|
const href = sanitizeWordsUrl(el.getAttribute('href') ?? '');
|
||||||
|
const texts = inlinesToText(parseInlines(el, marks));
|
||||||
|
if (href && texts.length > 0) result.push(createLink(href, texts));
|
||||||
|
else result.push(...parseInlines(el, marks));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
result.push(...parseInlines(el, marksForElement(tag, el, marks)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Marks contributed by an inline element (tag semantics + inline style). */
|
||||||
|
function marksForElement(
|
||||||
|
tag: string,
|
||||||
|
el: Element,
|
||||||
|
marks: readonly WordsMark[]
|
||||||
|
): readonly WordsMark[] {
|
||||||
|
const next: WordsMark[] = [...marks];
|
||||||
|
if (tag === 'strong' || tag === 'b') next.push('bold');
|
||||||
|
else if (tag === 'em' || tag === 'i' || tag === 'cite') next.push('italic');
|
||||||
|
else if (tag === 'u' || tag === 'ins') next.push('underline');
|
||||||
|
else if (tag === 's' || tag === 'strike' || tag === 'del') next.push('strike');
|
||||||
|
else if (tag === 'code' || tag === 'kbd' || tag === 'samp') next.push('code');
|
||||||
|
else if (tag === 'mark') next.push({ type: 'background', value: readColor(el, 'background-color') ?? '#ffc53d' });
|
||||||
|
|
||||||
|
// Inline style — covers `<span style="font-weight:bold; color:…">` etc.
|
||||||
|
const style = (el as Partial<HTMLElement>).style;
|
||||||
|
if (style) {
|
||||||
|
const weight = style.fontWeight;
|
||||||
|
if (weight === 'bold' || /^[6-9]00$/.test(weight)) next.push('bold');
|
||||||
|
if (style.fontStyle === 'italic') next.push('italic');
|
||||||
|
const deco = `${style.textDecorationLine || ''} ${style.textDecoration || ''}`;
|
||||||
|
if (deco.includes('underline')) next.push('underline');
|
||||||
|
if (deco.includes('line-through')) next.push('strike');
|
||||||
|
}
|
||||||
|
const color = readColor(el, 'color');
|
||||||
|
if (color) next.push({ type: 'color', value: color });
|
||||||
|
if (tag !== 'mark') {
|
||||||
|
const bg = readColor(el, 'background-color');
|
||||||
|
if (bg) next.push({ type: 'background', value: bg });
|
||||||
|
}
|
||||||
|
return next;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Small helpers ──────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
function headingLevel(tag: string): WordsHeadingLevel {
|
||||||
|
// h1-h3 map directly; h4-h6 collapse to h3 (the model caps at 3).
|
||||||
|
const n = Number(tag.slice(1));
|
||||||
|
return (n <= 1 ? 1 : n === 2 ? 2 : 3) as WordsHeadingLevel;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Collapse runs of whitespace (incl. newlines) to single spaces, HTML-style. */
|
||||||
|
function collapseWs(text: string): string {
|
||||||
|
return text.replace(/\s+/g, ' ');
|
||||||
|
}
|
||||||
|
|
||||||
|
function readColor(el: Element, prop: 'color' | 'background-color'): string | undefined {
|
||||||
|
const style = (el as Partial<HTMLElement>).style;
|
||||||
|
if (!style) return undefined;
|
||||||
|
const v = style.getPropertyValue(prop).trim();
|
||||||
|
if (!v || v === 'inherit' || v === 'initial' || v === 'transparent' || v === 'currentcolor') {
|
||||||
|
return undefined;
|
||||||
|
}
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Coerce parsed inline content to `WordsText[]` for a link's children. */
|
||||||
|
function inlinesToText(inlines: readonly WordsInline[]): WordsText[] {
|
||||||
|
const texts: WordsText[] = [];
|
||||||
|
for (const n of inlines) {
|
||||||
|
if (n.type === 'text') texts.push(n);
|
||||||
|
else if (n.type === 'link') for (const c of n.children) texts.push(c);
|
||||||
|
}
|
||||||
|
return texts.filter((t) => t.text.length > 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
function isEmptyText(n: WordsInline): boolean {
|
||||||
|
return n.type === 'text' && n.text.trim().length === 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
function inlineRunHasContent(run: readonly WordsInline[]): boolean {
|
||||||
|
return run.some((n) => (n.type === 'text' ? n.text.trim().length > 0 : true));
|
||||||
|
}
|
||||||
Loading…
Reference in new issue