feat(words): rich HTML paste — clipboard HTML becomes blocks + marks

Paste previously flattened everything to plain text (the signed import
doctrine deferred rich paste). Now the editor parses clipboard text/html
into the V2 model, the inverse of serialize-html.ts:

- parse-html.ts: parseWordsHtml(domRoot) -> WordsBlock[]. Maps p / h1-6 /
  blockquote / pre / ul-ol-li (incl. checkboxes) / hr / img / figure /
  table, with inline marks (bold/italic/underline/strike/code, colour +
  background from style) and links. URLs sanitised; unknown blocks fall
  back to a paragraph, unknown inline tags are transparent.
- insert-blocks.ts: insertBlocks(state, blocks) drops blocks at the caret.
  Single pasted paragraph inline-merges (a phrase stays in the sentence);
  multiple/block-level paste splits the host; a nested or non-inline caret
  degrades to a structure-preserving plain-text insert; a trailing
  paragraph guarantees a caret home after a terminal table/image. New
  insertBlocks command + dispatch.
- provider onpaste: parse text/html in an INERT document
  (createHTMLDocument off the active dom — no scripts run, no resources
  load), then applyCommand insertBlocks. Image-files / single-URL /
  plain-text paths unchanged; plain text is still the fallback.

Tests: insert-blocks 6/6 (merge/split/replace/append/no-op), parse-html
10/10 browser (marks, lists, tables, colour, javascript: URL sanitised).
Browser-verified: pasting an h2 + a bold/italic paragraph + a list inserts
the heading, the formatted paragraph and the list at the caret. Updated
the provider's unsupported-paste test (real HTML now parses; only
content-free fragments log unsupported-html).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
active-uix
dev 4 months ago
parent 48695cffad
commit 65883089b3

@ -66,6 +66,7 @@ import {
replaceDocument, replaceDocument,
setCodeLanguage setCodeLanguage
} from './insert-block-types'; } from './insert-block-types';
import { insertBlocks } from './insert-blocks';
import type { import type {
Block, Block,
WordsBlock, WordsBlock,
@ -146,6 +147,13 @@ export type WordsCommand =
readonly blockIndex: number; readonly blockIndex: number;
readonly block: Readonly<Record<string, unknown>>; readonly block: Readonly<Record<string, unknown>>;
} }
// Insert a list of blocks at the caret (rich paste — `parseWordsHtml`
// output). Splits the host, inline-merges a single paragraph, degrades a
// nested caret to a plain-text insert.
| {
readonly type: 'insertBlocks';
readonly blocks: readonly WordsBlock[];
}
// Caretless insert into a column slot (the `+` overlay in each // Caretless insert into a column slot (the `+` overlay in each
// column). Carries the new block AND its post-insert selection in // column). Carries the new block AND its post-insert selection in
// one transaction — see `insertBlockInColumn` for the placement // one transaction — see `insertBlockInColumn` for the placement
@ -282,6 +290,8 @@ export function applyWordsCommand(
command.blockIndex, command.blockIndex,
command.block as unknown as WordsBlock command.block as unknown as WordsBlock
); );
case 'insertBlocks':
return insertBlocks(state, command.blocks);
case 'insertBlockInColumn': case 'insertBlockInColumn':
return insertBlockInColumn( return insertBlockInColumn(
state, state,

@ -224,6 +224,9 @@ export {
// ── Plain-text serialization ───────────────────────────────────────────── // ── Plain-text serialization ─────────────────────────────────────────────
export { parseWordsPlainText, renderWordsPlainText } from './serialize-text'; export { parseWordsPlainText, renderWordsPlainText } from './serialize-text';
// ── HTML paste parsing (clipboard HTML → blocks) ─────────────────────────
export { parseWordsHtml } from './parse-html';
// ── DOM HTML emission ──────────────────────────────────────────────────── // ── DOM HTML emission ────────────────────────────────────────────────────
export { renderWordsDomHtml, renderWordsDomHtmlPretty } from './render-dom'; export { renderWordsDomHtml, renderWordsDomHtmlPretty } from './render-dom';

@ -0,0 +1,94 @@
/**
* insertBlocks — the rich-paste sink. Pure (no DOM): blocks come from
* factories, the op drops them at the caret.
*/
import { describe, expect, it } from 'vitest';
import { insertBlocks } from './insert-blocks';
import { createHeading, createParagraph, createText } from './factories';
import { createState } from './text';
import { createCollapsedSelection } from '../selection';
import { WORDS_VERSION, type WordsBlock, type WordsDocument } from '../types';
function doc(...children: WordsBlock[]): WordsDocument {
return { version: WORDS_VERSION, children };
}
const para = (text: string) => createParagraph([createText(text)]);
describe('insertBlocks (rich paste)', () => {
it('inline-merges a single pasted paragraph into the host (marks kept)', () => {
const d = doc(para('ab'));
// caret between a|b
const sel = createCollapsedSelection([0, 0], 1);
const pasted = [createParagraph([createText('X', ['bold'])])];
const r = insertBlocks(createState(d, sel), pasted);
expect(r.changed).toBe(true);
// Still ONE block — the phrase stayed in the sentence.
expect(r.state.document.children).toHaveLength(1);
const b = r.state.document.children[0];
if (b.type !== 'paragraph') throw new Error('expected paragraph');
expect(b.children.map((c) => (c.type === 'text' ? c.text : ''))).toEqual(['a', 'X', 'b']);
// The X carries bold.
const x = b.children[1];
expect(x.type === 'text' && x.marks).toContain('bold');
});
it('splits the host and inserts multiple blocks between the halves', () => {
const d = doc(para('hello world'));
// caret after "hello " (offset 6)
const sel = createCollapsedSelection([0, 0], 6);
const pasted = [createHeading(1, [createText('Title')]), para('Body')];
const r = insertBlocks(createState(d, sel), pasted);
expect(r.changed).toBe(true);
const types = r.state.document.children.map((b) => b.type);
// before("hello ") · heading · paragraph("Body") · after("world")
expect(types).toEqual(['paragraph', 'heading', 'paragraph', 'paragraph']);
const texts = r.state.document.children.map((b) =>
b.type === 'paragraph' || b.type === 'heading'
? b.children.map((c) => (c.type === 'text' ? c.text : '')).join('')
: ''
);
expect(texts).toEqual(['hello ', 'Title', 'Body', 'world']);
});
it('replaces a non-collapsed selection before inserting', () => {
const d = doc(para('abcde'));
// select "bcd" (offset 1..4)
const sel = {
anchor: { path: [0, 0], offset: 1 },
focus: { path: [0, 0], offset: 4 }
};
const r = insertBlocks(createState(d, sel), [createParagraph([createText('X')])]);
expect(r.changed).toBe(true);
expect(r.state.document.children).toHaveLength(1);
const b = r.state.document.children[0];
if (b.type !== 'paragraph') throw new Error('expected paragraph');
expect(b.children.map((c) => (c.type === 'text' ? c.text : '')).join('')).toBe('aXe');
});
it('replaces an empty host paragraph with the pasted block', () => {
const d = doc(createParagraph([createText('')]));
const sel = createCollapsedSelection([0, 0], 0);
const r = insertBlocks(createState(d, sel), [createHeading(2, [createText('H')])]);
expect(r.changed).toBe(true);
expect(r.state.document.children).toHaveLength(1);
expect(r.state.document.children[0].type).toBe('heading');
});
it('appends at the document end when there is no selection', () => {
const d = doc(para('one'));
const r = insertBlocks(createState(d, null), [para('two'), para('three')]);
expect(r.changed).toBe(true);
expect(r.state.document.children.map((b) => b.type)).toEqual([
'paragraph',
'paragraph',
'paragraph'
]);
});
it('no-ops on empty input', () => {
const d = doc(para('x'));
const r = insertBlocks(createState(d, createCollapsedSelection([0, 0], 0)), []);
expect(r.changed).toBe(false);
});
});

@ -0,0 +1,232 @@
/**
* V2 `insertBlocks` — insert a list of blocks at the selection.
*
* The paste sink: `parseWordsHtml` turns clipboard HTML into `WordsBlock[]`,
* this op drops them into the document at the caret.
*
* Behaviour:
* - A non-empty selection is deleted first (paste replaces it).
* - No caret → append at the document end.
* - Caret in a TOP-LEVEL inline block (paragraph / heading / quote):
* · a single pasted paragraph splices inline (a pasted phrase stays
* in the sentence);
* · otherwise the host splits at the caret and the blocks land
* between the two halves.
* - Caret NESTED (inside a column / table cell) or in a non-inline host:
* degrade to a structure-preserving plain-text insert (marks/blocks
* flatten, nothing breaks). Top-level rich paste is the 99% path.
*
* The caret lands at the end of the pasted content; when the paste ends in
* a non-inline block (table / image) a trailing paragraph is added so the
* caret always has a home.
*/
import { createCollapsedSelection, normalizeRange } from '../selection';
import { createParagraph, createText } from './factories';
import { deleteRange } from './delete';
import { editingContainerPath, pointFromInlineTextOffset, replaceAt } from './helpers';
import { splitContainerAtPoint } from './inline-split';
import { insertText } from './text';
import { normalizeDocument } from './normalize';
import { getActiveMarksForSelection } from './selection-walkers';
import { changed, noOp, type WordsEditorState, type WordsOperationResult } from './types';
import type { WordsBlock, WordsDocument, WordsInline } from '../types';
const INLINE_BLOCKS = new Set(['paragraph', 'heading', 'quote']);
export function insertBlocks(
state: WordsEditorState,
blocks: readonly WordsBlock[]
): WordsOperationResult {
if (blocks.length === 0) return noOp(state);
// 1. Replace a non-empty selection.
let cur = state;
if (cur.selection && !normalizeRange(cur.selection).collapsed) {
cur = deleteRange(cur).state;
}
const doc = cur.document;
const sel = cur.selection;
// 2. No caret → append at the end.
if (!sel) {
const children = [...doc.children, ...blocks];
return commit(doc, children, doc.children.length + blocks.length - 1);
}
const point = normalizeRange(sel).start;
const containerPath = editingContainerPath(doc, point.path);
// 3. Caret on an atomic block (no editable container) → insert after it.
if (!containerPath) {
const at = (point.path[0] ?? doc.children.length - 1) + 1;
const children = [...doc.children.slice(0, at), ...blocks, ...doc.children.slice(at)];
return commit(doc, children, at + blocks.length - 1);
}
const topIndex = containerPath[0] ?? 0;
const host = doc.children[topIndex];
const hostInline = !!host && INLINE_BLOCKS.has(host.type);
// 4. Nested caret OR non-inline host → degrade to a plain-text insert that
// splices safely at any depth (reuses the tested insertText path).
if (containerPath.length > 1 || !hostInline) {
return insertText(cur, flattenToPlainText(blocks));
}
// 5. Top-level inline host — split its inlines at the caret.
const split = splitContainerAtPoint(doc, containerPath, point);
// 5a. A single pasted paragraph → splice its inlines inline (one block).
if (blocks.length === 1 && blocks[0].type === 'paragraph') {
const inlines = [...split.before, ...blocks[0].children, ...split.after];
const merged = withInlines(host, inlines);
const children = replaceAt(doc.children, topIndex, merged);
const caretOffset = inlineTextLen(split.before) + inlineTextLen(blocks[0].children);
return commitAt(doc, children, topIndex, caretOffset);
}
// 5b. Multiple / block-level paste → split host, blocks between the halves.
const beforeBlocks = inlineRunHasContent(split.before) ? [withInlines(host, split.before)] : [];
const afterBlocks = inlineRunHasContent(split.after) ? [createParagraph(split.after)] : [];
// Guarantee a caret home when the paste ends in a non-inline block.
const lastPasted = blocks[blocks.length - 1];
const needsTrailing = afterBlocks.length === 0 && !INLINE_BLOCKS.has(lastPasted.type);
const trailing = needsTrailing ? [createParagraph()] : [];
const mid = [...beforeBlocks, ...blocks, ...afterBlocks, ...trailing];
const children = [
...doc.children.slice(0, topIndex),
...mid,
...doc.children.slice(topIndex + 1)
];
// Caret: end of the last inline-bearing pasted block, else the trailing /
// after paragraph.
const lastPastedIndex = topIndex + beforeBlocks.length + blocks.length - 1;
const caretIndex =
afterBlocks.length > 0 || trailing.length > 0 ? lastPastedIndex + 1 : lastPastedIndex;
return commit(doc, children, caretIndex);
}
// ── commit helpers ─────────────────────────────────────────────────────────
/** Normalize + place the caret at the END of the block at `caretIndex`
* (or the nearest inline-bearing block at/after it). */
function commit(
doc: WordsDocument,
children: readonly WordsBlock[],
caretIndex: number
): WordsOperationResult {
const normalized = normalizeDocument({ ...doc, children }).document;
const point = safeCaret(normalized, caretIndex);
const selection = createCollapsedSelection(point.path, point.offset);
return changed({
document: normalized,
selection,
activeMarks: getActiveMarksForSelection(normalized, selection)
});
}
/** Normalize + place the caret inside the block at `blockIndex` at a known
* character offset (used by the single-paragraph inline-merge). */
function commitAt(
doc: WordsDocument,
children: readonly WordsBlock[],
blockIndex: number,
offset: number
): WordsOperationResult {
const normalized = normalizeDocument({ ...doc, children }).document;
const point = pointFromInlineTextOffset(normalized, [blockIndex], offset);
const selection = createCollapsedSelection(point.path, point.offset);
return changed({
document: normalized,
selection,
activeMarks: getActiveMarksForSelection(normalized, selection)
});
}
/** A caret at the end of the inline-bearing block at/after `preferred`,
* else the nearest one before it, else the document start. */
function safeCaret(doc: WordsDocument, preferred: number): { path: number[]; offset: number } {
const n = doc.children.length;
for (let i = Math.max(0, preferred); i < n; i++) {
const b = doc.children[i];
if (b && INLINE_BLOCKS.has(b.type)) {
const off = i === preferred ? blockTextLen(b) : 0;
const p = pointFromInlineTextOffset(doc, [i], off);
return { path: [...p.path], offset: p.offset };
}
}
for (let i = Math.min(preferred, n - 1); i >= 0; i--) {
const b = doc.children[i];
if (b && INLINE_BLOCKS.has(b.type)) {
const p = pointFromInlineTextOffset(doc, [i], blockTextLen(b));
return { path: [...p.path], offset: p.offset };
}
}
const p = pointFromInlineTextOffset(doc, [0], 0);
return { path: [...p.path], offset: p.offset };
}
// ── small helpers ────────────────────────────────────────────────────────
function withInlines(block: WordsBlock, inlines: readonly WordsInline[]): WordsBlock {
const safe = inlines.length > 0 ? inlines : [createText('')];
return { ...block, children: safe } as WordsBlock;
}
function inlineTextLen(inlines: readonly WordsInline[]): number {
let n = 0;
for (const node of inlines) {
if (node.type === 'text') n += node.text.length;
else if (node.type === 'link') for (const c of node.children) n += c.text.length;
}
return n;
}
function blockTextLen(block: WordsBlock): number {
if (block.type === 'paragraph' || block.type === 'heading' || block.type === 'quote') {
return inlineTextLen(block.children);
}
return 0;
}
function inlineRunHasContent(run: readonly WordsInline[]): boolean {
return run.some((n) => (n.type === 'text' ? n.text.trim().length > 0 : true));
}
/** Flatten blocks to plain text (used for the nested-caret degrade path). */
function flattenToPlainText(blocks: readonly WordsBlock[]): string {
return blocks.map(blockToPlainText).join('\n');
}
function blockToPlainText(block: WordsBlock): string {
switch (block.type) {
case 'paragraph':
case 'heading':
case 'quote':
return inlinesToPlainText(block.children);
case 'code':
return block.children.map((c) => c.text).join('');
case 'list':
return block.items.map((i) => inlinesToPlainText(i.children)).join('\n');
case 'callout':
return block.children.map(blockToPlainText).join('\n');
case 'columns':
return block.columns.map((c) => c.children.map(blockToPlainText).join('\n')).join('\n');
case 'table':
return block.rows
.map((r) => r.cells.map((c) => c.children.map(blockToPlainText).join(' ')).join('\t'))
.join('\n');
default:
return '';
}
}
function inlinesToPlainText(inlines: readonly WordsInline[]): string {
return inlines
.map((n) =>
n.type === 'text' ? n.text : n.type === 'link' ? n.children.map((c) => c.text).join('') : ''
)
.join('');
}

@ -0,0 +1,110 @@
/**
* parseWordsHtml — clipboard HTML → blocks. Runs as a browser test
* (`*.svelte.test.ts`) because it walks a real DOM tree.
*/
import { describe, expect, it } from 'vitest';
import { parseWordsHtml } from './parse-html';
import type { WordsBlock, WordsInline, WordsMark } from '../types';
function parse(html: string): readonly WordsBlock[] {
const div = document.createElement('div');
div.innerHTML = html;
return parseWordsHtml(div);
}
function text(n: WordsInline): string {
return n.type === 'text' ? n.text : n.type === 'link' ? n.children.map((c) => c.text).join('') : '';
}
function marksOf(n: WordsInline): readonly WordsMark[] {
return n.type === 'text' ? (n.marks ?? []) : [];
}
describe('parseWordsHtml', () => {
it('paragraph with a bold run', () => {
const blocks = parse('<p>Hello <strong>world</strong></p>');
expect(blocks).toHaveLength(1);
const p = blocks[0];
if (p.type !== 'paragraph') throw new Error('expected paragraph');
expect(p.children.map(text).join('')).toBe('Hello world');
const bold = p.children.find((c) => text(c) === 'world');
expect(bold && marksOf(bold)).toContain('bold');
});
it('headings (h1-h3 direct, h4+ clamp to 3)', () => {
const blocks = parse('<h1>One</h1><h2>Two</h2><h5>Five</h5>');
expect(blocks.map((b) => b.type)).toEqual(['heading', 'heading', 'heading']);
expect(blocks.map((b) => (b.type === 'heading' ? b.level : 0))).toEqual([1, 2, 3]);
});
it('nested marks compose (italic + bold)', () => {
const blocks = parse('<p><em><strong>bi</strong></em></p>');
const p = blocks[0];
if (p.type !== 'paragraph') throw new Error('expected paragraph');
const node = p.children[0];
expect(text(node)).toBe('bi');
expect(marksOf(node)).toEqual(expect.arrayContaining(['bold', 'italic']));
});
it('unordered + ordered lists', () => {
const ul = parse('<ul><li>a</li><li>b</li></ul>');
expect(ul[0].type).toBe('list');
if (ul[0].type === 'list') {
expect(ul[0].kind).toBe('unordered');
expect(ul[0].items.map((i) => i.children.map(text).join(''))).toEqual(['a', 'b']);
}
const ol = parse('<ol><li>x</li></ol>');
expect(ol[0].type === 'list' && ol[0].kind).toBe('ordered');
});
it('links (sanitised href)', () => {
const blocks = parse('<p>see <a href="https://example.com">here</a></p>');
const p = blocks[0];
if (p.type !== 'paragraph') throw new Error('expected paragraph');
const link = p.children.find((c) => c.type === 'link');
expect(link?.type).toBe('link');
expect(link?.type === 'link' && link.href).toBe('https://example.com');
// A javascript: URL is dropped (sanitised) → no link, just its text.
const evil = parse('<p><a href="javascript:alert(1)">x</a></p>');
const ep = evil[0];
expect(ep.type === 'paragraph' && ep.children.every((c) => c.type !== 'link')).toBe(true);
});
it('quote + code', () => {
expect(parse('<blockquote>q</blockquote>')[0].type).toBe('quote');
const code = parse('<pre><code>const x = 1</code></pre>')[0];
expect(code.type).toBe('code');
if (code.type === 'code') expect(code.children.map((c) => c.text).join('')).toBe('const x = 1');
});
it('inline colour from style', () => {
const blocks = parse('<p><span style="color: rgb(255, 0, 0)">red</span></p>');
const p = blocks[0];
if (p.type !== 'paragraph') throw new Error('expected paragraph');
const node = p.children.find((c) => text(c) === 'red');
const colour = marksOf(node!).find((m) => typeof m === 'object' && m.type === 'color');
expect(colour).toBeTruthy();
});
it('table with a header row', () => {
const blocks = parse('<table><tr><th>H1</th><th>H2</th></tr><tr><td>a</td><td>b</td></tr></table>');
expect(blocks[0].type).toBe('table');
if (blocks[0].type === 'table') {
expect(blocks[0].headerRow).toBe(true);
expect(blocks[0].rows).toHaveLength(2);
expect(blocks[0].rows[0].cells).toHaveLength(2);
}
});
it('bare inline content (no wrapping <p>) becomes a paragraph', () => {
const blocks = parse('plain <b>bold</b> text');
expect(blocks).toHaveLength(1);
expect(blocks[0].type).toBe('paragraph');
});
it('hr → divider; img → image', () => {
expect(parse('<hr>')[0].type).toBe('divider');
const img = parse('<img src="https://example.com/a.png" alt="cat">')[0];
expect(img.type).toBe('image');
if (img.type === 'image') expect(img.src).toBe('https://example.com/a.png');
});
});

@ -0,0 +1,318 @@
/**
* V2 HTML paste parser — a pasted DOM tree → `WordsBlock[]`.
*
* The inverse of `serialize-html.ts`: walks an already-parsed DOM fragment
* and maps standard HTML into the V2 block model — paragraph / heading /
* quote / code / list / divider / image / table — with inline marks
* (bold / italic / underline / strike / code / colour / background) and
* links. URLs are sanitised through `sanitizeWordsUrl`.
*
* Pure DOM-walking: the caller supplies the parsed root (the provider via
* `DOMParser`, tests via happy-dom), so this module never reaches for a
* global `document`. Output feeds `insertBlocks` / `normalizeDocument`,
* which assign ids and coalesce adjacent same-mark text.
*
* Scope: the common tags browsers and editors emit on copy. Unknown block
* tags fall back to a paragraph of their inline content; unknown inline
* tags are transparent (their children are parsed with the same marks).
*/
import {
createCodeBlock,
createDivider,
createHeading,
createImage,
createLink,
createList,
createListItem,
createParagraph,
createQuote,
createTable,
createTableCell,
createTableRow,
createText
} from './factories';
import { sanitizeWordsUrl } from '../normalize';
import type {
ListItem,
TableCell,
TableRow,
WordsBlock,
WordsHeadingLevel,
WordsInline,
WordsListKind,
WordsMark,
WordsText
} from '../types';
const ELEMENT_NODE = 1;
const TEXT_NODE = 3;
const BLOCK_TAGS = new Set([
'p',
'h1',
'h2',
'h3',
'h4',
'h5',
'h6',
'blockquote',
'pre',
'ul',
'ol',
'hr',
'img',
'figure',
'table',
'div',
'section',
'article',
'main',
'header',
'footer',
'aside'
]);
// ── Entry point ──────────────────────────────────────────────────────────
export function parseWordsHtml(root: ParentNode): readonly WordsBlock[] {
const blocks: WordsBlock[] = [];
collectBlocks(root, blocks);
return blocks;
}
// ── Block level ──────────────────────────────────────────────────────────
/**
* Walk `parent`'s children, flushing runs of inline content into paragraphs
* around the block-level elements. Bare inline content (e.g. pasting
* "**a** b" with no wrapping `<p>`) becomes a paragraph.
*/
function collectBlocks(parent: ParentNode, out: WordsBlock[]): void {
let inlineRun: WordsInline[] = [];
const flush = () => {
if (inlineRunHasContent(inlineRun)) out.push(createParagraph(inlineRun));
inlineRun = [];
};
for (const node of Array.from(parent.childNodes)) {
if (node.nodeType === TEXT_NODE) {
const text = node.textContent ?? '';
// Whitespace-only text BETWEEN block elements is layout, not content.
if (text.trim().length > 0) inlineRun.push(createText(collapseWs(text)));
} else if (node.nodeType === ELEMENT_NODE) {
const el = node as Element;
const tag = el.tagName.toLowerCase();
if (BLOCK_TAGS.has(tag)) {
flush();
parseBlockElement(el, tag, out);
} else {
inlineRun.push(...parseInlines(el, []));
}
}
}
flush();
}
function parseBlockElement(el: Element, tag: string, out: WordsBlock[]): void {
switch (tag) {
case 'p':
out.push(createParagraph(parseInlines(el, [])));
return;
case 'h1':
case 'h2':
case 'h3':
case 'h4':
case 'h5':
case 'h6':
out.push(createHeading(headingLevel(tag), parseInlines(el, [])));
return;
case 'blockquote':
out.push(createQuote(parseInlines(el, [])));
return;
case 'pre':
out.push(createCodeBlock([createText(el.textContent ?? '')]));
return;
case 'ul':
out.push(createList('unordered', parseListItems(el, 'unordered')));
return;
case 'ol':
out.push(createList('ordered', parseListItems(el, 'ordered')));
return;
case 'hr':
out.push(createDivider());
return;
case 'img': {
const img = parseImage(el);
if (img) out.push(img);
return;
}
case 'figure': {
const inner = el.querySelector('img');
if (inner) {
const img = parseImage(inner);
if (img) out.push(img);
} else {
collectBlocks(el, out);
}
return;
}
case 'table': {
const table = parseTable(el);
if (table) out.push(table);
return;
}
default:
// div / section / article / … — unwrap structural containers.
collectBlocks(el, out);
return;
}
}
function parseListItems(listEl: Element, kind: WordsListKind): readonly ListItem[] {
const items: ListItem[] = [];
for (const li of Array.from(listEl.children)) {
if (li.tagName.toLowerCase() !== 'li') continue;
// Checkbox lists: GitHub / editors emit `<li><input type=checkbox>…`.
const checkbox = li.querySelector('input[type="checkbox"]');
const checked = checkbox ? (checkbox as HTMLInputElement).checked : undefined;
const inlines = parseInlines(li, []).filter((n) => !isEmptyText(n));
items.push(
createListItem(inlines, kind === 'check' || checkbox ? { checked: checked ?? false } : {})
);
}
return items;
}
function parseImage(el: Element): WordsBlock | undefined {
const src = sanitizeWordsUrl(el.getAttribute('src') ?? '');
if (!src) return undefined;
const alt = el.getAttribute('alt');
return createImage({ src, ...(alt ? { alt } : {}) });
}
function parseTable(tableEl: Element): WordsBlock | undefined {
const rows: TableRow[] = [];
let headerRow = false;
// Rows from thead/tbody/tfoot or direct <tr>.
const trs = Array.from(tableEl.querySelectorAll('tr'));
trs.forEach((tr, rowIdx) => {
const cells: TableCell[] = [];
for (const cell of Array.from(tr.children)) {
const ct = cell.tagName.toLowerCase();
if (ct !== 'td' && ct !== 'th') continue;
if (rowIdx === 0 && ct === 'th') headerRow = true;
// Cells hold blocks (since P5m). Parse the cell's content; seed a
// paragraph when it has only inline text.
const cellBlocks: WordsBlock[] = [];
collectBlocks(cell, cellBlocks);
cells.push(createTableCell(cellBlocks.length > 0 ? cellBlocks : [createParagraph()]));
}
if (cells.length > 0) rows.push(createTableRow(cells));
});
if (rows.length === 0) return undefined;
return createTable(rows, headerRow ? { headerRow: true } : {});
}
// ── Inline level ─────────────────────────────────────────────────────────
/** Parse `parent`'s inline content, accumulating marks down the tree. */
function parseInlines(parent: ParentNode, marks: readonly WordsMark[]): WordsInline[] {
const result: WordsInline[] = [];
for (const node of Array.from(parent.childNodes)) {
if (node.nodeType === TEXT_NODE) {
const text = collapseWs(node.textContent ?? '');
if (text.length > 0) result.push(createText(text, marks));
} else if (node.nodeType === ELEMENT_NODE) {
const el = node as Element;
const tag = el.tagName.toLowerCase();
if (tag === 'br') {
// V2 has no inline break node; a soft break reads as a space.
result.push(createText(' ', marks));
continue;
}
if (tag === 'a') {
const href = sanitizeWordsUrl(el.getAttribute('href') ?? '');
const texts = inlinesToText(parseInlines(el, marks));
if (href && texts.length > 0) result.push(createLink(href, texts));
else result.push(...parseInlines(el, marks));
continue;
}
result.push(...parseInlines(el, marksForElement(tag, el, marks)));
}
}
return result;
}
/** Marks contributed by an inline element (tag semantics + inline style). */
function marksForElement(
tag: string,
el: Element,
marks: readonly WordsMark[]
): readonly WordsMark[] {
const next: WordsMark[] = [...marks];
if (tag === 'strong' || tag === 'b') next.push('bold');
else if (tag === 'em' || tag === 'i' || tag === 'cite') next.push('italic');
else if (tag === 'u' || tag === 'ins') next.push('underline');
else if (tag === 's' || tag === 'strike' || tag === 'del') next.push('strike');
else if (tag === 'code' || tag === 'kbd' || tag === 'samp') next.push('code');
else if (tag === 'mark') next.push({ type: 'background', value: readColor(el, 'background-color') ?? '#ffc53d' });
// Inline style — covers `<span style="font-weight:bold; color:…">` etc.
const style = (el as Partial<HTMLElement>).style;
if (style) {
const weight = style.fontWeight;
if (weight === 'bold' || /^[6-9]00$/.test(weight)) next.push('bold');
if (style.fontStyle === 'italic') next.push('italic');
const deco = `${style.textDecorationLine || ''} ${style.textDecoration || ''}`;
if (deco.includes('underline')) next.push('underline');
if (deco.includes('line-through')) next.push('strike');
}
const color = readColor(el, 'color');
if (color) next.push({ type: 'color', value: color });
if (tag !== 'mark') {
const bg = readColor(el, 'background-color');
if (bg) next.push({ type: 'background', value: bg });
}
return next;
}
// ── Small helpers ──────────────────────────────────────────────────────────
function headingLevel(tag: string): WordsHeadingLevel {
// h1-h3 map directly; h4-h6 collapse to h3 (the model caps at 3).
const n = Number(tag.slice(1));
return (n <= 1 ? 1 : n === 2 ? 2 : 3) as WordsHeadingLevel;
}
/** Collapse runs of whitespace (incl. newlines) to single spaces, HTML-style. */
function collapseWs(text: string): string {
return text.replace(/\s+/g, ' ');
}
function readColor(el: Element, prop: 'color' | 'background-color'): string | undefined {
const style = (el as Partial<HTMLElement>).style;
if (!style) return undefined;
const v = style.getPropertyValue(prop).trim();
if (!v || v === 'inherit' || v === 'initial' || v === 'transparent' || v === 'currentcolor') {
return undefined;
}
return v;
}
/** Coerce parsed inline content to `WordsText[]` for a link's children. */
function inlinesToText(inlines: readonly WordsInline[]): WordsText[] {
const texts: WordsText[] = [];
for (const n of inlines) {
if (n.type === 'text') texts.push(n);
else if (n.type === 'link') for (const c of n.children) texts.push(c);
}
return texts.filter((t) => t.text.length > 0);
}
function isEmptyText(n: WordsInline): boolean {
return n.type === 'text' && n.text.trim().length === 0;
}
function inlineRunHasContent(run: readonly WordsInline[]): boolean {
return run.some((n) => (n.type === 'text' ? n.text.trim().length > 0 : true));
}

@ -2258,10 +2258,15 @@ describe('WordsProvider', () => {
return { content }; return { content };
}); });
// Rich paste now PARSES real HTML (`<p>…</p>` → a paragraph), so the
// "unsupported-html" log only fires for HTML that carries no extractable
// content AND no plain-text fallback — e.g. the bare clipboard fragment
// markers Office / Windows wrap an empty copy in.
result.content.props.onpaste({ result.content.props.onpaste({
currentTarget: contentEl, currentTarget: contentEl,
clipboardData: { clipboardData: {
getData: (type: string) => (type === 'text/html' ? '<p>unsafe</p>' : '') getData: (type: string) =>
type === 'text/html' ? '<!--StartFragment--><!--EndFragment-->' : ''
}, },
preventDefault: vi.fn() preventDefault: vi.fn()
} as never); } as never);

@ -58,6 +58,7 @@ import {
getLinkAtSelection, getLinkAtSelection,
getTextEntries, getTextEntries,
isFindQueryValid, isFindQueryValid,
parseWordsHtml,
parseWordsPlainText, parseWordsPlainText,
redoWords, redoWords,
renderWordsDomHtml, renderWordsDomHtml,
@ -80,6 +81,7 @@ import { serializeHtml } from './engine/serialize-html';
import { serializeMarkdown } from './engine/serialize-markdown'; import { serializeMarkdown } from './engine/serialize-markdown';
import type { import type {
WordsDocument, WordsDocument,
WordsBlock,
ImageBlock, ImageBlock,
WordsMark, WordsMark,
TableCell, TableCell,
@ -1175,14 +1177,23 @@ export class WordsProvider {
const plainText = e.clipboardData?.getData('text/plain') ?? ''; const plainText = e.clipboardData?.getData('text/plain') ?? '';
const html = e.clipboardData?.getData('text/html') ?? ''; const html = e.clipboardData?.getData('text/html') ?? '';
const action = actionFromPaste({
plainText,
html
});
e.preventDefault(); e.preventDefault();
this.syncSelectionFromDom({ source: 'input' }); this.syncSelectionFromDom({ source: 'input' });
// A single pasted URL → a link over the selection / at the caret.
const pastedUrl = normalizedSinglePastedUrl(plainText); const pastedUrl = normalizedSinglePastedUrl(plainText);
if (pastedUrl && this.applyCommand({ type: 'insertLink', href: pastedUrl })) return; if (pastedUrl && this.applyCommand({ type: 'insertLink', href: pastedUrl })) return;
// Rich paste: parse the clipboard HTML into V2 blocks + inline marks
// and insert them. Falls through to plain text when there's no usable
// HTML (or it parses to nothing).
if (html.trim().length > 0) {
const blocks = this.parseClipboardHtml(html, e.currentTarget);
if (blocks.length > 0 && this.applyCommand({ type: 'insertBlocks', blocks })) return;
}
// Plain-text fallback.
const action = actionFromPaste({ plainText, html });
if (action.type === 'ignore') { if (action.type === 'ignore') {
this.signalInvalidInput(action.reason, e.currentTarget); this.signalInvalidInput(action.reason, e.currentTarget);
return; return;
@ -1190,6 +1201,23 @@ export class WordsProvider {
this.applyCommand(action.command); this.applyCommand(action.command);
}; };
/**
* Parse clipboard `text/html` into V2 blocks. Parses inside an INERT
* document (`implementation.createHTMLDocument` off the active dom) so no
* scripts run and no `<img>` / resources load while parsing — the safe
* way to read foreign HTML. Returns [] on any failure.
*/
private parseClipboardHtml(html: string, target: HTMLElement): readonly WordsBlock[] {
try {
const doc = this.soma.dom.getDocument(target);
const inert = doc.implementation.createHTMLDocument('');
inert.body.innerHTML = html;
return parseWordsHtml(inert.body);
} catch {
return [];
}
}
readonly ondragover = (e: DragEvent & { currentTarget: HTMLElement }) => { readonly ondragover = (e: DragEvent & { currentTarget: HTMLElement }) => {
// Only intercept drags carrying image files; let everything else // Only intercept drags carrying image files; let everything else
// (text drags, internal selection drags) follow the browser // (text drags, internal selection drags) follow the browser

Loading…
Cancel
Save

Powered by TurnKey Linux.