From 65883089b3e0bc9077e83f6253824eaed3aae37c Mon Sep 17 00:00:00 2001 From: dev Date: Tue, 2 Jun 2026 23:19:16 +0200 Subject: [PATCH] =?UTF-8?q?feat(words):=20rich=20HTML=20paste=20=E2=80=94?= =?UTF-8?q?=20clipboard=20HTML=20becomes=20blocks=20+=20marks?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Paste previously flattened everything to plain text (the signed import doctrine deferred rich paste). Now the editor parses clipboard text/html into the V2 model, the inverse of serialize-html.ts: - parse-html.ts: parseWordsHtml(domRoot) -> WordsBlock[]. Maps p / h1-6 / blockquote / pre / ul-ol-li (incl. checkboxes) / hr / img / figure / table, with inline marks (bold/italic/underline/strike/code, colour + background from style) and links. URLs sanitised; unknown blocks fall back to a paragraph, unknown inline tags are transparent. - insert-blocks.ts: insertBlocks(state, blocks) drops blocks at the caret. Single pasted paragraph inline-merges (a phrase stays in the sentence); multiple/block-level paste splits the host; a nested or non-inline caret degrades to a structure-preserving plain-text insert; a trailing paragraph guarantees a caret home after a terminal table/image. New insertBlocks command + dispatch. - provider onpaste: parse text/html in an INERT document (createHTMLDocument off the active dom — no scripts run, no resources load), then applyCommand insertBlocks. Image-files / single-URL / plain-text paths unchanged; plain text is still the fallback. Tests: insert-blocks 6/6 (merge/split/replace/append/no-op), parse-html 10/10 browser (marks, lists, tables, colour, javascript: URL sanitised). Browser-verified: pasting an h2 + a bold/italic paragraph + a list inserts the heading, the formatted paragraph and the list at the caret. Updated the provider's unsupported-paste test (real HTML now parses; only content-free fragments log unsupported-html). Co-Authored-By: Claude Opus 4.8 (1M context) --- .../words/engine/operations/commands.ts | 10 + .../words/engine/operations/index.ts | 3 + .../engine/operations/insert-blocks.test.ts | 94 ++++++ .../words/engine/operations/insert-blocks.ts | 232 +++++++++++++ .../operations/parse-html.svelte.test.ts | 110 ++++++ .../words/engine/operations/parse-html.ts | 318 ++++++++++++++++++ .../words/words-provider.svelte.test.ts | 7 +- .../components/words/words-provider.svelte.ts | 36 +- 8 files changed, 805 insertions(+), 5 deletions(-) create mode 100644 src/uix/soma/components/words/engine/operations/insert-blocks.test.ts create mode 100644 src/uix/soma/components/words/engine/operations/insert-blocks.ts create mode 100644 src/uix/soma/components/words/engine/operations/parse-html.svelte.test.ts create mode 100644 src/uix/soma/components/words/engine/operations/parse-html.ts diff --git a/src/uix/soma/components/words/engine/operations/commands.ts b/src/uix/soma/components/words/engine/operations/commands.ts index 16118ee96..6b7d843b2 100644 --- a/src/uix/soma/components/words/engine/operations/commands.ts +++ b/src/uix/soma/components/words/engine/operations/commands.ts @@ -66,6 +66,7 @@ import { replaceDocument, setCodeLanguage } from './insert-block-types'; +import { insertBlocks } from './insert-blocks'; import type { Block, WordsBlock, @@ -146,6 +147,13 @@ export type WordsCommand = readonly blockIndex: number; readonly block: Readonly>; } + // Insert a list of blocks at the caret (rich paste — `parseWordsHtml` + // output). Splits the host, inline-merges a single paragraph, degrades a + // nested caret to a plain-text insert. + | { + readonly type: 'insertBlocks'; + readonly blocks: readonly WordsBlock[]; + } // Caretless insert into a column slot (the `+` overlay in each // column). Carries the new block AND its post-insert selection in // one transaction — see `insertBlockInColumn` for the placement @@ -282,6 +290,8 @@ export function applyWordsCommand( command.blockIndex, command.block as unknown as WordsBlock ); + case 'insertBlocks': + return insertBlocks(state, command.blocks); case 'insertBlockInColumn': return insertBlockInColumn( state, diff --git a/src/uix/soma/components/words/engine/operations/index.ts b/src/uix/soma/components/words/engine/operations/index.ts index af53f15ba..68284fa11 100644 --- a/src/uix/soma/components/words/engine/operations/index.ts +++ b/src/uix/soma/components/words/engine/operations/index.ts @@ -224,6 +224,9 @@ export { // ── Plain-text serialization ───────────────────────────────────────────── export { parseWordsPlainText, renderWordsPlainText } from './serialize-text'; +// ── HTML paste parsing (clipboard HTML → blocks) ───────────────────────── +export { parseWordsHtml } from './parse-html'; + // ── DOM HTML emission ──────────────────────────────────────────────────── export { renderWordsDomHtml, renderWordsDomHtmlPretty } from './render-dom'; diff --git a/src/uix/soma/components/words/engine/operations/insert-blocks.test.ts b/src/uix/soma/components/words/engine/operations/insert-blocks.test.ts new file mode 100644 index 000000000..d487cb8fd --- /dev/null +++ b/src/uix/soma/components/words/engine/operations/insert-blocks.test.ts @@ -0,0 +1,94 @@ +/** + * insertBlocks — the rich-paste sink. Pure (no DOM): blocks come from + * factories, the op drops them at the caret. + */ + +import { describe, expect, it } from 'vitest'; +import { insertBlocks } from './insert-blocks'; +import { createHeading, createParagraph, createText } from './factories'; +import { createState } from './text'; +import { createCollapsedSelection } from '../selection'; +import { WORDS_VERSION, type WordsBlock, type WordsDocument } from '../types'; + +function doc(...children: WordsBlock[]): WordsDocument { + return { version: WORDS_VERSION, children }; +} +const para = (text: string) => createParagraph([createText(text)]); + +describe('insertBlocks (rich paste)', () => { + it('inline-merges a single pasted paragraph into the host (marks kept)', () => { + const d = doc(para('ab')); + // caret between a|b + const sel = createCollapsedSelection([0, 0], 1); + const pasted = [createParagraph([createText('X', ['bold'])])]; + const r = insertBlocks(createState(d, sel), pasted); + expect(r.changed).toBe(true); + // Still ONE block — the phrase stayed in the sentence. + expect(r.state.document.children).toHaveLength(1); + const b = r.state.document.children[0]; + if (b.type !== 'paragraph') throw new Error('expected paragraph'); + expect(b.children.map((c) => (c.type === 'text' ? c.text : ''))).toEqual(['a', 'X', 'b']); + // The X carries bold. + const x = b.children[1]; + expect(x.type === 'text' && x.marks).toContain('bold'); + }); + + it('splits the host and inserts multiple blocks between the halves', () => { + const d = doc(para('hello world')); + // caret after "hello " (offset 6) + const sel = createCollapsedSelection([0, 0], 6); + const pasted = [createHeading(1, [createText('Title')]), para('Body')]; + const r = insertBlocks(createState(d, sel), pasted); + expect(r.changed).toBe(true); + const types = r.state.document.children.map((b) => b.type); + // before("hello ") · heading · paragraph("Body") · after("world") + expect(types).toEqual(['paragraph', 'heading', 'paragraph', 'paragraph']); + const texts = r.state.document.children.map((b) => + b.type === 'paragraph' || b.type === 'heading' + ? b.children.map((c) => (c.type === 'text' ? c.text : '')).join('') + : '' + ); + expect(texts).toEqual(['hello ', 'Title', 'Body', 'world']); + }); + + it('replaces a non-collapsed selection before inserting', () => { + const d = doc(para('abcde')); + // select "bcd" (offset 1..4) + const sel = { + anchor: { path: [0, 0], offset: 1 }, + focus: { path: [0, 0], offset: 4 } + }; + const r = insertBlocks(createState(d, sel), [createParagraph([createText('X')])]); + expect(r.changed).toBe(true); + expect(r.state.document.children).toHaveLength(1); + const b = r.state.document.children[0]; + if (b.type !== 'paragraph') throw new Error('expected paragraph'); + expect(b.children.map((c) => (c.type === 'text' ? c.text : '')).join('')).toBe('aXe'); + }); + + it('replaces an empty host paragraph with the pasted block', () => { + const d = doc(createParagraph([createText('')])); + const sel = createCollapsedSelection([0, 0], 0); + const r = insertBlocks(createState(d, sel), [createHeading(2, [createText('H')])]); + expect(r.changed).toBe(true); + expect(r.state.document.children).toHaveLength(1); + expect(r.state.document.children[0].type).toBe('heading'); + }); + + it('appends at the document end when there is no selection', () => { + const d = doc(para('one')); + const r = insertBlocks(createState(d, null), [para('two'), para('three')]); + expect(r.changed).toBe(true); + expect(r.state.document.children.map((b) => b.type)).toEqual([ + 'paragraph', + 'paragraph', + 'paragraph' + ]); + }); + + it('no-ops on empty input', () => { + const d = doc(para('x')); + const r = insertBlocks(createState(d, createCollapsedSelection([0, 0], 0)), []); + expect(r.changed).toBe(false); + }); +}); diff --git a/src/uix/soma/components/words/engine/operations/insert-blocks.ts b/src/uix/soma/components/words/engine/operations/insert-blocks.ts new file mode 100644 index 000000000..c4fb53f91 --- /dev/null +++ b/src/uix/soma/components/words/engine/operations/insert-blocks.ts @@ -0,0 +1,232 @@ +/** + * V2 `insertBlocks` — insert a list of blocks at the selection. + * + * The paste sink: `parseWordsHtml` turns clipboard HTML into `WordsBlock[]`, + * this op drops them into the document at the caret. + * + * Behaviour: + * - A non-empty selection is deleted first (paste replaces it). + * - No caret → append at the document end. + * - Caret in a TOP-LEVEL inline block (paragraph / heading / quote): + * · a single pasted paragraph splices inline (a pasted phrase stays + * in the sentence); + * · otherwise the host splits at the caret and the blocks land + * between the two halves. + * - Caret NESTED (inside a column / table cell) or in a non-inline host: + * degrade to a structure-preserving plain-text insert (marks/blocks + * flatten, nothing breaks). Top-level rich paste is the 99% path. + * + * The caret lands at the end of the pasted content; when the paste ends in + * a non-inline block (table / image) a trailing paragraph is added so the + * caret always has a home. + */ + +import { createCollapsedSelection, normalizeRange } from '../selection'; +import { createParagraph, createText } from './factories'; +import { deleteRange } from './delete'; +import { editingContainerPath, pointFromInlineTextOffset, replaceAt } from './helpers'; +import { splitContainerAtPoint } from './inline-split'; +import { insertText } from './text'; +import { normalizeDocument } from './normalize'; +import { getActiveMarksForSelection } from './selection-walkers'; +import { changed, noOp, type WordsEditorState, type WordsOperationResult } from './types'; +import type { WordsBlock, WordsDocument, WordsInline } from '../types'; + +const INLINE_BLOCKS = new Set(['paragraph', 'heading', 'quote']); + +export function insertBlocks( + state: WordsEditorState, + blocks: readonly WordsBlock[] +): WordsOperationResult { + if (blocks.length === 0) return noOp(state); + + // 1. Replace a non-empty selection. + let cur = state; + if (cur.selection && !normalizeRange(cur.selection).collapsed) { + cur = deleteRange(cur).state; + } + + const doc = cur.document; + const sel = cur.selection; + + // 2. No caret → append at the end. + if (!sel) { + const children = [...doc.children, ...blocks]; + return commit(doc, children, doc.children.length + blocks.length - 1); + } + + const point = normalizeRange(sel).start; + const containerPath = editingContainerPath(doc, point.path); + + // 3. Caret on an atomic block (no editable container) → insert after it. + if (!containerPath) { + const at = (point.path[0] ?? doc.children.length - 1) + 1; + const children = [...doc.children.slice(0, at), ...blocks, ...doc.children.slice(at)]; + return commit(doc, children, at + blocks.length - 1); + } + + const topIndex = containerPath[0] ?? 0; + const host = doc.children[topIndex]; + const hostInline = !!host && INLINE_BLOCKS.has(host.type); + + // 4. Nested caret OR non-inline host → degrade to a plain-text insert that + // splices safely at any depth (reuses the tested insertText path). + if (containerPath.length > 1 || !hostInline) { + return insertText(cur, flattenToPlainText(blocks)); + } + + // 5. Top-level inline host — split its inlines at the caret. + const split = splitContainerAtPoint(doc, containerPath, point); + + // 5a. A single pasted paragraph → splice its inlines inline (one block). + if (blocks.length === 1 && blocks[0].type === 'paragraph') { + const inlines = [...split.before, ...blocks[0].children, ...split.after]; + const merged = withInlines(host, inlines); + const children = replaceAt(doc.children, topIndex, merged); + const caretOffset = inlineTextLen(split.before) + inlineTextLen(blocks[0].children); + return commitAt(doc, children, topIndex, caretOffset); + } + + // 5b. Multiple / block-level paste → split host, blocks between the halves. + const beforeBlocks = inlineRunHasContent(split.before) ? [withInlines(host, split.before)] : []; + const afterBlocks = inlineRunHasContent(split.after) ? [createParagraph(split.after)] : []; + // Guarantee a caret home when the paste ends in a non-inline block. + const lastPasted = blocks[blocks.length - 1]; + const needsTrailing = afterBlocks.length === 0 && !INLINE_BLOCKS.has(lastPasted.type); + const trailing = needsTrailing ? [createParagraph()] : []; + const mid = [...beforeBlocks, ...blocks, ...afterBlocks, ...trailing]; + const children = [ + ...doc.children.slice(0, topIndex), + ...mid, + ...doc.children.slice(topIndex + 1) + ]; + // Caret: end of the last inline-bearing pasted block, else the trailing / + // after paragraph. + const lastPastedIndex = topIndex + beforeBlocks.length + blocks.length - 1; + const caretIndex = + afterBlocks.length > 0 || trailing.length > 0 ? lastPastedIndex + 1 : lastPastedIndex; + return commit(doc, children, caretIndex); +} + +// ── commit helpers ───────────────────────────────────────────────────────── + +/** Normalize + place the caret at the END of the block at `caretIndex` + * (or the nearest inline-bearing block at/after it). */ +function commit( + doc: WordsDocument, + children: readonly WordsBlock[], + caretIndex: number +): WordsOperationResult { + const normalized = normalizeDocument({ ...doc, children }).document; + const point = safeCaret(normalized, caretIndex); + const selection = createCollapsedSelection(point.path, point.offset); + return changed({ + document: normalized, + selection, + activeMarks: getActiveMarksForSelection(normalized, selection) + }); +} + +/** Normalize + place the caret inside the block at `blockIndex` at a known + * character offset (used by the single-paragraph inline-merge). */ +function commitAt( + doc: WordsDocument, + children: readonly WordsBlock[], + blockIndex: number, + offset: number +): WordsOperationResult { + const normalized = normalizeDocument({ ...doc, children }).document; + const point = pointFromInlineTextOffset(normalized, [blockIndex], offset); + const selection = createCollapsedSelection(point.path, point.offset); + return changed({ + document: normalized, + selection, + activeMarks: getActiveMarksForSelection(normalized, selection) + }); +} + +/** A caret at the end of the inline-bearing block at/after `preferred`, + * else the nearest one before it, else the document start. */ +function safeCaret(doc: WordsDocument, preferred: number): { path: number[]; offset: number } { + const n = doc.children.length; + for (let i = Math.max(0, preferred); i < n; i++) { + const b = doc.children[i]; + if (b && INLINE_BLOCKS.has(b.type)) { + const off = i === preferred ? blockTextLen(b) : 0; + const p = pointFromInlineTextOffset(doc, [i], off); + return { path: [...p.path], offset: p.offset }; + } + } + for (let i = Math.min(preferred, n - 1); i >= 0; i--) { + const b = doc.children[i]; + if (b && INLINE_BLOCKS.has(b.type)) { + const p = pointFromInlineTextOffset(doc, [i], blockTextLen(b)); + return { path: [...p.path], offset: p.offset }; + } + } + const p = pointFromInlineTextOffset(doc, [0], 0); + return { path: [...p.path], offset: p.offset }; +} + +// ── small helpers ──────────────────────────────────────────────────────── + +function withInlines(block: WordsBlock, inlines: readonly WordsInline[]): WordsBlock { + const safe = inlines.length > 0 ? inlines : [createText('')]; + return { ...block, children: safe } as WordsBlock; +} + +function inlineTextLen(inlines: readonly WordsInline[]): number { + let n = 0; + for (const node of inlines) { + if (node.type === 'text') n += node.text.length; + else if (node.type === 'link') for (const c of node.children) n += c.text.length; + } + return n; +} + +function blockTextLen(block: WordsBlock): number { + if (block.type === 'paragraph' || block.type === 'heading' || block.type === 'quote') { + return inlineTextLen(block.children); + } + return 0; +} + +function inlineRunHasContent(run: readonly WordsInline[]): boolean { + return run.some((n) => (n.type === 'text' ? n.text.trim().length > 0 : true)); +} + +/** Flatten blocks to plain text (used for the nested-caret degrade path). */ +function flattenToPlainText(blocks: readonly WordsBlock[]): string { + return blocks.map(blockToPlainText).join('\n'); +} + +function blockToPlainText(block: WordsBlock): string { + switch (block.type) { + case 'paragraph': + case 'heading': + case 'quote': + return inlinesToPlainText(block.children); + case 'code': + return block.children.map((c) => c.text).join(''); + case 'list': + return block.items.map((i) => inlinesToPlainText(i.children)).join('\n'); + case 'callout': + return block.children.map(blockToPlainText).join('\n'); + case 'columns': + return block.columns.map((c) => c.children.map(blockToPlainText).join('\n')).join('\n'); + case 'table': + return block.rows + .map((r) => r.cells.map((c) => c.children.map(blockToPlainText).join(' ')).join('\t')) + .join('\n'); + default: + return ''; + } +} + +function inlinesToPlainText(inlines: readonly WordsInline[]): string { + return inlines + .map((n) => + n.type === 'text' ? n.text : n.type === 'link' ? n.children.map((c) => c.text).join('') : '' + ) + .join(''); +} diff --git a/src/uix/soma/components/words/engine/operations/parse-html.svelte.test.ts b/src/uix/soma/components/words/engine/operations/parse-html.svelte.test.ts new file mode 100644 index 000000000..c4f595871 --- /dev/null +++ b/src/uix/soma/components/words/engine/operations/parse-html.svelte.test.ts @@ -0,0 +1,110 @@ +/** + * parseWordsHtml — clipboard HTML → blocks. Runs as a browser test + * (`*.svelte.test.ts`) because it walks a real DOM tree. + */ + +import { describe, expect, it } from 'vitest'; +import { parseWordsHtml } from './parse-html'; +import type { WordsBlock, WordsInline, WordsMark } from '../types'; + +function parse(html: string): readonly WordsBlock[] { + const div = document.createElement('div'); + div.innerHTML = html; + return parseWordsHtml(div); +} +function text(n: WordsInline): string { + return n.type === 'text' ? n.text : n.type === 'link' ? n.children.map((c) => c.text).join('') : ''; +} +function marksOf(n: WordsInline): readonly WordsMark[] { + return n.type === 'text' ? (n.marks ?? []) : []; +} + +describe('parseWordsHtml', () => { + it('paragraph with a bold run', () => { + const blocks = parse('

Hello world

'); + expect(blocks).toHaveLength(1); + const p = blocks[0]; + if (p.type !== 'paragraph') throw new Error('expected paragraph'); + expect(p.children.map(text).join('')).toBe('Hello world'); + const bold = p.children.find((c) => text(c) === 'world'); + expect(bold && marksOf(bold)).toContain('bold'); + }); + + it('headings (h1-h3 direct, h4+ clamp to 3)', () => { + const blocks = parse('

One

Two

Five
'); + expect(blocks.map((b) => b.type)).toEqual(['heading', 'heading', 'heading']); + expect(blocks.map((b) => (b.type === 'heading' ? b.level : 0))).toEqual([1, 2, 3]); + }); + + it('nested marks compose (italic + bold)', () => { + const blocks = parse('

bi

'); + const p = blocks[0]; + if (p.type !== 'paragraph') throw new Error('expected paragraph'); + const node = p.children[0]; + expect(text(node)).toBe('bi'); + expect(marksOf(node)).toEqual(expect.arrayContaining(['bold', 'italic'])); + }); + + it('unordered + ordered lists', () => { + const ul = parse('
  • a
  • b
'); + expect(ul[0].type).toBe('list'); + if (ul[0].type === 'list') { + expect(ul[0].kind).toBe('unordered'); + expect(ul[0].items.map((i) => i.children.map(text).join(''))).toEqual(['a', 'b']); + } + const ol = parse('
  1. x
'); + expect(ol[0].type === 'list' && ol[0].kind).toBe('ordered'); + }); + + it('links (sanitised href)', () => { + const blocks = parse('

see here

'); + const p = blocks[0]; + if (p.type !== 'paragraph') throw new Error('expected paragraph'); + const link = p.children.find((c) => c.type === 'link'); + expect(link?.type).toBe('link'); + expect(link?.type === 'link' && link.href).toBe('https://example.com'); + // A javascript: URL is dropped (sanitised) → no link, just its text. + const evil = parse('

x

'); + const ep = evil[0]; + expect(ep.type === 'paragraph' && ep.children.every((c) => c.type !== 'link')).toBe(true); + }); + + it('quote + code', () => { + expect(parse('
q
')[0].type).toBe('quote'); + const code = parse('
const x = 1
')[0]; + expect(code.type).toBe('code'); + if (code.type === 'code') expect(code.children.map((c) => c.text).join('')).toBe('const x = 1'); + }); + + it('inline colour from style', () => { + const blocks = parse('

red

'); + const p = blocks[0]; + if (p.type !== 'paragraph') throw new Error('expected paragraph'); + const node = p.children.find((c) => text(c) === 'red'); + const colour = marksOf(node!).find((m) => typeof m === 'object' && m.type === 'color'); + expect(colour).toBeTruthy(); + }); + + it('table with a header row', () => { + const blocks = parse('
H1H2
ab
'); + expect(blocks[0].type).toBe('table'); + if (blocks[0].type === 'table') { + expect(blocks[0].headerRow).toBe(true); + expect(blocks[0].rows).toHaveLength(2); + expect(blocks[0].rows[0].cells).toHaveLength(2); + } + }); + + it('bare inline content (no wrapping

) becomes a paragraph', () => { + const blocks = parse('plain bold text'); + expect(blocks).toHaveLength(1); + expect(blocks[0].type).toBe('paragraph'); + }); + + it('hr → divider; img → image', () => { + expect(parse('


')[0].type).toBe('divider'); + const img = parse('cat')[0]; + expect(img.type).toBe('image'); + if (img.type === 'image') expect(img.src).toBe('https://example.com/a.png'); + }); +}); diff --git a/src/uix/soma/components/words/engine/operations/parse-html.ts b/src/uix/soma/components/words/engine/operations/parse-html.ts new file mode 100644 index 000000000..4d1bc68b5 --- /dev/null +++ b/src/uix/soma/components/words/engine/operations/parse-html.ts @@ -0,0 +1,318 @@ +/** + * V2 HTML paste parser — a pasted DOM tree → `WordsBlock[]`. + * + * The inverse of `serialize-html.ts`: walks an already-parsed DOM fragment + * and maps standard HTML into the V2 block model — paragraph / heading / + * quote / code / list / divider / image / table — with inline marks + * (bold / italic / underline / strike / code / colour / background) and + * links. URLs are sanitised through `sanitizeWordsUrl`. + * + * Pure DOM-walking: the caller supplies the parsed root (the provider via + * `DOMParser`, tests via happy-dom), so this module never reaches for a + * global `document`. Output feeds `insertBlocks` / `normalizeDocument`, + * which assign ids and coalesce adjacent same-mark text. + * + * Scope: the common tags browsers and editors emit on copy. Unknown block + * tags fall back to a paragraph of their inline content; unknown inline + * tags are transparent (their children are parsed with the same marks). + */ + +import { + createCodeBlock, + createDivider, + createHeading, + createImage, + createLink, + createList, + createListItem, + createParagraph, + createQuote, + createTable, + createTableCell, + createTableRow, + createText +} from './factories'; +import { sanitizeWordsUrl } from '../normalize'; +import type { + ListItem, + TableCell, + TableRow, + WordsBlock, + WordsHeadingLevel, + WordsInline, + WordsListKind, + WordsMark, + WordsText +} from '../types'; + +const ELEMENT_NODE = 1; +const TEXT_NODE = 3; + +const BLOCK_TAGS = new Set([ + 'p', + 'h1', + 'h2', + 'h3', + 'h4', + 'h5', + 'h6', + 'blockquote', + 'pre', + 'ul', + 'ol', + 'hr', + 'img', + 'figure', + 'table', + 'div', + 'section', + 'article', + 'main', + 'header', + 'footer', + 'aside' +]); + +// ── Entry point ────────────────────────────────────────────────────────── + +export function parseWordsHtml(root: ParentNode): readonly WordsBlock[] { + const blocks: WordsBlock[] = []; + collectBlocks(root, blocks); + return blocks; +} + +// ── Block level ────────────────────────────────────────────────────────── + +/** + * Walk `parent`'s children, flushing runs of inline content into paragraphs + * around the block-level elements. Bare inline content (e.g. pasting + * "**a** b" with no wrapping `

`) becomes a paragraph. + */ +function collectBlocks(parent: ParentNode, out: WordsBlock[]): void { + let inlineRun: WordsInline[] = []; + const flush = () => { + if (inlineRunHasContent(inlineRun)) out.push(createParagraph(inlineRun)); + inlineRun = []; + }; + for (const node of Array.from(parent.childNodes)) { + if (node.nodeType === TEXT_NODE) { + const text = node.textContent ?? ''; + // Whitespace-only text BETWEEN block elements is layout, not content. + if (text.trim().length > 0) inlineRun.push(createText(collapseWs(text))); + } else if (node.nodeType === ELEMENT_NODE) { + const el = node as Element; + const tag = el.tagName.toLowerCase(); + if (BLOCK_TAGS.has(tag)) { + flush(); + parseBlockElement(el, tag, out); + } else { + inlineRun.push(...parseInlines(el, [])); + } + } + } + flush(); +} + +function parseBlockElement(el: Element, tag: string, out: WordsBlock[]): void { + switch (tag) { + case 'p': + out.push(createParagraph(parseInlines(el, []))); + return; + case 'h1': + case 'h2': + case 'h3': + case 'h4': + case 'h5': + case 'h6': + out.push(createHeading(headingLevel(tag), parseInlines(el, []))); + return; + case 'blockquote': + out.push(createQuote(parseInlines(el, []))); + return; + case 'pre': + out.push(createCodeBlock([createText(el.textContent ?? '')])); + return; + case 'ul': + out.push(createList('unordered', parseListItems(el, 'unordered'))); + return; + case 'ol': + out.push(createList('ordered', parseListItems(el, 'ordered'))); + return; + case 'hr': + out.push(createDivider()); + return; + case 'img': { + const img = parseImage(el); + if (img) out.push(img); + return; + } + case 'figure': { + const inner = el.querySelector('img'); + if (inner) { + const img = parseImage(inner); + if (img) out.push(img); + } else { + collectBlocks(el, out); + } + return; + } + case 'table': { + const table = parseTable(el); + if (table) out.push(table); + return; + } + default: + // div / section / article / … — unwrap structural containers. + collectBlocks(el, out); + return; + } +} + +function parseListItems(listEl: Element, kind: WordsListKind): readonly ListItem[] { + const items: ListItem[] = []; + for (const li of Array.from(listEl.children)) { + if (li.tagName.toLowerCase() !== 'li') continue; + // Checkbox lists: GitHub / editors emit `

  • …`. + const checkbox = li.querySelector('input[type="checkbox"]'); + const checked = checkbox ? (checkbox as HTMLInputElement).checked : undefined; + const inlines = parseInlines(li, []).filter((n) => !isEmptyText(n)); + items.push( + createListItem(inlines, kind === 'check' || checkbox ? { checked: checked ?? false } : {}) + ); + } + return items; +} + +function parseImage(el: Element): WordsBlock | undefined { + const src = sanitizeWordsUrl(el.getAttribute('src') ?? ''); + if (!src) return undefined; + const alt = el.getAttribute('alt'); + return createImage({ src, ...(alt ? { alt } : {}) }); +} + +function parseTable(tableEl: Element): WordsBlock | undefined { + const rows: TableRow[] = []; + let headerRow = false; + // Rows from thead/tbody/tfoot or direct . + const trs = Array.from(tableEl.querySelectorAll('tr')); + trs.forEach((tr, rowIdx) => { + const cells: TableCell[] = []; + for (const cell of Array.from(tr.children)) { + const ct = cell.tagName.toLowerCase(); + if (ct !== 'td' && ct !== 'th') continue; + if (rowIdx === 0 && ct === 'th') headerRow = true; + // Cells hold blocks (since P5m). Parse the cell's content; seed a + // paragraph when it has only inline text. + const cellBlocks: WordsBlock[] = []; + collectBlocks(cell, cellBlocks); + cells.push(createTableCell(cellBlocks.length > 0 ? cellBlocks : [createParagraph()])); + } + if (cells.length > 0) rows.push(createTableRow(cells)); + }); + if (rows.length === 0) return undefined; + return createTable(rows, headerRow ? { headerRow: true } : {}); +} + +// ── Inline level ───────────────────────────────────────────────────────── + +/** Parse `parent`'s inline content, accumulating marks down the tree. */ +function parseInlines(parent: ParentNode, marks: readonly WordsMark[]): WordsInline[] { + const result: WordsInline[] = []; + for (const node of Array.from(parent.childNodes)) { + if (node.nodeType === TEXT_NODE) { + const text = collapseWs(node.textContent ?? ''); + if (text.length > 0) result.push(createText(text, marks)); + } else if (node.nodeType === ELEMENT_NODE) { + const el = node as Element; + const tag = el.tagName.toLowerCase(); + if (tag === 'br') { + // V2 has no inline break node; a soft break reads as a space. + result.push(createText(' ', marks)); + continue; + } + if (tag === 'a') { + const href = sanitizeWordsUrl(el.getAttribute('href') ?? ''); + const texts = inlinesToText(parseInlines(el, marks)); + if (href && texts.length > 0) result.push(createLink(href, texts)); + else result.push(...parseInlines(el, marks)); + continue; + } + result.push(...parseInlines(el, marksForElement(tag, el, marks))); + } + } + return result; +} + +/** Marks contributed by an inline element (tag semantics + inline style). */ +function marksForElement( + tag: string, + el: Element, + marks: readonly WordsMark[] +): readonly WordsMark[] { + const next: WordsMark[] = [...marks]; + if (tag === 'strong' || tag === 'b') next.push('bold'); + else if (tag === 'em' || tag === 'i' || tag === 'cite') next.push('italic'); + else if (tag === 'u' || tag === 'ins') next.push('underline'); + else if (tag === 's' || tag === 'strike' || tag === 'del') next.push('strike'); + else if (tag === 'code' || tag === 'kbd' || tag === 'samp') next.push('code'); + else if (tag === 'mark') next.push({ type: 'background', value: readColor(el, 'background-color') ?? '#ffc53d' }); + + // Inline style — covers `` etc. + const style = (el as Partial).style; + if (style) { + const weight = style.fontWeight; + if (weight === 'bold' || /^[6-9]00$/.test(weight)) next.push('bold'); + if (style.fontStyle === 'italic') next.push('italic'); + const deco = `${style.textDecorationLine || ''} ${style.textDecoration || ''}`; + if (deco.includes('underline')) next.push('underline'); + if (deco.includes('line-through')) next.push('strike'); + } + const color = readColor(el, 'color'); + if (color) next.push({ type: 'color', value: color }); + if (tag !== 'mark') { + const bg = readColor(el, 'background-color'); + if (bg) next.push({ type: 'background', value: bg }); + } + return next; +} + +// ── Small helpers ────────────────────────────────────────────────────────── + +function headingLevel(tag: string): WordsHeadingLevel { + // h1-h3 map directly; h4-h6 collapse to h3 (the model caps at 3). + const n = Number(tag.slice(1)); + return (n <= 1 ? 1 : n === 2 ? 2 : 3) as WordsHeadingLevel; +} + +/** Collapse runs of whitespace (incl. newlines) to single spaces, HTML-style. */ +function collapseWs(text: string): string { + return text.replace(/\s+/g, ' '); +} + +function readColor(el: Element, prop: 'color' | 'background-color'): string | undefined { + const style = (el as Partial).style; + if (!style) return undefined; + const v = style.getPropertyValue(prop).trim(); + if (!v || v === 'inherit' || v === 'initial' || v === 'transparent' || v === 'currentcolor') { + return undefined; + } + return v; +} + +/** Coerce parsed inline content to `WordsText[]` for a link's children. */ +function inlinesToText(inlines: readonly WordsInline[]): WordsText[] { + const texts: WordsText[] = []; + for (const n of inlines) { + if (n.type === 'text') texts.push(n); + else if (n.type === 'link') for (const c of n.children) texts.push(c); + } + return texts.filter((t) => t.text.length > 0); +} + +function isEmptyText(n: WordsInline): boolean { + return n.type === 'text' && n.text.trim().length === 0; +} + +function inlineRunHasContent(run: readonly WordsInline[]): boolean { + return run.some((n) => (n.type === 'text' ? n.text.trim().length > 0 : true)); +} diff --git a/src/uix/soma/components/words/words-provider.svelte.test.ts b/src/uix/soma/components/words/words-provider.svelte.test.ts index a1f7eb682..981dfe483 100644 --- a/src/uix/soma/components/words/words-provider.svelte.test.ts +++ b/src/uix/soma/components/words/words-provider.svelte.test.ts @@ -2258,10 +2258,15 @@ describe('WordsProvider', () => { return { content }; }); + // Rich paste now PARSES real HTML (`

    …

    ` → a paragraph), so the + // "unsupported-html" log only fires for HTML that carries no extractable + // content AND no plain-text fallback — e.g. the bare clipboard fragment + // markers Office / Windows wrap an empty copy in. result.content.props.onpaste({ currentTarget: contentEl, clipboardData: { - getData: (type: string) => (type === 'text/html' ? '

    unsafe

    ' : '') + getData: (type: string) => + type === 'text/html' ? '' : '' }, preventDefault: vi.fn() } as never); diff --git a/src/uix/soma/components/words/words-provider.svelte.ts b/src/uix/soma/components/words/words-provider.svelte.ts index eda5e22e0..ba96cea67 100644 --- a/src/uix/soma/components/words/words-provider.svelte.ts +++ b/src/uix/soma/components/words/words-provider.svelte.ts @@ -58,6 +58,7 @@ import { getLinkAtSelection, getTextEntries, isFindQueryValid, + parseWordsHtml, parseWordsPlainText, redoWords, renderWordsDomHtml, @@ -80,6 +81,7 @@ import { serializeHtml } from './engine/serialize-html'; import { serializeMarkdown } from './engine/serialize-markdown'; import type { WordsDocument, + WordsBlock, ImageBlock, WordsMark, TableCell, @@ -1175,14 +1177,23 @@ export class WordsProvider { const plainText = e.clipboardData?.getData('text/plain') ?? ''; const html = e.clipboardData?.getData('text/html') ?? ''; - const action = actionFromPaste({ - plainText, - html - }); e.preventDefault(); this.syncSelectionFromDom({ source: 'input' }); + + // A single pasted URL → a link over the selection / at the caret. const pastedUrl = normalizedSinglePastedUrl(plainText); if (pastedUrl && this.applyCommand({ type: 'insertLink', href: pastedUrl })) return; + + // Rich paste: parse the clipboard HTML into V2 blocks + inline marks + // and insert them. Falls through to plain text when there's no usable + // HTML (or it parses to nothing). + if (html.trim().length > 0) { + const blocks = this.parseClipboardHtml(html, e.currentTarget); + if (blocks.length > 0 && this.applyCommand({ type: 'insertBlocks', blocks })) return; + } + + // Plain-text fallback. + const action = actionFromPaste({ plainText, html }); if (action.type === 'ignore') { this.signalInvalidInput(action.reason, e.currentTarget); return; @@ -1190,6 +1201,23 @@ export class WordsProvider { this.applyCommand(action.command); }; + /** + * Parse clipboard `text/html` into V2 blocks. Parses inside an INERT + * document (`implementation.createHTMLDocument` off the active dom) so no + * scripts run and no `` / resources load while parsing — the safe + * way to read foreign HTML. Returns [] on any failure. + */ + private parseClipboardHtml(html: string, target: HTMLElement): readonly WordsBlock[] { + try { + const doc = this.soma.dom.getDocument(target); + const inert = doc.implementation.createHTMLDocument(''); + inert.body.innerHTML = html; + return parseWordsHtml(inert.body); + } catch { + return []; + } + } + readonly ondragover = (e: DragEvent & { currentTarget: HTMLElement }) => { // Only intercept drags carrying image files; let everything else // (text drags, internal selection drags) follow the browser