words(paste): pre-filtro Office en parse-html - Word/GDocs limpios (F4.2)

Deteccion por fingerprint (mso-* / docs-internal-guid) y pre-limpieza
del buffer: tags con namespace (o:p, w:, v:) + style/script/meta fuera,
marcadores mso-list:Ignore fuera (su texto decide ordered/unordered),
parrafos mso-list consecutivos reagrupados en ListBlock reales
(level -> indent), parrafos separadores vacios fuera, y font-family
descartada en pastes office (default del editor origen, no intencion).
Fix universal: <b|strong style=font-weight:normal> no es bold (el
wrapper de GDocs) y un wrapper inline que CONTIENE bloques se atraviesa
como contenedor (el <b> bien anidado de GDocs no fragmenta). +5 tests
con fixtures estructuralmente fieles (515/515). Verificado en navegador
real: paste de fixture Word -> paragraph+list(indent)+paragraph sin
bullets/nbsp/Calibri/Times.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
menubar-v4-safe
dev 3 months ago
parent 31d49c88f6
commit 8444dd58dd

@ -129,3 +129,93 @@ describe('parseWordsHtml — F2 range typography', () => {
);
});
});
describe('parseWordsHtml — Office paste pre-filter (F4.2)', () => {
// Structurally faithful cut of a Word clipboard: Mso classes, <o:p>
// terminators, an EMPTY separator paragraph, and the per-span
// Calibri/pt soup Word stamps on every run.
const WORD_PARAS = [
`<p class=MsoNormal>Hola <b style='mso-bidi-font-weight:normal'>mundo</b><o:p></o:p></p>`,
`<p class=MsoNormal><o:p>&nbsp;</o:p></p>`,
`<p class=MsoNormal><span style='font-size:11.0pt;font-family:"Calibri",sans-serif;mso-fareast-font-family:Calibri;mso-fareast-theme-font:minor-latin'>Texto normal</span></p>`
].join('\n');
it('Word: strips o:p / empty separators / font soup, keeps real content and bold', () => {
const blocks = parse(WORD_PARAS);
expect(blocks.map((b) => b.type)).toEqual(['paragraph', 'paragraph']);
const first = blocks[0];
if (first.type !== 'paragraph') throw new Error('expected paragraph');
expect(first.children.map(text).join('')).toBe('Hola mundo');
const bold = first.children.find((c) => text(c) === 'mundo');
expect(bold && marksOf(bold)).toContain('bold');
const json = JSON.stringify(blocks);
expect(json).not.toContain('Calibri');
expect(json).not.toContain('fontFamily');
// Word pads its fake markers with NBSP runs — none may survive.
expect(json).not.toContain(' ');
expect(json.includes(String.fromCharCode(160))).toBe(false);
});
// Word fake lists: paragraphs with `mso-list:lN levelM` whose bullet /
// number is a literal `mso-list:Ignore` run behind a revealed
// conditional comment. l0 = bullet list (level 1 + nested level 2),
// l1 = a separate numbered list.
const WORD_LISTS = [
`<p class=MsoListParagraphCxSpFirst style='text-indent:-18.0pt;mso-list:l0 level1 lfo1'><![if !supportLists]><span style='font-family:Symbol;mso-fareast-font-family:Symbol;mso-bidi-font-family:Symbol'><span style='mso-list:Ignore'>·<span style='font:7.0pt "Times New Roman"'>&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;</span></span></span><![endif]>Uno</p>`,
`<p class=MsoListParagraphCxSpLast style='text-indent:-18.0pt;mso-list:l0 level2 lfo1'><![if !supportLists]><span style='font-family:"Courier New";mso-fareast-font-family:"Courier New"'><span style='mso-list:Ignore'>o<span style='font:7.0pt "Times New Roman"'>&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;</span></span></span><![endif]>Dos anidado</p>`,
`<p class=MsoListParagraph style='text-indent:-18.0pt;mso-list:l1 level1 lfo2'><![if !supportLists]><span style='mso-fareast-font-family:Calibri'><span style='mso-list:Ignore'>1.<span style='font:7.0pt "Times New Roman"'>&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;</span></span></span><![endif]>Primero</p>`
].join('\n');
it('Word: regroups fake-list paragraphs into real lists (marker → kind, level → indent)', () => {
const blocks = parse(WORD_LISTS);
expect(blocks.map((b) => b.type)).toEqual(['list', 'list']);
const [bullets, numbered] = blocks;
if (bullets.type !== 'list' || numbered.type !== 'list') throw new Error('expected lists');
expect(bullets.kind).toBe('unordered');
expect(bullets.items.map((i) => i.children.map(text).join(''))).toEqual([
'Uno',
'Dos anidado'
]);
expect(bullets.items[0].indent).toBeUndefined();
expect(bullets.items[1].indent).toBe(1);
expect(numbered.kind).toBe('ordered');
expect(numbered.items.map((i) => i.children.map(text).join(''))).toEqual(['Primero']);
const json = JSON.stringify(blocks);
expect(json).not.toContain('·');
expect(json).not.toContain('Times New Roman');
});
// Google Docs wraps the WHOLE clipboard in a no-op bold element
// (`<b style="font-weight:normal" id="docs-internal-guid-…">`; the HTML
// parser's formatting reconstruction clones it INTO each block) and
// stamps Arial + font-weight 400/700 on every span.
const GDOCS = `<b style="font-weight:normal;" id="docs-internal-guid-1a2b3c4d-5e6f"><p dir="ltr" style="line-height:1.38;margin-top:0pt;margin-bottom:0pt;"><span style="font-size:11pt;font-family:Arial,sans-serif;color:#000000;background-color:transparent;font-weight:400;font-style:normal;vertical-align:baseline;white-space:pre-wrap;">Texto plano </span><span style="font-size:11pt;font-family:Arial,sans-serif;color:#000000;background-color:transparent;font-weight:700;font-style:normal;vertical-align:baseline;white-space:pre-wrap;">negrita</span></p><ul style="margin-top:0;margin-bottom:0;padding-inline-start:48px;"><li dir="ltr" style="list-style-type:disc;font-size:11pt;font-family:Arial,sans-serif;font-weight:400;" aria-level="1"><p dir="ltr" style="line-height:1.38;margin-top:0pt;margin-bottom:0pt;" role="presentation"><span style="font-size:11pt;font-family:Arial,sans-serif;font-weight:400;">Item uno</span></p></li></ul></b>`;
it('GDocs: the font-weight:normal <b> wrapper is NOT bold; explicit 700 spans are', () => {
const blocks = parse(GDOCS);
expect(blocks.map((b) => b.type)).toEqual(['paragraph', 'list']);
const p = blocks[0];
if (p.type !== 'paragraph') throw new Error('expected paragraph');
const plain = p.children.find((c) => text(c).startsWith('Texto plano'));
expect(plain && marksOf(plain)).not.toContain('bold');
const bold = p.children.find((c) => text(c) === 'negrita');
expect(bold && marksOf(bold)).toContain('bold');
const list = blocks[1];
if (list.type !== 'list') throw new Error('expected list');
expect(list.kind).toBe('unordered');
expect(list.items.map((i) => i.children.map(text).join(''))).toEqual(['Item uno']);
expect(JSON.stringify(blocks)).not.toContain('Arial');
});
it('generic paste: <b style="font-weight:normal"> is not bold, font-family survives', () => {
const blocks = parse('<p><b style="font-weight:normal">x</b> <span style="font-family:Georgia">g</span></p>');
const p = blocks[0];
if (p.type !== 'paragraph') throw new Error('expected paragraph');
const x = p.children.find((c) => text(c) === 'x');
expect(x && marksOf(x)).not.toContain('bold');
const g = p.children.find((c) => text(c) === 'g');
expect(g && marksOf(g)).toEqual(
expect.arrayContaining([expect.objectContaining({ type: 'fontFamily' })])
);
});
});

@ -15,6 +15,18 @@
* Scope: the common tags browsers and editors emit on copy. Unknown block
* tags fall back to a paragraph of their inline content; unknown inline
* tags are transparent (their children are parsed with the same marks).
*
* Office pre-filter: pastes from Word / Google Docs are detected by their
* fingerprints (`mso-*` styles / `Mso*` classes; the `docs-internal-guid`
* wrapper) and pre-cleaned — namespaced Office tags (`<o:p>`, `w:`, `v:`)
* and style/script junk are dropped, Word's fake bullet runs
* (`mso-list:Ignore`) are stripped (their marker text decides
* ordered/unordered), consecutive `mso-list` paragraphs regroup into real
* list blocks (level → indent), empty separator paragraphs are skipped,
* and per-span `font-family` is discarded (it's the source editor's
* default, not user intent). The pre-filter MUTATES the supplied root —
* it's a parse buffer (inert document / detached container), never live
* DOM.
*/
import {
@ -73,70 +85,174 @@ const BLOCK_TAGS = new Set([
'aside'
]);
/** Selector form of BLOCK_TAGS — "does this subtree contain blocks?". */
const BLOCK_TAGS_SELECTOR = [...BLOCK_TAGS].join(',');
// ── Entry point ──────────────────────────────────────────────────────────
export function parseWordsHtml(root: ParentNode): readonly WordsBlock[] {
const source = detectPasteSource(root);
if (source) preCleanOfficeDom(root);
const blocks: WordsBlock[] = [];
collectBlocks(root, blocks);
collectBlocks(root, blocks, { source });
return blocks;
}
// ── Office pre-filter ──────────────────────────────────────────────────────
type PasteSource = 'word' | 'gdocs';
type ParseCtx = { readonly source?: PasteSource };
/** Elements that carry no pasteable content, ever. */
const OFFICE_JUNK_TAGS = new Set(['style', 'script', 'meta', 'link', 'title', 'xml']);
/** Fingerprint the clipboard's origin editor (undefined = generic HTML). */
function detectPasteSource(root: ParentNode): PasteSource | undefined {
if (root.querySelector('[id^="docs-internal-guid"]')) return 'gdocs';
if (root.querySelector('[class*="Mso"], [style*="mso-"]')) return 'word';
return undefined;
}
/**
* Strip Office junk from the parse buffer (mutates it): namespaced tags
* (`<o:p>`, `w:sdt`, `v:shape` — metadata, always content-free), style /
* script / meta elements (their text would leak into the paste), and
* Word's fake list markers (`mso-list:Ignore` runs: the literal "·" /
* "1." plus padding). The marker text is hoisted onto the host `<p>` as
* `data-words-mso-marker` so the block walk can pick ordered vs
* unordered without the junk.
*/
function preCleanOfficeDom(root: ParentNode): void {
for (const el of Array.from(root.querySelectorAll('*'))) {
const tag = el.tagName.toLowerCase();
if (tag.includes(':') || OFFICE_JUNK_TAGS.has(tag)) {
el.remove();
continue;
}
const style = el.getAttribute('style') ?? '';
if (/mso-list:\s*ignore/i.test(style)) {
const marker = (el.textContent ?? '').trim();
const host = el.closest('p');
if (host && marker) host.setAttribute('data-words-mso-marker', marker);
el.remove();
}
}
}
/** A Word fake-list paragraph (`style="…mso-list:l0 level2 lfo1…"`),
* decomposed for regrouping: which list it belongs to, its nesting
* level, and the item kind derived from the stripped marker text. */
function msoListItem(
el: Element,
ctx: ParseCtx
): { listId: string; kind: WordsListKind; item: ListItem } | undefined {
const style = el.getAttribute('style') ?? '';
const m = /mso-list:\s*(l\d+)\s+level(\d+)/i.exec(style);
if (!m) return undefined;
const marker = el.getAttribute('data-words-mso-marker') ?? '';
// "1." / "a)" / "(iv)" → ordered; "·" / "o" / "§" (no punctuation) → bullet.
const kind: WordsListKind = /^\(?\w{1,4}[.)]/.test(marker) ? 'ordered' : 'unordered';
const inlines = parseInlines(el, [], ctx).filter((n) => !isEmptyText(n));
const indent = Math.min(8, Math.max(0, Number(m[2]) - 1));
return { listId: m[1], kind, item: createListItem(inlines, indent > 0 ? { indent } : {}) };
}
// ── Block level ──────────────────────────────────────────────────────────
/**
* Walk `parent`'s children, flushing runs of inline content into paragraphs
* around the block-level elements. Bare inline content (e.g. pasting
* "**a** b" with no wrapping `<p>`) becomes a paragraph.
* "**a** b" with no wrapping `<p>`) becomes a paragraph. Under a Word
* paste, consecutive `mso-list` paragraphs of the same list id regroup
* into a single list block (kind from the first item's marker).
*/
function collectBlocks(parent: ParentNode, out: WordsBlock[]): void {
function collectBlocks(parent: ParentNode, out: WordsBlock[], ctx: ParseCtx): void {
let inlineRun: WordsInline[] = [];
let listRun: { listId: string; kind: WordsListKind; item: ListItem }[] = [];
const flush = () => {
if (inlineRunHasContent(inlineRun)) out.push(createParagraph(inlineRun));
inlineRun = [];
};
const flushList = () => {
if (listRun.length > 0) {
out.push(
createList(
listRun[0].kind,
listRun.map((r) => r.item)
)
);
listRun = [];
}
};
for (const node of Array.from(parent.childNodes)) {
if (node.nodeType === TEXT_NODE) {
const text = node.textContent ?? '';
// Whitespace-only text BETWEEN block elements is layout, not content.
if (text.trim().length > 0) inlineRun.push(createText(collapseWs(text)));
if (text.trim().length > 0) {
flushList();
inlineRun.push(createText(collapseWs(text)));
}
} else if (node.nodeType === ELEMENT_NODE) {
const el = node as Element;
const tag = el.tagName.toLowerCase();
const listItem = ctx.source === 'word' && tag === 'p' ? msoListItem(el, ctx) : undefined;
if (listItem) {
flush();
if (listRun.length > 0 && listRun[0].listId !== listItem.listId) flushList();
listRun.push(listItem);
continue;
}
flushList();
if (BLOCK_TAGS.has(tag)) {
flush();
parseBlockElement(el, tag, out);
parseBlockElement(el, tag, out, ctx);
} else if (el.querySelector(BLOCK_TAGS_SELECTOR)) {
// An inline/formatting element that CONTAINS blocks is a
// structural wrapper, not emphasis — GDocs wraps the whole
// clipboard in a well-nested `<b>` (which the HTML parser
// keeps intact). Unwrap it into the block walk instead of
// flattening its paragraphs/lists into one inline run.
flush();
collectBlocks(el, out, ctx);
} else {
inlineRun.push(...parseInlines(el, []));
inlineRun.push(...parseInlines(el, [], ctx));
}
}
}
flushList();
flush();
}
function parseBlockElement(el: Element, tag: string, out: WordsBlock[]): void {
function parseBlockElement(el: Element, tag: string, out: WordsBlock[], ctx: ParseCtx): void {
switch (tag) {
case 'p':
out.push(createParagraph(parseInlines(el, [])));
case 'p': {
const inlines = parseInlines(el, [], ctx);
// Office pastes separate paragraphs with EMPTY ones (Word:
// `<p><o:p>&nbsp;</o:p></p>`, GDocs: `<p><br></p>`) — layout
// junk, not content. Generic pastes keep them (user intent).
if (ctx.source && !inlineRunHasContent(inlines)) return;
out.push(createParagraph(inlines));
return;
}
case 'h1':
case 'h2':
case 'h3':
case 'h4':
case 'h5':
case 'h6':
out.push(createHeading(headingLevel(tag), parseInlines(el, [])));
out.push(createHeading(headingLevel(tag), parseInlines(el, [], ctx)));
return;
case 'blockquote':
out.push(createQuote(parseInlines(el, [])));
out.push(createQuote(parseInlines(el, [], ctx)));
return;
case 'pre':
out.push(createCodeBlock([createText(el.textContent ?? '')]));
return;
case 'ul':
out.push(createList('unordered', parseListItems(el, 'unordered')));
out.push(createList('unordered', parseListItems(el, 'unordered', ctx)));
return;
case 'ol':
out.push(createList('ordered', parseListItems(el, 'ordered')));
out.push(createList('ordered', parseListItems(el, 'ordered', ctx)));
return;
case 'hr':
out.push(createDivider());
@ -152,30 +268,34 @@ function parseBlockElement(el: Element, tag: string, out: WordsBlock[]): void {
const img = parseImage(inner);
if (img) out.push(img);
} else {
collectBlocks(el, out);
collectBlocks(el, out, ctx);
}
return;
}
case 'table': {
const table = parseTable(el);
const table = parseTable(el, ctx);
if (table) out.push(table);
return;
}
default:
// div / section / article / … — unwrap structural containers.
collectBlocks(el, out);
collectBlocks(el, out, ctx);
return;
}
}
function parseListItems(listEl: Element, kind: WordsListKind): readonly ListItem[] {
function parseListItems(
listEl: Element,
kind: WordsListKind,
ctx: ParseCtx
): readonly ListItem[] {
const items: ListItem[] = [];
for (const li of Array.from(listEl.children)) {
if (li.tagName.toLowerCase() !== 'li') continue;
// Checkbox lists: GitHub / editors emit `<li><input type=checkbox>…`.
const checkbox = li.querySelector('input[type="checkbox"]');
const checked = checkbox ? (checkbox as HTMLInputElement).checked : undefined;
const inlines = parseInlines(li, []).filter((n) => !isEmptyText(n));
const inlines = parseInlines(li, [], ctx).filter((n) => !isEmptyText(n));
items.push(
createListItem(inlines, kind === 'check' || checkbox ? { checked: checked ?? false } : {})
);
@ -190,7 +310,7 @@ function parseImage(el: Element): WordsBlock | undefined {
return createImage({ src, ...(alt ? { alt } : {}) });
}
function parseTable(tableEl: Element): WordsBlock | undefined {
function parseTable(tableEl: Element, ctx: ParseCtx): WordsBlock | undefined {
const rows: TableRow[] = [];
let headerRow = false;
// Rows from thead/tbody/tfoot or direct <tr>.
@ -204,7 +324,7 @@ function parseTable(tableEl: Element): WordsBlock | undefined {
// Cells hold blocks (since P5m). Parse the cell's content; seed a
// paragraph when it has only inline text.
const cellBlocks: WordsBlock[] = [];
collectBlocks(cell, cellBlocks);
collectBlocks(cell, cellBlocks, ctx);
cells.push(createTableCell(cellBlocks.length > 0 ? cellBlocks : [createParagraph()]));
}
if (cells.length > 0) rows.push(createTableRow(cells));
@ -216,7 +336,11 @@ function parseTable(tableEl: Element): WordsBlock | undefined {
// ── Inline level ─────────────────────────────────────────────────────────
/** Parse `parent`'s inline content, accumulating marks down the tree. */
function parseInlines(parent: ParentNode, marks: readonly WordsMark[]): WordsInline[] {
function parseInlines(
parent: ParentNode,
marks: readonly WordsMark[],
ctx: ParseCtx
): WordsInline[] {
const result: WordsInline[] = [];
for (const node of Array.from(parent.childNodes)) {
if (node.nodeType === TEXT_NODE) {
@ -232,12 +356,12 @@ function parseInlines(parent: ParentNode, marks: readonly WordsMark[]): WordsInl
}
if (tag === 'a') {
const href = sanitizeWordsUrl(el.getAttribute('href') ?? '');
const texts = inlinesToText(parseInlines(el, marks));
const texts = inlinesToText(parseInlines(el, marks, ctx));
if (href && texts.length > 0) result.push(createLink(href, texts));
else result.push(...parseInlines(el, marks));
else result.push(...parseInlines(el, marks, ctx));
continue;
}
result.push(...parseInlines(el, marksForElement(tag, el, marks)));
result.push(...parseInlines(el, marksForElement(tag, el, marks, ctx), ctx));
}
}
return result;
@ -247,10 +371,16 @@ function parseInlines(parent: ParentNode, marks: readonly WordsMark[]): WordsInl
function marksForElement(
tag: string,
el: Element,
marks: readonly WordsMark[]
marks: readonly WordsMark[],
ctx: ParseCtx
): readonly WordsMark[] {
const next: WordsMark[] = [...marks];
if (tag === 'strong' || tag === 'b') next.push('bold');
// A <b>/<strong> whose own style says "not bold" is a structural
// wrapper, not emphasis — Google Docs wraps the WHOLE clipboard in
// `<b style="font-weight:normal" id="docs-internal-guid-…">`.
const ownWeight = (el as Partial<HTMLElement>).style?.fontWeight ?? '';
const tagNotBold = ownWeight === 'normal' || /^[1-4]00$/.test(ownWeight);
if ((tag === 'strong' || tag === 'b') && !tagNotBold) next.push('bold');
else if (tag === 'em' || tag === 'i' || tag === 'cite') next.push('italic');
else if (tag === 'u' || tag === 'ins') next.push('underline');
else if (tag === 's' || tag === 'strike' || tag === 'del') next.push('strike');
@ -271,10 +401,14 @@ function marksForElement(
// Range typography (Q6: preserve the user's raw intent) — px sizes
// only (relative/keyword sizes have no universal raw value), and a
// raw font-family as long as it isn't a token/var() reference.
// Office pastes stamp the SOURCE editor's default font on every
// span (Word: Calibri, GDocs: Arial) — noise, not intent → dropped.
const sizePx = /^(\d+(?:\.\d+)?)px$/.exec(style.fontSize?.trim() ?? '');
if (sizePx) next.push({ type: 'fontSize', value: Number(sizePx[1]) });
const family = style.fontFamily?.trim();
if (family && !family.includes('var(')) next.push({ type: 'fontFamily', value: family });
if (family && !family.includes('var(') && !ctx.source) {
next.push({ type: 'fontFamily', value: family });
}
}
const color = readColor(el, 'color');
if (color) next.push({ type: 'color', value: color });

Loading…
Cancel
Save

Powered by TurnKey Linux.