@ -15,6 +15,18 @@
* Scope : the common tags browsers and editors emit on copy . Unknown block
* tags fall back to a paragraph of their inline content ; unknown inline
* tags are transparent ( their children are parsed with the same marks ) .
*
* Office pre - filter : pastes from Word / Google Docs are detected by their
* fingerprints ( ` mso-* ` styles / ` Mso* ` classes ; the ` docs-internal-guid `
* wrapper ) and pre - cleaned — namespaced Office tags ( ` <o:p> ` , ` w: ` , ` v: ` )
* and style / script junk are dropped , Word ' s fake bullet runs
* ( ` mso-list:Ignore ` ) are stripped ( their marker text decides
* ordered / unordered ) , consecutive ` mso-list ` paragraphs regroup into real
* list blocks ( level → indent ) , empty separator paragraphs are skipped ,
* and per - span ` font-family ` is discarded ( it 's the source editor' s
* default , not user intent ) . The pre - filter MUTATES the supplied root —
* it ' s a parse buffer ( inert document / detached container ) , never live
* DOM .
* /
import {
@ -73,70 +85,174 @@ const BLOCK_TAGS = new Set([
'aside'
] ) ;
/** Selector form of BLOCK_TAGS — "does this subtree contain blocks?". */
const BLOCK_TAGS_SELECTOR = [ . . . BLOCK_TAGS ] . join ( ',' ) ;
// ── Entry point ──────────────────────────────────────────────────────────
export function parseWordsHtml ( root : ParentNode ) : readonly WordsBlock [ ] {
const source = detectPasteSource ( root ) ;
if ( source ) preCleanOfficeDom ( root ) ;
const blocks : WordsBlock [ ] = [ ] ;
collectBlocks ( root , blocks ) ;
collectBlocks ( root , blocks , { source } );
return blocks ;
}
// ── Office pre-filter ──────────────────────────────────────────────────────
type PasteSource = 'word' | 'gdocs' ;
type ParseCtx = { readonly source? : PasteSource } ;
/** Elements that carry no pasteable content, ever. */
const OFFICE_JUNK_TAGS = new Set ( [ 'style' , 'script' , 'meta' , 'link' , 'title' , 'xml' ] ) ;
/** Fingerprint the clipboard's origin editor (undefined = generic HTML). */
function detectPasteSource ( root : ParentNode ) : PasteSource | undefined {
if ( root . querySelector ( '[id^="docs-internal-guid"]' ) ) return 'gdocs' ;
if ( root . querySelector ( '[class*="Mso"], [style*="mso-"]' ) ) return 'word' ;
return undefined ;
}
/ * *
* Strip Office junk from the parse buffer ( mutates it ) : namespaced tags
* ( ` <o:p> ` , ` w:sdt ` , ` v:shape ` — metadata , always content - free ) , style /
* script / meta elements ( their text would leak into the paste ) , and
* Word ' s fake list markers ( ` mso-list:Ignore ` runs : the literal "·" /
* "1." plus padding ) . The marker text is hoisted onto the host ` <p> ` as
* ` data-words-mso-marker ` so the block walk can pick ordered vs
* unordered without the junk .
* /
function preCleanOfficeDom ( root : ParentNode ) : void {
for ( const el of Array . from ( root . querySelectorAll ( '*' ) ) ) {
const tag = el . tagName . toLowerCase ( ) ;
if ( tag . includes ( ':' ) || OFFICE_JUNK_TAGS . has ( tag ) ) {
el . remove ( ) ;
continue ;
}
const style = el . getAttribute ( 'style' ) ? ? '' ;
if ( /mso-list:\s*ignore/i . test ( style ) ) {
const marker = ( el . textContent ? ? '' ) . trim ( ) ;
const host = el . closest ( 'p' ) ;
if ( host && marker ) host . setAttribute ( 'data-words-mso-marker' , marker ) ;
el . remove ( ) ;
}
}
}
/ * * A W o r d f a k e - l i s t p a r a g r a p h ( ` s t y l e = " … m s o - l i s t : l 0 l e v e l 2 l f o 1 … " ` ) ,
* decomposed for regrouping : which list it belongs to , its nesting
* level , and the item kind derived from the stripped marker text . * /
function msoListItem (
el : Element ,
ctx : ParseCtx
) : { listId : string ; kind : WordsListKind ; item : ListItem } | undefined {
const style = el . getAttribute ( 'style' ) ? ? '' ;
const m = /mso-list:\s*(l\d+)\s+level(\d+)/i . exec ( style ) ;
if ( ! m ) return undefined ;
const marker = el . getAttribute ( 'data-words-mso-marker' ) ? ? '' ;
// "1." / "a)" / "(iv)" → ordered; "·" / "o" / "§" (no punctuation) → bullet.
const kind : WordsListKind = /^\(?\w{1,4}[.)]/ . test ( marker ) ? 'ordered' : 'unordered' ;
const inlines = parseInlines ( el , [ ] , ctx ) . filter ( ( n ) = > ! isEmptyText ( n ) ) ;
const indent = Math . min ( 8 , Math . max ( 0 , Number ( m [ 2 ] ) - 1 ) ) ;
return { listId : m [ 1 ] , kind , item : createListItem ( inlines , indent > 0 ? { indent } : { } ) } ;
}
// ── Block level ──────────────────────────────────────────────────────────
/ * *
* Walk ` parent ` ' s children , flushing runs of inline content into paragraphs
* around the block - level elements . Bare inline content ( e . g . pasting
* "**a** b" with no wrapping ` <p> ` ) becomes a paragraph .
* "**a** b" with no wrapping ` <p> ` ) becomes a paragraph . Under a Word
* paste , consecutive ` mso-list ` paragraphs of the same list id regroup
* into a single list block ( kind from the first item ' s marker ) .
* /
function collectBlocks ( parent : ParentNode , out : WordsBlock [ ] ) : void {
function collectBlocks ( parent : ParentNode , out : WordsBlock [ ] , ctx : ParseCtx ): void {
let inlineRun : WordsInline [ ] = [ ] ;
let listRun : { listId : string ; kind : WordsListKind ; item : ListItem } [ ] = [ ] ;
const flush = ( ) = > {
if ( inlineRunHasContent ( inlineRun ) ) out . push ( createParagraph ( inlineRun ) ) ;
inlineRun = [ ] ;
} ;
const flushList = ( ) = > {
if ( listRun . length > 0 ) {
out . push (
createList (
listRun [ 0 ] . kind ,
listRun . map ( ( r ) = > r . item )
)
) ;
listRun = [ ] ;
}
} ;
for ( const node of Array . from ( parent . childNodes ) ) {
if ( node . nodeType === TEXT_NODE ) {
const text = node . textContent ? ? '' ;
// Whitespace-only text BETWEEN block elements is layout, not content.
if ( text . trim ( ) . length > 0 ) inlineRun . push ( createText ( collapseWs ( text ) ) ) ;
if ( text . trim ( ) . length > 0 ) {
flushList ( ) ;
inlineRun . push ( createText ( collapseWs ( text ) ) ) ;
}
} else if ( node . nodeType === ELEMENT_NODE ) {
const el = node as Element ;
const tag = el . tagName . toLowerCase ( ) ;
const listItem = ctx . source === 'word' && tag === 'p' ? msoListItem ( el , ctx ) : undefined ;
if ( listItem ) {
flush ( ) ;
if ( listRun . length > 0 && listRun [ 0 ] . listId !== listItem . listId ) flushList ( ) ;
listRun . push ( listItem ) ;
continue ;
}
flushList ( ) ;
if ( BLOCK_TAGS . has ( tag ) ) {
flush ( ) ;
parseBlockElement ( el , tag , out ) ;
parseBlockElement ( el , tag , out , ctx ) ;
} else if ( el . querySelector ( BLOCK_TAGS_SELECTOR ) ) {
// An inline/formatting element that CONTAINS blocks is a
// structural wrapper, not emphasis — GDocs wraps the whole
// clipboard in a well-nested `<b>` (which the HTML parser
// keeps intact). Unwrap it into the block walk instead of
// flattening its paragraphs/lists into one inline run.
flush ( ) ;
collectBlocks ( el , out , ctx ) ;
} else {
inlineRun . push ( . . . parseInlines ( el , [ ] ) ) ;
inlineRun . push ( . . . parseInlines ( el , [ ] , ctx )) ;
}
}
}
flushList ( ) ;
flush ( ) ;
}
function parseBlockElement ( el : Element , tag : string , out : WordsBlock [ ] ) : void {
function parseBlockElement ( el : Element , tag : string , out : WordsBlock [ ] , ctx : ParseCtx ): void {
switch ( tag ) {
case 'p' :
out . push ( createParagraph ( parseInlines ( el , [ ] ) ) ) ;
case 'p' : {
const inlines = parseInlines ( el , [ ] , ctx ) ;
// Office pastes separate paragraphs with EMPTY ones (Word:
// `<p><o:p> </o:p></p>`, GDocs: `<p><br></p>`) — layout
// junk, not content. Generic pastes keep them (user intent).
if ( ctx . source && ! inlineRunHasContent ( inlines ) ) return ;
out . push ( createParagraph ( inlines ) ) ;
return ;
}
case 'h1' :
case 'h2' :
case 'h3' :
case 'h4' :
case 'h5' :
case 'h6' :
out . push ( createHeading ( headingLevel ( tag ) , parseInlines ( el , [ ] ) ) ) ;
out . push ( createHeading ( headingLevel ( tag ) , parseInlines ( el , [ ] , ctx )) ) ;
return ;
case 'blockquote' :
out . push ( createQuote ( parseInlines ( el , [ ] ) ) ) ;
out . push ( createQuote ( parseInlines ( el , [ ] , ctx )) ) ;
return ;
case 'pre' :
out . push ( createCodeBlock ( [ createText ( el . textContent ? ? '' ) ] ) ) ;
return ;
case 'ul' :
out . push ( createList ( 'unordered' , parseListItems ( el , 'unordered' ) ) ) ;
out . push ( createList ( 'unordered' , parseListItems ( el , 'unordered' , ctx )) ) ;
return ;
case 'ol' :
out . push ( createList ( 'ordered' , parseListItems ( el , 'ordered' ) ) ) ;
out . push ( createList ( 'ordered' , parseListItems ( el , 'ordered' , ctx )) ) ;
return ;
case 'hr' :
out . push ( createDivider ( ) ) ;
@ -152,30 +268,34 @@ function parseBlockElement(el: Element, tag: string, out: WordsBlock[]): void {
const img = parseImage ( inner ) ;
if ( img ) out . push ( img ) ;
} else {
collectBlocks ( el , out );
collectBlocks ( el , out , ctx );
}
return ;
}
case 'table' : {
const table = parseTable ( el );
const table = parseTable ( el , ctx );
if ( table ) out . push ( table ) ;
return ;
}
default :
// div / section / article / … — unwrap structural containers.
collectBlocks ( el , out );
collectBlocks ( el , out , ctx );
return ;
}
}
function parseListItems ( listEl : Element , kind : WordsListKind ) : readonly ListItem [ ] {
function parseListItems (
listEl : Element ,
kind : WordsListKind ,
ctx : ParseCtx
) : readonly ListItem [ ] {
const items : ListItem [ ] = [ ] ;
for ( const li of Array . from ( listEl . children ) ) {
if ( li . tagName . toLowerCase ( ) !== 'li' ) continue ;
// Checkbox lists: GitHub / editors emit `<li><input type=checkbox>…`.
const checkbox = li . querySelector ( 'input[type="checkbox"]' ) ;
const checked = checkbox ? ( checkbox as HTMLInputElement ) . checked : undefined ;
const inlines = parseInlines ( li , [ ] ). filter ( ( n ) = > ! isEmptyText ( n ) ) ;
const inlines = parseInlines ( li , [ ] , ctx ). filter ( ( n ) = > ! isEmptyText ( n ) ) ;
items . push (
createListItem ( inlines , kind === 'check' || checkbox ? { checked : checked ? ? false } : { } )
) ;
@ -190,7 +310,7 @@ function parseImage(el: Element): WordsBlock | undefined {
return createImage ( { src , . . . ( alt ? { alt } : { } ) } ) ;
}
function parseTable ( tableEl : Element ): WordsBlock | undefined {
function parseTable ( tableEl : Element , ctx : ParseCtx ): WordsBlock | undefined {
const rows : TableRow [ ] = [ ] ;
let headerRow = false ;
// Rows from thead/tbody/tfoot or direct <tr>.
@ -204,7 +324,7 @@ function parseTable(tableEl: Element): WordsBlock | undefined {
// Cells hold blocks (since P5m). Parse the cell's content; seed a
// paragraph when it has only inline text.
const cellBlocks : WordsBlock [ ] = [ ] ;
collectBlocks ( cell , cellBlocks );
collectBlocks ( cell , cellBlocks , ctx );
cells . push ( createTableCell ( cellBlocks . length > 0 ? cellBlocks : [ createParagraph ( ) ] ) ) ;
}
if ( cells . length > 0 ) rows . push ( createTableRow ( cells ) ) ;
@ -216,7 +336,11 @@ function parseTable(tableEl: Element): WordsBlock | undefined {
// ── Inline level ─────────────────────────────────────────────────────────
/** Parse `parent`'s inline content, accumulating marks down the tree. */
function parseInlines ( parent : ParentNode , marks : readonly WordsMark [ ] ) : WordsInline [ ] {
function parseInlines (
parent : ParentNode ,
marks : readonly WordsMark [ ] ,
ctx : ParseCtx
) : WordsInline [ ] {
const result : WordsInline [ ] = [ ] ;
for ( const node of Array . from ( parent . childNodes ) ) {
if ( node . nodeType === TEXT_NODE ) {
@ -232,12 +356,12 @@ function parseInlines(parent: ParentNode, marks: readonly WordsMark[]): WordsInl
}
if ( tag === 'a' ) {
const href = sanitizeWordsUrl ( el . getAttribute ( 'href' ) ? ? '' ) ;
const texts = inlinesToText ( parseInlines ( el , marks )) ;
const texts = inlinesToText ( parseInlines ( el , marks , ctx )) ;
if ( href && texts . length > 0 ) result . push ( createLink ( href , texts ) ) ;
else result . push ( . . . parseInlines ( el , marks )) ;
else result . push ( . . . parseInlines ( el , marks , ctx )) ;
continue ;
}
result . push ( . . . parseInlines ( el , marksForElement ( tag , el , marks )) ) ;
result . push ( . . . parseInlines ( el , marksForElement ( tag , el , marks , ctx ), ctx ) ) ;
}
}
return result ;
@ -247,10 +371,16 @@ function parseInlines(parent: ParentNode, marks: readonly WordsMark[]): WordsInl
function marksForElement (
tag : string ,
el : Element ,
marks : readonly WordsMark [ ]
marks : readonly WordsMark [ ] ,
ctx : ParseCtx
) : readonly WordsMark [ ] {
const next : WordsMark [ ] = [ . . . marks ] ;
if ( tag === 'strong' || tag === 'b' ) next . push ( 'bold' ) ;
// A <b>/<strong> whose own style says "not bold" is a structural
// wrapper, not emphasis — Google Docs wraps the WHOLE clipboard in
// `<b style="font-weight:normal" id="docs-internal-guid-…">`.
const ownWeight = ( el as Partial < HTMLElement > ) . style ? . fontWeight ? ? '' ;
const tagNotBold = ownWeight === 'normal' || /^[1-4]00$/ . test ( ownWeight ) ;
if ( ( tag === 'strong' || tag === 'b' ) && ! tagNotBold ) next . push ( 'bold' ) ;
else if ( tag === 'em' || tag === 'i' || tag === 'cite' ) next . push ( 'italic' ) ;
else if ( tag === 'u' || tag === 'ins' ) next . push ( 'underline' ) ;
else if ( tag === 's' || tag === 'strike' || tag === 'del' ) next . push ( 'strike' ) ;
@ -271,10 +401,14 @@ function marksForElement(
// Range typography (Q6: preserve the user's raw intent) — px sizes
// only (relative/keyword sizes have no universal raw value), and a
// raw font-family as long as it isn't a token/var() reference.
// Office pastes stamp the SOURCE editor's default font on every
// span (Word: Calibri, GDocs: Arial) — noise, not intent → dropped.
const sizePx = /^(\d+(?:\.\d+)?)px$/ . exec ( style . fontSize ? . trim ( ) ? ? '' ) ;
if ( sizePx ) next . push ( { type : 'fontSize' , value : Number ( sizePx [ 1 ] ) } ) ;
const family = style . fontFamily ? . trim ( ) ;
if ( family && ! family . includes ( 'var(' ) ) next . push ( { type : 'fontFamily' , value : family } ) ;
if ( family && ! family . includes ( 'var(' ) && ! ctx . source ) {
next . push ( { type : 'fontFamily' , value : family } ) ;
}
}
const color = readColor ( el , 'color' ) ;
if ( color ) next . push ( { type : 'color' , value : color } ) ;