officeparser 7.6.1 → 7.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -706,6 +706,22 @@ const parseHtml = async (buffer, config) => {
706
706
  admonitionNode.rawContent = '<div class="admonition">...</div>';
707
707
  return admonitionNode;
708
708
  }
709
+ // Blockquote. Previously dropped entirely (its children were lifted out unquoted), so a
710
+ // <blockquote> lost its `> ` on the Markdown hop. Mark each block child with the 'Quote'
711
+ // style the styleMapper maps back to a blockquote; loose inline content is wrapped in one
712
+ // Quote-styled paragraph so it isn't emitted as an ordinary line.
713
+ if (tagName === 'blockquote') {
714
+ const kids = parseChildren(node, newFormatting, listContext);
715
+ const isBlock = (t) => t === 'paragraph' || t === 'heading' || t === 'list';
716
+ if (!kids.some(k => isBlock(k.type))) {
717
+ return { type: 'paragraph', metadata: { style: 'Quote' }, children: kids };
718
+ }
719
+ kids.forEach(k => {
720
+ if (isBlock(k.type))
721
+ k.metadata = { ...k.metadata, style: 'Quote' };
722
+ });
723
+ return kids;
724
+ }
709
725
  // Mermaid diagrams. Attribute-driven producers render a
710
726
  // <div class="mermaid" data-mermaid="<code>"> with the code also as text content.
711
727
  // Map either shape to a fenced code node with language `mermaid`, so it round-trips as
@@ -931,11 +947,21 @@ const parseHtml = async (buffer, config) => {
931
947
  const rowSpanAttr = node.attributes?.rowspan;
932
948
  const colSpan = colSpanAttr ? parseInt(colSpanAttr, 10) : undefined;
933
949
  const rowSpan = rowSpanAttr ? parseInt(rowSpanAttr, 10) : undefined;
950
+ // Per-column GFM alignment: read the cell's own `text-align` (or a legacy `align=`
951
+ // attribute) into `CellMetadata.align`, so the `:---`/`:---:`/`---:` markers survive
952
+ // AST -> HTML -> AST. `justify` has no pipe-table marker, so it is not a cell align.
953
+ // The table-level `<table data-align>` form is read separately in the `table` branch.
954
+ const cellTextAlign = (getDeclaration(parseStyleDeclarations(node.attributes?.style || ''), 'text-align')
955
+ || node.attributes?.align || '').toLowerCase();
956
+ const cellAlign = ['left', 'center', 'right'].includes(cellTextAlign)
957
+ ? cellTextAlign
958
+ : undefined;
934
959
  const cellNode = {
935
960
  type: 'cell',
936
961
  metadata: {
937
962
  colSpan: colSpan && !isNaN(colSpan) ? colSpan : undefined,
938
- rowSpan: rowSpan && !isNaN(rowSpan) ? rowSpan : undefined
963
+ rowSpan: rowSpan && !isNaN(rowSpan) ? rowSpan : undefined,
964
+ align: cellAlign
939
965
  },
940
966
  children: parseChildren(node, newFormatting, listContext),
941
967
  htmlAttributes: collectHtmlAttributes(node, ['colspan', 'rowspan', 'align'])
@@ -991,6 +1017,7 @@ const parseHtml = async (buffer, config) => {
991
1017
  metadata: {
992
1018
  attachmentName: name,
993
1019
  altText: alt,
1020
+ title: node.attributes?.title,
994
1021
  width,
995
1022
  align
996
1023
  }
@@ -1002,6 +1029,7 @@ const parseHtml = async (buffer, config) => {
1002
1029
  metadata: {
1003
1030
  url: src,
1004
1031
  altText: alt,
1032
+ title: node.attributes?.title,
1005
1033
  width,
1006
1034
  align
1007
1035
  }
@@ -1014,6 +1042,7 @@ const parseHtml = async (buffer, config) => {
1014
1042
  metadata: {
1015
1043
  url: src,
1016
1044
  altText: alt,
1045
+ title: node.attributes?.title,
1017
1046
  anchorIds: anchorIds.length > 0 ? anchorIds : undefined,
1018
1047
  width,
1019
1048
  align
@@ -1055,16 +1084,20 @@ const parseHtml = async (buffer, config) => {
1055
1084
  }
1056
1085
  else if (href) {
1057
1086
  const linkType = href.startsWith('#') ? 'internal' : 'external';
1087
+ const linkTitle = node.attributes?.title;
1058
1088
  children.forEach(c => {
1059
1089
  if (c.type === 'text') {
1060
- c.metadata = { ...c.metadata, link: href, linkType };
1090
+ c.metadata = { ...c.metadata, link: href, linkType, title: linkTitle };
1061
1091
  }
1062
1092
  });
1063
1093
  }
1064
1094
  return children;
1065
1095
  }
1066
1096
  if (tagName === 'br') {
1067
- const brNode = { type: 'break', metadata: { breakType: 'textWrapping' } };
1097
+ // A <br> is a hard line break: `carriageReturn` so the Markdown generator emits a
1098
+ // hard break (` \n`, or a `<br>` inside a table cell) that re-imports as a <br>.
1099
+ // `textWrapping` emitted a bare `\n` in a paragraph, which re-imports as a space.
1100
+ const brNode = { type: 'break', metadata: { breakType: 'carriageReturn' } };
1068
1101
  if (config.includeRawContent) {
1069
1102
  brNode.rawContent = '<br/>';
1070
1103
  }
@@ -166,6 +166,15 @@ const parseMarkdown = async (buffer, config) => {
166
166
  mathBlocks.push(latex);
167
167
  return `\n\n${id}\n\n`;
168
168
  });
169
+ // Single-line `$$...$$` occupying its own line is display (block) math too. Without this it
170
+ // falls through to the inline `$...$` tokenizer, which matches the INNER `$\int$` and leaks
171
+ // the outer pair as two stray literal `$`. Runs after the multi-line pass, whose placeholders
172
+ // carry no `$$` and so can't be re-matched. `(?!\$)` rejects `$$$...`/empty `$$$$`.
173
+ textStr = textStr.replace(/^\$\$(?!\$)([^\n]+?)\$\$[ \t]*$/gm, (_match, latex) => {
174
+ const id = `__MATH_BLOCK_${mathBlocks.length}__`;
175
+ mathBlocks.push(latex);
176
+ return `\n\n${id}\n\n`;
177
+ });
169
178
  // Extract GLFM-style fenced-div admonitions (`:::note ... :::`) before block splitting,
170
179
  // since their body may itself contain blank lines that would otherwise fragment them.
171
180
  // The `> [!NOTE]` GitHub form doesn't need this - it's detected inline in the blockquote
@@ -259,7 +268,15 @@ const parseMarkdown = async (buffer, config) => {
259
268
  // Builds the same image/link node shape regardless of whether the URL came from
260
269
  // an inline `(url)` or a resolved reference definition - shared by the inline
261
270
  // image/link branch and the two reference-style branches below.
262
- const buildLinkOrImageNodes = (isImage, altText, url, attrsStr) => {
271
+ // Split a Markdown inline destination `url "title"` (also `'title'` / `(title)`) into its
272
+ // URL and optional title. The inline parser previously kept the whole thing as the URL, so
273
+ // `[t](u "T")` produced href `u "T"`; reference-style `[t][id]` already split it correctly.
274
+ const splitUrlTitle = (raw) => {
275
+ const m = raw.trim().match(/^(.*?)\s+(?:"([^"]*)"|'([^']*)'|\(([^)]*)\))\s*$/);
276
+ return m ? { url: m[1].trim(), title: m[2] ?? m[3] ?? m[4] } : { url: raw };
277
+ };
278
+ const buildLinkOrImageNodes = (isImage, altText, rawUrl, attrsStr) => {
279
+ const { url, title } = splitUrlTitle(rawUrl);
263
280
  if (isImage) {
264
281
  // Pandoc-style attribute list immediately after an image, e.g. {width=50% .centered}
265
282
  const attrs = attrsStr !== undefined ? parseAttributeList(attrsStr) : undefined;
@@ -276,15 +293,15 @@ const parseMarkdown = async (buffer, config) => {
276
293
  name,
277
294
  extension: mimeType.split('/')[1]
278
295
  });
279
- return [{ type: 'image', metadata: { attachmentName: name, altText, ...attrs } }];
296
+ return [{ type: 'image', metadata: { attachmentName: name, altText, title, ...attrs } }];
280
297
  }
281
298
  }
282
- return [{ type: 'image', metadata: { url, altText, ...attrs } }];
299
+ return [{ type: 'image', metadata: { url, altText, title, ...attrs } }];
283
300
  }
284
301
  const linkNodes = parseInline(altText, currentFormatting);
285
302
  linkNodes.forEach(n => {
286
303
  if (n.type === 'text') {
287
- n.metadata = { link: url, linkType: 'external' };
304
+ n.metadata = { link: url, linkType: 'external', title };
288
305
  }
289
306
  });
290
307
  return linkNodes;
@@ -314,7 +331,7 @@ const parseMarkdown = async (buffer, config) => {
314
331
  // Inline math requires no whitespace right after the opening $ or right before the
315
332
  // closing $, the common heuristic (matching Pandoc/KaTeX) for avoiding false
316
333
  // positives on currency like "$5 and $10".
317
- const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)~~|(?<codeFence>`+)(?<codeContent>(?:(?!\k<codeFence>)[\s\S])+?)\k<codeFence>(?!`)|<u>(?<underline>.+?)<\/u>|<sub>(?<subscript>.+?)<\/sub>|<sup>(?<superscript>.+?)<\/sup>|<span\s+style="(?<spanStyle>[^"]*)">(?<spanContent>.+?)<\/span>|\[\^(?<footnoteId>[^\]]+)\]|\[@(?<citationKey>[a-zA-Z0-9_:.-]+)\]|\[\[(?<wikiPage>[^\]|]+)(?:\|(?<wikiAlias>[^\]]+))?\]\]|(?<refBang>!?)\[(?<refText>[^\]]*)\]\[(?<refId>[^\]]*)\]|(?<shortBang>!?)\[(?<shortText>[^\]]+)\]|<(?<autolinkUrl>(?:https?|mailto):[^\s<>]+)>|\$(?!\s)(?<mathInline>[^$\n]+?)(?<!\s)\$/g;
334
+ const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)~~|==(?<highlight>.+?)==|(?<codeFence>`+)(?<codeContent>(?:(?!\k<codeFence>)[\s\S])+?)\k<codeFence>(?!`)|<u>(?<underline>.+?)<\/u>|<sub>(?<subscript>.+?)<\/sub>|<sup>(?<superscript>.+?)<\/sup>|(?<lineBreak><br\s*\/?>)|<span\s+style="(?<spanStyle>[^"]*)">(?<spanContent>.+?)<\/span>|\[\^(?<footnoteId>[^\]]+)\]|\[@(?<citationKey>[a-zA-Z0-9_:.-]+)\]|\[\[(?<wikiPage>[^\]|]+)(?:\|(?<wikiAlias>[^\]]+))?\]\]|(?<refBang>!?)\[(?<refText>[^\]]*)\]\[(?<refId>[^\]]*)\]|(?<shortBang>!?)\[(?<shortText>[^\]]+)\]|<(?<autolinkUrl>(?:https?|mailto):[^\s<>]+)>|\$(?!\s)(?<mathInline>[^$\n]+?)(?<!\s)\$/g;
318
335
  let lastIndex = 0;
319
336
  let match;
320
337
  while ((match = regex.exec(text)) !== null) {
@@ -343,6 +360,9 @@ const parseMarkdown = async (buffer, config) => {
343
360
  else if (g.strike !== undefined) { // Strikethrough
344
361
  nodes.push(...parseInline(g.strike, { ...currentFormatting, strikethrough: true }));
345
362
  }
363
+ else if (g.highlight !== undefined) { // ==highlight== (Obsidian/extended); additive on import
364
+ nodes.push(...parseInline(g.highlight, { ...currentFormatting, backgroundColor: '#ffff00' }));
365
+ }
346
366
  else if (g.codeContent !== undefined) { // Inline code (any matching backtick-run length)
347
367
  nodes.push({ type: 'text', text: g.codeContent, formatting: { ...currentFormatting, font: 'monospace' } });
348
368
  }
@@ -355,6 +375,12 @@ const parseMarkdown = async (buffer, config) => {
355
375
  else if (g.superscript !== undefined) { // Superscript
356
376
  nodes.push(...parseInline(g.superscript, { ...currentFormatting, superscript: true }));
357
377
  }
378
+ else if (g.lineBreak !== undefined) { // Raw inline <br>/<br/>/<br /> - a hard line break.
379
+ // MarkdownGenerator emits a raw <br> for a line break inside a table cell (a GFM pipe
380
+ // cell can't hold a newline), so the parser must read it back symmetrically as a break
381
+ // node instead of escaping it to literal `&lt;br&gt;` text and destroying it.
382
+ nodes.push({ type: 'break', metadata: { breakType: 'carriageReturn' } });
383
+ }
358
384
  else if (g.spanContent !== undefined) { // Inline styled span: color / highlight / font-size
359
385
  const style = g.spanStyle || '';
360
386
  const styled = { ...currentFormatting };
@@ -604,6 +630,31 @@ const parseMarkdown = async (buffer, config) => {
604
630
  blocks.push(currentSubBlock.join('\n'));
605
631
  }
606
632
  }
633
+ // Re-join a list block that a blank line tore away from its parent. The generator's older
634
+ // loose output (`- a\n\n\n - a1`) and foreign editors both split a nested item into its
635
+ // own block; left apart, the child's leading indent is stripped by the per-block `trim()`
636
+ // below and it reparses as a flat top-level item under a fresh listId. Merge a block back
637
+ // into the preceding one only when the previous block is itself a list (its FIRST line is a
638
+ // marker - the sub-splitter guarantees such a block holds only marker/continuation lines) and
639
+ // the current block OPENS with an INDENTED marker. An unindented `- b` after a blank line is
640
+ // deliberately left split (a flat loose list keeps its own listId), and anything that is not
641
+ // an indented marker (continuation text, indented code, placeholders) never triggers a merge.
642
+ const listMarkerStart = /^(\s*)([-*+]|\d+[.)])\s+/;
643
+ const indentedMarkerStart = /^(?: {2,}|\t)\s*(?:[-*+]|\d+[.)])\s+/;
644
+ const mergedBlocks = [];
645
+ for (const block of blocks) {
646
+ const prev = mergedBlocks[mergedBlocks.length - 1];
647
+ if (prev !== undefined
648
+ && listMarkerStart.test(prev.split('\n', 1)[0])
649
+ && indentedMarkerStart.test(block.split('\n', 1)[0])) {
650
+ mergedBlocks[mergedBlocks.length - 1] = `${prev}\n${block}`;
651
+ }
652
+ else {
653
+ mergedBlocks.push(block);
654
+ }
655
+ }
656
+ blocks.length = 0;
657
+ blocks.push(...mergedBlocks);
607
658
  let listIdCounter = 1;
608
659
  let currentAlignment = undefined;
609
660
  for (let block of blocks) {
@@ -968,11 +1019,23 @@ const parseMarkdown = async (buffer, config) => {
968
1019
  else {
969
1020
  const lines = block.trim().split('\n');
970
1021
  const rows = [];
1022
+ // Pre-scan the separator row for per-column GFM alignment (`:--` left, `:-:` center,
1023
+ // `--:` right; a bare `--` column has none), so every cell can carry its column's
1024
+ // alignment on CellMetadata.align (the header row precedes the separator, so a
1025
+ // per-cell pass alone could not see it).
1026
+ const sepLine = lines.find(l => l.match(/^\|?\s*:?-+:?\s*(?:\|\s*:?-+:?\s*)*\|?$/));
1027
+ const columnAligns = sepLine
1028
+ ? sepLine.replace(/^\||\|$/g, '').split('|').map(c => {
1029
+ const t = c.trim();
1030
+ const l = t.startsWith(':'), r = t.endsWith(':');
1031
+ return (l && r) ? 'center' : r ? 'right' : l ? 'left' : null;
1032
+ })
1033
+ : [];
971
1034
  for (let i = 0; i < lines.length; i++) {
972
1035
  if (lines[i].match(/^\|?\s*:?-+:?\s*(?:\|\s*:?-+:?\s*)*\|?$/))
973
1036
  continue; // Separator row (per-cell `:?-+:?`, GFM-style; accepts short cells like `|-|-|`)
974
1037
  const cellsStr = lines[i].replace(/^\||\|$/g, '').split('|');
975
- const cells = cellsStr.map(c => {
1038
+ const cells = cellsStr.map((c, colIdx) => {
976
1039
  // Recognize the MarkdownGenerator's own cell-alignment fallback,
977
1040
  // `<div style="text-align: X">…</div>`, and lift it into an aligned
978
1041
  // paragraph so it round-trips as alignment instead of being escaped to
@@ -981,17 +1044,25 @@ const parseMarkdown = async (buffer, config) => {
981
1044
  let cellAlign;
982
1045
  cellText = cellText.replace(/<div\s+style="text-align:\s*(left|center|right|justify);?"\s*>([\s\S]*?)<\/div>/gi, (_m, a, inner) => { cellAlign = a.toLowerCase(); return inner; });
983
1046
  const inline = parseInline(cellText, i === 0 ? { bold: true } : {});
1047
+ const colAlign = columnAligns[colIdx] ?? undefined;
1048
+ const cellMeta = colAlign ? { col: colIdx, align: colAlign } : undefined;
984
1049
  if (cellAlign && cellAlign !== 'left') {
985
1050
  return {
986
1051
  type: 'cell',
1052
+ metadata: cellMeta,
987
1053
  children: [{ type: 'paragraph', metadata: { alignment: cellAlign }, children: inline }]
988
1054
  };
989
1055
  }
990
- return { type: 'cell', children: inline };
1056
+ return { type: 'cell', metadata: cellMeta, children: inline };
991
1057
  });
992
1058
  rows.push({ type: 'row', children: cells });
993
1059
  }
994
- content.push({ type: 'table', metadata: tableAlign ? { align: tableAlign } : undefined, children: rows });
1060
+ // If every explicitly-aligned column agrees, also expose it as the table-level align,
1061
+ // so an editor that models one alignment per table (and HTML data-align) round-trips.
1062
+ const explicitAligns = columnAligns.filter((a) => a !== null);
1063
+ const uniformAlign = explicitAligns.length > 0 && explicitAligns.every(a => a === explicitAligns[0]) ? explicitAligns[0] : undefined;
1064
+ const resolvedTableAlign = tableAlign || uniformAlign;
1065
+ content.push({ type: 'table', metadata: resolvedTableAlign ? { align: resolvedTableAlign } : undefined, children: rows });
995
1066
  continue;
996
1067
  }
997
1068
  }