officeparser 7.5.1 → 7.6.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -12,6 +12,19 @@ const sanitize_js_1 = require("../utils/sanitize.js");
12
12
  * `parseNode` for why this value and not a larger one.
13
13
  */
14
14
  const MAX_HTML_NESTING_DEPTH = 256;
15
+ /**
16
+ * Decode the handful of HTML entities this parser leaves intact. Text nodes and attribute
17
+ * values are kept in their raw escaped form during parsing (see `parseAttributes`), so any
18
+ * branch that lifts text or an attribute into AST content has to decode first - `<` inside
19
+ * a code/math body is a less-than operator, not markup.
20
+ */
21
+ const decodeEntities = (s) => s
22
+ .replace(/ /g, ' ')
23
+ .replace(/&lt;/g, '<')
24
+ .replace(/&gt;/g, '>')
25
+ .replace(/&amp;/g, '&')
26
+ .replace(/&quot;/g, '"')
27
+ .replace(/&#39;/g, "'");
15
28
  /**
16
29
  * Presents an `HtmlNode` as a `MathNode` for the shared MathML converter.
17
30
  *
@@ -22,13 +35,7 @@ const MAX_HTML_NESTING_DEPTH = 256;
22
35
  const toMathNode = (node) => ({
23
36
  tagName: node.tagName,
24
37
  attributes: node.attributes,
25
- text: node.text === undefined ? undefined : node.text
26
- .replace(/&nbsp;/g, ' ')
27
- .replace(/&lt;/g, '<')
28
- .replace(/&gt;/g, '>')
29
- .replace(/&amp;/g, '&')
30
- .replace(/&quot;/g, '"')
31
- .replace(/&#39;/g, "'"),
38
+ text: node.text === undefined ? undefined : decodeEntities(node.text),
32
39
  children: (node.children || []).map(toMathNode),
33
40
  });
34
41
  const parseAttributes = (attrString) => {
@@ -165,9 +172,36 @@ const parseHtmlTree = (html) => {
165
172
  cursor = commentEnd !== -1 ? commentEnd + 3 : html.length;
166
173
  continue;
167
174
  }
168
- // indexOf (not substring().match) so scanning for the tag end is O(1) in
169
- // allocation — a document with many "<" chars would otherwise be O(n^2).
170
- const tagEndIdx = html.indexOf('>', tagStart);
175
+ // Scan for the tag's closing '>', skipping any that appear inside a quoted attribute value.
176
+ // Browsers do NOT escape '>' inside attribute values on serialization, so a literal '>' there
177
+ // (e.g. a mermaid diagram's `-->` in data-mermaid) must not be read as the tag end. The scan
178
+ // is linear in the tag's length and the cursor never rewinds, so the parse stays O(n) overall
179
+ // (no substring().match allocation per '<').
180
+ let tagEndIdx = -1;
181
+ let attrQuote = '';
182
+ for (let i = tagStart + 1; i < html.length; i++) {
183
+ const ch = html[i];
184
+ if (attrQuote) {
185
+ if (ch === attrQuote)
186
+ attrQuote = '';
187
+ }
188
+ else if (ch === '"' || ch === '\'') {
189
+ attrQuote = ch;
190
+ }
191
+ else if (ch === '>') {
192
+ tagEndIdx = i;
193
+ break;
194
+ }
195
+ }
196
+ if (tagEndIdx === -1) {
197
+ // The quote-aware scan ran to the end without closing the tag. That is almost always an
198
+ // unbalanced quote from a stray unescaped '<' in prose (e.g. "a < b's weight"), not a
199
+ // genuinely truncated tag. Retry naively for the next literal '>': the resulting
200
+ // pseudo-tag is then dropped, so a malformed run degrades exactly as it did before the
201
+ // quote-aware scan existed instead of swallowing the rest of the document into one text
202
+ // node. Well-formed input with balanced quotes never reaches here.
203
+ tagEndIdx = html.indexOf('>', tagStart);
204
+ }
171
205
  if (tagEndIdx === -1) {
172
206
  const text = html.substring(tagStart);
173
207
  current.children.push({ type: 'text', text, children: [], parent: current });
@@ -335,6 +369,9 @@ const parseHtml = async (buffer, config) => {
335
369
  // body loop, since references can appear anywhere earlier in the document) and
336
370
  // consulted by parseChildren's <sup data-footnote-ref> handling below.
337
371
  const footnoteDefinitions = new Map();
372
+ // Keys a `<sup data-footnote-ref>` actually consumed, so definitions in the section that no
373
+ // reference points at (orphans) can be recovered at the end instead of silently dropped.
374
+ const referencedFootnoteKeys = new Set();
338
375
  // --- Generic attribute pass-through (htmlParserConfig.preserveAttributes) ---------------
339
376
  // Captures attributes no typed metadata field consumed, so they can be replayed on
340
377
  // generation. Everything here is a *defence-in-depth* filter: HtmlGenerator sanitizes again
@@ -395,13 +432,7 @@ const parseHtml = async (buffer, config) => {
395
432
  throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.MAX_NESTING_DEPTH_EXCEEDED);
396
433
  }
397
434
  if (node.type === 'text') {
398
- let decodedText = (node.text || '')
399
- .replace(/&nbsp;/g, ' ')
400
- .replace(/&lt;/g, '<')
401
- .replace(/&gt;/g, '>')
402
- .replace(/&amp;/g, '&')
403
- .replace(/&quot;/g, '"')
404
- .replace(/&#39;/g, "'");
435
+ let decodedText = decodeEntities(node.text || '');
405
436
  if (!config.preserveXmlWhitespace) {
406
437
  decodedText = decodedText.replace(/\s+/g, ' ');
407
438
  }
@@ -436,6 +467,12 @@ const parseHtml = async (buffer, config) => {
436
467
  newFormatting.superscript = true;
437
468
  if (tagName === 'code')
438
469
  newFormatting.font = 'monospace';
470
+ if (tagName === 'mark') {
471
+ // <mark> is a highlight. Use its data-color when present (an inline
472
+ // background-color style, read below, still wins); a bare <mark> falls back to the
473
+ // conventional yellow so it round-trips as a highlight rather than plain text.
474
+ newFormatting.backgroundColor = node.attributes?.['data-color'] || '#ffff00';
475
+ }
439
476
  const styleAttr = node.attributes?.style || '';
440
477
  const alignAttr = node.attributes?.align || '';
441
478
  if (styleAttr || alignAttr) {
@@ -492,6 +529,7 @@ const parseHtml = async (buffer, config) => {
492
529
  // instead of inserting a visible node, matching WordParser's convention.
493
530
  if (child.type === 'element' && child.tagName === 'sup' && child.attributes?.['data-footnote-ref'] !== undefined) {
494
531
  const key = child.attributes['data-footnote-ref'];
532
+ referencedFootnoteKeys.add(key);
495
533
  const definition = footnoteDefinitions.get(key);
496
534
  const noteNode = {
497
535
  type: 'note',
@@ -520,7 +558,7 @@ const parseHtml = async (buffer, config) => {
520
558
  }
521
559
  return kids;
522
560
  };
523
- // YouTube embeds: inscript-editor's Youtube node renders
561
+ // YouTube embeds: attribute-driven editors render
524
562
  // <div data-youtube-video="ID" data-width="…" data-align="…">…<iframe…></div>.
525
563
  // Recognise both the wrapper div and a bare iframe so externally-authored HTML
526
564
  // (and a saved-then-reopened .md that fell back to raw HTML) both round-trip.
@@ -561,6 +599,27 @@ const parseHtml = async (buffer, config) => {
561
599
  embedNode.rawContent = '<iframe>...</iframe>';
562
600
  return embedNode;
563
601
  }
602
+ // Non-YouTube iframes are dropped by default (a deliberate security posture).
603
+ // preserveIframes opts back in, keeping the src as a generic 'iframe' embed; the
604
+ // src is scheme-checked again on generation, so this only widens what is retained.
605
+ // Decode the src (attribute values are stored entity-encoded) so it isn't
606
+ // double-escaped when the generator re-escapes it, which would corrupt query strings.
607
+ const decodedSrc = decodeEntities(src);
608
+ if ((0, sanitize_js_1.iframeAllowed)(decodedSrc, config.htmlParserConfig?.preserveIframes)) {
609
+ const iframeNode = {
610
+ type: 'embed',
611
+ text: decodedSrc,
612
+ metadata: {
613
+ embedType: 'iframe',
614
+ url: decodedSrc,
615
+ width: node.attributes?.width,
616
+ height: node.attributes?.height
617
+ }
618
+ };
619
+ if (config.includeRawContent)
620
+ iframeNode.rawContent = '<iframe>...</iframe>';
621
+ return iframeNode;
622
+ }
564
623
  return null;
565
624
  }
566
625
  // Footnotes section: its definitions were already extracted up front (see
@@ -570,22 +629,42 @@ const parseHtml = async (buffer, config) => {
570
629
  if (tagName === 'section' && node.attributes?.['data-footnotes'] !== undefined) {
571
630
  return null;
572
631
  }
573
- // Math: proposed contract (no editor node built yet) - HtmlGenerator emits
574
- // <span/div class="math math-inline|math-block" data-math="inline|block">
575
- // with the $-delimited LaTeX as the visible (escaped) text content.
632
+ // Math. Two accepted shapes, disambiguated by the `data-math` value:
633
+ // 1. This library's own output - `data-math="inline|block"` names the mode, and the
634
+ // LaTeX is the visible ($-delimited, escaped) text content.
635
+ // 2. Attribute-driven producers that put the raw LaTeX in `data-math` and signal the
636
+ // mode through the class (`math-inline`/`math-block`) or the tag.
637
+ // Anything whose `data-math` is exactly `inline`/`block` takes path 1 unchanged; every
638
+ // other value is treated as LaTeX (path 2). LaTeX literally equal to `inline`/`block`
639
+ // is the only ambiguous input, and its text content wins there anyway.
576
640
  if ((tagName === 'div' || tagName === 'span') && node.attributes?.['data-math'] !== undefined) {
577
- const mathMode = node.attributes['data-math'] === 'block' ? 'block' : 'inline';
578
- const rawText = node.children.map(c => c.text || '').join('')
579
- .replace(/&nbsp;/g, ' ')
580
- .replace(/&lt;/g, '<')
581
- .replace(/&gt;/g, '>')
582
- .replace(/&amp;/g, '&')
583
- .replace(/&quot;/g, '"')
584
- .replace(/&#39;/g, '\'');
585
- const delimiter = mathMode === 'block' ? '$$' : '$';
586
- const latex = rawText.startsWith(delimiter) && rawText.endsWith(delimiter)
587
- ? rawText.slice(delimiter.length, -delimiter.length)
588
- : rawText;
641
+ const dataMath = node.attributes['data-math'];
642
+ const modeIsExplicit = dataMath === 'inline' || dataMath === 'block';
643
+ const classTokens = (node.attributes?.class || '').split(/\s+/);
644
+ const rawText = decodeEntities(node.children.map(c => c.text || '').join(''));
645
+ // Prefer the text content; fall back to the attribute value (path 2 producers may
646
+ // emit an empty body).
647
+ const source = rawText || (modeIsExplicit ? '' : decodeEntities(dataMath));
648
+ // Strip whichever `$`/`$$` delimiters are actually present, independent of the
649
+ // resolved mode - a `$`-delimited body inside a <div> must not keep its delimiters.
650
+ // The delimiter also disambiguates the mode when neither an explicit `data-math` nor
651
+ // a `math-inline`/`math-block` class settles it (so `<div data-math="x">$x$</div>`
652
+ // reads as inline, not block-via-tag).
653
+ let latex = source;
654
+ let delimiterMode;
655
+ if (source.length >= 4 && source.startsWith('$$') && source.endsWith('$$')) {
656
+ latex = source.slice(2, -2);
657
+ delimiterMode = 'block';
658
+ }
659
+ else if (source.length >= 2 && source.startsWith('$') && source.endsWith('$')) {
660
+ latex = source.slice(1, -1);
661
+ delimiterMode = 'inline';
662
+ }
663
+ const mathMode = modeIsExplicit
664
+ ? dataMath
665
+ : classTokens.includes('math-block') ? 'block'
666
+ : classTokens.includes('math-inline') ? 'inline'
667
+ : delimiterMode ?? (tagName === 'div' ? 'block' : 'inline');
589
668
  return {
590
669
  type: 'code',
591
670
  text: latex,
@@ -611,7 +690,7 @@ const parseHtml = async (buffer, config) => {
611
690
  metadata: { math: isBlock ? 'block' : 'inline' }
612
691
  };
613
692
  }
614
- // Admonition: inscript-editor's Admonition node renders
693
+ // Admonition: attribute-driven editors render
615
694
  // <div class="admonition admonition-note" data-type="note">…children…</div>.
616
695
  if (tagName === 'div' && (node.attributes?.class || '').split(/\s+/).includes('admonition')) {
617
696
  const admonitionTypeAttr = node.attributes?.['data-type'];
@@ -627,6 +706,29 @@ const parseHtml = async (buffer, config) => {
627
706
  admonitionNode.rawContent = '<div class="admonition">...</div>';
628
707
  return admonitionNode;
629
708
  }
709
+ // Mermaid diagrams. Attribute-driven producers render a
710
+ // <div class="mermaid" data-mermaid="<code>"> with the code also as text content.
711
+ // Map either shape to a fenced code node with language `mermaid`, so it round-trips as
712
+ // a ```mermaid block. Previously this div fell through to generic handling and its
713
+ // code flattened to paragraph text.
714
+ if (tagName === 'div' && (node.attributes?.['data-mermaid'] !== undefined || (node.attributes?.class || '').split(/\s+/).includes('mermaid'))) {
715
+ const code = decodeEntities(node.children.map(c => c.text || '').join('')).trim()
716
+ || decodeEntities(node.attributes?.['data-mermaid'] || '');
717
+ // Only claim this as a mermaid code node when there is actual diagram source.
718
+ // A bare `class="mermaid"` div with nested elements (a mermaid.js-rendered <svg>,
719
+ // or a div merely reusing the class for styling) has no direct text and no
720
+ // data-mermaid; fall through to generic handling so its content is not dropped.
721
+ if (code) {
722
+ const mermaidNode = {
723
+ type: 'code',
724
+ text: code,
725
+ metadata: { language: 'mermaid' }
726
+ };
727
+ if (config.includeRawContent)
728
+ mermaidNode.rawContent = '<div data-mermaid>...</div>';
729
+ return mermaidNode;
730
+ }
731
+ }
630
732
  // Skip structural containers produced by HtmlGenerator to avoid deep AST nesting
631
733
  if (tagName === 'div' && (node.attributes?.class === 'container' ||
632
734
  node.attributes?.class === 'spreadsheet-container' ||
@@ -724,6 +826,21 @@ const parseHtml = async (buffer, config) => {
724
826
  metadata: { citationKey }
725
827
  };
726
828
  }
829
+ // Attribute-driven citation shape: a <span> carrying the `citation` class token (among
830
+ // any others) and a non-empty data-key. Produces the same bare-key text node as the
831
+ // <cite> form above. An empty/absent data-key falls through to generic span handling,
832
+ // so the span's visible text still survives.
833
+ if (tagName === 'span'
834
+ && (node.attributes?.class || '').split(/\s+/).includes('citation')
835
+ && node.attributes?.['data-key']) {
836
+ const citationKey = decodeEntities(node.attributes['data-key']);
837
+ return {
838
+ type: 'text',
839
+ text: citationKey,
840
+ formatting: Object.keys(newFormatting).length > 0 ? { ...newFormatting } : undefined,
841
+ metadata: { citationKey }
842
+ };
843
+ }
727
844
  if (tagName === 'ul' || tagName === 'ol') {
728
845
  const isNewTopLevel = !listContext;
729
846
  const newListContext = {
@@ -782,7 +899,7 @@ const parseHtml = async (buffer, config) => {
782
899
  return [selfNode, ...nestedLists];
783
900
  }
784
901
  if (tagName === 'table') {
785
- // CustomTable (inscript-editor) renders data-align on the <table> itself.
902
+ // Attribute-driven editors render data-align on the <table> itself.
786
903
  const tableAlignAttr = node.attributes?.['data-align'];
787
904
  const tableAlign = ['left', 'center', 'right'].includes(tableAlignAttr) ? tableAlignAttr : undefined;
788
905
  const tableNode = {
@@ -831,7 +948,7 @@ const parseHtml = async (buffer, config) => {
831
948
  if (tagName === 'img') {
832
949
  const src = node.attributes?.src;
833
950
  const alt = node.attributes?.alt;
834
- // CustomImage (inscript-editor) renders data-width/data-align, falling back to
951
+ // Attribute-driven editors render data-width/data-align, falling back to
835
952
  // parsing the inline style for consumers that only emit the CSS.
836
953
  const imgDecls = parseStyleDeclarations(node.attributes?.style || '');
837
954
  // Exact lookup, so `max-width: 100%` - the standard responsive-image style, and by
@@ -919,6 +1036,23 @@ const parseHtml = async (buffer, config) => {
919
1036
  }
920
1037
  });
921
1038
  }
1039
+ else if (node.attributes?.['data-wikilink'] !== undefined) {
1040
+ // Attribute-driven wikilink shape: the page lives in data-target, the display
1041
+ // text is the anchor's own content (or data-alias/data-target when the anchor
1042
+ // is empty). data-wikilink-page above keeps precedence over this form.
1043
+ const page = decodeEntities(node.attributes['data-target'] || '');
1044
+ if (!children.some(c => c.type === 'text')) {
1045
+ children.push({
1046
+ type: 'text',
1047
+ text: decodeEntities(node.attributes['data-alias'] || node.attributes['data-target'] || ''),
1048
+ });
1049
+ }
1050
+ children.forEach(c => {
1051
+ if (c.type === 'text') {
1052
+ c.metadata = { ...c.metadata, link: page, linkType: 'internal', wikilink: true };
1053
+ }
1054
+ });
1055
+ }
922
1056
  else if (href) {
923
1057
  const linkType = href.startsWith('#') ? 'internal' : 'external';
924
1058
  children.forEach(c => {
@@ -936,6 +1070,17 @@ const parseHtml = async (buffer, config) => {
936
1070
  }
937
1071
  return brNode;
938
1072
  }
1073
+ if (tagName === 'hr') {
1074
+ // A horizontal rule is a thematic break. This library tags an office page break
1075
+ // as <hr class="page-break"> on emission, so that variant round-trips back to a
1076
+ // page break; every other <hr> is thematic. Previously <hr> was dropped entirely.
1077
+ const isPageBreak = (node.attributes?.class || '').split(/\s+/).includes('page-break');
1078
+ const hrNode = { type: 'break', metadata: { breakType: isPageBreak ? 'page' : 'thematic' } };
1079
+ if (config.includeRawContent) {
1080
+ hrNode.rawContent = '<hr/>';
1081
+ }
1082
+ return hrNode;
1083
+ }
939
1084
  if (tagName === 'pre') {
940
1085
  const codeNode = node.children.find(c => c.tagName === 'code');
941
1086
  let language;
@@ -945,10 +1090,18 @@ const parseHtml = async (buffer, config) => {
945
1090
  const langMatch = classAttr.split(' ').find((c) => c.startsWith('language-'));
946
1091
  if (langMatch)
947
1092
  language = langMatch.replace('language-', '');
948
- codeText = codeNode.children.map(c => c.text || '').join('');
1093
+ // Decode entities: the code body is stored raw, so `&lt;`/`&gt;`/`&amp;` (e.g. a
1094
+ // mermaid `-->` arrow, or `a < b` in a snippet) must be turned back into text.
1095
+ codeText = decodeEntities(codeNode.children.map(c => c.text || '').join(''));
949
1096
  }
950
1097
  else {
951
- codeText = node.children.map(c => c.text || '').join('');
1098
+ codeText = decodeEntities(node.children.map(c => c.text || '').join(''));
1099
+ }
1100
+ // A `mermaid` class token (on the <pre> or its <code>) names the language when no
1101
+ // explicit language-* class is present - some producers emit <pre class="mermaid">.
1102
+ if (!language && ((node.attributes?.class || '').split(/\s+/).includes('mermaid') ||
1103
+ (codeNode?.attributes?.class || '').split(/\s+/).includes('mermaid'))) {
1104
+ language = 'mermaid';
952
1105
  }
953
1106
  const preNode = {
954
1107
  type: 'code',
@@ -1027,6 +1180,21 @@ const parseHtml = async (buffer, config) => {
1027
1180
  }
1028
1181
  }
1029
1182
  }
1183
+ // Orphan footnote definitions: a `<section data-footnotes>` entry that no `<sup
1184
+ // data-footnote-ref>` consumed would otherwise be dropped (it is skipped in the body walk and
1185
+ // only materialised via a reference). Recover them as trailing `unreferenced` note nodes, the
1186
+ // same shape MarkdownParser produces, so md -> html -> md preserves the definition instead of
1187
+ // turning it into junk text with a dead back-link.
1188
+ for (const [key, definition] of footnoteDefinitions) {
1189
+ if (referencedFootnoteKeys.has(key))
1190
+ continue;
1191
+ content.push({
1192
+ type: 'note',
1193
+ text: (definition || []).map(d => d.text || '').join(''),
1194
+ children: definition || [],
1195
+ metadata: { noteType: 'footnote', noteId: key, unreferenced: true },
1196
+ });
1197
+ }
1030
1198
  const toTextSync = () => content.map(n => {
1031
1199
  const getText = (node) => {
1032
1200
  if (node.type === 'text' || node.type === 'code')
@@ -3,6 +3,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.parseMarkdown = void 0;
4
4
  const astUtils_js_1 = require("../utils/astUtils.js");
5
5
  const errorUtils_js_1 = require("../utils/errorUtils.js");
6
+ const sanitize_js_1 = require("../utils/sanitize.js");
6
7
  // Sentinel node type for a standalone bookmark-anchor block (e.g. `<a id="x"></a>` on its
7
8
  // own line). A post-parse pass folds these into the following node's anchorIds so they
8
9
  // round-trip as real anchors rather than being escaped to visible text on regeneration.
@@ -61,7 +62,12 @@ const parseMarkdown = async (buffer, config) => {
61
62
  const metadata = {};
62
63
  const attachments = [];
63
64
  // Parse YAML Front Matter
64
- if (textStr.startsWith('---\n')) {
65
+ if (/^---\n---[ \t]*(?:\n|$)/.test(textStr)) {
66
+ // Empty frontmatter block: strip it so `---\n---` isn't misread as a setext `## ---`
67
+ // heading (empty metadata used to emit exactly this shape, and other producers do too).
68
+ textStr = textStr.replace(/^---\n---[ \t]*(?:\n|$)/, '');
69
+ }
70
+ else if (textStr.startsWith('---\n')) {
65
71
  const endIdx = textStr.indexOf('\n---\n', 4);
66
72
  if (endIdx !== -1) {
67
73
  const frontMatter = textStr.substring(4, endIdx);
@@ -74,9 +80,15 @@ const parseMarkdown = async (buffer, config) => {
74
80
  if (match) {
75
81
  const key = match[1].trim();
76
82
  const rawVal = match[2].trim();
77
- const val = rawVal.replace(/^"(.*)"$/, '$1');
83
+ // A quoted scalar is explicitly a string in YAML: strip the quotes but never
84
+ // coerce it, so `version: "123"` / `flag: "true"` keep their string-ness across
85
+ // a save/reload cycle instead of silently degrading to a number/boolean on the
86
+ // next parse (which the generator would then re-emit unquoted, losing the type
87
+ // permanently). Only bare, unquoted scalars coerce.
88
+ const isQuoted = /^"(.*)"$/.test(rawVal) || /^'(.*)'$/.test(rawVal);
89
+ const val = rawVal.replace(/^"(.*)"$/, '$1').replace(/^'(.*)'$/, '$1');
78
90
  let parsedVal = val;
79
- if (rawVal.startsWith('[') && rawVal.endsWith(']')) {
91
+ if (!isQuoted && rawVal.startsWith('[') && rawVal.endsWith(']')) {
80
92
  // Flow-array (`tags: [a, b]`) or JSON-array (`tags: ["a","b"]`) value -
81
93
  // parse into a real array instead of storing the literal bracket string,
82
94
  // so it round-trips symmetrically with MarkdownGenerator's frontmatter output.
@@ -89,6 +101,8 @@ const parseMarkdown = async (buffer, config) => {
89
101
  parsedVal = inner === '' ? [] : splitFlowArrayItems(inner).map(item => item.replace(/^['"](.*)['"]$/, '$1'));
90
102
  }
91
103
  }
104
+ else if (isQuoted)
105
+ parsedVal = val;
92
106
  else if (val === 'true')
93
107
  parsedVal = true;
94
108
  else if (val === 'false')
@@ -167,11 +181,32 @@ const parseMarkdown = async (buffer, config) => {
167
181
  });
168
182
  // Extract footnote definitions (`[^id]: text`) before block splitting, since
169
183
  // definitions conventionally live at the end of the document, after every place
170
- // they're referenced - inline parsing below needs the full map upfront. v1 only
171
- // supports single-line definitions (MultiMarkdown/Pandoc/GLFM's common baseline).
184
+ // they're referenced - inline parsing below needs the full map upfront. The first line
185
+ // may be followed by continuation lines indented one level (4 spaces or a tab), which are
186
+ // dedented and joined onto the definition (Pandoc/GFM). A 4-space-indented block right after
187
+ // a definition is therefore read as its continuation rather than as a standalone code block.
188
+ // Supported (lossless) shape: contiguous continuation - the indented lines follow the
189
+ // definition with no blank line between them. Known limitation (6.E.1): a continuation
190
+ // separated from the definition by a BLANK line is not folded in - the regex below stops at
191
+ // the blank line, and the indented block after it re-parses as a fenced/indented code block on
192
+ // save. Multi-paragraph footnotes should therefore use the contiguous form.
172
193
  const footnoteDefinitions = new Map();
173
- textStr = textStr.replace(/^\[\^([^\]]+)\]:[ \t]*(.*)$/gm, (_match, id, definition) => {
174
- footnoteDefinitions.set(id, definition.trim());
194
+ // Every id a `[^id]` reference consumes, so definitions that are never referenced can be
195
+ // detected at the end and preserved rather than silently dropped (see the orphan sweep below).
196
+ const referencedFootnoteIds = new Set();
197
+ // One reused note node per referenced id. Repeated `[^id]` references are a single shared
198
+ // footnote in Markdown, so they must not each materialise a full copy of the body - the
199
+ // generators would otherwise renumber them to [^1]/[^2] and duplicate the definition. Office
200
+ // notes reach the generators as distinct objects even when they share a numeric id, so those
201
+ // stay separate; only genuinely shared Markdown references collapse.
202
+ const footnoteNodesById = new Map();
203
+ textStr = textStr.replace(/^\[\^([^\]]+)\]:[ \t]*(.*(?:\n(?: {4}|\t).*)*)$/gm, (_match, id, definition) => {
204
+ const dedented = String(definition)
205
+ .split('\n')
206
+ .map((line, i) => i === 0 ? line : line.replace(/^(?: {4}|\t)/, ''))
207
+ .join('\n')
208
+ .trim();
209
+ footnoteDefinitions.set(id, dedented);
175
210
  return '';
176
211
  });
177
212
  // Extract Markdown Extra abbreviation definitions (`*[HTML]: Hypertext Markup Language`)
@@ -279,7 +314,7 @@ const parseMarkdown = async (buffer, config) => {
279
314
  // Inline math requires no whitespace right after the opening $ or right before the
280
315
  // closing $, the common heuristic (matching Pandoc/KaTeX) for avoiding false
281
316
  // positives on currency like "$5 and $10".
282
- const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)~~|(?<codeFence>`+)(?<codeContent>(?:(?!\k<codeFence>)[\s\S])+?)\k<codeFence>(?!`)|<u>(?<underline>.+?)<\/u>|<sub>(?<subscript>.+?)<\/sub>|<sup>(?<superscript>.+?)<\/sup>|\[\^(?<footnoteId>[^\]]+)\]|\[@(?<citationKey>[a-zA-Z0-9_:.-]+)\]|\[\[(?<wikiPage>[^\]|]+)(?:\|(?<wikiAlias>[^\]]+))?\]\]|(?<refBang>!?)\[(?<refText>[^\]]*)\]\[(?<refId>[^\]]*)\]|(?<shortBang>!?)\[(?<shortText>[^\]]+)\]|<(?<autolinkUrl>(?:https?|mailto):[^\s<>]+)>|\$(?!\s)(?<mathInline>[^$\n]+?)(?<!\s)\$/g;
317
+ const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)~~|(?<codeFence>`+)(?<codeContent>(?:(?!\k<codeFence>)[\s\S])+?)\k<codeFence>(?!`)|<u>(?<underline>.+?)<\/u>|<sub>(?<subscript>.+?)<\/sub>|<sup>(?<superscript>.+?)<\/sup>|(?<lineBreak><br\s*\/?>)|<span\s+style="(?<spanStyle>[^"]*)">(?<spanContent>.+?)<\/span>|\[\^(?<footnoteId>[^\]]+)\]|\[@(?<citationKey>[a-zA-Z0-9_:.-]+)\]|\[\[(?<wikiPage>[^\]|]+)(?:\|(?<wikiAlias>[^\]]+))?\]\]|(?<refBang>!?)\[(?<refText>[^\]]*)\]\[(?<refId>[^\]]*)\]|(?<shortBang>!?)\[(?<shortText>[^\]]+)\]|<(?<autolinkUrl>(?:https?|mailto):[^\s<>]+)>|\$(?!\s)(?<mathInline>[^$\n]+?)(?<!\s)\$/g;
283
318
  let lastIndex = 0;
284
319
  let match;
285
320
  while ((match = regex.exec(text)) !== null) {
@@ -320,16 +355,50 @@ const parseMarkdown = async (buffer, config) => {
320
355
  else if (g.superscript !== undefined) { // Superscript
321
356
  nodes.push(...parseInline(g.superscript, { ...currentFormatting, superscript: true }));
322
357
  }
358
+ else if (g.lineBreak !== undefined) { // Raw inline <br>/<br/>/<br /> - a hard line break.
359
+ // MarkdownGenerator emits a raw <br> for a line break inside a table cell (a GFM pipe
360
+ // cell can't hold a newline), so the parser must read it back symmetrically as a break
361
+ // node instead of escaping it to literal `&lt;br&gt;` text and destroying it.
362
+ nodes.push({ type: 'break', metadata: { breakType: 'carriageReturn' } });
363
+ }
364
+ else if (g.spanContent !== undefined) { // Inline styled span: color / highlight / font-size
365
+ const style = g.spanStyle || '';
366
+ const styled = { ...currentFormatting };
367
+ // Anchor each property to a declaration boundary so `color` doesn't match inside
368
+ // `background-color`.
369
+ const prop = (name) => {
370
+ const m = style.match(new RegExp(`(?:^|;)\\s*${name}\\s*:\\s*([^;]+)`, 'i'));
371
+ return m ? m[1].trim() : undefined;
372
+ };
373
+ const color = prop('color');
374
+ if (color)
375
+ styled.color = color;
376
+ const background = prop('background-color');
377
+ if (background)
378
+ styled.backgroundColor = background;
379
+ const size = prop('font-size');
380
+ if (size)
381
+ styled.size = size;
382
+ nodes.push(...parseInline(g.spanContent, styled));
383
+ }
323
384
  else if (g.footnoteId !== undefined) { // Footnote reference
324
385
  const noteId = g.footnoteId;
325
- const definition = footnoteDefinitions.get(noteId);
326
- const noteChildren = definition !== undefined ? parseInline(definition) : [];
327
- const noteNode = {
328
- type: 'note',
329
- text: noteChildren.map(c => c.text || '').join(''),
330
- children: noteChildren,
331
- metadata: { noteType: 'footnote', noteId }
332
- };
386
+ referencedFootnoteIds.add(noteId);
387
+ // Reuse the same note object across every reference to this id (see the map's
388
+ // declaration): the first reference builds the body, the rest share it, so the
389
+ // generators assign one key and emit one definition.
390
+ let noteNode = footnoteNodesById.get(noteId);
391
+ if (!noteNode) {
392
+ const definition = footnoteDefinitions.get(noteId);
393
+ const noteChildren = definition !== undefined ? parseInline(definition) : [];
394
+ noteNode = {
395
+ type: 'note',
396
+ text: noteChildren.map(c => c.text || '').join(''),
397
+ children: noteChildren,
398
+ metadata: { noteType: 'footnote', noteId }
399
+ };
400
+ footnoteNodesById.set(noteId, noteNode);
401
+ }
333
402
  // Notes attach to the preceding text node (matches WordParser's convention);
334
403
  // fall back to an empty text node if the reference opens the inline run.
335
404
  if (nodes.length > 0) {
@@ -609,6 +678,36 @@ const parseMarkdown = async (buffer, config) => {
609
678
  });
610
679
  continue;
611
680
  }
681
+ // Generic iframe fallback: MarkdownGenerator's 'embed' case emits a single-line
682
+ // <iframe src="…"></iframe> for a preserved iframe when fallbackToHtml is on. Recognise
683
+ // it only when the caller opted into iframe preservation, so default parsing is unchanged.
684
+ const iframeMatch = block.match(/^<iframe\s+([^>]*?)\/?>(?:\s*<\/iframe>)?$/i);
685
+ if (iframeMatch) {
686
+ const attrsStr = iframeMatch[1];
687
+ // The emitted <iframe> HTML-escapes its attribute values (sanitizeUrl -> escapeHtml), so
688
+ // decode them back; otherwise the src double-escapes (`&amp;` -> `&amp;amp;`) and its
689
+ // query string is corrupted a little more on every save/reload cycle. `&amp;` is decoded
690
+ // last so a genuinely double-escaped value only unwinds one level per parse.
691
+ const decodeAttr = (s) => (s || '')
692
+ .replace(/&lt;/g, '<').replace(/&gt;/g, '>').replace(/&quot;/g, '"')
693
+ .replace(/&#39;/g, '\'').replace(/&amp;/g, '&');
694
+ const src = decodeAttr(attrsStr.match(/\bsrc="([^"]*)"/i)?.[1]);
695
+ if (src && (0, sanitize_js_1.iframeAllowed)(src, config.htmlParserConfig?.preserveIframes)) {
696
+ const width = attrsStr.match(/\bwidth="([^"]*)"/i)?.[1];
697
+ const height = attrsStr.match(/\bheight="([^"]*)"/i)?.[1];
698
+ content.push({
699
+ type: 'embed',
700
+ text: src,
701
+ metadata: {
702
+ embedType: 'iframe',
703
+ url: src,
704
+ width: width !== undefined ? decodeAttr(width) : undefined,
705
+ height: height !== undefined ? decodeAttr(height) : undefined
706
+ }
707
+ });
708
+ continue;
709
+ }
710
+ }
612
711
  // Code Block
613
712
  const codeMatch = block.match(/^__CODE_BLOCK_(\d+)__$/);
614
713
  if (codeMatch) {
@@ -918,9 +1017,10 @@ const parseMarkdown = async (buffer, config) => {
918
1017
  continue;
919
1018
  }
920
1019
  }
921
- // Hr
1020
+ // Hr - a thematic break (horizontal rule), not a page break, so it survives a save as
1021
+ // `---` rather than collapsing to a bare newline.
922
1022
  if (block.match(/^---+$|^\*\*\*+$|^___+$/)) {
923
- content.push({ type: 'break', metadata: { breakType: 'page' } });
1023
+ content.push({ type: 'break', metadata: { breakType: 'thematic' } });
924
1024
  continue;
925
1025
  }
926
1026
  // Paragraph
@@ -956,6 +1056,23 @@ const parseMarkdown = async (buffer, config) => {
956
1056
  content.length = 0;
957
1057
  content.push(...merged);
958
1058
  }
1059
+ // Orphan footnote definitions (defined but never referenced) would otherwise vanish entirely -
1060
+ // a user who deletes a `[^x]` reference but keeps its `[^x]: ...` definition loses the
1061
+ // definition on the next save. Preserve them as trailing note nodes, marked `unreferenced` so
1062
+ // the generators route them into their footnotes section (not inline) and emit no citation
1063
+ // marker or dangling back-link. Both generators still emit the definition (md: a `[^x]:` line;
1064
+ // html: a `div[data-footnote-id]` inside `section[data-footnotes]`, which re-parses on import).
1065
+ for (const [id, definition] of footnoteDefinitions) {
1066
+ if (referencedFootnoteIds.has(id))
1067
+ continue;
1068
+ const noteChildren = parseInline(definition);
1069
+ content.push({
1070
+ type: 'note',
1071
+ text: noteChildren.map(c => c.text || '').join(''),
1072
+ children: noteChildren,
1073
+ metadata: { noteType: 'footnote', noteId: id, unreferenced: true },
1074
+ });
1075
+ }
959
1076
  const toTextSync = () => content.map(n => {
960
1077
  const getText = (node) => {
961
1078
  if (node.type === 'text' || node.type === 'code')
@@ -458,23 +458,13 @@ const parseWord = async (buffer, config) => {
458
458
  }
459
459
  }
460
460
  }
461
- // Extract paragraph-level run properties
462
- let paragraphRunFormatting = { ...styleProps.formatting };
463
- if (pPr) {
464
- const pPrRPr = (0, xmlUtils_js_1.getFirstElementByTagName)(pPr, "w:rPr");
465
- if (pPrRPr) {
466
- const pPrFormatting = extractFormattingFromXml(pPrRPr);
467
- for (const key in pPrFormatting) {
468
- const value = pPrFormatting[key];
469
- if (value === false) {
470
- delete paragraphRunFormatting[key];
471
- }
472
- else if (value !== undefined) {
473
- paragraphRunFormatting[key] = value;
474
- }
475
- }
476
- }
477
- }
461
+ // Runs inherit their base formatting from the style chain: the paragraph style (seeded
462
+ // here, and re-applied via the run-style path below), then any character style, then the
463
+ // run's own properties. The paragraph-mark run properties (`<w:pPr><w:rPr>`) format only
464
+ // the paragraph mark glyph itself per OOXML ISO 29500 §17.3.1.29, so they are deliberately
465
+ // NOT folded into the run base - doing so bled the paragraph mark's bold/italic/color/etc.
466
+ // onto every run in the paragraph (issue #109).
467
+ const paragraphRunFormatting = { ...styleProps.formatting };
478
468
  // Extract text and children
479
469
  let text = '';
480
470
  const children = [];