officeparser 7.5.1 → 7.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -6
- package/dist/OfficeConverter.d.ts +2 -2
- package/dist/OfficeParser.d.ts +2 -2
- package/dist/OfficeParser.js +10 -0
- package/dist/cli.js +3 -0
- package/dist/defaults.js +3 -1
- package/dist/generators/ChunkingGenerator.d.ts +2 -1
- package/dist/generators/ChunkingGenerator.js +73 -6
- package/dist/generators/EpubGenerator.js +4 -1
- package/dist/generators/HtmlGenerator.js +117 -29
- package/dist/generators/MarkdownGenerator.js +120 -29
- package/dist/generators/PdfGenerator.js +4 -1
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +70 -12
- package/dist/officeparser.browser.iife.js +211 -202
- package/dist/officeparser.browser.mjs +211 -202
- package/dist/officeparser.browser.slim.d.ts +70 -12
- package/dist/officeparser.browser.slim.iife.js +213 -204
- package/dist/officeparser.browser.slim.mjs +213 -204
- package/dist/parsers/HtmlParser.js +206 -38
- package/dist/parsers/MarkdownParser.js +135 -18
- package/dist/parsers/WordParser.js +7 -17
- package/dist/sbom.cdx.json +95 -95
- package/dist/types.d.ts +68 -10
- package/dist/utils/configUtils.js +3 -1
- package/dist/utils/sanitize.d.ts +9 -0
- package/dist/utils/sanitize.js +26 -0
- package/package.json +5 -5
|
@@ -12,6 +12,19 @@ const sanitize_js_1 = require("../utils/sanitize.js");
|
|
|
12
12
|
* `parseNode` for why this value and not a larger one.
|
|
13
13
|
*/
|
|
14
14
|
const MAX_HTML_NESTING_DEPTH = 256;
|
|
15
|
+
/**
|
|
16
|
+
* Decode the handful of HTML entities this parser leaves intact. Text nodes and attribute
|
|
17
|
+
* values are kept in their raw escaped form during parsing (see `parseAttributes`), so any
|
|
18
|
+
* branch that lifts text or an attribute into AST content has to decode first - `<` inside
|
|
19
|
+
* a code/math body is a less-than operator, not markup.
|
|
20
|
+
*/
|
|
21
|
+
const decodeEntities = (s) => s
|
|
22
|
+
.replace(/ /g, ' ')
|
|
23
|
+
.replace(/</g, '<')
|
|
24
|
+
.replace(/>/g, '>')
|
|
25
|
+
.replace(/&/g, '&')
|
|
26
|
+
.replace(/"/g, '"')
|
|
27
|
+
.replace(/'/g, "'");
|
|
15
28
|
/**
|
|
16
29
|
* Presents an `HtmlNode` as a `MathNode` for the shared MathML converter.
|
|
17
30
|
*
|
|
@@ -22,13 +35,7 @@ const MAX_HTML_NESTING_DEPTH = 256;
|
|
|
22
35
|
const toMathNode = (node) => ({
|
|
23
36
|
tagName: node.tagName,
|
|
24
37
|
attributes: node.attributes,
|
|
25
|
-
text: node.text === undefined ? undefined : node.text
|
|
26
|
-
.replace(/ /g, ' ')
|
|
27
|
-
.replace(/</g, '<')
|
|
28
|
-
.replace(/>/g, '>')
|
|
29
|
-
.replace(/&/g, '&')
|
|
30
|
-
.replace(/"/g, '"')
|
|
31
|
-
.replace(/'/g, "'"),
|
|
38
|
+
text: node.text === undefined ? undefined : decodeEntities(node.text),
|
|
32
39
|
children: (node.children || []).map(toMathNode),
|
|
33
40
|
});
|
|
34
41
|
const parseAttributes = (attrString) => {
|
|
@@ -165,9 +172,36 @@ const parseHtmlTree = (html) => {
|
|
|
165
172
|
cursor = commentEnd !== -1 ? commentEnd + 3 : html.length;
|
|
166
173
|
continue;
|
|
167
174
|
}
|
|
168
|
-
//
|
|
169
|
-
//
|
|
170
|
-
|
|
175
|
+
// Scan for the tag's closing '>', skipping any that appear inside a quoted attribute value.
|
|
176
|
+
// Browsers do NOT escape '>' inside attribute values on serialization, so a literal '>' there
|
|
177
|
+
// (e.g. a mermaid diagram's `-->` in data-mermaid) must not be read as the tag end. The scan
|
|
178
|
+
// is linear in the tag's length and the cursor never rewinds, so the parse stays O(n) overall
|
|
179
|
+
// (no substring().match allocation per '<').
|
|
180
|
+
let tagEndIdx = -1;
|
|
181
|
+
let attrQuote = '';
|
|
182
|
+
for (let i = tagStart + 1; i < html.length; i++) {
|
|
183
|
+
const ch = html[i];
|
|
184
|
+
if (attrQuote) {
|
|
185
|
+
if (ch === attrQuote)
|
|
186
|
+
attrQuote = '';
|
|
187
|
+
}
|
|
188
|
+
else if (ch === '"' || ch === '\'') {
|
|
189
|
+
attrQuote = ch;
|
|
190
|
+
}
|
|
191
|
+
else if (ch === '>') {
|
|
192
|
+
tagEndIdx = i;
|
|
193
|
+
break;
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
if (tagEndIdx === -1) {
|
|
197
|
+
// The quote-aware scan ran to the end without closing the tag. That is almost always an
|
|
198
|
+
// unbalanced quote from a stray unescaped '<' in prose (e.g. "a < b's weight"), not a
|
|
199
|
+
// genuinely truncated tag. Retry naively for the next literal '>': the resulting
|
|
200
|
+
// pseudo-tag is then dropped, so a malformed run degrades exactly as it did before the
|
|
201
|
+
// quote-aware scan existed instead of swallowing the rest of the document into one text
|
|
202
|
+
// node. Well-formed input with balanced quotes never reaches here.
|
|
203
|
+
tagEndIdx = html.indexOf('>', tagStart);
|
|
204
|
+
}
|
|
171
205
|
if (tagEndIdx === -1) {
|
|
172
206
|
const text = html.substring(tagStart);
|
|
173
207
|
current.children.push({ type: 'text', text, children: [], parent: current });
|
|
@@ -335,6 +369,9 @@ const parseHtml = async (buffer, config) => {
|
|
|
335
369
|
// body loop, since references can appear anywhere earlier in the document) and
|
|
336
370
|
// consulted by parseChildren's <sup data-footnote-ref> handling below.
|
|
337
371
|
const footnoteDefinitions = new Map();
|
|
372
|
+
// Keys a `<sup data-footnote-ref>` actually consumed, so definitions in the section that no
|
|
373
|
+
// reference points at (orphans) can be recovered at the end instead of silently dropped.
|
|
374
|
+
const referencedFootnoteKeys = new Set();
|
|
338
375
|
// --- Generic attribute pass-through (htmlParserConfig.preserveAttributes) ---------------
|
|
339
376
|
// Captures attributes no typed metadata field consumed, so they can be replayed on
|
|
340
377
|
// generation. Everything here is a *defence-in-depth* filter: HtmlGenerator sanitizes again
|
|
@@ -395,13 +432,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
395
432
|
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.MAX_NESTING_DEPTH_EXCEEDED);
|
|
396
433
|
}
|
|
397
434
|
if (node.type === 'text') {
|
|
398
|
-
let decodedText = (node.text || '')
|
|
399
|
-
.replace(/ /g, ' ')
|
|
400
|
-
.replace(/</g, '<')
|
|
401
|
-
.replace(/>/g, '>')
|
|
402
|
-
.replace(/&/g, '&')
|
|
403
|
-
.replace(/"/g, '"')
|
|
404
|
-
.replace(/'/g, "'");
|
|
435
|
+
let decodedText = decodeEntities(node.text || '');
|
|
405
436
|
if (!config.preserveXmlWhitespace) {
|
|
406
437
|
decodedText = decodedText.replace(/\s+/g, ' ');
|
|
407
438
|
}
|
|
@@ -436,6 +467,12 @@ const parseHtml = async (buffer, config) => {
|
|
|
436
467
|
newFormatting.superscript = true;
|
|
437
468
|
if (tagName === 'code')
|
|
438
469
|
newFormatting.font = 'monospace';
|
|
470
|
+
if (tagName === 'mark') {
|
|
471
|
+
// <mark> is a highlight. Use its data-color when present (an inline
|
|
472
|
+
// background-color style, read below, still wins); a bare <mark> falls back to the
|
|
473
|
+
// conventional yellow so it round-trips as a highlight rather than plain text.
|
|
474
|
+
newFormatting.backgroundColor = node.attributes?.['data-color'] || '#ffff00';
|
|
475
|
+
}
|
|
439
476
|
const styleAttr = node.attributes?.style || '';
|
|
440
477
|
const alignAttr = node.attributes?.align || '';
|
|
441
478
|
if (styleAttr || alignAttr) {
|
|
@@ -492,6 +529,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
492
529
|
// instead of inserting a visible node, matching WordParser's convention.
|
|
493
530
|
if (child.type === 'element' && child.tagName === 'sup' && child.attributes?.['data-footnote-ref'] !== undefined) {
|
|
494
531
|
const key = child.attributes['data-footnote-ref'];
|
|
532
|
+
referencedFootnoteKeys.add(key);
|
|
495
533
|
const definition = footnoteDefinitions.get(key);
|
|
496
534
|
const noteNode = {
|
|
497
535
|
type: 'note',
|
|
@@ -520,7 +558,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
520
558
|
}
|
|
521
559
|
return kids;
|
|
522
560
|
};
|
|
523
|
-
// YouTube embeds:
|
|
561
|
+
// YouTube embeds: attribute-driven editors render
|
|
524
562
|
// <div data-youtube-video="ID" data-width="…" data-align="…">…<iframe…></div>.
|
|
525
563
|
// Recognise both the wrapper div and a bare iframe so externally-authored HTML
|
|
526
564
|
// (and a saved-then-reopened .md that fell back to raw HTML) both round-trip.
|
|
@@ -561,6 +599,27 @@ const parseHtml = async (buffer, config) => {
|
|
|
561
599
|
embedNode.rawContent = '<iframe>...</iframe>';
|
|
562
600
|
return embedNode;
|
|
563
601
|
}
|
|
602
|
+
// Non-YouTube iframes are dropped by default (a deliberate security posture).
|
|
603
|
+
// preserveIframes opts back in, keeping the src as a generic 'iframe' embed; the
|
|
604
|
+
// src is scheme-checked again on generation, so this only widens what is retained.
|
|
605
|
+
// Decode the src (attribute values are stored entity-encoded) so it isn't
|
|
606
|
+
// double-escaped when the generator re-escapes it, which would corrupt query strings.
|
|
607
|
+
const decodedSrc = decodeEntities(src);
|
|
608
|
+
if ((0, sanitize_js_1.iframeAllowed)(decodedSrc, config.htmlParserConfig?.preserveIframes)) {
|
|
609
|
+
const iframeNode = {
|
|
610
|
+
type: 'embed',
|
|
611
|
+
text: decodedSrc,
|
|
612
|
+
metadata: {
|
|
613
|
+
embedType: 'iframe',
|
|
614
|
+
url: decodedSrc,
|
|
615
|
+
width: node.attributes?.width,
|
|
616
|
+
height: node.attributes?.height
|
|
617
|
+
}
|
|
618
|
+
};
|
|
619
|
+
if (config.includeRawContent)
|
|
620
|
+
iframeNode.rawContent = '<iframe>...</iframe>';
|
|
621
|
+
return iframeNode;
|
|
622
|
+
}
|
|
564
623
|
return null;
|
|
565
624
|
}
|
|
566
625
|
// Footnotes section: its definitions were already extracted up front (see
|
|
@@ -570,22 +629,42 @@ const parseHtml = async (buffer, config) => {
|
|
|
570
629
|
if (tagName === 'section' && node.attributes?.['data-footnotes'] !== undefined) {
|
|
571
630
|
return null;
|
|
572
631
|
}
|
|
573
|
-
// Math
|
|
574
|
-
//
|
|
575
|
-
//
|
|
632
|
+
// Math. Two accepted shapes, disambiguated by the `data-math` value:
|
|
633
|
+
// 1. This library's own output - `data-math="inline|block"` names the mode, and the
|
|
634
|
+
// LaTeX is the visible ($-delimited, escaped) text content.
|
|
635
|
+
// 2. Attribute-driven producers that put the raw LaTeX in `data-math` and signal the
|
|
636
|
+
// mode through the class (`math-inline`/`math-block`) or the tag.
|
|
637
|
+
// Anything whose `data-math` is exactly `inline`/`block` takes path 1 unchanged; every
|
|
638
|
+
// other value is treated as LaTeX (path 2). LaTeX literally equal to `inline`/`block`
|
|
639
|
+
// is the only ambiguous input, and its text content wins there anyway.
|
|
576
640
|
if ((tagName === 'div' || tagName === 'span') && node.attributes?.['data-math'] !== undefined) {
|
|
577
|
-
const
|
|
578
|
-
const
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
641
|
+
const dataMath = node.attributes['data-math'];
|
|
642
|
+
const modeIsExplicit = dataMath === 'inline' || dataMath === 'block';
|
|
643
|
+
const classTokens = (node.attributes?.class || '').split(/\s+/);
|
|
644
|
+
const rawText = decodeEntities(node.children.map(c => c.text || '').join(''));
|
|
645
|
+
// Prefer the text content; fall back to the attribute value (path 2 producers may
|
|
646
|
+
// emit an empty body).
|
|
647
|
+
const source = rawText || (modeIsExplicit ? '' : decodeEntities(dataMath));
|
|
648
|
+
// Strip whichever `$`/`$$` delimiters are actually present, independent of the
|
|
649
|
+
// resolved mode - a `$`-delimited body inside a <div> must not keep its delimiters.
|
|
650
|
+
// The delimiter also disambiguates the mode when neither an explicit `data-math` nor
|
|
651
|
+
// a `math-inline`/`math-block` class settles it (so `<div data-math="x">$x$</div>`
|
|
652
|
+
// reads as inline, not block-via-tag).
|
|
653
|
+
let latex = source;
|
|
654
|
+
let delimiterMode;
|
|
655
|
+
if (source.length >= 4 && source.startsWith('$$') && source.endsWith('$$')) {
|
|
656
|
+
latex = source.slice(2, -2);
|
|
657
|
+
delimiterMode = 'block';
|
|
658
|
+
}
|
|
659
|
+
else if (source.length >= 2 && source.startsWith('$') && source.endsWith('$')) {
|
|
660
|
+
latex = source.slice(1, -1);
|
|
661
|
+
delimiterMode = 'inline';
|
|
662
|
+
}
|
|
663
|
+
const mathMode = modeIsExplicit
|
|
664
|
+
? dataMath
|
|
665
|
+
: classTokens.includes('math-block') ? 'block'
|
|
666
|
+
: classTokens.includes('math-inline') ? 'inline'
|
|
667
|
+
: delimiterMode ?? (tagName === 'div' ? 'block' : 'inline');
|
|
589
668
|
return {
|
|
590
669
|
type: 'code',
|
|
591
670
|
text: latex,
|
|
@@ -611,7 +690,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
611
690
|
metadata: { math: isBlock ? 'block' : 'inline' }
|
|
612
691
|
};
|
|
613
692
|
}
|
|
614
|
-
// Admonition:
|
|
693
|
+
// Admonition: attribute-driven editors render
|
|
615
694
|
// <div class="admonition admonition-note" data-type="note">…children…</div>.
|
|
616
695
|
if (tagName === 'div' && (node.attributes?.class || '').split(/\s+/).includes('admonition')) {
|
|
617
696
|
const admonitionTypeAttr = node.attributes?.['data-type'];
|
|
@@ -627,6 +706,29 @@ const parseHtml = async (buffer, config) => {
|
|
|
627
706
|
admonitionNode.rawContent = '<div class="admonition">...</div>';
|
|
628
707
|
return admonitionNode;
|
|
629
708
|
}
|
|
709
|
+
// Mermaid diagrams. Attribute-driven producers render a
|
|
710
|
+
// <div class="mermaid" data-mermaid="<code>"> with the code also as text content.
|
|
711
|
+
// Map either shape to a fenced code node with language `mermaid`, so it round-trips as
|
|
712
|
+
// a ```mermaid block. Previously this div fell through to generic handling and its
|
|
713
|
+
// code flattened to paragraph text.
|
|
714
|
+
if (tagName === 'div' && (node.attributes?.['data-mermaid'] !== undefined || (node.attributes?.class || '').split(/\s+/).includes('mermaid'))) {
|
|
715
|
+
const code = decodeEntities(node.children.map(c => c.text || '').join('')).trim()
|
|
716
|
+
|| decodeEntities(node.attributes?.['data-mermaid'] || '');
|
|
717
|
+
// Only claim this as a mermaid code node when there is actual diagram source.
|
|
718
|
+
// A bare `class="mermaid"` div with nested elements (a mermaid.js-rendered <svg>,
|
|
719
|
+
// or a div merely reusing the class for styling) has no direct text and no
|
|
720
|
+
// data-mermaid; fall through to generic handling so its content is not dropped.
|
|
721
|
+
if (code) {
|
|
722
|
+
const mermaidNode = {
|
|
723
|
+
type: 'code',
|
|
724
|
+
text: code,
|
|
725
|
+
metadata: { language: 'mermaid' }
|
|
726
|
+
};
|
|
727
|
+
if (config.includeRawContent)
|
|
728
|
+
mermaidNode.rawContent = '<div data-mermaid>...</div>';
|
|
729
|
+
return mermaidNode;
|
|
730
|
+
}
|
|
731
|
+
}
|
|
630
732
|
// Skip structural containers produced by HtmlGenerator to avoid deep AST nesting
|
|
631
733
|
if (tagName === 'div' && (node.attributes?.class === 'container' ||
|
|
632
734
|
node.attributes?.class === 'spreadsheet-container' ||
|
|
@@ -724,6 +826,21 @@ const parseHtml = async (buffer, config) => {
|
|
|
724
826
|
metadata: { citationKey }
|
|
725
827
|
};
|
|
726
828
|
}
|
|
829
|
+
// Attribute-driven citation shape: a <span> carrying the `citation` class token (among
|
|
830
|
+
// any others) and a non-empty data-key. Produces the same bare-key text node as the
|
|
831
|
+
// <cite> form above. An empty/absent data-key falls through to generic span handling,
|
|
832
|
+
// so the span's visible text still survives.
|
|
833
|
+
if (tagName === 'span'
|
|
834
|
+
&& (node.attributes?.class || '').split(/\s+/).includes('citation')
|
|
835
|
+
&& node.attributes?.['data-key']) {
|
|
836
|
+
const citationKey = decodeEntities(node.attributes['data-key']);
|
|
837
|
+
return {
|
|
838
|
+
type: 'text',
|
|
839
|
+
text: citationKey,
|
|
840
|
+
formatting: Object.keys(newFormatting).length > 0 ? { ...newFormatting } : undefined,
|
|
841
|
+
metadata: { citationKey }
|
|
842
|
+
};
|
|
843
|
+
}
|
|
727
844
|
if (tagName === 'ul' || tagName === 'ol') {
|
|
728
845
|
const isNewTopLevel = !listContext;
|
|
729
846
|
const newListContext = {
|
|
@@ -782,7 +899,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
782
899
|
return [selfNode, ...nestedLists];
|
|
783
900
|
}
|
|
784
901
|
if (tagName === 'table') {
|
|
785
|
-
//
|
|
902
|
+
// Attribute-driven editors render data-align on the <table> itself.
|
|
786
903
|
const tableAlignAttr = node.attributes?.['data-align'];
|
|
787
904
|
const tableAlign = ['left', 'center', 'right'].includes(tableAlignAttr) ? tableAlignAttr : undefined;
|
|
788
905
|
const tableNode = {
|
|
@@ -831,7 +948,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
831
948
|
if (tagName === 'img') {
|
|
832
949
|
const src = node.attributes?.src;
|
|
833
950
|
const alt = node.attributes?.alt;
|
|
834
|
-
//
|
|
951
|
+
// Attribute-driven editors render data-width/data-align, falling back to
|
|
835
952
|
// parsing the inline style for consumers that only emit the CSS.
|
|
836
953
|
const imgDecls = parseStyleDeclarations(node.attributes?.style || '');
|
|
837
954
|
// Exact lookup, so `max-width: 100%` - the standard responsive-image style, and by
|
|
@@ -919,6 +1036,23 @@ const parseHtml = async (buffer, config) => {
|
|
|
919
1036
|
}
|
|
920
1037
|
});
|
|
921
1038
|
}
|
|
1039
|
+
else if (node.attributes?.['data-wikilink'] !== undefined) {
|
|
1040
|
+
// Attribute-driven wikilink shape: the page lives in data-target, the display
|
|
1041
|
+
// text is the anchor's own content (or data-alias/data-target when the anchor
|
|
1042
|
+
// is empty). data-wikilink-page above keeps precedence over this form.
|
|
1043
|
+
const page = decodeEntities(node.attributes['data-target'] || '');
|
|
1044
|
+
if (!children.some(c => c.type === 'text')) {
|
|
1045
|
+
children.push({
|
|
1046
|
+
type: 'text',
|
|
1047
|
+
text: decodeEntities(node.attributes['data-alias'] || node.attributes['data-target'] || ''),
|
|
1048
|
+
});
|
|
1049
|
+
}
|
|
1050
|
+
children.forEach(c => {
|
|
1051
|
+
if (c.type === 'text') {
|
|
1052
|
+
c.metadata = { ...c.metadata, link: page, linkType: 'internal', wikilink: true };
|
|
1053
|
+
}
|
|
1054
|
+
});
|
|
1055
|
+
}
|
|
922
1056
|
else if (href) {
|
|
923
1057
|
const linkType = href.startsWith('#') ? 'internal' : 'external';
|
|
924
1058
|
children.forEach(c => {
|
|
@@ -936,6 +1070,17 @@ const parseHtml = async (buffer, config) => {
|
|
|
936
1070
|
}
|
|
937
1071
|
return brNode;
|
|
938
1072
|
}
|
|
1073
|
+
if (tagName === 'hr') {
|
|
1074
|
+
// A horizontal rule is a thematic break. This library tags an office page break
|
|
1075
|
+
// as <hr class="page-break"> on emission, so that variant round-trips back to a
|
|
1076
|
+
// page break; every other <hr> is thematic. Previously <hr> was dropped entirely.
|
|
1077
|
+
const isPageBreak = (node.attributes?.class || '').split(/\s+/).includes('page-break');
|
|
1078
|
+
const hrNode = { type: 'break', metadata: { breakType: isPageBreak ? 'page' : 'thematic' } };
|
|
1079
|
+
if (config.includeRawContent) {
|
|
1080
|
+
hrNode.rawContent = '<hr/>';
|
|
1081
|
+
}
|
|
1082
|
+
return hrNode;
|
|
1083
|
+
}
|
|
939
1084
|
if (tagName === 'pre') {
|
|
940
1085
|
const codeNode = node.children.find(c => c.tagName === 'code');
|
|
941
1086
|
let language;
|
|
@@ -945,10 +1090,18 @@ const parseHtml = async (buffer, config) => {
|
|
|
945
1090
|
const langMatch = classAttr.split(' ').find((c) => c.startsWith('language-'));
|
|
946
1091
|
if (langMatch)
|
|
947
1092
|
language = langMatch.replace('language-', '');
|
|
948
|
-
|
|
1093
|
+
// Decode entities: the code body is stored raw, so `<`/`>`/`&` (e.g. a
|
|
1094
|
+
// mermaid `-->` arrow, or `a < b` in a snippet) must be turned back into text.
|
|
1095
|
+
codeText = decodeEntities(codeNode.children.map(c => c.text || '').join(''));
|
|
949
1096
|
}
|
|
950
1097
|
else {
|
|
951
|
-
codeText = node.children.map(c => c.text || '').join('');
|
|
1098
|
+
codeText = decodeEntities(node.children.map(c => c.text || '').join(''));
|
|
1099
|
+
}
|
|
1100
|
+
// A `mermaid` class token (on the <pre> or its <code>) names the language when no
|
|
1101
|
+
// explicit language-* class is present - some producers emit <pre class="mermaid">.
|
|
1102
|
+
if (!language && ((node.attributes?.class || '').split(/\s+/).includes('mermaid') ||
|
|
1103
|
+
(codeNode?.attributes?.class || '').split(/\s+/).includes('mermaid'))) {
|
|
1104
|
+
language = 'mermaid';
|
|
952
1105
|
}
|
|
953
1106
|
const preNode = {
|
|
954
1107
|
type: 'code',
|
|
@@ -1027,6 +1180,21 @@ const parseHtml = async (buffer, config) => {
|
|
|
1027
1180
|
}
|
|
1028
1181
|
}
|
|
1029
1182
|
}
|
|
1183
|
+
// Orphan footnote definitions: a `<section data-footnotes>` entry that no `<sup
|
|
1184
|
+
// data-footnote-ref>` consumed would otherwise be dropped (it is skipped in the body walk and
|
|
1185
|
+
// only materialised via a reference). Recover them as trailing `unreferenced` note nodes, the
|
|
1186
|
+
// same shape MarkdownParser produces, so md -> html -> md preserves the definition instead of
|
|
1187
|
+
// turning it into junk text with a dead back-link.
|
|
1188
|
+
for (const [key, definition] of footnoteDefinitions) {
|
|
1189
|
+
if (referencedFootnoteKeys.has(key))
|
|
1190
|
+
continue;
|
|
1191
|
+
content.push({
|
|
1192
|
+
type: 'note',
|
|
1193
|
+
text: (definition || []).map(d => d.text || '').join(''),
|
|
1194
|
+
children: definition || [],
|
|
1195
|
+
metadata: { noteType: 'footnote', noteId: key, unreferenced: true },
|
|
1196
|
+
});
|
|
1197
|
+
}
|
|
1030
1198
|
const toTextSync = () => content.map(n => {
|
|
1031
1199
|
const getText = (node) => {
|
|
1032
1200
|
if (node.type === 'text' || node.type === 'code')
|
|
@@ -3,6 +3,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
|
|
|
3
3
|
exports.parseMarkdown = void 0;
|
|
4
4
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
5
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
6
|
+
const sanitize_js_1 = require("../utils/sanitize.js");
|
|
6
7
|
// Sentinel node type for a standalone bookmark-anchor block (e.g. `<a id="x"></a>` on its
|
|
7
8
|
// own line). A post-parse pass folds these into the following node's anchorIds so they
|
|
8
9
|
// round-trip as real anchors rather than being escaped to visible text on regeneration.
|
|
@@ -61,7 +62,12 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
61
62
|
const metadata = {};
|
|
62
63
|
const attachments = [];
|
|
63
64
|
// Parse YAML Front Matter
|
|
64
|
-
if (
|
|
65
|
+
if (/^---\n---[ \t]*(?:\n|$)/.test(textStr)) {
|
|
66
|
+
// Empty frontmatter block: strip it so `---\n---` isn't misread as a setext `## ---`
|
|
67
|
+
// heading (empty metadata used to emit exactly this shape, and other producers do too).
|
|
68
|
+
textStr = textStr.replace(/^---\n---[ \t]*(?:\n|$)/, '');
|
|
69
|
+
}
|
|
70
|
+
else if (textStr.startsWith('---\n')) {
|
|
65
71
|
const endIdx = textStr.indexOf('\n---\n', 4);
|
|
66
72
|
if (endIdx !== -1) {
|
|
67
73
|
const frontMatter = textStr.substring(4, endIdx);
|
|
@@ -74,9 +80,15 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
74
80
|
if (match) {
|
|
75
81
|
const key = match[1].trim();
|
|
76
82
|
const rawVal = match[2].trim();
|
|
77
|
-
|
|
83
|
+
// A quoted scalar is explicitly a string in YAML: strip the quotes but never
|
|
84
|
+
// coerce it, so `version: "123"` / `flag: "true"` keep their string-ness across
|
|
85
|
+
// a save/reload cycle instead of silently degrading to a number/boolean on the
|
|
86
|
+
// next parse (which the generator would then re-emit unquoted, losing the type
|
|
87
|
+
// permanently). Only bare, unquoted scalars coerce.
|
|
88
|
+
const isQuoted = /^"(.*)"$/.test(rawVal) || /^'(.*)'$/.test(rawVal);
|
|
89
|
+
const val = rawVal.replace(/^"(.*)"$/, '$1').replace(/^'(.*)'$/, '$1');
|
|
78
90
|
let parsedVal = val;
|
|
79
|
-
if (rawVal.startsWith('[') && rawVal.endsWith(']')) {
|
|
91
|
+
if (!isQuoted && rawVal.startsWith('[') && rawVal.endsWith(']')) {
|
|
80
92
|
// Flow-array (`tags: [a, b]`) or JSON-array (`tags: ["a","b"]`) value -
|
|
81
93
|
// parse into a real array instead of storing the literal bracket string,
|
|
82
94
|
// so it round-trips symmetrically with MarkdownGenerator's frontmatter output.
|
|
@@ -89,6 +101,8 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
89
101
|
parsedVal = inner === '' ? [] : splitFlowArrayItems(inner).map(item => item.replace(/^['"](.*)['"]$/, '$1'));
|
|
90
102
|
}
|
|
91
103
|
}
|
|
104
|
+
else if (isQuoted)
|
|
105
|
+
parsedVal = val;
|
|
92
106
|
else if (val === 'true')
|
|
93
107
|
parsedVal = true;
|
|
94
108
|
else if (val === 'false')
|
|
@@ -167,11 +181,32 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
167
181
|
});
|
|
168
182
|
// Extract footnote definitions (`[^id]: text`) before block splitting, since
|
|
169
183
|
// definitions conventionally live at the end of the document, after every place
|
|
170
|
-
// they're referenced - inline parsing below needs the full map upfront.
|
|
171
|
-
//
|
|
184
|
+
// they're referenced - inline parsing below needs the full map upfront. The first line
|
|
185
|
+
// may be followed by continuation lines indented one level (4 spaces or a tab), which are
|
|
186
|
+
// dedented and joined onto the definition (Pandoc/GFM). A 4-space-indented block right after
|
|
187
|
+
// a definition is therefore read as its continuation rather than as a standalone code block.
|
|
188
|
+
// Supported (lossless) shape: contiguous continuation - the indented lines follow the
|
|
189
|
+
// definition with no blank line between them. Known limitation (6.E.1): a continuation
|
|
190
|
+
// separated from the definition by a BLANK line is not folded in - the regex below stops at
|
|
191
|
+
// the blank line, and the indented block after it re-parses as a fenced/indented code block on
|
|
192
|
+
// save. Multi-paragraph footnotes should therefore use the contiguous form.
|
|
172
193
|
const footnoteDefinitions = new Map();
|
|
173
|
-
|
|
174
|
-
|
|
194
|
+
// Every id a `[^id]` reference consumes, so definitions that are never referenced can be
|
|
195
|
+
// detected at the end and preserved rather than silently dropped (see the orphan sweep below).
|
|
196
|
+
const referencedFootnoteIds = new Set();
|
|
197
|
+
// One reused note node per referenced id. Repeated `[^id]` references are a single shared
|
|
198
|
+
// footnote in Markdown, so they must not each materialise a full copy of the body - the
|
|
199
|
+
// generators would otherwise renumber them to [^1]/[^2] and duplicate the definition. Office
|
|
200
|
+
// notes reach the generators as distinct objects even when they share a numeric id, so those
|
|
201
|
+
// stay separate; only genuinely shared Markdown references collapse.
|
|
202
|
+
const footnoteNodesById = new Map();
|
|
203
|
+
textStr = textStr.replace(/^\[\^([^\]]+)\]:[ \t]*(.*(?:\n(?: {4}|\t).*)*)$/gm, (_match, id, definition) => {
|
|
204
|
+
const dedented = String(definition)
|
|
205
|
+
.split('\n')
|
|
206
|
+
.map((line, i) => i === 0 ? line : line.replace(/^(?: {4}|\t)/, ''))
|
|
207
|
+
.join('\n')
|
|
208
|
+
.trim();
|
|
209
|
+
footnoteDefinitions.set(id, dedented);
|
|
175
210
|
return '';
|
|
176
211
|
});
|
|
177
212
|
// Extract Markdown Extra abbreviation definitions (`*[HTML]: Hypertext Markup Language`)
|
|
@@ -279,7 +314,7 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
279
314
|
// Inline math requires no whitespace right after the opening $ or right before the
|
|
280
315
|
// closing $, the common heuristic (matching Pandoc/KaTeX) for avoiding false
|
|
281
316
|
// positives on currency like "$5 and $10".
|
|
282
|
-
const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)~~|(?<codeFence>`+)(?<codeContent>(?:(?!\k<codeFence>)[\s\S])+?)\k<codeFence>(?!`)|<u>(?<underline>.+?)<\/u>|<sub>(?<subscript>.+?)<\/sub>|<sup>(?<superscript>.+?)<\/sup>|\[\^(?<footnoteId>[^\]]+)\]|\[@(?<citationKey>[a-zA-Z0-9_:.-]+)\]|\[\[(?<wikiPage>[^\]|]+)(?:\|(?<wikiAlias>[^\]]+))?\]\]|(?<refBang>!?)\[(?<refText>[^\]]*)\]\[(?<refId>[^\]]*)\]|(?<shortBang>!?)\[(?<shortText>[^\]]+)\]|<(?<autolinkUrl>(?:https?|mailto):[^\s<>]+)>|\$(?!\s)(?<mathInline>[^$\n]+?)(?<!\s)\$/g;
|
|
317
|
+
const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)~~|(?<codeFence>`+)(?<codeContent>(?:(?!\k<codeFence>)[\s\S])+?)\k<codeFence>(?!`)|<u>(?<underline>.+?)<\/u>|<sub>(?<subscript>.+?)<\/sub>|<sup>(?<superscript>.+?)<\/sup>|(?<lineBreak><br\s*\/?>)|<span\s+style="(?<spanStyle>[^"]*)">(?<spanContent>.+?)<\/span>|\[\^(?<footnoteId>[^\]]+)\]|\[@(?<citationKey>[a-zA-Z0-9_:.-]+)\]|\[\[(?<wikiPage>[^\]|]+)(?:\|(?<wikiAlias>[^\]]+))?\]\]|(?<refBang>!?)\[(?<refText>[^\]]*)\]\[(?<refId>[^\]]*)\]|(?<shortBang>!?)\[(?<shortText>[^\]]+)\]|<(?<autolinkUrl>(?:https?|mailto):[^\s<>]+)>|\$(?!\s)(?<mathInline>[^$\n]+?)(?<!\s)\$/g;
|
|
283
318
|
let lastIndex = 0;
|
|
284
319
|
let match;
|
|
285
320
|
while ((match = regex.exec(text)) !== null) {
|
|
@@ -320,16 +355,50 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
320
355
|
else if (g.superscript !== undefined) { // Superscript
|
|
321
356
|
nodes.push(...parseInline(g.superscript, { ...currentFormatting, superscript: true }));
|
|
322
357
|
}
|
|
358
|
+
else if (g.lineBreak !== undefined) { // Raw inline <br>/<br/>/<br /> - a hard line break.
|
|
359
|
+
// MarkdownGenerator emits a raw <br> for a line break inside a table cell (a GFM pipe
|
|
360
|
+
// cell can't hold a newline), so the parser must read it back symmetrically as a break
|
|
361
|
+
// node instead of escaping it to literal `<br>` text and destroying it.
|
|
362
|
+
nodes.push({ type: 'break', metadata: { breakType: 'carriageReturn' } });
|
|
363
|
+
}
|
|
364
|
+
else if (g.spanContent !== undefined) { // Inline styled span: color / highlight / font-size
|
|
365
|
+
const style = g.spanStyle || '';
|
|
366
|
+
const styled = { ...currentFormatting };
|
|
367
|
+
// Anchor each property to a declaration boundary so `color` doesn't match inside
|
|
368
|
+
// `background-color`.
|
|
369
|
+
const prop = (name) => {
|
|
370
|
+
const m = style.match(new RegExp(`(?:^|;)\\s*${name}\\s*:\\s*([^;]+)`, 'i'));
|
|
371
|
+
return m ? m[1].trim() : undefined;
|
|
372
|
+
};
|
|
373
|
+
const color = prop('color');
|
|
374
|
+
if (color)
|
|
375
|
+
styled.color = color;
|
|
376
|
+
const background = prop('background-color');
|
|
377
|
+
if (background)
|
|
378
|
+
styled.backgroundColor = background;
|
|
379
|
+
const size = prop('font-size');
|
|
380
|
+
if (size)
|
|
381
|
+
styled.size = size;
|
|
382
|
+
nodes.push(...parseInline(g.spanContent, styled));
|
|
383
|
+
}
|
|
323
384
|
else if (g.footnoteId !== undefined) { // Footnote reference
|
|
324
385
|
const noteId = g.footnoteId;
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
386
|
+
referencedFootnoteIds.add(noteId);
|
|
387
|
+
// Reuse the same note object across every reference to this id (see the map's
|
|
388
|
+
// declaration): the first reference builds the body, the rest share it, so the
|
|
389
|
+
// generators assign one key and emit one definition.
|
|
390
|
+
let noteNode = footnoteNodesById.get(noteId);
|
|
391
|
+
if (!noteNode) {
|
|
392
|
+
const definition = footnoteDefinitions.get(noteId);
|
|
393
|
+
const noteChildren = definition !== undefined ? parseInline(definition) : [];
|
|
394
|
+
noteNode = {
|
|
395
|
+
type: 'note',
|
|
396
|
+
text: noteChildren.map(c => c.text || '').join(''),
|
|
397
|
+
children: noteChildren,
|
|
398
|
+
metadata: { noteType: 'footnote', noteId }
|
|
399
|
+
};
|
|
400
|
+
footnoteNodesById.set(noteId, noteNode);
|
|
401
|
+
}
|
|
333
402
|
// Notes attach to the preceding text node (matches WordParser's convention);
|
|
334
403
|
// fall back to an empty text node if the reference opens the inline run.
|
|
335
404
|
if (nodes.length > 0) {
|
|
@@ -609,6 +678,36 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
609
678
|
});
|
|
610
679
|
continue;
|
|
611
680
|
}
|
|
681
|
+
// Generic iframe fallback: MarkdownGenerator's 'embed' case emits a single-line
|
|
682
|
+
// <iframe src="…"></iframe> for a preserved iframe when fallbackToHtml is on. Recognise
|
|
683
|
+
// it only when the caller opted into iframe preservation, so default parsing is unchanged.
|
|
684
|
+
const iframeMatch = block.match(/^<iframe\s+([^>]*?)\/?>(?:\s*<\/iframe>)?$/i);
|
|
685
|
+
if (iframeMatch) {
|
|
686
|
+
const attrsStr = iframeMatch[1];
|
|
687
|
+
// The emitted <iframe> HTML-escapes its attribute values (sanitizeUrl -> escapeHtml), so
|
|
688
|
+
// decode them back; otherwise the src double-escapes (`&` -> `&amp;`) and its
|
|
689
|
+
// query string is corrupted a little more on every save/reload cycle. `&` is decoded
|
|
690
|
+
// last so a genuinely double-escaped value only unwinds one level per parse.
|
|
691
|
+
const decodeAttr = (s) => (s || '')
|
|
692
|
+
.replace(/</g, '<').replace(/>/g, '>').replace(/"/g, '"')
|
|
693
|
+
.replace(/'/g, '\'').replace(/&/g, '&');
|
|
694
|
+
const src = decodeAttr(attrsStr.match(/\bsrc="([^"]*)"/i)?.[1]);
|
|
695
|
+
if (src && (0, sanitize_js_1.iframeAllowed)(src, config.htmlParserConfig?.preserveIframes)) {
|
|
696
|
+
const width = attrsStr.match(/\bwidth="([^"]*)"/i)?.[1];
|
|
697
|
+
const height = attrsStr.match(/\bheight="([^"]*)"/i)?.[1];
|
|
698
|
+
content.push({
|
|
699
|
+
type: 'embed',
|
|
700
|
+
text: src,
|
|
701
|
+
metadata: {
|
|
702
|
+
embedType: 'iframe',
|
|
703
|
+
url: src,
|
|
704
|
+
width: width !== undefined ? decodeAttr(width) : undefined,
|
|
705
|
+
height: height !== undefined ? decodeAttr(height) : undefined
|
|
706
|
+
}
|
|
707
|
+
});
|
|
708
|
+
continue;
|
|
709
|
+
}
|
|
710
|
+
}
|
|
612
711
|
// Code Block
|
|
613
712
|
const codeMatch = block.match(/^__CODE_BLOCK_(\d+)__$/);
|
|
614
713
|
if (codeMatch) {
|
|
@@ -918,9 +1017,10 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
918
1017
|
continue;
|
|
919
1018
|
}
|
|
920
1019
|
}
|
|
921
|
-
// Hr
|
|
1020
|
+
// Hr - a thematic break (horizontal rule), not a page break, so it survives a save as
|
|
1021
|
+
// `---` rather than collapsing to a bare newline.
|
|
922
1022
|
if (block.match(/^---+$|^\*\*\*+$|^___+$/)) {
|
|
923
|
-
content.push({ type: 'break', metadata: { breakType: '
|
|
1023
|
+
content.push({ type: 'break', metadata: { breakType: 'thematic' } });
|
|
924
1024
|
continue;
|
|
925
1025
|
}
|
|
926
1026
|
// Paragraph
|
|
@@ -956,6 +1056,23 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
956
1056
|
content.length = 0;
|
|
957
1057
|
content.push(...merged);
|
|
958
1058
|
}
|
|
1059
|
+
// Orphan footnote definitions (defined but never referenced) would otherwise vanish entirely -
|
|
1060
|
+
// a user who deletes a `[^x]` reference but keeps its `[^x]: ...` definition loses the
|
|
1061
|
+
// definition on the next save. Preserve them as trailing note nodes, marked `unreferenced` so
|
|
1062
|
+
// the generators route them into their footnotes section (not inline) and emit no citation
|
|
1063
|
+
// marker or dangling back-link. Both generators still emit the definition (md: a `[^x]:` line;
|
|
1064
|
+
// html: a `div[data-footnote-id]` inside `section[data-footnotes]`, which re-parses on import).
|
|
1065
|
+
for (const [id, definition] of footnoteDefinitions) {
|
|
1066
|
+
if (referencedFootnoteIds.has(id))
|
|
1067
|
+
continue;
|
|
1068
|
+
const noteChildren = parseInline(definition);
|
|
1069
|
+
content.push({
|
|
1070
|
+
type: 'note',
|
|
1071
|
+
text: noteChildren.map(c => c.text || '').join(''),
|
|
1072
|
+
children: noteChildren,
|
|
1073
|
+
metadata: { noteType: 'footnote', noteId: id, unreferenced: true },
|
|
1074
|
+
});
|
|
1075
|
+
}
|
|
959
1076
|
const toTextSync = () => content.map(n => {
|
|
960
1077
|
const getText = (node) => {
|
|
961
1078
|
if (node.type === 'text' || node.type === 'code')
|
|
@@ -458,23 +458,13 @@ const parseWord = async (buffer, config) => {
|
|
|
458
458
|
}
|
|
459
459
|
}
|
|
460
460
|
}
|
|
461
|
-
//
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
const value = pPrFormatting[key];
|
|
469
|
-
if (value === false) {
|
|
470
|
-
delete paragraphRunFormatting[key];
|
|
471
|
-
}
|
|
472
|
-
else if (value !== undefined) {
|
|
473
|
-
paragraphRunFormatting[key] = value;
|
|
474
|
-
}
|
|
475
|
-
}
|
|
476
|
-
}
|
|
477
|
-
}
|
|
461
|
+
// Runs inherit their base formatting from the style chain: the paragraph style (seeded
|
|
462
|
+
// here, and re-applied via the run-style path below), then any character style, then the
|
|
463
|
+
// run's own properties. The paragraph-mark run properties (`<w:pPr><w:rPr>`) format only
|
|
464
|
+
// the paragraph mark glyph itself per OOXML ISO 29500 §17.3.1.29, so they are deliberately
|
|
465
|
+
// NOT folded into the run base - doing so bled the paragraph mark's bold/italic/color/etc.
|
|
466
|
+
// onto every run in the paragraph (issue #109).
|
|
467
|
+
const paragraphRunFormatting = { ...styleProps.formatting };
|
|
478
468
|
// Extract text and children
|
|
479
469
|
let text = '';
|
|
480
470
|
const children = [];
|