officeparser 7.6.2 → 7.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -2
- package/dist/generators/HtmlGenerator.js +53 -14
- package/dist/generators/MarkdownGenerator.js +111 -40
- package/dist/officeparser.browser.d.ts +99 -18
- package/dist/officeparser.browser.iife.js +187 -180
- package/dist/officeparser.browser.mjs +187 -180
- package/dist/officeparser.browser.slim.d.ts +99 -18
- package/dist/officeparser.browser.slim.iife.js +168 -161
- package/dist/officeparser.browser.slim.mjs +168 -161
- package/dist/parsers/HtmlParser.js +36 -3
- package/dist/parsers/MarkdownParser.js +73 -8
- package/dist/sbom.cdx.json +161 -93
- package/dist/types.d.ts +99 -18
- package/package.json +1 -1
|
@@ -706,6 +706,22 @@ const parseHtml = async (buffer, config) => {
|
|
|
706
706
|
admonitionNode.rawContent = '<div class="admonition">...</div>';
|
|
707
707
|
return admonitionNode;
|
|
708
708
|
}
|
|
709
|
+
// Blockquote. Previously dropped entirely (its children were lifted out unquoted), so a
|
|
710
|
+
// <blockquote> lost its `> ` on the Markdown hop. Mark each block child with the 'Quote'
|
|
711
|
+
// style the styleMapper maps back to a blockquote; loose inline content is wrapped in one
|
|
712
|
+
// Quote-styled paragraph so it isn't emitted as an ordinary line.
|
|
713
|
+
if (tagName === 'blockquote') {
|
|
714
|
+
const kids = parseChildren(node, newFormatting, listContext);
|
|
715
|
+
const isBlock = (t) => t === 'paragraph' || t === 'heading' || t === 'list';
|
|
716
|
+
if (!kids.some(k => isBlock(k.type))) {
|
|
717
|
+
return { type: 'paragraph', metadata: { style: 'Quote' }, children: kids };
|
|
718
|
+
}
|
|
719
|
+
kids.forEach(k => {
|
|
720
|
+
if (isBlock(k.type))
|
|
721
|
+
k.metadata = { ...k.metadata, style: 'Quote' };
|
|
722
|
+
});
|
|
723
|
+
return kids;
|
|
724
|
+
}
|
|
709
725
|
// Mermaid diagrams. Attribute-driven producers render a
|
|
710
726
|
// <div class="mermaid" data-mermaid="<code>"> with the code also as text content.
|
|
711
727
|
// Map either shape to a fenced code node with language `mermaid`, so it round-trips as
|
|
@@ -931,11 +947,21 @@ const parseHtml = async (buffer, config) => {
|
|
|
931
947
|
const rowSpanAttr = node.attributes?.rowspan;
|
|
932
948
|
const colSpan = colSpanAttr ? parseInt(colSpanAttr, 10) : undefined;
|
|
933
949
|
const rowSpan = rowSpanAttr ? parseInt(rowSpanAttr, 10) : undefined;
|
|
950
|
+
// Per-column GFM alignment: read the cell's own `text-align` (or a legacy `align=`
|
|
951
|
+
// attribute) into `CellMetadata.align`, so the `:---`/`:---:`/`---:` markers survive
|
|
952
|
+
// AST -> HTML -> AST. `justify` has no pipe-table marker, so it is not a cell align.
|
|
953
|
+
// The table-level `<table data-align>` form is read separately in the `table` branch.
|
|
954
|
+
const cellTextAlign = (getDeclaration(parseStyleDeclarations(node.attributes?.style || ''), 'text-align')
|
|
955
|
+
|| node.attributes?.align || '').toLowerCase();
|
|
956
|
+
const cellAlign = ['left', 'center', 'right'].includes(cellTextAlign)
|
|
957
|
+
? cellTextAlign
|
|
958
|
+
: undefined;
|
|
934
959
|
const cellNode = {
|
|
935
960
|
type: 'cell',
|
|
936
961
|
metadata: {
|
|
937
962
|
colSpan: colSpan && !isNaN(colSpan) ? colSpan : undefined,
|
|
938
|
-
rowSpan: rowSpan && !isNaN(rowSpan) ? rowSpan : undefined
|
|
963
|
+
rowSpan: rowSpan && !isNaN(rowSpan) ? rowSpan : undefined,
|
|
964
|
+
align: cellAlign
|
|
939
965
|
},
|
|
940
966
|
children: parseChildren(node, newFormatting, listContext),
|
|
941
967
|
htmlAttributes: collectHtmlAttributes(node, ['colspan', 'rowspan', 'align'])
|
|
@@ -991,6 +1017,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
991
1017
|
metadata: {
|
|
992
1018
|
attachmentName: name,
|
|
993
1019
|
altText: alt,
|
|
1020
|
+
title: node.attributes?.title,
|
|
994
1021
|
width,
|
|
995
1022
|
align
|
|
996
1023
|
}
|
|
@@ -1002,6 +1029,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
1002
1029
|
metadata: {
|
|
1003
1030
|
url: src,
|
|
1004
1031
|
altText: alt,
|
|
1032
|
+
title: node.attributes?.title,
|
|
1005
1033
|
width,
|
|
1006
1034
|
align
|
|
1007
1035
|
}
|
|
@@ -1014,6 +1042,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
1014
1042
|
metadata: {
|
|
1015
1043
|
url: src,
|
|
1016
1044
|
altText: alt,
|
|
1045
|
+
title: node.attributes?.title,
|
|
1017
1046
|
anchorIds: anchorIds.length > 0 ? anchorIds : undefined,
|
|
1018
1047
|
width,
|
|
1019
1048
|
align
|
|
@@ -1055,16 +1084,20 @@ const parseHtml = async (buffer, config) => {
|
|
|
1055
1084
|
}
|
|
1056
1085
|
else if (href) {
|
|
1057
1086
|
const linkType = href.startsWith('#') ? 'internal' : 'external';
|
|
1087
|
+
const linkTitle = node.attributes?.title;
|
|
1058
1088
|
children.forEach(c => {
|
|
1059
1089
|
if (c.type === 'text') {
|
|
1060
|
-
c.metadata = { ...c.metadata, link: href, linkType };
|
|
1090
|
+
c.metadata = { ...c.metadata, link: href, linkType, title: linkTitle };
|
|
1061
1091
|
}
|
|
1062
1092
|
});
|
|
1063
1093
|
}
|
|
1064
1094
|
return children;
|
|
1065
1095
|
}
|
|
1066
1096
|
if (tagName === 'br') {
|
|
1067
|
-
|
|
1097
|
+
// A <br> is a hard line break: `carriageReturn` so the Markdown generator emits a
|
|
1098
|
+
// hard break (` \n`, or a `<br>` inside a table cell) that re-imports as a <br>.
|
|
1099
|
+
// `textWrapping` emitted a bare `\n` in a paragraph, which re-imports as a space.
|
|
1100
|
+
const brNode = { type: 'break', metadata: { breakType: 'carriageReturn' } };
|
|
1068
1101
|
if (config.includeRawContent) {
|
|
1069
1102
|
brNode.rawContent = '<br/>';
|
|
1070
1103
|
}
|
|
@@ -166,6 +166,15 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
166
166
|
mathBlocks.push(latex);
|
|
167
167
|
return `\n\n${id}\n\n`;
|
|
168
168
|
});
|
|
169
|
+
// Single-line `$$...$$` occupying its own line is display (block) math too. Without this it
|
|
170
|
+
// falls through to the inline `$...$` tokenizer, which matches the INNER `$\int$` and leaks
|
|
171
|
+
// the outer pair as two stray literal `$`. Runs after the multi-line pass, whose placeholders
|
|
172
|
+
// carry no `$$` and so can't be re-matched. `(?!\$)` rejects `$$$...`/empty `$$$$`.
|
|
173
|
+
textStr = textStr.replace(/^\$\$(?!\$)([^\n]+?)\$\$[ \t]*$/gm, (_match, latex) => {
|
|
174
|
+
const id = `__MATH_BLOCK_${mathBlocks.length}__`;
|
|
175
|
+
mathBlocks.push(latex);
|
|
176
|
+
return `\n\n${id}\n\n`;
|
|
177
|
+
});
|
|
169
178
|
// Extract GLFM-style fenced-div admonitions (`:::note ... :::`) before block splitting,
|
|
170
179
|
// since their body may itself contain blank lines that would otherwise fragment them.
|
|
171
180
|
// The `> [!NOTE]` GitHub form doesn't need this - it's detected inline in the blockquote
|
|
@@ -259,7 +268,15 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
259
268
|
// Builds the same image/link node shape regardless of whether the URL came from
|
|
260
269
|
// an inline `(url)` or a resolved reference definition - shared by the inline
|
|
261
270
|
// image/link branch and the two reference-style branches below.
|
|
262
|
-
|
|
271
|
+
// Split a Markdown inline destination `url "title"` (also `'title'` / `(title)`) into its
|
|
272
|
+
// URL and optional title. The inline parser previously kept the whole thing as the URL, so
|
|
273
|
+
// `[t](u "T")` produced href `u "T"`; reference-style `[t][id]` already split it correctly.
|
|
274
|
+
const splitUrlTitle = (raw) => {
|
|
275
|
+
const m = raw.trim().match(/^(.*?)\s+(?:"([^"]*)"|'([^']*)'|\(([^)]*)\))\s*$/);
|
|
276
|
+
return m ? { url: m[1].trim(), title: m[2] ?? m[3] ?? m[4] } : { url: raw };
|
|
277
|
+
};
|
|
278
|
+
const buildLinkOrImageNodes = (isImage, altText, rawUrl, attrsStr) => {
|
|
279
|
+
const { url, title } = splitUrlTitle(rawUrl);
|
|
263
280
|
if (isImage) {
|
|
264
281
|
// Pandoc-style attribute list immediately after an image, e.g. {width=50% .centered}
|
|
265
282
|
const attrs = attrsStr !== undefined ? parseAttributeList(attrsStr) : undefined;
|
|
@@ -276,15 +293,15 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
276
293
|
name,
|
|
277
294
|
extension: mimeType.split('/')[1]
|
|
278
295
|
});
|
|
279
|
-
return [{ type: 'image', metadata: { attachmentName: name, altText, ...attrs } }];
|
|
296
|
+
return [{ type: 'image', metadata: { attachmentName: name, altText, title, ...attrs } }];
|
|
280
297
|
}
|
|
281
298
|
}
|
|
282
|
-
return [{ type: 'image', metadata: { url, altText, ...attrs } }];
|
|
299
|
+
return [{ type: 'image', metadata: { url, altText, title, ...attrs } }];
|
|
283
300
|
}
|
|
284
301
|
const linkNodes = parseInline(altText, currentFormatting);
|
|
285
302
|
linkNodes.forEach(n => {
|
|
286
303
|
if (n.type === 'text') {
|
|
287
|
-
n.metadata = { link: url, linkType: 'external' };
|
|
304
|
+
n.metadata = { link: url, linkType: 'external', title };
|
|
288
305
|
}
|
|
289
306
|
});
|
|
290
307
|
return linkNodes;
|
|
@@ -314,7 +331,7 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
314
331
|
// Inline math requires no whitespace right after the opening $ or right before the
|
|
315
332
|
// closing $, the common heuristic (matching Pandoc/KaTeX) for avoiding false
|
|
316
333
|
// positives on currency like "$5 and $10".
|
|
317
|
-
const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)
|
|
334
|
+
const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)~~|==(?<highlight>.+?)==|(?<codeFence>`+)(?<codeContent>(?:(?!\k<codeFence>)[\s\S])+?)\k<codeFence>(?!`)|<u>(?<underline>.+?)<\/u>|<sub>(?<subscript>.+?)<\/sub>|<sup>(?<superscript>.+?)<\/sup>|(?<lineBreak><br\s*\/?>)|<span\s+style="(?<spanStyle>[^"]*)">(?<spanContent>.+?)<\/span>|\[\^(?<footnoteId>[^\]]+)\]|\[@(?<citationKey>[a-zA-Z0-9_:.-]+)\]|\[\[(?<wikiPage>[^\]|]+)(?:\|(?<wikiAlias>[^\]]+))?\]\]|(?<refBang>!?)\[(?<refText>[^\]]*)\]\[(?<refId>[^\]]*)\]|(?<shortBang>!?)\[(?<shortText>[^\]]+)\]|<(?<autolinkUrl>(?:https?|mailto):[^\s<>]+)>|\$(?!\s)(?<mathInline>[^$\n]+?)(?<!\s)\$/g;
|
|
318
335
|
let lastIndex = 0;
|
|
319
336
|
let match;
|
|
320
337
|
while ((match = regex.exec(text)) !== null) {
|
|
@@ -343,6 +360,9 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
343
360
|
else if (g.strike !== undefined) { // Strikethrough
|
|
344
361
|
nodes.push(...parseInline(g.strike, { ...currentFormatting, strikethrough: true }));
|
|
345
362
|
}
|
|
363
|
+
else if (g.highlight !== undefined) { // ==highlight== (Obsidian/extended); additive on import
|
|
364
|
+
nodes.push(...parseInline(g.highlight, { ...currentFormatting, backgroundColor: '#ffff00' }));
|
|
365
|
+
}
|
|
346
366
|
else if (g.codeContent !== undefined) { // Inline code (any matching backtick-run length)
|
|
347
367
|
nodes.push({ type: 'text', text: g.codeContent, formatting: { ...currentFormatting, font: 'monospace' } });
|
|
348
368
|
}
|
|
@@ -610,6 +630,31 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
610
630
|
blocks.push(currentSubBlock.join('\n'));
|
|
611
631
|
}
|
|
612
632
|
}
|
|
633
|
+
// Re-join a list block that a blank line tore away from its parent. The generator's older
|
|
634
|
+
// loose output (`- a\n\n\n - a1`) and foreign editors both split a nested item into its
|
|
635
|
+
// own block; left apart, the child's leading indent is stripped by the per-block `trim()`
|
|
636
|
+
// below and it reparses as a flat top-level item under a fresh listId. Merge a block back
|
|
637
|
+
// into the preceding one only when the previous block is itself a list (its FIRST line is a
|
|
638
|
+
// marker - the sub-splitter guarantees such a block holds only marker/continuation lines) and
|
|
639
|
+
// the current block OPENS with an INDENTED marker. An unindented `- b` after a blank line is
|
|
640
|
+
// deliberately left split (a flat loose list keeps its own listId), and anything that is not
|
|
641
|
+
// an indented marker (continuation text, indented code, placeholders) never triggers a merge.
|
|
642
|
+
const listMarkerStart = /^(\s*)([-*+]|\d+[.)])\s+/;
|
|
643
|
+
const indentedMarkerStart = /^(?: {2,}|\t)\s*(?:[-*+]|\d+[.)])\s+/;
|
|
644
|
+
const mergedBlocks = [];
|
|
645
|
+
for (const block of blocks) {
|
|
646
|
+
const prev = mergedBlocks[mergedBlocks.length - 1];
|
|
647
|
+
if (prev !== undefined
|
|
648
|
+
&& listMarkerStart.test(prev.split('\n', 1)[0])
|
|
649
|
+
&& indentedMarkerStart.test(block.split('\n', 1)[0])) {
|
|
650
|
+
mergedBlocks[mergedBlocks.length - 1] = `${prev}\n${block}`;
|
|
651
|
+
}
|
|
652
|
+
else {
|
|
653
|
+
mergedBlocks.push(block);
|
|
654
|
+
}
|
|
655
|
+
}
|
|
656
|
+
blocks.length = 0;
|
|
657
|
+
blocks.push(...mergedBlocks);
|
|
613
658
|
let listIdCounter = 1;
|
|
614
659
|
let currentAlignment = undefined;
|
|
615
660
|
for (let block of blocks) {
|
|
@@ -974,11 +1019,23 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
974
1019
|
else {
|
|
975
1020
|
const lines = block.trim().split('\n');
|
|
976
1021
|
const rows = [];
|
|
1022
|
+
// Pre-scan the separator row for per-column GFM alignment (`:--` left, `:-:` center,
|
|
1023
|
+
// `--:` right; a bare `--` column has none), so every cell can carry its column's
|
|
1024
|
+
// alignment on CellMetadata.align (the header row precedes the separator, so a
|
|
1025
|
+
// per-cell pass alone could not see it).
|
|
1026
|
+
const sepLine = lines.find(l => l.match(/^\|?\s*:?-+:?\s*(?:\|\s*:?-+:?\s*)*\|?$/));
|
|
1027
|
+
const columnAligns = sepLine
|
|
1028
|
+
? sepLine.replace(/^\||\|$/g, '').split('|').map(c => {
|
|
1029
|
+
const t = c.trim();
|
|
1030
|
+
const l = t.startsWith(':'), r = t.endsWith(':');
|
|
1031
|
+
return (l && r) ? 'center' : r ? 'right' : l ? 'left' : null;
|
|
1032
|
+
})
|
|
1033
|
+
: [];
|
|
977
1034
|
for (let i = 0; i < lines.length; i++) {
|
|
978
1035
|
if (lines[i].match(/^\|?\s*:?-+:?\s*(?:\|\s*:?-+:?\s*)*\|?$/))
|
|
979
1036
|
continue; // Separator row (per-cell `:?-+:?`, GFM-style; accepts short cells like `|-|-|`)
|
|
980
1037
|
const cellsStr = lines[i].replace(/^\||\|$/g, '').split('|');
|
|
981
|
-
const cells = cellsStr.map(c => {
|
|
1038
|
+
const cells = cellsStr.map((c, colIdx) => {
|
|
982
1039
|
// Recognize the MarkdownGenerator's own cell-alignment fallback,
|
|
983
1040
|
// `<div style="text-align: X">…</div>`, and lift it into an aligned
|
|
984
1041
|
// paragraph so it round-trips as alignment instead of being escaped to
|
|
@@ -987,17 +1044,25 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
987
1044
|
let cellAlign;
|
|
988
1045
|
cellText = cellText.replace(/<div\s+style="text-align:\s*(left|center|right|justify);?"\s*>([\s\S]*?)<\/div>/gi, (_m, a, inner) => { cellAlign = a.toLowerCase(); return inner; });
|
|
989
1046
|
const inline = parseInline(cellText, i === 0 ? { bold: true } : {});
|
|
1047
|
+
const colAlign = columnAligns[colIdx] ?? undefined;
|
|
1048
|
+
const cellMeta = colAlign ? { col: colIdx, align: colAlign } : undefined;
|
|
990
1049
|
if (cellAlign && cellAlign !== 'left') {
|
|
991
1050
|
return {
|
|
992
1051
|
type: 'cell',
|
|
1052
|
+
metadata: cellMeta,
|
|
993
1053
|
children: [{ type: 'paragraph', metadata: { alignment: cellAlign }, children: inline }]
|
|
994
1054
|
};
|
|
995
1055
|
}
|
|
996
|
-
return { type: 'cell', children: inline };
|
|
1056
|
+
return { type: 'cell', metadata: cellMeta, children: inline };
|
|
997
1057
|
});
|
|
998
1058
|
rows.push({ type: 'row', children: cells });
|
|
999
1059
|
}
|
|
1000
|
-
|
|
1060
|
+
// If every explicitly-aligned column agrees, also expose it as the table-level align,
|
|
1061
|
+
// so an editor that models one alignment per table (and HTML data-align) round-trips.
|
|
1062
|
+
const explicitAligns = columnAligns.filter((a) => a !== null);
|
|
1063
|
+
const uniformAlign = explicitAligns.length > 0 && explicitAligns.every(a => a === explicitAligns[0]) ? explicitAligns[0] : undefined;
|
|
1064
|
+
const resolvedTableAlign = tableAlign || uniformAlign;
|
|
1065
|
+
content.push({ type: 'table', metadata: resolvedTableAlign ? { align: resolvedTableAlign } : undefined, children: rows });
|
|
1001
1066
|
continue;
|
|
1002
1067
|
}
|
|
1003
1068
|
}
|