officeparser 7.6.2 → 7.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -4
- package/dist/defaults.js +2 -0
- package/dist/generators/HtmlGenerator.js +72 -17
- package/dist/generators/MarkdownGenerator.d.ts +1 -0
- package/dist/generators/MarkdownGenerator.js +169 -54
- package/dist/officeparser.browser.d.ts +151 -20
- package/dist/officeparser.browser.iife.js +194 -181
- package/dist/officeparser.browser.mjs +194 -181
- package/dist/officeparser.browser.slim.d.ts +151 -20
- package/dist/officeparser.browser.slim.iife.js +194 -181
- package/dist/officeparser.browser.slim.mjs +194 -181
- package/dist/parsers/HtmlParser.js +73 -5
- package/dist/parsers/MarkdownParser.js +186 -11
- package/dist/sbom.cdx.json +161 -93
- package/dist/types.d.ts +151 -20
- package/package.json +1 -1
|
@@ -558,6 +558,32 @@ const parseHtml = async (buffer, config) => {
|
|
|
558
558
|
}
|
|
559
559
|
return kids;
|
|
560
560
|
};
|
|
561
|
+
// Gated generic-iframe embed (HtmlGenerator's `gatedEmbeds` shape): an inert
|
|
562
|
+
// click-to-load placeholder that never auto-loads its src. Read it back to the same
|
|
563
|
+
// `embed` node unconditionally - capturing metadata is safe (the src is scheme-checked
|
|
564
|
+
// again on any re-emit); it is the trusted, already-gated counterpart to a raw <iframe>.
|
|
565
|
+
if (tagName === 'div' && node.attributes?.['data-embed-gated'] !== undefined) {
|
|
566
|
+
const gatedSrc = decodeEntities(node.attributes?.['data-embed-src'] || '');
|
|
567
|
+
if (!gatedSrc)
|
|
568
|
+
return null;
|
|
569
|
+
const gatedAlignAttr = node.attributes?.['data-embed-align'];
|
|
570
|
+
const gatedAlign = ['left', 'center', 'right'].includes(gatedAlignAttr) ? gatedAlignAttr : undefined;
|
|
571
|
+
const gatedNode = {
|
|
572
|
+
type: 'embed',
|
|
573
|
+
text: gatedSrc,
|
|
574
|
+
metadata: {
|
|
575
|
+
embedType: 'iframe',
|
|
576
|
+
url: gatedSrc,
|
|
577
|
+
width: node.attributes?.['data-embed-width'],
|
|
578
|
+
height: node.attributes?.['data-embed-height'],
|
|
579
|
+
align: gatedAlign,
|
|
580
|
+
label: node.attributes?.['data-embed-label']
|
|
581
|
+
}
|
|
582
|
+
};
|
|
583
|
+
if (config.includeRawContent)
|
|
584
|
+
gatedNode.rawContent = '<div data-embed-gated>...</div>';
|
|
585
|
+
return gatedNode;
|
|
586
|
+
}
|
|
561
587
|
// YouTube embeds: attribute-driven editors render
|
|
562
588
|
// <div data-youtube-video="ID" data-width="…" data-align="…">…<iframe…></div>.
|
|
563
589
|
// Recognise both the wrapper div and a bare iframe so externally-authored HTML
|
|
@@ -578,7 +604,8 @@ const parseHtml = async (buffer, config) => {
|
|
|
578
604
|
videoId,
|
|
579
605
|
url: embedUrl,
|
|
580
606
|
width,
|
|
581
|
-
align: embedAlign
|
|
607
|
+
align: embedAlign,
|
|
608
|
+
label: node.attributes?.['data-embed-label']
|
|
582
609
|
}
|
|
583
610
|
};
|
|
584
611
|
if (config.includeRawContent)
|
|
@@ -593,7 +620,15 @@ const parseHtml = async (buffer, config) => {
|
|
|
593
620
|
const embedNode = {
|
|
594
621
|
type: 'embed',
|
|
595
622
|
text: embedUrl,
|
|
596
|
-
|
|
623
|
+
// Carry the iframe's own width/height so a YouTube iframe's dimensions are not
|
|
624
|
+
// dropped (metadata is unified across embedTypes; align has no source here).
|
|
625
|
+
metadata: {
|
|
626
|
+
embedType: 'youtube',
|
|
627
|
+
videoId: ytMatch[1],
|
|
628
|
+
url: embedUrl,
|
|
629
|
+
width: node.attributes?.width,
|
|
630
|
+
height: node.attributes?.height
|
|
631
|
+
}
|
|
597
632
|
};
|
|
598
633
|
if (config.includeRawContent)
|
|
599
634
|
embedNode.rawContent = '<iframe>...</iframe>';
|
|
@@ -706,6 +741,22 @@ const parseHtml = async (buffer, config) => {
|
|
|
706
741
|
admonitionNode.rawContent = '<div class="admonition">...</div>';
|
|
707
742
|
return admonitionNode;
|
|
708
743
|
}
|
|
744
|
+
// Blockquote. Previously dropped entirely (its children were lifted out unquoted), so a
|
|
745
|
+
// <blockquote> lost its `> ` on the Markdown hop. Mark each block child with the 'Quote'
|
|
746
|
+
// style the styleMapper maps back to a blockquote; loose inline content is wrapped in one
|
|
747
|
+
// Quote-styled paragraph so it isn't emitted as an ordinary line.
|
|
748
|
+
if (tagName === 'blockquote') {
|
|
749
|
+
const kids = parseChildren(node, newFormatting, listContext);
|
|
750
|
+
const isBlock = (t) => t === 'paragraph' || t === 'heading' || t === 'list';
|
|
751
|
+
if (!kids.some(k => isBlock(k.type))) {
|
|
752
|
+
return { type: 'paragraph', metadata: { style: 'Quote' }, children: kids };
|
|
753
|
+
}
|
|
754
|
+
kids.forEach(k => {
|
|
755
|
+
if (isBlock(k.type))
|
|
756
|
+
k.metadata = { ...k.metadata, style: 'Quote' };
|
|
757
|
+
});
|
|
758
|
+
return kids;
|
|
759
|
+
}
|
|
709
760
|
// Mermaid diagrams. Attribute-driven producers render a
|
|
710
761
|
// <div class="mermaid" data-mermaid="<code>"> with the code also as text content.
|
|
711
762
|
// Map either shape to a fenced code node with language `mermaid`, so it round-trips as
|
|
@@ -931,11 +982,21 @@ const parseHtml = async (buffer, config) => {
|
|
|
931
982
|
const rowSpanAttr = node.attributes?.rowspan;
|
|
932
983
|
const colSpan = colSpanAttr ? parseInt(colSpanAttr, 10) : undefined;
|
|
933
984
|
const rowSpan = rowSpanAttr ? parseInt(rowSpanAttr, 10) : undefined;
|
|
985
|
+
// Per-column GFM alignment: read the cell's own `text-align` (or a legacy `align=`
|
|
986
|
+
// attribute) into `CellMetadata.align`, so the `:---`/`:---:`/`---:` markers survive
|
|
987
|
+
// AST -> HTML -> AST. `justify` has no pipe-table marker, so it is not a cell align.
|
|
988
|
+
// The table-level `<table data-align>` form is read separately in the `table` branch.
|
|
989
|
+
const cellTextAlign = (getDeclaration(parseStyleDeclarations(node.attributes?.style || ''), 'text-align')
|
|
990
|
+
|| node.attributes?.align || '').toLowerCase();
|
|
991
|
+
const cellAlign = ['left', 'center', 'right'].includes(cellTextAlign)
|
|
992
|
+
? cellTextAlign
|
|
993
|
+
: undefined;
|
|
934
994
|
const cellNode = {
|
|
935
995
|
type: 'cell',
|
|
936
996
|
metadata: {
|
|
937
997
|
colSpan: colSpan && !isNaN(colSpan) ? colSpan : undefined,
|
|
938
|
-
rowSpan: rowSpan && !isNaN(rowSpan) ? rowSpan : undefined
|
|
998
|
+
rowSpan: rowSpan && !isNaN(rowSpan) ? rowSpan : undefined,
|
|
999
|
+
align: cellAlign
|
|
939
1000
|
},
|
|
940
1001
|
children: parseChildren(node, newFormatting, listContext),
|
|
941
1002
|
htmlAttributes: collectHtmlAttributes(node, ['colspan', 'rowspan', 'align'])
|
|
@@ -991,6 +1052,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
991
1052
|
metadata: {
|
|
992
1053
|
attachmentName: name,
|
|
993
1054
|
altText: alt,
|
|
1055
|
+
title: node.attributes?.title,
|
|
994
1056
|
width,
|
|
995
1057
|
align
|
|
996
1058
|
}
|
|
@@ -1002,6 +1064,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
1002
1064
|
metadata: {
|
|
1003
1065
|
url: src,
|
|
1004
1066
|
altText: alt,
|
|
1067
|
+
title: node.attributes?.title,
|
|
1005
1068
|
width,
|
|
1006
1069
|
align
|
|
1007
1070
|
}
|
|
@@ -1014,6 +1077,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
1014
1077
|
metadata: {
|
|
1015
1078
|
url: src,
|
|
1016
1079
|
altText: alt,
|
|
1080
|
+
title: node.attributes?.title,
|
|
1017
1081
|
anchorIds: anchorIds.length > 0 ? anchorIds : undefined,
|
|
1018
1082
|
width,
|
|
1019
1083
|
align
|
|
@@ -1055,16 +1119,20 @@ const parseHtml = async (buffer, config) => {
|
|
|
1055
1119
|
}
|
|
1056
1120
|
else if (href) {
|
|
1057
1121
|
const linkType = href.startsWith('#') ? 'internal' : 'external';
|
|
1122
|
+
const linkTitle = node.attributes?.title;
|
|
1058
1123
|
children.forEach(c => {
|
|
1059
1124
|
if (c.type === 'text') {
|
|
1060
|
-
c.metadata = { ...c.metadata, link: href, linkType };
|
|
1125
|
+
c.metadata = { ...c.metadata, link: href, linkType, title: linkTitle };
|
|
1061
1126
|
}
|
|
1062
1127
|
});
|
|
1063
1128
|
}
|
|
1064
1129
|
return children;
|
|
1065
1130
|
}
|
|
1066
1131
|
if (tagName === 'br') {
|
|
1067
|
-
|
|
1132
|
+
// A <br> is a hard line break: `carriageReturn` so the Markdown generator emits a
|
|
1133
|
+
// hard break (` \n`, or a `<br>` inside a table cell) that re-imports as a <br>.
|
|
1134
|
+
// `textWrapping` emitted a bare `\n` in a paragraph, which re-imports as a space.
|
|
1135
|
+
const brNode = { type: 'break', metadata: { breakType: 'carriageReturn' } };
|
|
1068
1136
|
if (config.includeRawContent) {
|
|
1069
1137
|
brNode.rawContent = '<br/>';
|
|
1070
1138
|
}
|
|
@@ -166,6 +166,15 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
166
166
|
mathBlocks.push(latex);
|
|
167
167
|
return `\n\n${id}\n\n`;
|
|
168
168
|
});
|
|
169
|
+
// Single-line `$$...$$` occupying its own line is display (block) math too. Without this it
|
|
170
|
+
// falls through to the inline `$...$` tokenizer, which matches the INNER `$\int$` and leaks
|
|
171
|
+
// the outer pair as two stray literal `$`. Runs after the multi-line pass, whose placeholders
|
|
172
|
+
// carry no `$$` and so can't be re-matched. `(?!\$)` rejects `$$$...`/empty `$$$$`.
|
|
173
|
+
textStr = textStr.replace(/^\$\$(?!\$)([^\n]+?)\$\$[ \t]*$/gm, (_match, latex) => {
|
|
174
|
+
const id = `__MATH_BLOCK_${mathBlocks.length}__`;
|
|
175
|
+
mathBlocks.push(latex);
|
|
176
|
+
return `\n\n${id}\n\n`;
|
|
177
|
+
});
|
|
169
178
|
// Extract GLFM-style fenced-div admonitions (`:::note ... :::`) before block splitting,
|
|
170
179
|
// since their body may itself contain blank lines that would otherwise fragment them.
|
|
171
180
|
// The `> [!NOTE]` GitHub form doesn't need this - it's detected inline in the blockquote
|
|
@@ -253,13 +262,52 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
253
262
|
}
|
|
254
263
|
return result;
|
|
255
264
|
};
|
|
265
|
+
// Attribute list for an embed leaf directive `{id=... src=... width=... height=... align=...}`.
|
|
266
|
+
// Superset of parseAttributeList (adds id/src/height); space-separated `k=v` tokens.
|
|
267
|
+
const parseEmbedDirectiveAttrs = (attrStr) => {
|
|
268
|
+
const result = {};
|
|
269
|
+
for (const token of attrStr.trim().split(/\s+/).filter(Boolean)) {
|
|
270
|
+
const kv = token.match(/^([a-zA-Z-]+)=(.+)$/);
|
|
271
|
+
if (!kv)
|
|
272
|
+
continue;
|
|
273
|
+
const [, key, val] = kv;
|
|
274
|
+
if (key === 'id')
|
|
275
|
+
result.id = val;
|
|
276
|
+
else if (key === 'src')
|
|
277
|
+
result.src = val;
|
|
278
|
+
else if (key === 'width')
|
|
279
|
+
result.width = val;
|
|
280
|
+
else if (key === 'height')
|
|
281
|
+
result.height = val;
|
|
282
|
+
else if (key === 'align' && ['left', 'center', 'right'].includes(val))
|
|
283
|
+
result.align = val;
|
|
284
|
+
}
|
|
285
|
+
return result;
|
|
286
|
+
};
|
|
287
|
+
// Extracts a YouTube video id from any of its URL shapes (watch?v=, youtu.be/, /embed/,
|
|
288
|
+
// img.youtube.com/vi/). Returns undefined for a non-YouTube URL. Used only by the opt-in
|
|
289
|
+
// folk-form import (embedFolkForms).
|
|
290
|
+
const extractYoutubeId = (url) => {
|
|
291
|
+
if (!url || !/(?:youtu\.be|youtube(?:-nocookie)?\.com)/.test(url))
|
|
292
|
+
return undefined;
|
|
293
|
+
const m = url.match(/(?:youtu\.be\/|\/embed\/|[?&]v=|\/vi\/)([A-Za-z0-9_-]+)/);
|
|
294
|
+
return m ? m[1] : undefined;
|
|
295
|
+
};
|
|
256
296
|
const parseInline = (text, currentFormatting = {}) => {
|
|
257
297
|
const nodes = [];
|
|
258
298
|
const plainText = (t) => ({ type: 'text', text: t, formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined });
|
|
259
299
|
// Builds the same image/link node shape regardless of whether the URL came from
|
|
260
300
|
// an inline `(url)` or a resolved reference definition - shared by the inline
|
|
261
301
|
// image/link branch and the two reference-style branches below.
|
|
262
|
-
|
|
302
|
+
// Split a Markdown inline destination `url "title"` (also `'title'` / `(title)`) into its
|
|
303
|
+
// URL and optional title. The inline parser previously kept the whole thing as the URL, so
|
|
304
|
+
// `[t](u "T")` produced href `u "T"`; reference-style `[t][id]` already split it correctly.
|
|
305
|
+
const splitUrlTitle = (raw) => {
|
|
306
|
+
const m = raw.trim().match(/^(.*?)\s+(?:"([^"]*)"|'([^']*)'|\(([^)]*)\))\s*$/);
|
|
307
|
+
return m ? { url: m[1].trim(), title: m[2] ?? m[3] ?? m[4] } : { url: raw };
|
|
308
|
+
};
|
|
309
|
+
const buildLinkOrImageNodes = (isImage, altText, rawUrl, attrsStr) => {
|
|
310
|
+
const { url, title } = splitUrlTitle(rawUrl);
|
|
263
311
|
if (isImage) {
|
|
264
312
|
// Pandoc-style attribute list immediately after an image, e.g. {width=50% .centered}
|
|
265
313
|
const attrs = attrsStr !== undefined ? parseAttributeList(attrsStr) : undefined;
|
|
@@ -276,15 +324,15 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
276
324
|
name,
|
|
277
325
|
extension: mimeType.split('/')[1]
|
|
278
326
|
});
|
|
279
|
-
return [{ type: 'image', metadata: { attachmentName: name, altText, ...attrs } }];
|
|
327
|
+
return [{ type: 'image', metadata: { attachmentName: name, altText, title, ...attrs } }];
|
|
280
328
|
}
|
|
281
329
|
}
|
|
282
|
-
return [{ type: 'image', metadata: { url, altText, ...attrs } }];
|
|
330
|
+
return [{ type: 'image', metadata: { url, altText, title, ...attrs } }];
|
|
283
331
|
}
|
|
284
332
|
const linkNodes = parseInline(altText, currentFormatting);
|
|
285
333
|
linkNodes.forEach(n => {
|
|
286
334
|
if (n.type === 'text') {
|
|
287
|
-
n.metadata = { link: url, linkType: 'external' };
|
|
335
|
+
n.metadata = { link: url, linkType: 'external', title };
|
|
288
336
|
}
|
|
289
337
|
});
|
|
290
338
|
return linkNodes;
|
|
@@ -314,7 +362,7 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
314
362
|
// Inline math requires no whitespace right after the opening $ or right before the
|
|
315
363
|
// closing $, the common heuristic (matching Pandoc/KaTeX) for avoiding false
|
|
316
364
|
// positives on currency like "$5 and $10".
|
|
317
|
-
const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)
|
|
365
|
+
const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)~~|==(?<highlight>.+?)==|(?<codeFence>`+)(?<codeContent>(?:(?!\k<codeFence>)[\s\S])+?)\k<codeFence>(?!`)|<u>(?<underline>.+?)<\/u>|<sub>(?<subscript>.+?)<\/sub>|<sup>(?<superscript>.+?)<\/sup>|(?<lineBreak><br\s*\/?>)|<span\s+style="(?<spanStyle>[^"]*)">(?<spanContent>.+?)<\/span>|\[\^(?<footnoteId>[^\]]+)\]|\[@(?<citationKey>[a-zA-Z0-9_:.-]+)\]|\[\[(?<wikiPage>[^\]|]+)(?:\|(?<wikiAlias>[^\]]+))?\]\]|(?<refBang>!?)\[(?<refText>[^\]]*)\]\[(?<refId>[^\]]*)\]|(?<shortBang>!?)\[(?<shortText>[^\]]+)\]|<(?<autolinkUrl>(?:https?|mailto):[^\s<>]+)>|\$(?!\s)(?<mathInline>[^$\n]+?)(?<!\s)\$/g;
|
|
318
366
|
let lastIndex = 0;
|
|
319
367
|
let match;
|
|
320
368
|
while ((match = regex.exec(text)) !== null) {
|
|
@@ -343,6 +391,9 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
343
391
|
else if (g.strike !== undefined) { // Strikethrough
|
|
344
392
|
nodes.push(...parseInline(g.strike, { ...currentFormatting, strikethrough: true }));
|
|
345
393
|
}
|
|
394
|
+
else if (g.highlight !== undefined) { // ==highlight== (Obsidian/extended); additive on import
|
|
395
|
+
nodes.push(...parseInline(g.highlight, { ...currentFormatting, backgroundColor: '#ffff00' }));
|
|
396
|
+
}
|
|
346
397
|
else if (g.codeContent !== undefined) { // Inline code (any matching backtick-run length)
|
|
347
398
|
nodes.push({ type: 'text', text: g.codeContent, formatting: { ...currentFormatting, font: 'monospace' } });
|
|
348
399
|
}
|
|
@@ -610,6 +661,31 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
610
661
|
blocks.push(currentSubBlock.join('\n'));
|
|
611
662
|
}
|
|
612
663
|
}
|
|
664
|
+
// Re-join a list block that a blank line tore away from its parent. The generator's older
|
|
665
|
+
// loose output (`- a\n\n\n - a1`) and foreign editors both split a nested item into its
|
|
666
|
+
// own block; left apart, the child's leading indent is stripped by the per-block `trim()`
|
|
667
|
+
// below and it reparses as a flat top-level item under a fresh listId. Merge a block back
|
|
668
|
+
// into the preceding one only when the previous block is itself a list (its FIRST line is a
|
|
669
|
+
// marker - the sub-splitter guarantees such a block holds only marker/continuation lines) and
|
|
670
|
+
// the current block OPENS with an INDENTED marker. An unindented `- b` after a blank line is
|
|
671
|
+
// deliberately left split (a flat loose list keeps its own listId), and anything that is not
|
|
672
|
+
// an indented marker (continuation text, indented code, placeholders) never triggers a merge.
|
|
673
|
+
const listMarkerStart = /^(\s*)([-*+]|\d+[.)])\s+/;
|
|
674
|
+
const indentedMarkerStart = /^(?: {2,}|\t)\s*(?:[-*+]|\d+[.)])\s+/;
|
|
675
|
+
const mergedBlocks = [];
|
|
676
|
+
for (const block of blocks) {
|
|
677
|
+
const prev = mergedBlocks[mergedBlocks.length - 1];
|
|
678
|
+
if (prev !== undefined
|
|
679
|
+
&& listMarkerStart.test(prev.split('\n', 1)[0])
|
|
680
|
+
&& indentedMarkerStart.test(block.split('\n', 1)[0])) {
|
|
681
|
+
mergedBlocks[mergedBlocks.length - 1] = `${prev}\n${block}`;
|
|
682
|
+
}
|
|
683
|
+
else {
|
|
684
|
+
mergedBlocks.push(block);
|
|
685
|
+
}
|
|
686
|
+
}
|
|
687
|
+
blocks.length = 0;
|
|
688
|
+
blocks.push(...mergedBlocks);
|
|
613
689
|
let listIdCounter = 1;
|
|
614
690
|
let currentAlignment = undefined;
|
|
615
691
|
for (let block of blocks) {
|
|
@@ -652,6 +728,63 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
652
728
|
alignment = (alignMatch[1] || alignMatch[2]).toLowerCase();
|
|
653
729
|
block = alignMatch[3];
|
|
654
730
|
}
|
|
731
|
+
// Embed leaf directive (remark-directive family): `::youtube[Label]{id=... width=... align=...}`
|
|
732
|
+
// or `::embed[Label]{src=... width=... height=... align=...}`. Only these two names are
|
|
733
|
+
// recognised; any other `::name` stays literal text (no catch-all). `::youtube` renders from
|
|
734
|
+
// a validated id via a fixed template, so it is unconditional; `::embed` carries an arbitrary
|
|
735
|
+
// src, so it is gated behind `preserveIframes` (the trust input) exactly like a raw <iframe>,
|
|
736
|
+
// and stays literal text otherwise. New input only; nothing that parsed before changes.
|
|
737
|
+
const embedDirectiveMatch = block.match(/^::(youtube|embed)(?:\[([^\]]*)\])?\{([^}]*)\}$/);
|
|
738
|
+
if (embedDirectiveMatch) {
|
|
739
|
+
const kind = embedDirectiveMatch[1];
|
|
740
|
+
const label = (embedDirectiveMatch[2] || '').trim() || undefined;
|
|
741
|
+
const attrs = parseEmbedDirectiveAttrs(embedDirectiveMatch[3]);
|
|
742
|
+
if (kind === 'youtube' && attrs.id) {
|
|
743
|
+
const embedUrl = `https://www.youtube.com/watch?v=${attrs.id}`;
|
|
744
|
+
content.push({
|
|
745
|
+
type: 'embed',
|
|
746
|
+
text: embedUrl,
|
|
747
|
+
metadata: { embedType: 'youtube', videoId: attrs.id, url: embedUrl, width: attrs.width, align: attrs.align, label }
|
|
748
|
+
});
|
|
749
|
+
continue;
|
|
750
|
+
}
|
|
751
|
+
if (kind === 'embed' && attrs.src && (0, sanitize_js_1.iframeAllowed)(attrs.src, config.htmlParserConfig?.preserveIframes)) {
|
|
752
|
+
content.push({
|
|
753
|
+
type: 'embed',
|
|
754
|
+
text: attrs.src,
|
|
755
|
+
metadata: { embedType: 'iframe', url: attrs.src, width: attrs.width, height: attrs.height, align: attrs.align, label }
|
|
756
|
+
});
|
|
757
|
+
continue;
|
|
758
|
+
}
|
|
759
|
+
// Recognised name but not a usable/allowed directive: fall through so the line becomes
|
|
760
|
+
// ordinary text rather than being dropped.
|
|
761
|
+
}
|
|
762
|
+
// Ambiguous "folk" embed forms, imported only under the opt-in (embedFolkForms), since
|
|
763
|
+
// auto-upgrading an image/link to an embed is a heuristic that could mangle a genuine image
|
|
764
|
+
// link. Both become a safe youtube embed (rendered from the validated id). A standalone line
|
|
765
|
+
// only; anything not matching falls through to ordinary image/link parsing.
|
|
766
|
+
if (config.htmlParserConfig?.embedFolkForms) {
|
|
767
|
+
// Clickable thumbnail: [](watch), youtube when either URL is a youtube link.
|
|
768
|
+
const thumbMatch = block.match(/^\[!\[([^\]]*)\]\(([^)\s]+)\)\]\(([^)\s]+)\)$/);
|
|
769
|
+
if (thumbMatch) {
|
|
770
|
+
const fid = extractYoutubeId(thumbMatch[2]) || extractYoutubeId(thumbMatch[3]);
|
|
771
|
+
if (fid) {
|
|
772
|
+
const embedUrl = `https://www.youtube.com/watch?v=${fid}`;
|
|
773
|
+
content.push({ type: 'embed', text: embedUrl, metadata: { embedType: 'youtube', videoId: fid, url: embedUrl, label: thumbMatch[1].trim() || undefined } });
|
|
774
|
+
continue;
|
|
775
|
+
}
|
|
776
|
+
}
|
|
777
|
+
// Obsidian-style: a standalone image whose URL is a youtube link.
|
|
778
|
+
const obsMatch = block.match(/^!\[([^\]]*)\]\(([^)\s]+)\)$/);
|
|
779
|
+
if (obsMatch) {
|
|
780
|
+
const fid = extractYoutubeId(obsMatch[2]);
|
|
781
|
+
if (fid) {
|
|
782
|
+
const embedUrl = `https://www.youtube.com/watch?v=${fid}`;
|
|
783
|
+
content.push({ type: 'embed', text: embedUrl, metadata: { embedType: 'youtube', videoId: fid, url: embedUrl, label: obsMatch[1].trim() || undefined } });
|
|
784
|
+
continue;
|
|
785
|
+
}
|
|
786
|
+
}
|
|
787
|
+
}
|
|
655
788
|
// YouTube embed fallback: MarkdownGenerator's 'embed' case emits a single-line
|
|
656
789
|
// <div data-youtube-video="ID" data-width="…" data-align="…"></div> when fallbackToHtml
|
|
657
790
|
// is on; recognise it here so a saved-then-reopened .md keeps the video.
|
|
@@ -661,6 +794,7 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
661
794
|
const attrsStr = youtubeMatch[2];
|
|
662
795
|
const widthMatch = attrsStr.match(/data-width="([^"]*)"/i);
|
|
663
796
|
const youtubeAlignMatch = attrsStr.match(/data-align="([^"]*)"/i);
|
|
797
|
+
const youtubeLabelMatch = attrsStr.match(/data-embed-label="([^"]*)"/i);
|
|
664
798
|
const embedAlign = youtubeAlignMatch && ['left', 'center', 'right'].includes(youtubeAlignMatch[1]) ? youtubeAlignMatch[1] : undefined;
|
|
665
799
|
const embedUrl = videoId ? `https://www.youtube.com/watch?v=${videoId}` : undefined;
|
|
666
800
|
content.push({
|
|
@@ -673,7 +807,8 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
673
807
|
videoId,
|
|
674
808
|
url: embedUrl,
|
|
675
809
|
width: widthMatch?.[1],
|
|
676
|
-
align: embedAlign
|
|
810
|
+
align: embedAlign,
|
|
811
|
+
label: youtubeLabelMatch?.[1]
|
|
677
812
|
}
|
|
678
813
|
});
|
|
679
814
|
continue;
|
|
@@ -692,9 +827,29 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
692
827
|
.replace(/</g, '<').replace(/>/g, '>').replace(/"/g, '"')
|
|
693
828
|
.replace(/'/g, '\'').replace(/&/g, '&');
|
|
694
829
|
const src = decodeAttr(attrsStr.match(/\bsrc="([^"]*)"/i)?.[1]);
|
|
830
|
+
const width = attrsStr.match(/\bwidth="([^"]*)"/i)?.[1];
|
|
831
|
+
const height = attrsStr.match(/\bheight="([^"]*)"/i)?.[1];
|
|
832
|
+
// A YouTube iframe is recognised the same way the HTML parser does it (host + id
|
|
833
|
+
// capture), BEFORE and INDEPENDENT of the preserveIframes gate, so the same iframe
|
|
834
|
+
// yields the same 'youtube' embed whichever parser sees it. Only a generic (non-YouTube)
|
|
835
|
+
// iframe is gated behind preserveIframes and kept as an 'iframe' embed.
|
|
836
|
+
const ytMatch = src && /youtube(?:-nocookie)?\.com/.test(src) ? src.match(/(?:embed\/|v=)([^&?/\s]+)/) : null;
|
|
837
|
+
if (ytMatch) {
|
|
838
|
+
const embedUrl = `https://www.youtube.com/watch?v=${ytMatch[1]}`;
|
|
839
|
+
content.push({
|
|
840
|
+
type: 'embed',
|
|
841
|
+
text: embedUrl,
|
|
842
|
+
metadata: {
|
|
843
|
+
embedType: 'youtube',
|
|
844
|
+
videoId: ytMatch[1],
|
|
845
|
+
url: embedUrl,
|
|
846
|
+
width: width !== undefined ? decodeAttr(width) : undefined,
|
|
847
|
+
height: height !== undefined ? decodeAttr(height) : undefined
|
|
848
|
+
}
|
|
849
|
+
});
|
|
850
|
+
continue;
|
|
851
|
+
}
|
|
695
852
|
if (src && (0, sanitize_js_1.iframeAllowed)(src, config.htmlParserConfig?.preserveIframes)) {
|
|
696
|
-
const width = attrsStr.match(/\bwidth="([^"]*)"/i)?.[1];
|
|
697
|
-
const height = attrsStr.match(/\bheight="([^"]*)"/i)?.[1];
|
|
698
853
|
content.push({
|
|
699
854
|
type: 'embed',
|
|
700
855
|
text: src,
|
|
@@ -974,11 +1129,23 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
974
1129
|
else {
|
|
975
1130
|
const lines = block.trim().split('\n');
|
|
976
1131
|
const rows = [];
|
|
1132
|
+
// Pre-scan the separator row for per-column GFM alignment (`:--` left, `:-:` center,
|
|
1133
|
+
// `--:` right; a bare `--` column has none), so every cell can carry its column's
|
|
1134
|
+
// alignment on CellMetadata.align (the header row precedes the separator, so a
|
|
1135
|
+
// per-cell pass alone could not see it).
|
|
1136
|
+
const sepLine = lines.find(l => l.match(/^\|?\s*:?-+:?\s*(?:\|\s*:?-+:?\s*)*\|?$/));
|
|
1137
|
+
const columnAligns = sepLine
|
|
1138
|
+
? sepLine.replace(/^\||\|$/g, '').split('|').map(c => {
|
|
1139
|
+
const t = c.trim();
|
|
1140
|
+
const l = t.startsWith(':'), r = t.endsWith(':');
|
|
1141
|
+
return (l && r) ? 'center' : r ? 'right' : l ? 'left' : null;
|
|
1142
|
+
})
|
|
1143
|
+
: [];
|
|
977
1144
|
for (let i = 0; i < lines.length; i++) {
|
|
978
1145
|
if (lines[i].match(/^\|?\s*:?-+:?\s*(?:\|\s*:?-+:?\s*)*\|?$/))
|
|
979
1146
|
continue; // Separator row (per-cell `:?-+:?`, GFM-style; accepts short cells like `|-|-|`)
|
|
980
1147
|
const cellsStr = lines[i].replace(/^\||\|$/g, '').split('|');
|
|
981
|
-
const cells = cellsStr.map(c => {
|
|
1148
|
+
const cells = cellsStr.map((c, colIdx) => {
|
|
982
1149
|
// Recognize the MarkdownGenerator's own cell-alignment fallback,
|
|
983
1150
|
// `<div style="text-align: X">…</div>`, and lift it into an aligned
|
|
984
1151
|
// paragraph so it round-trips as alignment instead of being escaped to
|
|
@@ -987,17 +1154,25 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
987
1154
|
let cellAlign;
|
|
988
1155
|
cellText = cellText.replace(/<div\s+style="text-align:\s*(left|center|right|justify);?"\s*>([\s\S]*?)<\/div>/gi, (_m, a, inner) => { cellAlign = a.toLowerCase(); return inner; });
|
|
989
1156
|
const inline = parseInline(cellText, i === 0 ? { bold: true } : {});
|
|
1157
|
+
const colAlign = columnAligns[colIdx] ?? undefined;
|
|
1158
|
+
const cellMeta = colAlign ? { col: colIdx, align: colAlign } : undefined;
|
|
990
1159
|
if (cellAlign && cellAlign !== 'left') {
|
|
991
1160
|
return {
|
|
992
1161
|
type: 'cell',
|
|
1162
|
+
metadata: cellMeta,
|
|
993
1163
|
children: [{ type: 'paragraph', metadata: { alignment: cellAlign }, children: inline }]
|
|
994
1164
|
};
|
|
995
1165
|
}
|
|
996
|
-
return { type: 'cell', children: inline };
|
|
1166
|
+
return { type: 'cell', metadata: cellMeta, children: inline };
|
|
997
1167
|
});
|
|
998
1168
|
rows.push({ type: 'row', children: cells });
|
|
999
1169
|
}
|
|
1000
|
-
|
|
1170
|
+
// If every explicitly-aligned column agrees, also expose it as the table-level align,
|
|
1171
|
+
// so an editor that models one alignment per table (and HTML data-align) round-trips.
|
|
1172
|
+
const explicitAligns = columnAligns.filter((a) => a !== null);
|
|
1173
|
+
const uniformAlign = explicitAligns.length > 0 && explicitAligns.every(a => a === explicitAligns[0]) ? explicitAligns[0] : undefined;
|
|
1174
|
+
const resolvedTableAlign = tableAlign || uniformAlign;
|
|
1175
|
+
content.push({ type: 'table', metadata: resolvedTableAlign ? { align: resolvedTableAlign } : undefined, children: rows });
|
|
1001
1176
|
continue;
|
|
1002
1177
|
}
|
|
1003
1178
|
}
|