officeparser 7.6.2 → 7.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -558,6 +558,32 @@ const parseHtml = async (buffer, config) => {
558
558
  }
559
559
  return kids;
560
560
  };
561
+ // Gated generic-iframe embed (HtmlGenerator's `gatedEmbeds` shape): an inert
562
+ // click-to-load placeholder that never auto-loads its src. Read it back to the same
563
+ // `embed` node unconditionally - capturing metadata is safe (the src is scheme-checked
564
+ // again on any re-emit); it is the trusted, already-gated counterpart to a raw <iframe>.
565
+ if (tagName === 'div' && node.attributes?.['data-embed-gated'] !== undefined) {
566
+ const gatedSrc = decodeEntities(node.attributes?.['data-embed-src'] || '');
567
+ if (!gatedSrc)
568
+ return null;
569
+ const gatedAlignAttr = node.attributes?.['data-embed-align'];
570
+ const gatedAlign = ['left', 'center', 'right'].includes(gatedAlignAttr) ? gatedAlignAttr : undefined;
571
+ const gatedNode = {
572
+ type: 'embed',
573
+ text: gatedSrc,
574
+ metadata: {
575
+ embedType: 'iframe',
576
+ url: gatedSrc,
577
+ width: node.attributes?.['data-embed-width'],
578
+ height: node.attributes?.['data-embed-height'],
579
+ align: gatedAlign,
580
+ label: node.attributes?.['data-embed-label']
581
+ }
582
+ };
583
+ if (config.includeRawContent)
584
+ gatedNode.rawContent = '<div data-embed-gated>...</div>';
585
+ return gatedNode;
586
+ }
561
587
  // YouTube embeds: attribute-driven editors render
562
588
  // <div data-youtube-video="ID" data-width="…" data-align="…">…<iframe…></div>.
563
589
  // Recognise both the wrapper div and a bare iframe so externally-authored HTML
@@ -578,7 +604,8 @@ const parseHtml = async (buffer, config) => {
578
604
  videoId,
579
605
  url: embedUrl,
580
606
  width,
581
- align: embedAlign
607
+ align: embedAlign,
608
+ label: node.attributes?.['data-embed-label']
582
609
  }
583
610
  };
584
611
  if (config.includeRawContent)
@@ -593,7 +620,15 @@ const parseHtml = async (buffer, config) => {
593
620
  const embedNode = {
594
621
  type: 'embed',
595
622
  text: embedUrl,
596
- metadata: { embedType: 'youtube', videoId: ytMatch[1], url: embedUrl }
623
+ // Carry the iframe's own width/height so a YouTube iframe's dimensions are not
624
+ // dropped (metadata is unified across embedTypes; align has no source here).
625
+ metadata: {
626
+ embedType: 'youtube',
627
+ videoId: ytMatch[1],
628
+ url: embedUrl,
629
+ width: node.attributes?.width,
630
+ height: node.attributes?.height
631
+ }
597
632
  };
598
633
  if (config.includeRawContent)
599
634
  embedNode.rawContent = '<iframe>...</iframe>';
@@ -706,6 +741,22 @@ const parseHtml = async (buffer, config) => {
706
741
  admonitionNode.rawContent = '<div class="admonition">...</div>';
707
742
  return admonitionNode;
708
743
  }
744
+ // Blockquote. Previously dropped entirely (its children were lifted out unquoted), so a
745
+ // <blockquote> lost its `> ` on the Markdown hop. Mark each block child with the 'Quote'
746
+ // style the styleMapper maps back to a blockquote; loose inline content is wrapped in one
747
+ // Quote-styled paragraph so it isn't emitted as an ordinary line.
748
+ if (tagName === 'blockquote') {
749
+ const kids = parseChildren(node, newFormatting, listContext);
750
+ const isBlock = (t) => t === 'paragraph' || t === 'heading' || t === 'list';
751
+ if (!kids.some(k => isBlock(k.type))) {
752
+ return { type: 'paragraph', metadata: { style: 'Quote' }, children: kids };
753
+ }
754
+ kids.forEach(k => {
755
+ if (isBlock(k.type))
756
+ k.metadata = { ...k.metadata, style: 'Quote' };
757
+ });
758
+ return kids;
759
+ }
709
760
  // Mermaid diagrams. Attribute-driven producers render a
710
761
  // <div class="mermaid" data-mermaid="<code>"> with the code also as text content.
711
762
  // Map either shape to a fenced code node with language `mermaid`, so it round-trips as
@@ -931,11 +982,21 @@ const parseHtml = async (buffer, config) => {
931
982
  const rowSpanAttr = node.attributes?.rowspan;
932
983
  const colSpan = colSpanAttr ? parseInt(colSpanAttr, 10) : undefined;
933
984
  const rowSpan = rowSpanAttr ? parseInt(rowSpanAttr, 10) : undefined;
985
+ // Per-column GFM alignment: read the cell's own `text-align` (or a legacy `align=`
986
+ // attribute) into `CellMetadata.align`, so the `:---`/`:---:`/`---:` markers survive
987
+ // AST -> HTML -> AST. `justify` has no pipe-table marker, so it is not a cell align.
988
+ // The table-level `<table data-align>` form is read separately in the `table` branch.
989
+ const cellTextAlign = (getDeclaration(parseStyleDeclarations(node.attributes?.style || ''), 'text-align')
990
+ || node.attributes?.align || '').toLowerCase();
991
+ const cellAlign = ['left', 'center', 'right'].includes(cellTextAlign)
992
+ ? cellTextAlign
993
+ : undefined;
934
994
  const cellNode = {
935
995
  type: 'cell',
936
996
  metadata: {
937
997
  colSpan: colSpan && !isNaN(colSpan) ? colSpan : undefined,
938
- rowSpan: rowSpan && !isNaN(rowSpan) ? rowSpan : undefined
998
+ rowSpan: rowSpan && !isNaN(rowSpan) ? rowSpan : undefined,
999
+ align: cellAlign
939
1000
  },
940
1001
  children: parseChildren(node, newFormatting, listContext),
941
1002
  htmlAttributes: collectHtmlAttributes(node, ['colspan', 'rowspan', 'align'])
@@ -991,6 +1052,7 @@ const parseHtml = async (buffer, config) => {
991
1052
  metadata: {
992
1053
  attachmentName: name,
993
1054
  altText: alt,
1055
+ title: node.attributes?.title,
994
1056
  width,
995
1057
  align
996
1058
  }
@@ -1002,6 +1064,7 @@ const parseHtml = async (buffer, config) => {
1002
1064
  metadata: {
1003
1065
  url: src,
1004
1066
  altText: alt,
1067
+ title: node.attributes?.title,
1005
1068
  width,
1006
1069
  align
1007
1070
  }
@@ -1014,6 +1077,7 @@ const parseHtml = async (buffer, config) => {
1014
1077
  metadata: {
1015
1078
  url: src,
1016
1079
  altText: alt,
1080
+ title: node.attributes?.title,
1017
1081
  anchorIds: anchorIds.length > 0 ? anchorIds : undefined,
1018
1082
  width,
1019
1083
  align
@@ -1055,16 +1119,20 @@ const parseHtml = async (buffer, config) => {
1055
1119
  }
1056
1120
  else if (href) {
1057
1121
  const linkType = href.startsWith('#') ? 'internal' : 'external';
1122
+ const linkTitle = node.attributes?.title;
1058
1123
  children.forEach(c => {
1059
1124
  if (c.type === 'text') {
1060
- c.metadata = { ...c.metadata, link: href, linkType };
1125
+ c.metadata = { ...c.metadata, link: href, linkType, title: linkTitle };
1061
1126
  }
1062
1127
  });
1063
1128
  }
1064
1129
  return children;
1065
1130
  }
1066
1131
  if (tagName === 'br') {
1067
- const brNode = { type: 'break', metadata: { breakType: 'textWrapping' } };
1132
+ // A <br> is a hard line break: `carriageReturn` so the Markdown generator emits a
1133
+ // hard break (` \n`, or a `<br>` inside a table cell) that re-imports as a <br>.
1134
+ // `textWrapping` emitted a bare `\n` in a paragraph, which re-imports as a space.
1135
+ const brNode = { type: 'break', metadata: { breakType: 'carriageReturn' } };
1068
1136
  if (config.includeRawContent) {
1069
1137
  brNode.rawContent = '<br/>';
1070
1138
  }
@@ -166,6 +166,15 @@ const parseMarkdown = async (buffer, config) => {
166
166
  mathBlocks.push(latex);
167
167
  return `\n\n${id}\n\n`;
168
168
  });
169
+ // Single-line `$$...$$` occupying its own line is display (block) math too. Without this it
170
+ // falls through to the inline `$...$` tokenizer, which matches the INNER `$\int$` and leaks
171
+ // the outer pair as two stray literal `$`. Runs after the multi-line pass, whose placeholders
172
+ // carry no `$$` and so can't be re-matched. `(?!\$)` rejects `$$$...`/empty `$$$$`.
173
+ textStr = textStr.replace(/^\$\$(?!\$)([^\n]+?)\$\$[ \t]*$/gm, (_match, latex) => {
174
+ const id = `__MATH_BLOCK_${mathBlocks.length}__`;
175
+ mathBlocks.push(latex);
176
+ return `\n\n${id}\n\n`;
177
+ });
169
178
  // Extract GLFM-style fenced-div admonitions (`:::note ... :::`) before block splitting,
170
179
  // since their body may itself contain blank lines that would otherwise fragment them.
171
180
  // The `> [!NOTE]` GitHub form doesn't need this - it's detected inline in the blockquote
@@ -253,13 +262,52 @@ const parseMarkdown = async (buffer, config) => {
253
262
  }
254
263
  return result;
255
264
  };
265
+ // Attribute list for an embed leaf directive `{id=... src=... width=... height=... align=...}`.
266
+ // Superset of parseAttributeList (adds id/src/height); space-separated `k=v` tokens.
267
+ const parseEmbedDirectiveAttrs = (attrStr) => {
268
+ const result = {};
269
+ for (const token of attrStr.trim().split(/\s+/).filter(Boolean)) {
270
+ const kv = token.match(/^([a-zA-Z-]+)=(.+)$/);
271
+ if (!kv)
272
+ continue;
273
+ const [, key, val] = kv;
274
+ if (key === 'id')
275
+ result.id = val;
276
+ else if (key === 'src')
277
+ result.src = val;
278
+ else if (key === 'width')
279
+ result.width = val;
280
+ else if (key === 'height')
281
+ result.height = val;
282
+ else if (key === 'align' && ['left', 'center', 'right'].includes(val))
283
+ result.align = val;
284
+ }
285
+ return result;
286
+ };
287
+ // Extracts a YouTube video id from any of its URL shapes (watch?v=, youtu.be/, /embed/,
288
+ // img.youtube.com/vi/). Returns undefined for a non-YouTube URL. Used only by the opt-in
289
+ // folk-form import (embedFolkForms).
290
+ const extractYoutubeId = (url) => {
291
+ if (!url || !/(?:youtu\.be|youtube(?:-nocookie)?\.com)/.test(url))
292
+ return undefined;
293
+ const m = url.match(/(?:youtu\.be\/|\/embed\/|[?&]v=|\/vi\/)([A-Za-z0-9_-]+)/);
294
+ return m ? m[1] : undefined;
295
+ };
256
296
  const parseInline = (text, currentFormatting = {}) => {
257
297
  const nodes = [];
258
298
  const plainText = (t) => ({ type: 'text', text: t, formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined });
259
299
  // Builds the same image/link node shape regardless of whether the URL came from
260
300
  // an inline `(url)` or a resolved reference definition - shared by the inline
261
301
  // image/link branch and the two reference-style branches below.
262
- const buildLinkOrImageNodes = (isImage, altText, url, attrsStr) => {
302
+ // Split a Markdown inline destination `url "title"` (also `'title'` / `(title)`) into its
303
+ // URL and optional title. The inline parser previously kept the whole thing as the URL, so
304
+ // `[t](u "T")` produced href `u "T"`; reference-style `[t][id]` already split it correctly.
305
+ const splitUrlTitle = (raw) => {
306
+ const m = raw.trim().match(/^(.*?)\s+(?:"([^"]*)"|'([^']*)'|\(([^)]*)\))\s*$/);
307
+ return m ? { url: m[1].trim(), title: m[2] ?? m[3] ?? m[4] } : { url: raw };
308
+ };
309
+ const buildLinkOrImageNodes = (isImage, altText, rawUrl, attrsStr) => {
310
+ const { url, title } = splitUrlTitle(rawUrl);
263
311
  if (isImage) {
264
312
  // Pandoc-style attribute list immediately after an image, e.g. {width=50% .centered}
265
313
  const attrs = attrsStr !== undefined ? parseAttributeList(attrsStr) : undefined;
@@ -276,15 +324,15 @@ const parseMarkdown = async (buffer, config) => {
276
324
  name,
277
325
  extension: mimeType.split('/')[1]
278
326
  });
279
- return [{ type: 'image', metadata: { attachmentName: name, altText, ...attrs } }];
327
+ return [{ type: 'image', metadata: { attachmentName: name, altText, title, ...attrs } }];
280
328
  }
281
329
  }
282
- return [{ type: 'image', metadata: { url, altText, ...attrs } }];
330
+ return [{ type: 'image', metadata: { url, altText, title, ...attrs } }];
283
331
  }
284
332
  const linkNodes = parseInline(altText, currentFormatting);
285
333
  linkNodes.forEach(n => {
286
334
  if (n.type === 'text') {
287
- n.metadata = { link: url, linkType: 'external' };
335
+ n.metadata = { link: url, linkType: 'external', title };
288
336
  }
289
337
  });
290
338
  return linkNodes;
@@ -314,7 +362,7 @@ const parseMarkdown = async (buffer, config) => {
314
362
  // Inline math requires no whitespace right after the opening $ or right before the
315
363
  // closing $, the common heuristic (matching Pandoc/KaTeX) for avoiding false
316
364
  // positives on currency like "$5 and $10".
317
- const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)~~|(?<codeFence>`+)(?<codeContent>(?:(?!\k<codeFence>)[\s\S])+?)\k<codeFence>(?!`)|<u>(?<underline>.+?)<\/u>|<sub>(?<subscript>.+?)<\/sub>|<sup>(?<superscript>.+?)<\/sup>|(?<lineBreak><br\s*\/?>)|<span\s+style="(?<spanStyle>[^"]*)">(?<spanContent>.+?)<\/span>|\[\^(?<footnoteId>[^\]]+)\]|\[@(?<citationKey>[a-zA-Z0-9_:.-]+)\]|\[\[(?<wikiPage>[^\]|]+)(?:\|(?<wikiAlias>[^\]]+))?\]\]|(?<refBang>!?)\[(?<refText>[^\]]*)\]\[(?<refId>[^\]]*)\]|(?<shortBang>!?)\[(?<shortText>[^\]]+)\]|<(?<autolinkUrl>(?:https?|mailto):[^\s<>]+)>|\$(?!\s)(?<mathInline>[^$\n]+?)(?<!\s)\$/g;
365
+ const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)~~|==(?<highlight>.+?)==|(?<codeFence>`+)(?<codeContent>(?:(?!\k<codeFence>)[\s\S])+?)\k<codeFence>(?!`)|<u>(?<underline>.+?)<\/u>|<sub>(?<subscript>.+?)<\/sub>|<sup>(?<superscript>.+?)<\/sup>|(?<lineBreak><br\s*\/?>)|<span\s+style="(?<spanStyle>[^"]*)">(?<spanContent>.+?)<\/span>|\[\^(?<footnoteId>[^\]]+)\]|\[@(?<citationKey>[a-zA-Z0-9_:.-]+)\]|\[\[(?<wikiPage>[^\]|]+)(?:\|(?<wikiAlias>[^\]]+))?\]\]|(?<refBang>!?)\[(?<refText>[^\]]*)\]\[(?<refId>[^\]]*)\]|(?<shortBang>!?)\[(?<shortText>[^\]]+)\]|<(?<autolinkUrl>(?:https?|mailto):[^\s<>]+)>|\$(?!\s)(?<mathInline>[^$\n]+?)(?<!\s)\$/g;
318
366
  let lastIndex = 0;
319
367
  let match;
320
368
  while ((match = regex.exec(text)) !== null) {
@@ -343,6 +391,9 @@ const parseMarkdown = async (buffer, config) => {
343
391
  else if (g.strike !== undefined) { // Strikethrough
344
392
  nodes.push(...parseInline(g.strike, { ...currentFormatting, strikethrough: true }));
345
393
  }
394
+ else if (g.highlight !== undefined) { // ==highlight== (Obsidian/extended); additive on import
395
+ nodes.push(...parseInline(g.highlight, { ...currentFormatting, backgroundColor: '#ffff00' }));
396
+ }
346
397
  else if (g.codeContent !== undefined) { // Inline code (any matching backtick-run length)
347
398
  nodes.push({ type: 'text', text: g.codeContent, formatting: { ...currentFormatting, font: 'monospace' } });
348
399
  }
@@ -610,6 +661,31 @@ const parseMarkdown = async (buffer, config) => {
610
661
  blocks.push(currentSubBlock.join('\n'));
611
662
  }
612
663
  }
664
+ // Re-join a list block that a blank line tore away from its parent. The generator's older
665
+ // loose output (`- a\n\n\n - a1`) and foreign editors both split a nested item into its
666
+ // own block; left apart, the child's leading indent is stripped by the per-block `trim()`
667
+ // below and it reparses as a flat top-level item under a fresh listId. Merge a block back
668
+ // into the preceding one only when the previous block is itself a list (its FIRST line is a
669
+ // marker - the sub-splitter guarantees such a block holds only marker/continuation lines) and
670
+ // the current block OPENS with an INDENTED marker. An unindented `- b` after a blank line is
671
+ // deliberately left split (a flat loose list keeps its own listId), and anything that is not
672
+ // an indented marker (continuation text, indented code, placeholders) never triggers a merge.
673
+ const listMarkerStart = /^(\s*)([-*+]|\d+[.)])\s+/;
674
+ const indentedMarkerStart = /^(?: {2,}|\t)\s*(?:[-*+]|\d+[.)])\s+/;
675
+ const mergedBlocks = [];
676
+ for (const block of blocks) {
677
+ const prev = mergedBlocks[mergedBlocks.length - 1];
678
+ if (prev !== undefined
679
+ && listMarkerStart.test(prev.split('\n', 1)[0])
680
+ && indentedMarkerStart.test(block.split('\n', 1)[0])) {
681
+ mergedBlocks[mergedBlocks.length - 1] = `${prev}\n${block}`;
682
+ }
683
+ else {
684
+ mergedBlocks.push(block);
685
+ }
686
+ }
687
+ blocks.length = 0;
688
+ blocks.push(...mergedBlocks);
613
689
  let listIdCounter = 1;
614
690
  let currentAlignment = undefined;
615
691
  for (let block of blocks) {
@@ -652,6 +728,63 @@ const parseMarkdown = async (buffer, config) => {
652
728
  alignment = (alignMatch[1] || alignMatch[2]).toLowerCase();
653
729
  block = alignMatch[3];
654
730
  }
731
+ // Embed leaf directive (remark-directive family): `::youtube[Label]{id=... width=... align=...}`
732
+ // or `::embed[Label]{src=... width=... height=... align=...}`. Only these two names are
733
+ // recognised; any other `::name` stays literal text (no catch-all). `::youtube` renders from
734
+ // a validated id via a fixed template, so it is unconditional; `::embed` carries an arbitrary
735
+ // src, so it is gated behind `preserveIframes` (the trust input) exactly like a raw <iframe>,
736
+ // and stays literal text otherwise. New input only; nothing that parsed before changes.
737
+ const embedDirectiveMatch = block.match(/^::(youtube|embed)(?:\[([^\]]*)\])?\{([^}]*)\}$/);
738
+ if (embedDirectiveMatch) {
739
+ const kind = embedDirectiveMatch[1];
740
+ const label = (embedDirectiveMatch[2] || '').trim() || undefined;
741
+ const attrs = parseEmbedDirectiveAttrs(embedDirectiveMatch[3]);
742
+ if (kind === 'youtube' && attrs.id) {
743
+ const embedUrl = `https://www.youtube.com/watch?v=${attrs.id}`;
744
+ content.push({
745
+ type: 'embed',
746
+ text: embedUrl,
747
+ metadata: { embedType: 'youtube', videoId: attrs.id, url: embedUrl, width: attrs.width, align: attrs.align, label }
748
+ });
749
+ continue;
750
+ }
751
+ if (kind === 'embed' && attrs.src && (0, sanitize_js_1.iframeAllowed)(attrs.src, config.htmlParserConfig?.preserveIframes)) {
752
+ content.push({
753
+ type: 'embed',
754
+ text: attrs.src,
755
+ metadata: { embedType: 'iframe', url: attrs.src, width: attrs.width, height: attrs.height, align: attrs.align, label }
756
+ });
757
+ continue;
758
+ }
759
+ // Recognised name but not a usable/allowed directive: fall through so the line becomes
760
+ // ordinary text rather than being dropped.
761
+ }
762
+ // Ambiguous "folk" embed forms, imported only under the opt-in (embedFolkForms), since
763
+ // auto-upgrading an image/link to an embed is a heuristic that could mangle a genuine image
764
+ // link. Both become a safe youtube embed (rendered from the validated id). A standalone line
765
+ // only; anything not matching falls through to ordinary image/link parsing.
766
+ if (config.htmlParserConfig?.embedFolkForms) {
767
+ // Clickable thumbnail: [![alt](thumb)](watch), youtube when either URL is a youtube link.
768
+ const thumbMatch = block.match(/^\[!\[([^\]]*)\]\(([^)\s]+)\)\]\(([^)\s]+)\)$/);
769
+ if (thumbMatch) {
770
+ const fid = extractYoutubeId(thumbMatch[2]) || extractYoutubeId(thumbMatch[3]);
771
+ if (fid) {
772
+ const embedUrl = `https://www.youtube.com/watch?v=${fid}`;
773
+ content.push({ type: 'embed', text: embedUrl, metadata: { embedType: 'youtube', videoId: fid, url: embedUrl, label: thumbMatch[1].trim() || undefined } });
774
+ continue;
775
+ }
776
+ }
777
+ // Obsidian-style: a standalone image whose URL is a youtube link.
778
+ const obsMatch = block.match(/^!\[([^\]]*)\]\(([^)\s]+)\)$/);
779
+ if (obsMatch) {
780
+ const fid = extractYoutubeId(obsMatch[2]);
781
+ if (fid) {
782
+ const embedUrl = `https://www.youtube.com/watch?v=${fid}`;
783
+ content.push({ type: 'embed', text: embedUrl, metadata: { embedType: 'youtube', videoId: fid, url: embedUrl, label: obsMatch[1].trim() || undefined } });
784
+ continue;
785
+ }
786
+ }
787
+ }
655
788
  // YouTube embed fallback: MarkdownGenerator's 'embed' case emits a single-line
656
789
  // <div data-youtube-video="ID" data-width="…" data-align="…"></div> when fallbackToHtml
657
790
  // is on; recognise it here so a saved-then-reopened .md keeps the video.
@@ -661,6 +794,7 @@ const parseMarkdown = async (buffer, config) => {
661
794
  const attrsStr = youtubeMatch[2];
662
795
  const widthMatch = attrsStr.match(/data-width="([^"]*)"/i);
663
796
  const youtubeAlignMatch = attrsStr.match(/data-align="([^"]*)"/i);
797
+ const youtubeLabelMatch = attrsStr.match(/data-embed-label="([^"]*)"/i);
664
798
  const embedAlign = youtubeAlignMatch && ['left', 'center', 'right'].includes(youtubeAlignMatch[1]) ? youtubeAlignMatch[1] : undefined;
665
799
  const embedUrl = videoId ? `https://www.youtube.com/watch?v=${videoId}` : undefined;
666
800
  content.push({
@@ -673,7 +807,8 @@ const parseMarkdown = async (buffer, config) => {
673
807
  videoId,
674
808
  url: embedUrl,
675
809
  width: widthMatch?.[1],
676
- align: embedAlign
810
+ align: embedAlign,
811
+ label: youtubeLabelMatch?.[1]
677
812
  }
678
813
  });
679
814
  continue;
@@ -692,9 +827,29 @@ const parseMarkdown = async (buffer, config) => {
692
827
  .replace(/&lt;/g, '<').replace(/&gt;/g, '>').replace(/&quot;/g, '"')
693
828
  .replace(/&#39;/g, '\'').replace(/&amp;/g, '&');
694
829
  const src = decodeAttr(attrsStr.match(/\bsrc="([^"]*)"/i)?.[1]);
830
+ const width = attrsStr.match(/\bwidth="([^"]*)"/i)?.[1];
831
+ const height = attrsStr.match(/\bheight="([^"]*)"/i)?.[1];
832
+ // A YouTube iframe is recognised the same way the HTML parser does it (host + id
833
+ // capture), BEFORE and INDEPENDENT of the preserveIframes gate, so the same iframe
834
+ // yields the same 'youtube' embed whichever parser sees it. Only a generic (non-YouTube)
835
+ // iframe is gated behind preserveIframes and kept as an 'iframe' embed.
836
+ const ytMatch = src && /youtube(?:-nocookie)?\.com/.test(src) ? src.match(/(?:embed\/|v=)([^&?/\s]+)/) : null;
837
+ if (ytMatch) {
838
+ const embedUrl = `https://www.youtube.com/watch?v=${ytMatch[1]}`;
839
+ content.push({
840
+ type: 'embed',
841
+ text: embedUrl,
842
+ metadata: {
843
+ embedType: 'youtube',
844
+ videoId: ytMatch[1],
845
+ url: embedUrl,
846
+ width: width !== undefined ? decodeAttr(width) : undefined,
847
+ height: height !== undefined ? decodeAttr(height) : undefined
848
+ }
849
+ });
850
+ continue;
851
+ }
695
852
  if (src && (0, sanitize_js_1.iframeAllowed)(src, config.htmlParserConfig?.preserveIframes)) {
696
- const width = attrsStr.match(/\bwidth="([^"]*)"/i)?.[1];
697
- const height = attrsStr.match(/\bheight="([^"]*)"/i)?.[1];
698
853
  content.push({
699
854
  type: 'embed',
700
855
  text: src,
@@ -974,11 +1129,23 @@ const parseMarkdown = async (buffer, config) => {
974
1129
  else {
975
1130
  const lines = block.trim().split('\n');
976
1131
  const rows = [];
1132
+ // Pre-scan the separator row for per-column GFM alignment (`:--` left, `:-:` center,
1133
+ // `--:` right; a bare `--` column has none), so every cell can carry its column's
1134
+ // alignment on CellMetadata.align (the header row precedes the separator, so a
1135
+ // per-cell pass alone could not see it).
1136
+ const sepLine = lines.find(l => l.match(/^\|?\s*:?-+:?\s*(?:\|\s*:?-+:?\s*)*\|?$/));
1137
+ const columnAligns = sepLine
1138
+ ? sepLine.replace(/^\||\|$/g, '').split('|').map(c => {
1139
+ const t = c.trim();
1140
+ const l = t.startsWith(':'), r = t.endsWith(':');
1141
+ return (l && r) ? 'center' : r ? 'right' : l ? 'left' : null;
1142
+ })
1143
+ : [];
977
1144
  for (let i = 0; i < lines.length; i++) {
978
1145
  if (lines[i].match(/^\|?\s*:?-+:?\s*(?:\|\s*:?-+:?\s*)*\|?$/))
979
1146
  continue; // Separator row (per-cell `:?-+:?`, GFM-style; accepts short cells like `|-|-|`)
980
1147
  const cellsStr = lines[i].replace(/^\||\|$/g, '').split('|');
981
- const cells = cellsStr.map(c => {
1148
+ const cells = cellsStr.map((c, colIdx) => {
982
1149
  // Recognize the MarkdownGenerator's own cell-alignment fallback,
983
1150
  // `<div style="text-align: X">…</div>`, and lift it into an aligned
984
1151
  // paragraph so it round-trips as alignment instead of being escaped to
@@ -987,17 +1154,25 @@ const parseMarkdown = async (buffer, config) => {
987
1154
  let cellAlign;
988
1155
  cellText = cellText.replace(/<div\s+style="text-align:\s*(left|center|right|justify);?"\s*>([\s\S]*?)<\/div>/gi, (_m, a, inner) => { cellAlign = a.toLowerCase(); return inner; });
989
1156
  const inline = parseInline(cellText, i === 0 ? { bold: true } : {});
1157
+ const colAlign = columnAligns[colIdx] ?? undefined;
1158
+ const cellMeta = colAlign ? { col: colIdx, align: colAlign } : undefined;
990
1159
  if (cellAlign && cellAlign !== 'left') {
991
1160
  return {
992
1161
  type: 'cell',
1162
+ metadata: cellMeta,
993
1163
  children: [{ type: 'paragraph', metadata: { alignment: cellAlign }, children: inline }]
994
1164
  };
995
1165
  }
996
- return { type: 'cell', children: inline };
1166
+ return { type: 'cell', metadata: cellMeta, children: inline };
997
1167
  });
998
1168
  rows.push({ type: 'row', children: cells });
999
1169
  }
1000
- content.push({ type: 'table', metadata: tableAlign ? { align: tableAlign } : undefined, children: rows });
1170
+ // If every explicitly-aligned column agrees, also expose it as the table-level align,
1171
+ // so an editor that models one alignment per table (and HTML data-align) round-trips.
1172
+ const explicitAligns = columnAligns.filter((a) => a !== null);
1173
+ const uniformAlign = explicitAligns.length > 0 && explicitAligns.every(a => a === explicitAligns[0]) ? explicitAligns[0] : undefined;
1174
+ const resolvedTableAlign = tableAlign || uniformAlign;
1175
+ content.push({ type: 'table', metadata: resolvedTableAlign ? { align: resolvedTableAlign } : undefined, children: rows });
1001
1176
  continue;
1002
1177
  }
1003
1178
  }