officeparser 7.2.3 → 7.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +148 -9
  2. package/dist/OfficeGenerator.js +4 -0
  3. package/dist/OfficeParser.d.ts +2 -0
  4. package/dist/OfficeParser.js +6 -0
  5. package/dist/cli.d.ts +1 -1
  6. package/dist/cli.js +3 -2
  7. package/dist/generators/BaseGenerator.d.ts +11 -0
  8. package/dist/generators/BaseGenerator.js +29 -0
  9. package/dist/generators/CsvGenerator.d.ts +9 -1
  10. package/dist/generators/CsvGenerator.js +24 -14
  11. package/dist/generators/EpubGenerator.d.ts +18 -0
  12. package/dist/generators/EpubGenerator.js +242 -0
  13. package/dist/generators/HtmlGenerator.d.ts +12 -0
  14. package/dist/generators/HtmlGenerator.js +266 -51
  15. package/dist/generators/MarkdownGenerator.d.ts +16 -0
  16. package/dist/generators/MarkdownGenerator.js +173 -24
  17. package/dist/generators/PdfGenerator.js +32 -0
  18. package/dist/generators/RtfGenerator.js +12 -15
  19. package/dist/generators/TextGenerator.js +11 -0
  20. package/dist/index.d.ts +1 -0
  21. package/dist/index.js +1 -0
  22. package/dist/officeparser.browser.d.ts +143 -6
  23. package/dist/officeparser.browser.iife.js +284 -188
  24. package/dist/officeparser.browser.mjs +284 -188
  25. package/dist/officeparser.browser.slim.d.ts +143 -6
  26. package/dist/officeparser.browser.slim.iife.js +284 -188
  27. package/dist/officeparser.browser.slim.mjs +284 -188
  28. package/dist/parsers/EpubParser.d.ts +8 -0
  29. package/dist/parsers/EpubParser.js +217 -0
  30. package/dist/parsers/HtmlParser.js +284 -20
  31. package/dist/parsers/MarkdownParser.js +424 -33
  32. package/dist/parsers/PdfParser.js +4 -1
  33. package/dist/sbom.cdx.json +1695 -0
  34. package/dist/types.d.ts +142 -6
  35. package/dist/types.js +2 -0
  36. package/dist/utils/errorUtils.js +3 -2
  37. package/dist/utils/sanitize.d.ts +99 -0
  38. package/dist/utils/sanitize.js +228 -0
  39. package/dist/utils/zipUtils.js +76 -26
  40. package/package.json +9 -5
@@ -1,6 +1,7 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.parseHtml = void 0;
4
+ const types_js_1 = require("../types.js");
4
5
  const astUtils_js_1 = require("../utils/astUtils.js");
5
6
  const errorUtils_js_1 = require("../utils/errorUtils.js");
6
7
  const parseAttributes = (attrString) => {
@@ -36,14 +37,16 @@ const parseHtmlTree = (html) => {
36
37
  cursor = commentEnd !== -1 ? commentEnd + 3 : html.length;
37
38
  continue;
38
39
  }
39
- const tagEndMatch = html.substring(tagStart).match(/>/);
40
- if (!tagEndMatch) {
40
+ // indexOf (not substring().match) so scanning for the tag end is O(1) in
41
+ // allocation — a document with many "<" chars would otherwise be O(n^2).
42
+ const tagEndIdx = html.indexOf('>', tagStart);
43
+ if (tagEndIdx === -1) {
41
44
  const text = html.substring(tagStart);
42
45
  current.children.push({ type: 'text', text, children: [], parent: current });
43
46
  break;
44
47
  }
45
- const tagContent = html.substring(tagStart + 1, tagStart + tagEndMatch.index);
46
- cursor = tagStart + tagEndMatch.index + 1;
48
+ const tagContent = html.substring(tagStart + 1, tagEndIdx);
49
+ cursor = tagEndIdx + 1;
47
50
  const isClosing = tagContent.startsWith('/');
48
51
  const isSelfClosing = tagContent.endsWith('/');
49
52
  const tagCore = tagContent.replace(/^\/|\/$/g, '').trim();
@@ -77,16 +80,20 @@ const parseHtmlTree = (html) => {
77
80
  if (!isSelfClosing && !voidElements.has(tagName)) {
78
81
  current = node;
79
82
  if (tagName === 'script' || tagName === 'style') {
80
- const closeTag = `</${tagName}>`;
81
- const closeIdx = html.toLowerCase().indexOf(closeTag, cursor);
82
- if (closeIdx !== -1) {
83
+ // Case-insensitive search from `cursor` via a sticky-ish regex, instead of
84
+ // lower-casing the whole document on every <script>/<style> (was O(n^2)).
85
+ // tagName is validated to /^[a-z0-9-]+$/ above, so it's safe to interpolate.
86
+ const closeRe = new RegExp(`</${tagName}>`, 'gi');
87
+ closeRe.lastIndex = cursor;
88
+ const closeMatch = closeRe.exec(html);
89
+ if (closeMatch) {
83
90
  node.children.push({
84
91
  type: 'text',
85
- text: html.substring(cursor, closeIdx),
92
+ text: html.substring(cursor, closeMatch.index),
86
93
  children: [],
87
94
  parent: node
88
95
  });
89
- cursor = closeIdx + closeTag.length;
96
+ cursor = closeMatch.index + closeMatch[0].length;
90
97
  current = node.parent;
91
98
  }
92
99
  }
@@ -183,7 +190,30 @@ const parseHtml = async (buffer, config) => {
183
190
  }
184
191
  const content = [];
185
192
  let htmlListIdCounter = 1;
186
- const parseNode = (node, currentFormatting = {}, listContext) => {
193
+ // Finds the checked state from a nested <input type="checkbox"> (GFM task-list items
194
+ // nest it inside a <label>, so it isn't a direct child of the <li>).
195
+ const findNestedCheckboxChecked = (n) => {
196
+ if (n.tagName === 'input' && (n.attributes?.type || '').toLowerCase() === 'checkbox') {
197
+ return 'checked' in (n.attributes || {});
198
+ }
199
+ for (const child of n.children) {
200
+ const found = findNestedCheckboxChecked(child);
201
+ if (found !== undefined)
202
+ return found;
203
+ }
204
+ return undefined;
205
+ };
206
+ // Populated from a <section data-footnotes> block (found and parsed before the main
207
+ // body loop, since references can appear anywhere earlier in the document) and
208
+ // consulted by parseChildren's <sup data-footnote-ref> handling below.
209
+ const footnoteDefinitions = new Map();
210
+ const parseNode = (node, currentFormatting = {}, listContext, depth = 0) => {
211
+ // Guard against a maliciously deep element tree (e.g. tens of thousands of
212
+ // nested <div>) recursing until the call stack overflows. Real documents
213
+ // nest only a few dozen levels; this trips well before a RangeError.
214
+ if (depth > 1000) {
215
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.MAX_NESTING_DEPTH_EXCEEDED);
216
+ }
187
217
  if (node.type === 'text') {
188
218
  let decodedText = (node.text || '')
189
219
  .replace(/&nbsp;/g, ' ')
@@ -264,7 +294,29 @@ const parseHtml = async (buffer, config) => {
264
294
  const parseChildren = (n, fmt, lCtx) => {
265
295
  const kids = [];
266
296
  for (const child of n.children) {
267
- const parsed = parseNode(child, fmt, lCtx);
297
+ // Footnote/endnote reference: attach as .notes on the preceding node
298
+ // instead of inserting a visible node, matching WordParser's convention.
299
+ if (child.type === 'element' && child.tagName === 'sup' && child.attributes?.['data-footnote-ref'] !== undefined) {
300
+ const key = child.attributes['data-footnote-ref'];
301
+ const definition = footnoteDefinitions.get(key);
302
+ const noteNode = {
303
+ type: 'note',
304
+ text: (definition || []).map(d => d.text || '').join(''),
305
+ children: definition || [],
306
+ metadata: { noteType: 'footnote', noteId: key }
307
+ };
308
+ if (kids.length > 0) {
309
+ const target = kids[kids.length - 1];
310
+ if (!target.notes)
311
+ target.notes = [];
312
+ target.notes.push(noteNode);
313
+ }
314
+ else {
315
+ kids.push({ type: 'text', text: '', notes: [noteNode] });
316
+ }
317
+ continue;
318
+ }
319
+ const parsed = parseNode(child, fmt, lCtx, depth + 1);
268
320
  if (parsed) {
269
321
  if (Array.isArray(parsed))
270
322
  kids.push(...parsed);
@@ -274,6 +326,94 @@ const parseHtml = async (buffer, config) => {
274
326
  }
275
327
  return kids;
276
328
  };
329
+ // YouTube embeds: inscript-editor's Youtube node renders
330
+ // <div data-youtube-video="ID" data-width="…" data-align="…">…<iframe…></div>.
331
+ // Recognise both the wrapper div and a bare iframe so externally-authored HTML
332
+ // (and a saved-then-reopened .md that fell back to raw HTML) both round-trip.
333
+ if (tagName === 'div' && node.attributes?.['data-youtube-video'] !== undefined) {
334
+ const videoId = node.attributes['data-youtube-video'] || '';
335
+ const width = node.attributes?.['data-width'];
336
+ const embedAlignAttr = node.attributes?.['data-align'];
337
+ const embedAlign = ['left', 'center', 'right'].includes(embedAlignAttr) ? embedAlignAttr : undefined;
338
+ const embedUrl = videoId ? `https://www.youtube.com/watch?v=${videoId}` : undefined;
339
+ const embedNode = {
340
+ type: 'embed',
341
+ // Childless nodes need .text so generic AST consumers (toText, chunking)
342
+ // don't silently drop them.
343
+ text: embedUrl,
344
+ metadata: {
345
+ embedType: 'youtube',
346
+ videoId,
347
+ url: embedUrl,
348
+ width,
349
+ align: embedAlign
350
+ }
351
+ };
352
+ if (config.includeRawContent)
353
+ embedNode.rawContent = '<div data-youtube-video>...</div>';
354
+ return embedNode;
355
+ }
356
+ if (tagName === 'iframe') {
357
+ const src = node.attributes?.src || '';
358
+ const ytMatch = /youtube(?:-nocookie)?\.com/.test(src) ? src.match(/(?:embed\/|v=)([^&?/\s]+)/) : null;
359
+ if (ytMatch) {
360
+ const embedUrl = `https://www.youtube.com/watch?v=${ytMatch[1]}`;
361
+ const embedNode = {
362
+ type: 'embed',
363
+ text: embedUrl,
364
+ metadata: { embedType: 'youtube', videoId: ytMatch[1], url: embedUrl }
365
+ };
366
+ if (config.includeRawContent)
367
+ embedNode.rawContent = '<iframe>...</iframe>';
368
+ return embedNode;
369
+ }
370
+ return null;
371
+ }
372
+ // Footnotes section: its definitions were already extracted up front (see
373
+ // footnoteDefinitions below), so skip it here wherever it appears in the tree -
374
+ // it isn't necessarily a direct child of <body> (e.g. it may be nested inside
375
+ // a non-standalone HtmlGenerator output's wrapping <div>).
376
+ if (tagName === 'section' && node.attributes?.['data-footnotes'] !== undefined) {
377
+ return null;
378
+ }
379
+ // Math: proposed contract (no editor node built yet) - HtmlGenerator emits
380
+ // <span/div class="math math-inline|math-block" data-math="inline|block">
381
+ // with the $-delimited LaTeX as the visible (escaped) text content.
382
+ if ((tagName === 'div' || tagName === 'span') && node.attributes?.['data-math'] !== undefined) {
383
+ const mathMode = node.attributes['data-math'] === 'block' ? 'block' : 'inline';
384
+ const rawText = node.children.map(c => c.text || '').join('')
385
+ .replace(/&nbsp;/g, ' ')
386
+ .replace(/&lt;/g, '<')
387
+ .replace(/&gt;/g, '>')
388
+ .replace(/&amp;/g, '&')
389
+ .replace(/&quot;/g, '"')
390
+ .replace(/&#39;/g, '\'');
391
+ const delimiter = mathMode === 'block' ? '$$' : '$';
392
+ const latex = rawText.startsWith(delimiter) && rawText.endsWith(delimiter)
393
+ ? rawText.slice(delimiter.length, -delimiter.length)
394
+ : rawText;
395
+ return {
396
+ type: 'code',
397
+ text: latex,
398
+ metadata: { math: mathMode }
399
+ };
400
+ }
401
+ // Admonition: inscript-editor's Admonition node renders
402
+ // <div class="admonition admonition-note" data-type="note">…children…</div>.
403
+ if (tagName === 'div' && (node.attributes?.class || '').split(/\s+/).includes('admonition')) {
404
+ const admonitionTypeAttr = node.attributes?.['data-type'];
405
+ const admonitionType = ['note', 'tip', 'important', 'warning', 'caution'].includes(admonitionTypeAttr)
406
+ ? admonitionTypeAttr
407
+ : 'note';
408
+ const admonitionNode = {
409
+ type: 'admonition',
410
+ metadata: { admonitionType },
411
+ children: parseChildren(node, newFormatting, listContext)
412
+ };
413
+ if (config.includeRawContent)
414
+ admonitionNode.rawContent = '<div class="admonition">...</div>';
415
+ return admonitionNode;
416
+ }
277
417
  // Skip structural containers produced by HtmlGenerator to avoid deep AST nesting
278
418
  if (tagName === 'div' && (node.attributes?.class === 'container' ||
279
419
  node.attributes?.class === 'spreadsheet-container' ||
@@ -296,7 +436,7 @@ const parseHtml = async (buffer, config) => {
296
436
  if (tagName === 'p' || tagName === 'div') {
297
437
  const children = parseChildren(node, newFormatting, listContext);
298
438
  // If it's a div and contains block elements, return children directly
299
- const hasBlockElements = children.some(c => ['paragraph', 'table', 'heading', 'list', 'image', 'chart', 'code'].includes(c.type));
439
+ const hasBlockElements = children.some(c => ['paragraph', 'table', 'heading', 'list', 'image', 'chart', 'code', 'embed', 'admonition', 'definitionList'].includes(c.type));
300
440
  if (tagName === 'div' && hasBlockElements) {
301
441
  return children;
302
442
  }
@@ -330,13 +470,53 @@ const parseHtml = async (buffer, config) => {
330
470
  };
331
471
  return hNode;
332
472
  }
473
+ if (tagName === 'dl') {
474
+ return {
475
+ type: 'definitionList',
476
+ children: parseChildren(node, newFormatting, listContext)
477
+ };
478
+ }
479
+ if (tagName === 'dt') {
480
+ return {
481
+ type: 'definitionTerm',
482
+ children: parseChildren(node, newFormatting, listContext)
483
+ };
484
+ }
485
+ if (tagName === 'dd') {
486
+ return {
487
+ type: 'definitionDescription',
488
+ children: parseChildren(node, newFormatting, listContext)
489
+ };
490
+ }
491
+ if (tagName === 'abbr') {
492
+ const title = node.attributes?.title;
493
+ const children = parseChildren(node, newFormatting, listContext);
494
+ if (title) {
495
+ children.forEach(c => {
496
+ if (c.type === 'text') {
497
+ c.metadata = { ...c.metadata, abbreviationTitle: title };
498
+ }
499
+ });
500
+ }
501
+ return children;
502
+ }
503
+ if (tagName === 'cite' && node.attributes?.['data-citation-key'] !== undefined) {
504
+ const citationKey = node.attributes['data-citation-key'];
505
+ return {
506
+ type: 'text',
507
+ text: citationKey,
508
+ formatting: Object.keys(newFormatting).length > 0 ? { ...newFormatting } : undefined,
509
+ metadata: { citationKey }
510
+ };
511
+ }
333
512
  if (tagName === 'ul' || tagName === 'ol') {
334
513
  const isNewTopLevel = !listContext;
335
514
  const newListContext = {
336
515
  listId: isNewTopLevel ? `html-list-${htmlListIdCounter++}` : listContext.listId,
337
516
  type: tagName === 'ol' ? 'ordered' : 'unordered',
338
517
  level: isNewTopLevel ? 0 : listContext.level + 1,
339
- counters: isNewTopLevel ? {} : { ...listContext.counters } // Clone to avoid side effects on parent levels
518
+ counters: isNewTopLevel ? {} : { ...listContext.counters }, // Clone to avoid side effects on parent levels
519
+ isTask: node.attributes?.['data-type'] === 'taskList'
340
520
  };
341
521
  // Initialize counter for this level
342
522
  if (tagName === 'ol' && node.attributes?.start) {
@@ -362,6 +542,13 @@ const parseHtml = async (buffer, config) => {
362
542
  const children = parseChildren(node, newFormatting, listContext);
363
543
  const nestedLists = children.filter(c => c.type === 'list');
364
544
  const selfChildren = children.filter(c => c.type !== 'list');
545
+ let isTask;
546
+ let checked;
547
+ if (listContext?.isTask) {
548
+ isTask = true;
549
+ const dataChecked = node.attributes?.['data-checked'];
550
+ checked = dataChecked !== undefined ? dataChecked === 'true' : (findNestedCheckboxChecked(node) ?? false);
551
+ }
365
552
  const selfNode = {
366
553
  type: 'list',
367
554
  text: selfChildren.map(c => c.text || '').join(''),
@@ -371,16 +558,21 @@ const parseHtml = async (buffer, config) => {
371
558
  alignment: newFormatting.alignment || 'left',
372
559
  listId: listContext?.listId || 'html-list-none',
373
560
  itemIndex: (listContext?.counters[listContext.level] ?? 1) - 1,
374
- anchorIds: anchorIds.length > 0 ? anchorIds : undefined
561
+ anchorIds: anchorIds.length > 0 ? anchorIds : undefined,
562
+ isTask,
563
+ checked
375
564
  },
376
565
  children: selfChildren
377
566
  };
378
567
  return [selfNode, ...nestedLists];
379
568
  }
380
569
  if (tagName === 'table') {
570
+ // CustomTable (inscript-editor) renders data-align on the <table> itself.
571
+ const tableAlignAttr = node.attributes?.['data-align'];
572
+ const tableAlign = ['left', 'center', 'right'].includes(tableAlignAttr) ? tableAlignAttr : undefined;
381
573
  const tableNode = {
382
574
  type: 'table',
383
- metadata: { anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
575
+ metadata: { anchorIds: anchorIds.length > 0 ? anchorIds : undefined, align: tableAlign },
384
576
  children: parseChildren(node, newFormatting, listContext)
385
577
  };
386
578
  if (config.includeRawContent) {
@@ -399,8 +591,18 @@ const parseHtml = async (buffer, config) => {
399
591
  return rowNode;
400
592
  }
401
593
  if (tagName === 'td' || tagName === 'th') {
594
+ // Merged cells: mirrors the colspan/rowspan reading already done in
595
+ // MarkdownParser's inline HTML-table handler.
596
+ const colSpanAttr = node.attributes?.colspan;
597
+ const rowSpanAttr = node.attributes?.rowspan;
598
+ const colSpan = colSpanAttr ? parseInt(colSpanAttr, 10) : undefined;
599
+ const rowSpan = rowSpanAttr ? parseInt(rowSpanAttr, 10) : undefined;
402
600
  const cellNode = {
403
601
  type: 'cell',
602
+ metadata: {
603
+ colSpan: colSpan && !isNaN(colSpan) ? colSpan : undefined,
604
+ rowSpan: rowSpan && !isNaN(rowSpan) ? rowSpan : undefined
605
+ },
404
606
  children: parseChildren(node, newFormatting, listContext)
405
607
  };
406
608
  if (config.includeRawContent) {
@@ -411,6 +613,14 @@ const parseHtml = async (buffer, config) => {
411
613
  if (tagName === 'img') {
412
614
  const src = node.attributes?.src;
413
615
  const alt = node.attributes?.alt;
616
+ // CustomImage (inscript-editor) renders data-width/data-align, falling back to
617
+ // parsing the inline style for consumers that only emit the CSS.
618
+ const imgStyle = node.attributes?.style || '';
619
+ const width = node.attributes?.['data-width'] || imgStyle.match(/width:\s*([^;]+)/)?.[1]?.trim();
620
+ const alignAttr = node.attributes?.['data-align']
621
+ || (imgStyle.includes('margin-left: 0') && !imgStyle.includes('margin-right: 0') ? 'left'
622
+ : (imgStyle.includes('margin-right: 0') && !imgStyle.includes('margin-left: 0') ? 'right' : undefined));
623
+ const align = ['left', 'center', 'right'].includes(alignAttr) ? alignAttr : undefined;
414
624
  let imageNode;
415
625
  if (src?.startsWith('data:')) {
416
626
  const match = src.match(/^data:([^;]+);base64,(.*)$/);
@@ -429,7 +639,9 @@ const parseHtml = async (buffer, config) => {
429
639
  type: 'image',
430
640
  metadata: {
431
641
  attachmentName: name,
432
- altText: alt
642
+ altText: alt,
643
+ width,
644
+ align
433
645
  }
434
646
  };
435
647
  }
@@ -438,7 +650,9 @@ const parseHtml = async (buffer, config) => {
438
650
  type: 'image',
439
651
  metadata: {
440
652
  url: src,
441
- altText: alt
653
+ altText: alt,
654
+ width,
655
+ align
442
656
  }
443
657
  };
444
658
  }
@@ -449,7 +663,9 @@ const parseHtml = async (buffer, config) => {
449
663
  metadata: {
450
664
  url: src,
451
665
  altText: alt,
452
- anchorIds: anchorIds.length > 0 ? anchorIds : undefined
666
+ anchorIds: anchorIds.length > 0 ? anchorIds : undefined,
667
+ width,
668
+ align
453
669
  }
454
670
  };
455
671
  }
@@ -460,8 +676,16 @@ const parseHtml = async (buffer, config) => {
460
676
  }
461
677
  if (tagName === 'a') {
462
678
  const href = node.attributes?.href;
679
+ const wikilinkPage = node.attributes?.['data-wikilink-page'];
463
680
  const children = parseChildren(node, newFormatting, listContext);
464
- if (href) {
681
+ if (wikilinkPage !== undefined) {
682
+ children.forEach(c => {
683
+ if (c.type === 'text') {
684
+ c.metadata = { ...c.metadata, link: wikilinkPage, linkType: 'internal', wikilink: true };
685
+ }
686
+ });
687
+ }
688
+ else if (href) {
465
689
  const linkType = href.startsWith('#') ? 'internal' : 'external';
466
690
  children.forEach(c => {
467
691
  if (c.type === 'text') {
@@ -509,6 +733,42 @@ const parseHtml = async (buffer, config) => {
509
733
  }
510
734
  return null;
511
735
  };
736
+ // Extract <section data-footnotes> up front so its definitions are available to
737
+ // <sup data-footnote-ref> references encountered anywhere earlier in the body.
738
+ const findFootnotesSection = (n) => {
739
+ if (n.tagName === 'section' && n.attributes?.['data-footnotes'] !== undefined)
740
+ return n;
741
+ for (const child of n.children) {
742
+ const found = findFootnotesSection(child);
743
+ if (found)
744
+ return found;
745
+ }
746
+ return undefined;
747
+ };
748
+ const footnotesSectionNode = findFootnotesSection(body);
749
+ if (footnotesSectionNode) {
750
+ for (const item of footnotesSectionNode.children) {
751
+ if (item.type !== 'element')
752
+ continue;
753
+ const key = item.attributes?.['data-footnote-id'];
754
+ if (!key)
755
+ continue;
756
+ // Strip the generated back-reference link ("↩") - it's round-trip plumbing,
757
+ // not part of the footnote's actual content.
758
+ const filteredChildren = item.children.filter(c => !(c.tagName === 'a' && (c.attributes?.href || '').startsWith('#footnote-ref-')));
759
+ const contentNodes = [];
760
+ for (const child of filteredChildren) {
761
+ const parsed = parseNode(child);
762
+ if (parsed) {
763
+ if (Array.isArray(parsed))
764
+ contentNodes.push(...parsed);
765
+ else
766
+ contentNodes.push(parsed);
767
+ }
768
+ }
769
+ footnoteDefinitions.set(key, contentNodes);
770
+ }
771
+ }
512
772
  for (const child of body.children) {
513
773
  const parsed = parseNode(child);
514
774
  if (parsed) {
@@ -539,8 +799,12 @@ const parseHtml = async (buffer, config) => {
539
799
  return node.text || '';
540
800
  if (node.type === 'break')
541
801
  return '\n';
802
+ // Childless nodes still carry meaningful text - fall back to it instead of
803
+ // silently vanishing from plain-text/RAG-chunk output.
804
+ if (node.type === 'embed')
805
+ return node.metadata?.url || '';
542
806
  if (node.children) {
543
- const isBlock = ['table', 'row', 'list', 'sheet', 'slide'].includes(node.type);
807
+ const isBlock = ['table', 'row', 'list', 'sheet', 'slide', 'admonition', 'definitionList'].includes(node.type);
544
808
  return node.children.map(getText).join(isBlock ? config.newlineDelimiter : '');
545
809
  }
546
810
  return '';