officeparser 7.5.1 → 7.6.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -68,7 +68,10 @@ function resolveDialect(dialect) {
68
68
  */
69
69
  function resolveFallbackToHtml(fallbackToHtml) {
70
70
  const uniform = (on) => ({
71
+ // inlineFormatting is opt-in only: it is never enabled by the boolean form, since it changes
72
+ // default output. Every other field follows the boolean.
71
73
  textFormatting: on, alignment: on, anchors: on, tables: on, embeds: on, cellLineBreaks: on,
74
+ inlineFormatting: false,
72
75
  });
73
76
  if (fallbackToHtml === undefined || typeof fallbackToHtml === 'boolean')
74
77
  return uniform(fallbackToHtml ?? true);
@@ -80,6 +83,7 @@ function resolveFallbackToHtml(fallbackToHtml) {
80
83
  tables: fallbackToHtml.tables ?? on.tables,
81
84
  embeds: fallbackToHtml.embeds ?? on.embeds,
82
85
  cellLineBreaks: fallbackToHtml.cellLineBreaks ?? on.cellLineBreaks,
86
+ inlineFormatting: fallbackToHtml.inlineFormatting ?? on.inlineFormatting,
83
87
  };
84
88
  }
85
89
  /**
@@ -203,35 +207,40 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
203
207
  // Add Metadata (YAML Front Matter)
204
208
  const meta = this.effectiveMetadata;
205
209
  if (meta) {
206
- output += '---\n';
207
- // JSON-encode scalar values so a title/author/description containing a
208
- // quote or newline can't break out of the YAML string and inject
209
- // arbitrary front-matter keys. (JSON.stringify of a benign value yields
210
- // the same `"..."` form as before, so normal output is unchanged.)
210
+ // Build the field lines first. JSON-encode scalar values so a title/author/description
211
+ // containing a quote or newline can't break out of the YAML string and inject arbitrary
212
+ // front-matter keys. (JSON.stringify of a benign value yields the same `"..."` form as
213
+ // before, so normal output is unchanged.)
214
+ let fields = '';
211
215
  if (meta.title)
212
- output += `title: ${JSON.stringify(meta.title)}\n`;
216
+ fields += `title: ${JSON.stringify(meta.title)}\n`;
213
217
  if (meta.author)
214
- output += `author: ${JSON.stringify(meta.author)}\n`;
218
+ fields += `author: ${JSON.stringify(meta.author)}\n`;
215
219
  const createdIso = this.toIsoDate(meta.created);
216
220
  if (createdIso)
217
- output += `created: ${createdIso}\n`;
221
+ fields += `created: ${createdIso}\n`;
218
222
  const modifiedIso = this.toIsoDate(meta.modified);
219
223
  if (modifiedIso)
220
- output += `modified: ${modifiedIso}\n`;
224
+ fields += `modified: ${modifiedIso}\n`;
221
225
  if (meta.description)
222
- output += `description: ${JSON.stringify(meta.description)}\n`;
226
+ fields += `description: ${JSON.stringify(meta.description)}\n`;
223
227
  if (meta.subject)
224
- output += `subject: ${JSON.stringify(meta.subject)}\n`;
228
+ fields += `subject: ${JSON.stringify(meta.subject)}\n`;
225
229
  if (meta.keywords)
226
- output += `keywords: ${JSON.stringify(meta.keywords)}\n`;
230
+ fields += `keywords: ${JSON.stringify(meta.keywords)}\n`;
227
231
  if (meta.customProperties) {
228
232
  for (const [key, val] of Object.entries(meta.customProperties)) {
229
233
  // Strip newlines/colons from the key so it can't inject a new mapping.
230
234
  const safeKey = String(key).replace(/[\r\n:]+/g, ' ').trim();
231
- output += `${safeKey}: ${Array.isArray(val) ? this.serializeFrontmatterArray(val) : JSON.stringify(val)}\n`;
235
+ fields += `${safeKey}: ${Array.isArray(val) ? this.serializeFrontmatterArray(val) : JSON.stringify(val)}\n`;
232
236
  }
233
237
  }
234
- output += '---\n\n';
238
+ // Only emit the frontmatter fence when at least one field is present. Empty metadata
239
+ // (a bare Tiptap/HTML fragment with no <head>) would otherwise emit `---\n---`, which
240
+ // reparses as a setext `## ---` heading and corrupts the document on every save/reload.
241
+ if (fields) {
242
+ output += `---\n${fields}---\n\n`;
243
+ }
235
244
  }
236
245
  const processor = async (node, childrenOutput) => {
237
246
  // Handle Style Mapping for Markdown using the semantic mapping helper
@@ -256,6 +265,19 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
256
265
  // HTML tag (e.g. <script>) when the Markdown is rendered to HTML.
257
266
  let text = (0, sanitize_js_1.markdownEscapeText)(node.text || '');
258
267
  if (this.config.includeFormatting && node.formatting) {
268
+ // Inline code: re-wrap the RAW text in backticks. The content is literal
269
+ // inside a code span, so the entity-escaped form above must not show through.
270
+ // The fence is one backtick longer than the longest embedded run so an inner
271
+ // backtick can't close the span early, padded when the content touches a
272
+ // backtick. Done before emphasis so bold/italic wrap the span (`**`code`**`).
273
+ // Previously a monospace text node emitted its bare text, dropping the code.
274
+ if (node.formatting.font === 'monospace') {
275
+ const raw = node.text || '';
276
+ const longestRun = Math.max(0, ...(raw.match(/`+/g) || []).map(s => s.length));
277
+ const fence = '`'.repeat(longestRun + 1);
278
+ const pad = (raw.startsWith('`') || raw.endsWith('`')) ? ' ' : '';
279
+ text = `${fence}${pad}${raw}${pad}${fence}`;
280
+ }
259
281
  const emphasisAsterisk = this.resolvedDialect.emphasisMarker === 'asterisk';
260
282
  if (node.formatting.bold && !this.inImplicitBold)
261
283
  text = emphasisAsterisk ? `**${text}**` : `__${text}__`;
@@ -272,6 +294,24 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
272
294
  if (node.formatting.superscript)
273
295
  text = `<sup>${text}</sup>`;
274
296
  }
297
+ // Inline color / highlight / font size have no Markdown syntax; emit a styled
298
+ // <span> (outermost, so the inner Markdown markers survive) only when opted in,
299
+ // so default output is unchanged. Values are CSS-sanitized against injection.
300
+ if (this.resolvedFallbackToHtml.inlineFormatting) {
301
+ const styles = [];
302
+ const pushStyle = (prop, val) => {
303
+ if (!val)
304
+ return;
305
+ const safe = (0, sanitize_js_1.sanitizeCssValue)(val); // drops url()/expression()/<>/quotes
306
+ if (safe)
307
+ styles.push(`${prop}: ${safe}`);
308
+ };
309
+ pushStyle('color', node.formatting.color);
310
+ pushStyle('background-color', node.formatting.backgroundColor);
311
+ pushStyle('font-size', node.formatting.size);
312
+ if (styles.length)
313
+ text = `<span style="${styles.join('; ')}">${text}</span>`;
314
+ }
275
315
  }
276
316
  const meta = node.metadata;
277
317
  if (meta?.wikilink && this.resolvedDialect.wikilinks) {
@@ -411,13 +451,17 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
411
451
  return childrenOutput;
412
452
  }
413
453
  case 'break': {
414
- // A hard line break (CommonMark: two trailing spaces before the
415
- // newline) round-trips back to a distinct 'break' node on reparse;
416
- // every other breakType (including 'page', used for a thematic-break
417
- // HR) keeps emitting a bare newline, unchanged.
454
+ // A hard line break (CommonMark: two trailing spaces before the newline)
455
+ // round-trips back to a distinct 'break' node on reparse. A thematic break
456
+ // emits `---` as its own block (the top-level loop supplies the surrounding
457
+ // blank lines), so a Markdown `---` / HTML `<hr>` survives a save instead of
458
+ // collapsing to whitespace. Every other breakType - notably 'page', which
459
+ // Markdown has no syntax for - keeps emitting a bare newline, unchanged.
418
460
  const meta = node.metadata;
419
461
  if (meta?.breakType === 'carriageReturn')
420
462
  return ' \n';
463
+ if (meta?.breakType === 'thematic')
464
+ return '---';
421
465
  return '\n';
422
466
  }
423
467
  case 'code': {
@@ -447,17 +491,20 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
447
491
  return this.resolvedDialect.math === 'dollar' ? `$${mathInline}$` : mathInline;
448
492
  }
449
493
  const lang = (meta?.language || '').replace(/[\r\n`]+/g, '');
450
- // Block code if it contains a line break, else inline. Testing only for `\n`
451
- // routed a CR-only string to the inline branch, where a renderer that
452
- // normalizes `\r` to a line ending sees a blank line, the span dies, and the
453
- // remainder is exposed as raw Markdown. The fence sizing below is correct and
454
- // needs no change; code content itself is not an HTML context.
455
- if (node.text && /[\r\n]/.test(node.text)) {
494
+ // A `code` node is always block-level: genuinely inline code is a monospace
495
+ // text node, never a `code` node. So emit a fenced block whenever the node
496
+ // carries a language OR spans multiple lines. Previously the decision keyed only
497
+ // off a line break, so a single-line code node with a language - `const x = 1;`
498
+ // tagged `js`, or a one-line `mermaid` diagram - collapsed to an inline span,
499
+ // silently dropping both its language and its block-ness. (Testing `[\r\n]`, not
500
+ // just `\n`, still routes a CR-only body to the fenced branch, where a renderer
501
+ // that normalizes `\r` to a line ending would otherwise kill an inline span.)
502
+ if (lang || (node.text && /[\r\n]/.test(node.text))) {
456
503
  // Fence with one more backtick than the longest run inside the content
457
504
  // so an embedded ``` can't close the block early and inject markup.
458
- const longestRun = Math.max(0, ...(node.text.match(/`+/g) || []).map(s => s.length));
505
+ const longestRun = Math.max(0, ...((node.text || '').match(/`+/g) || []).map(s => s.length));
459
506
  const fence = '`'.repeat(Math.max(3, longestRun + 1));
460
- return `\n${fence}${lang}\n${node.text}\n${fence}\n\n`;
507
+ return `\n${fence}${lang}\n${node.text || ''}\n${fence}\n\n`;
461
508
  }
462
509
  else {
463
510
  const t = node.text || '';
@@ -489,7 +536,10 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
489
536
  // into an end-of-document "### Notes" section under a [^id] marker.
490
537
  return childrenOutput.trim();
491
538
  }
492
- return `[^${this.getFootnoteKey(node)}]: ${childrenOutput.trim()}\n\n`;
539
+ // Indent continuation lines one level so a multi-line body re-parses as a
540
+ // single definition (a bare newline would end it). Single-line bodies, the
541
+ // common case, are unaffected.
542
+ return `[^${this.getFootnoteKey(node)}]: ${childrenOutput.trim().replace(/\n/g, '\n ')}\n\n`;
493
543
  }
494
544
  return `> **Note:** ${childrenOutput.trim()}\n\n`;
495
545
  }
@@ -498,6 +548,20 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
498
548
  // save default), emit the exact single-line div MarkdownParser recognises on
499
549
  // reimport; otherwise degrade to a plain link.
500
550
  const meta = node.metadata;
551
+ if (meta?.embedType === 'iframe') {
552
+ // sanitizeUrl scheme-checks and HTML-escapes the src (hostile schemes drop
553
+ // the node). The single-line <iframe> is what MarkdownParser recognises on
554
+ // reimport, gated there on preserveIframes.
555
+ const safe = (0, sanitize_js_1.sanitizeUrl)(meta?.url || '');
556
+ if (!safe)
557
+ return '';
558
+ if (this.resolvedFallbackToHtml.embeds) {
559
+ const w = meta?.width ? ` width="${(0, sanitize_js_1.escapeHtml)(meta.width)}"` : '';
560
+ const h = meta?.height ? ` height="${(0, sanitize_js_1.escapeHtml)(meta.height)}"` : '';
561
+ return `\n<iframe src="${safe}"${w}${h}></iframe>\n\n`;
562
+ }
563
+ return `[Embed](${(0, sanitize_js_1.sanitizeMarkdownUrl)(meta?.url || '')})\n\n`;
564
+ }
501
565
  const id = meta?.videoId || '';
502
566
  if (this.resolvedFallbackToHtml.embeds) {
503
567
  const width = meta?.width ? ` data-width="${(0, sanitize_js_1.escapeHtml)(meta.width)}"` : '';
@@ -569,6 +633,14 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
569
633
  for (let i = 0; i < optimizedContent.length; i++) {
570
634
  const node = optimizedContent[i];
571
635
  const nextNode = optimizedContent[i + 1];
636
+ // A top-level footnote/endnote note is an orphan definition (unreferenced `[^id]: ...`
637
+ // the MarkdownParser recovered). Collect it so it's emitted with the other definitions
638
+ // at the document end rather than inline before them.
639
+ const orphanNoteType = node.metadata?.noteType;
640
+ if (node.type === 'note' && (orphanNoteType === 'footnote' || orphanNoteType === 'endnote')) {
641
+ this.collectedNotes.push(node);
642
+ continue;
643
+ }
572
644
  let result = await this.processNodeRecursive(node, processor);
573
645
  // Ensure lists and other block elements are separated from non-similar content by a blank line
574
646
  if (nextNode) {
@@ -585,8 +657,21 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
585
657
  output += result;
586
658
  }
587
659
  if (this.collectedNotes.length > 0) {
588
- let notesMd = '\n\n---\n\n### Notes\n\n';
589
- for (const note of this.collectedNotes) {
660
+ // No decorative `---\n\n### Notes` preamble: `[^id]:` definitions are valid on their own
661
+ // (GitHub/Pandoc render the footnotes section and its rule automatically), and the
662
+ // literal heading round-tripped as a real `###` node - so every save/reload re-emitted
663
+ // the parsed heading AND a fresh one, growing the document unbounded. Emitting the bare
664
+ // definitions makes the cycle byte-stable. Behaviour change, noted in the changelog.
665
+ // Collapse the preceding block's trailing blank lines so exactly one blank line separates
666
+ // the body from the definitions (rather than the doubled `\n\n\n\n` the concatenation
667
+ // would otherwise leave).
668
+ output = output.replace(/\n+$/, '');
669
+ let notesMd = '\n\n';
670
+ // De-duplicate by node identity: a footnote referenced more than once shares a single
671
+ // note object (see MarkdownParser), pushed here once per reference. Emit its definition
672
+ // just once. Distinct notes - even two office notes that happen to share a numeric id -
673
+ // are separate objects and are all kept.
674
+ for (const note of [...new Set(this.collectedNotes)]) {
590
675
  notesMd += await this.processNodeRecursive(note, processor);
591
676
  }
592
677
  output += notesMd;
@@ -689,6 +774,12 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
689
774
  let current = null;
690
775
  for (const node of nodes) {
691
776
  if (node.type === 'text' && current && current.type === 'text' &&
777
+ // A note anchors to the end of its text run and its `[^id]` marker is emitted there;
778
+ // merging a following run onto a note-carrying run would slide the marker past it
779
+ // (`Body[^1].` -> `Body.[^1]`). Keep such runs separate so the marker stays put and
780
+ // matches where HtmlGenerator emits it.
781
+ (!current.notes || current.notes.length === 0) &&
782
+ (!node.notes || node.notes.length === 0) &&
692
783
  this.areFormattingEqual(node.formatting, current.formatting) &&
693
784
  JSON.stringify(node.metadata) === JSON.stringify(current.metadata)) {
694
785
  current.text = (current.text || '') + (node.text || '');
@@ -22,7 +22,10 @@ class PdfGenerator extends BaseGenerator_js_1.BaseGenerator {
22
22
  // We reuse the current configuration but ensure standalone mode is on for HTML
23
23
  const htmlGenerator = new HtmlGenerator_js_1.HtmlGenerator(this.ast, {
24
24
  ...this.config,
25
- htmlConfig: { ...this.config.htmlConfig, standalone: true },
25
+ // Force sourceAttributes off: those data-* attributes are wire-format plumbing for
26
+ // structured consumers and change the mermaid shape's rendered appearance, neither of
27
+ // which belongs in a printed PDF.
28
+ htmlConfig: { ...this.config.htmlConfig, standalone: true, sourceAttributes: false },
26
29
  });
27
30
  const htmlResult = await htmlGenerator.generate();
28
31
  const html = typeof htmlResult.value === 'string' ? htmlResult.value : '';
package/dist/index.d.ts CHANGED
@@ -51,10 +51,10 @@
51
51
  import { OfficeParser } from './OfficeParser.js';
52
52
  import { OfficeGenerator } from './OfficeGenerator.js';
53
53
  import { OfficeConverter } from './OfficeConverter.js';
54
- import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverterConfig, OfficeErrorType, OfficeWarningType, OfficeError, ConversionResult } from './types.js';
54
+ import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, BlobLike, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverterConfig, OfficeErrorType, OfficeWarningType, OfficeError, ConversionResult, HtmlGeneratorConfig, HtmlParserConfig, MdGeneratorConfig, CsvGeneratorConfig, PdfGeneratorConfig, RtfGeneratorConfig, TextGeneratorConfig, MarkdownDialectConfig, MarkdownDialectPreset, StandaloneConfig, MetadataOverrides, FallbackToHtmlConfig, HtmlInjectionConfig, OcrConfig, DecompressionLimits, TextMetadata, TableMetadata, CodeMetadata, NoteMetadata, AdmonitionMetadata, EmbedMetadata, ChartMetadata, CommentMetadata, HeaderFooterMetadata, IndentationMetadata, ParagraphMetadata } from './types.js';
55
55
  declare const parseOffice: typeof OfficeParser.parseOffice;
56
56
  declare const terminateOcr: typeof OfficeParser.terminateOcr;
57
57
  declare const convert: typeof OfficeConverter.convert;
58
58
  declare const generate: typeof OfficeGenerator.generate;
59
- export { OfficeParser, parseOffice, terminateOcr, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, OfficeGenerator, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverter, OfficeConverterConfig, convert, generate, OfficeErrorType, OfficeWarningType, OfficeError, ConversionResult, };
59
+ export { OfficeParser, parseOffice, terminateOcr, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, BlobLike, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, OfficeGenerator, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverter, OfficeConverterConfig, convert, generate, OfficeErrorType, OfficeWarningType, OfficeError, ConversionResult, HtmlGeneratorConfig, HtmlParserConfig, MdGeneratorConfig, CsvGeneratorConfig, PdfGeneratorConfig, RtfGeneratorConfig, TextGeneratorConfig, MarkdownDialectConfig, MarkdownDialectPreset, StandaloneConfig, MetadataOverrides, FallbackToHtmlConfig, HtmlInjectionConfig, OcrConfig, DecompressionLimits, TextMetadata, TableMetadata, CodeMetadata, NoteMetadata, AdmonitionMetadata, EmbedMetadata, ChartMetadata, CommentMetadata, HeaderFooterMetadata, IndentationMetadata, ParagraphMetadata, };
60
60
  export default OfficeParser;
@@ -403,6 +403,18 @@ export interface HtmlParserConfig {
403
403
  * Defaults to false.
404
404
  */
405
405
  preserveAttributes?: boolean;
406
+ /**
407
+ * Preserve `<iframe>` embeds that aren't recognized as a known provider (YouTube is always
408
+ * recognized). By default every non-YouTube iframe is dropped, which is a deliberate security
409
+ * posture other consumers rely on; set this to opt back in. `true` preserves any iframe; an
410
+ * array is a hostname allowlist (an entry matches the src's host exactly or as a `.`-suffix,
411
+ * so `"vimeo.com"` also matches `player.vimeo.com`). Preserved iframes become `embed` nodes
412
+ * with `embedType: 'iframe'`; on generation the `src` is still scheme-checked (only http/https
413
+ * survive). This also governs a raw `<iframe>` block encountered in Markdown input.
414
+ *
415
+ * Defaults to false.
416
+ */
417
+ preserveIframes?: boolean | string[];
406
418
  }
407
419
  /**
408
420
  * Maps an input format string to its corresponding format-specific parser configuration, mirroring
@@ -859,6 +871,18 @@ export interface HtmlGeneratorConfig {
859
871
  * Granular injection points for custom HTML, scripts, and styles.
860
872
  */
861
873
  injections?: HtmlInjectionConfig;
874
+ /**
875
+ * Carry each rich node's raw source in a `data-*` attribute, with undelimited text content,
876
+ * so attribute-driven structured consumers (rich-text editors, custom viewers) can rehydrate
877
+ * the node from the markup rather than re-parsing the display text. Affects wikilinks
878
+ * (adds `data-wikilink`/`data-target`/`data-alias`), citations (a `<span class="citation">`
879
+ * carrying `data-key` instead of `<cite>`), math (the LaTeX in `data-math`, undelimited) and
880
+ * mermaid (a `<div class="mermaid" data-mermaid>` instead of `<pre><code>`).
881
+ *
882
+ * Off by default; the default output is byte-identical to previous releases. The widened
883
+ * `HtmlParser` reads every shape this emits, so output stays self-round-trippable.
884
+ */
885
+ sourceAttributes?: boolean;
862
886
  }
863
887
  /**
864
888
  * Configuration options for PDF generation.
@@ -1074,6 +1098,14 @@ export interface FallbackToHtmlConfig {
1074
1098
  embeds?: boolean;
1075
1099
  /** Multi-line table cell content joined with `<br>` instead of a space. */
1076
1100
  cellLineBreaks?: boolean;
1101
+ /**
1102
+ * Inline text color, highlight, and font size via a `<span style="color:...;background-color:...;
1103
+ * font-size:...">` run, which the Markdown parser reads back. These have no Markdown syntax and
1104
+ * are silently lost otherwise. Unlike the other fields this is **off by default even when
1105
+ * `fallbackToHtml` is `true`**, because it changes default output; enable it explicitly with
1106
+ * `fallbackToHtml: { inlineFormatting: true }`.
1107
+ */
1108
+ inlineFormatting?: boolean;
1077
1109
  }
1078
1110
  /**
1079
1111
  * Configuration options for Markdown generation.
@@ -1345,6 +1377,16 @@ export interface OfficeChunk {
1345
1377
  * Supported file types for parsing.
1346
1378
  */
1347
1379
  export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods" | "pdf" | "rtf" | "md" | "html" | "csv" | "epub";
1380
+ /**
1381
+ * A structural stand-in for the web `Blob`/`File` so `parseOffice`/`convert` accept them in the
1382
+ * browser without pulling the DOM lib into this package's types. Any object with an
1383
+ * `arrayBuffer()` method qualifies. When `name` is present (as on a `File`) it is used only for
1384
+ * extension-based type detection, never as a filesystem path.
1385
+ */
1386
+ export interface BlobLike {
1387
+ arrayBuffer(): Promise<ArrayBuffer>;
1388
+ name?: string;
1389
+ }
1348
1390
  /**
1349
1391
  * Types of content nodes in the AST.
1350
1392
  */
@@ -1586,7 +1628,7 @@ export interface TableMetadata {
1586
1628
  /** Unique anchor IDs for internal linking. */
1587
1629
  anchorIds?: string[];
1588
1630
  /**
1589
- * Layout alignment of the table on the page (e.g. inscript-editor's `CustomTable`).
1631
+ * Layout alignment of the table on the page (e.g. an editor's custom table node).
1590
1632
  * @example 'center'
1591
1633
  */
1592
1634
  align?: "left" | "center" | "right";
@@ -1631,12 +1673,12 @@ export interface ImageMetadata {
1631
1673
  /** Unique anchor IDs for internal linking. */
1632
1674
  anchorIds?: string[];
1633
1675
  /**
1634
- * Display width of the image (e.g. inscript-editor's `CustomImage`), as a CSS length or percentage.
1676
+ * Display width of the image (e.g. an editor's custom image node), as a CSS length or percentage.
1635
1677
  * @example "50%"
1636
1678
  */
1637
1679
  width?: string;
1638
1680
  /**
1639
- * Layout alignment of the image (e.g. inscript-editor's `CustomImage`).
1681
+ * Layout alignment of the image (e.g. an editor's custom image node).
1640
1682
  * @example 'center'
1641
1683
  */
1642
1684
  align?: "left" | "center" | "right";
@@ -1646,14 +1688,19 @@ export interface ImageMetadata {
1646
1688
  * Markdown has no native syntax for this - see `MarkdownGenerator`'s `embed` case.
1647
1689
  */
1648
1690
  export interface EmbedMetadata {
1649
- /** The kind of embed. Only 'youtube' is supported today; the shape is generic for future providers. */
1650
- embedType: "youtube";
1651
- /** The provider-specific video ID (e.g. the 11-character YouTube video ID). */
1652
- videoId: string;
1653
- /** The original/canonical URL of the embedded media, if known. */
1691
+ /**
1692
+ * The kind of embed. 'youtube' is recognized from a `data-youtube-video` wrapper or a YouTube
1693
+ * iframe; 'iframe' is a generic preserved iframe (opt-in via `HtmlParserConfig.preserveIframes`).
1694
+ */
1695
+ embedType: "youtube" | "iframe";
1696
+ /** The provider-specific video ID (e.g. the 11-character YouTube video ID). Absent for generic iframes. */
1697
+ videoId?: string;
1698
+ /** The original/canonical URL of the embedded media, if known. For a generic iframe, its `src`. */
1654
1699
  url?: string;
1655
1700
  /** Display width, as a CSS length or percentage. */
1656
1701
  width?: string;
1702
+ /** Display height, as a CSS length or percentage (generic iframes). */
1703
+ height?: string;
1657
1704
  /** Layout alignment of the embed. */
1658
1705
  align?: "left" | "center" | "right";
1659
1706
  }
@@ -1737,6 +1784,14 @@ export interface NoteMetadata {
1737
1784
  anchorIds?: string[];
1738
1785
  /** The slide number this note is associated with (used in PowerPoint). */
1739
1786
  slideNumber?: number;
1787
+ /**
1788
+ * True for a footnote/endnote definition that no reference points at (an "orphan").
1789
+ * The Markdown parser sets this when it recovers a `[^id]: ...` definition with no matching
1790
+ * `[^id]` reference so the definition is preserved rather than dropped. Generators route such
1791
+ * notes into their footnotes section without a citation marker, and the HTML generator omits
1792
+ * the (otherwise dangling) back-link.
1793
+ */
1794
+ unreferenced?: boolean;
1740
1795
  }
1741
1796
  /**
1742
1797
  * Metadata for break nodes.
@@ -1751,8 +1806,11 @@ export interface BreakMetadata {
1751
1806
  * - 'lastRenderedPage': The editing application has inserted a soft break on the last save.
1752
1807
  * - 'textWrapping' (default, assumed when not specified): The next text will be placed on the next line.
1753
1808
  * - 'carriageReturn': An explicit carriage return (w:cr) equivalent to a hard line break.
1809
+ * - 'thematic': A thematic break (Markdown `---`/`***`/`___`, HTML `<hr>`) - a horizontal
1810
+ * rule separating sections, distinct from a page break. Emitted as `---` in Markdown and
1811
+ * `<hr>` in HTML.
1754
1812
  */
1755
- breakType: "column" | "page" | "lastRenderedPage" | "textWrapping" | "carriageReturn";
1813
+ breakType: "column" | "page" | "lastRenderedPage" | "textWrapping" | "carriageReturn" | "thematic";
1756
1814
  /**
1757
1815
  * Specifies the location which shall be used as the next available line when breakType
1758
1816
  * has a value of 'textWrapping'. Should be ignored for other break types.
@@ -1774,7 +1832,7 @@ export interface CodeMetadata {
1774
1832
  /**
1775
1833
  * When set, this node is a LaTeX math expression rather than a code block. `node.text`
1776
1834
  * holds the bare LaTeX (delimiters excluded); 'inline' round-trips as `$...$`,
1777
- * 'block' as `$$...$$`. Matches inscript-editor's math node (Roadmap Step 11.5).
1835
+ * 'block' as `$$...$$`. Matches attribute-driven editors' math nodes.
1778
1836
  */
1779
1837
  math?: "inline" | "block";
1780
1838
  }
@@ -2342,7 +2400,7 @@ export declare class OfficeParser {
2342
2400
  * const text = ast.toText();
2343
2401
  * ```
2344
2402
  */
2345
- static parseOffice(file: string | Buffer | ArrayBuffer | Uint8Array, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
2403
+ static parseOffice(file: string | Buffer | ArrayBuffer | Uint8Array | BlobLike, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
2346
2404
  /**
2347
2405
  * Terminates all active OCR workers and cleans up resources.
2348
2406
  *
@@ -2418,7 +2476,7 @@ export declare class OfficeConverter {
2418
2476
  * });
2419
2477
  * ```
2420
2478
  */
2421
- static convert<F extends string | Buffer | ArrayBuffer | Uint8Array, T extends SupportedFileType = InferFileTypeFromPath<F>, D extends SupportedDestination<T> = SupportedDestination<T>>(file: F, destination: D, config?: OfficeConverterConfig<D, T>): Promise<ConversionResult<D>>;
2479
+ static convert<F extends string | Buffer | ArrayBuffer | Uint8Array | BlobLike, T extends SupportedFileType = InferFileTypeFromPath<F>, D extends SupportedDestination<T> = SupportedDestination<T>>(file: F, destination: D, config?: OfficeConverterConfig<D, T>): Promise<ConversionResult<D>>;
2422
2480
  }
2423
2481
  export declare const parseOffice: typeof OfficeParser.parseOffice;
2424
2482
  export declare const terminateOcr: typeof OfficeParser.terminateOcr;