officeparser 7.2.3 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +277 -17
  2. package/dist/OfficeConverter.d.ts +1 -1
  3. package/dist/OfficeConverter.js +3 -0
  4. package/dist/OfficeGenerator.js +4 -0
  5. package/dist/OfficeParser.d.ts +2 -0
  6. package/dist/OfficeParser.js +6 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +5 -2
  9. package/dist/defaults.js +12 -0
  10. package/dist/generators/BaseGenerator.d.ts +34 -1
  11. package/dist/generators/BaseGenerator.js +98 -0
  12. package/dist/generators/CsvGenerator.d.ts +9 -1
  13. package/dist/generators/CsvGenerator.js +28 -16
  14. package/dist/generators/EpubGenerator.d.ts +43 -0
  15. package/dist/generators/EpubGenerator.js +312 -0
  16. package/dist/generators/HtmlGenerator.d.ts +12 -0
  17. package/dist/generators/HtmlGenerator.js +378 -61
  18. package/dist/generators/MarkdownGenerator.d.ts +28 -5
  19. package/dist/generators/MarkdownGenerator.js +432 -51
  20. package/dist/generators/PdfGenerator.js +32 -0
  21. package/dist/generators/RtfGenerator.js +47 -22
  22. package/dist/generators/TextGenerator.js +98 -11
  23. package/dist/index.d.ts +1 -0
  24. package/dist/index.js +1 -0
  25. package/dist/officeparser.browser.d.ts +427 -20
  26. package/dist/officeparser.browser.iife.js +338 -206
  27. package/dist/officeparser.browser.mjs +346 -214
  28. package/dist/officeparser.browser.slim.d.ts +427 -20
  29. package/dist/officeparser.browser.slim.iife.js +346 -214
  30. package/dist/officeparser.browser.slim.mjs +346 -214
  31. package/dist/parsers/EpubParser.d.ts +8 -0
  32. package/dist/parsers/EpubParser.js +217 -0
  33. package/dist/parsers/ExcelParser.js +2 -0
  34. package/dist/parsers/HtmlParser.js +507 -48
  35. package/dist/parsers/MarkdownParser.js +704 -92
  36. package/dist/parsers/OpenOfficeParser.js +128 -20
  37. package/dist/parsers/PdfParser.js +4 -1
  38. package/dist/parsers/PowerPointParser.js +1 -0
  39. package/dist/parsers/WordParser.js +1 -0
  40. package/dist/sbom.cdx.json +1695 -0
  41. package/dist/types.d.ts +427 -20
  42. package/dist/types.js +8 -0
  43. package/dist/utils/configUtils.js +53 -4
  44. package/dist/utils/errorUtils.js +7 -3
  45. package/dist/utils/sanitize.d.ts +139 -0
  46. package/dist/utils/sanitize.js +318 -0
  47. package/dist/utils/xmlUtils.js +2 -2
  48. package/dist/utils/zipUtils.js +76 -26
  49. package/package.json +16 -12
@@ -41,6 +41,8 @@ export declare enum OfficeErrorType {
41
41
  ZIP_ENTRY_INVALID_SIZE = "ZIP_ENTRY_INVALID_SIZE",
42
42
  /** ZIP uncompressed size limit exceeded */
43
43
  ZIP_SIZE_LIMIT_EXCEEDED = "ZIP_SIZE_LIMIT_EXCEEDED",
44
+ /** Document element/structure nesting exceeded the safe recursion depth */
45
+ MAX_NESTING_DEPTH_EXCEEDED = "MAX_NESTING_DEPTH_EXCEEDED",
44
46
  /** Embedding call timed out */
45
47
  EMBEDDING_TIMEOUT = "EMBEDDING_TIMEOUT"
46
48
  }
@@ -82,7 +84,13 @@ export declare enum OfficeWarningType {
82
84
  /** A node was skipped because it only contained whitespace */
83
85
  WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED",
84
86
  /** The HTML generator containerWidth option is invalid */
85
- INVALID_CONTAINER_WIDTH = "INVALID_CONTAINER_WIDTH"
87
+ INVALID_CONTAINER_WIDTH = "INVALID_CONTAINER_WIDTH",
88
+ /** A document's repeated-cell expansion hit the configured cell limit and was truncated */
89
+ TABLE_CELL_LIMIT_EXCEEDED = "TABLE_CELL_LIMIT_EXCEEDED",
90
+ /** A metadata override could not be represented in the destination format's vocabulary */
91
+ METADATA_NOT_REPRESENTABLE = "METADATA_NOT_REPRESENTABLE",
92
+ /** A styleMap output.tag was not an allowed element name and was ignored */
93
+ INVALID_STYLE_MAP_TAG = "INVALID_STYLE_MAP_TAG"
86
94
  }
87
95
  /**
88
96
  * Consolidated timeout settings for OCR operations.
@@ -209,9 +217,9 @@ export interface OcrConfig {
209
217
  abortSignal?: AbortSignal | null;
210
218
  }
211
219
  /**
212
- * Configuration options for the OfficeParser.
220
+ * Configuration options shared across every input format.
213
221
  */
214
- export interface OfficeParserConfig {
222
+ export interface CommonOfficeParserConfig {
215
223
  /**
216
224
  * @deprecated Use `onWarning` instead.
217
225
  * Flag to show all the logs to console in case of an error irrespective of your own handling.
@@ -363,6 +371,44 @@ export interface OfficeParserConfig {
363
371
  */
364
372
  decompressionLimits?: DecompressionLimits;
365
373
  }
374
+ /**
375
+ * Format-specific options for HTML (and XHTML/EPUB, which parse through the same code path).
376
+ *
377
+ * Note there is deliberately no `MdParserConfig`: the Markdown parser populates its
378
+ * dialect-provenance metadata (e.g. `AdmonitionMetadata.sourceSyntax`) unconditionally because
379
+ * doing so costs nothing and changes no existing field's value, so it has nothing to configure.
380
+ * An empty placeholder interface would be worse than useless here - `interface X {}` accepts any
381
+ * non-nullish value in TypeScript, so `mdParserConfig: 5` would type-check.
382
+ */
383
+ export interface HtmlParserConfig {
384
+ /**
385
+ * Preserve source HTML attributes that no typed metadata field consumed, on
386
+ * `OfficeContentNode.htmlAttributes`, so they can be replayed on generation.
387
+ *
388
+ * Off by default: with it off nothing is populated, so the AST is byte-identical to previous
389
+ * releases, and the attribute-replay surface stays something a consumer opts into rather than
390
+ * something switched on for every existing caller. Captured values are sanitized on the way in
391
+ * *and* on the way out - see `BaseContentNode.htmlAttributes`.
392
+ *
393
+ * Defaults to false.
394
+ */
395
+ preserveAttributes?: boolean;
396
+ }
397
+ /**
398
+ * Maps an input format string to its corresponding format-specific parser configuration, mirroring
399
+ * `GeneratorSpecificConfig<D>` on the generator side. Unlike the generator side, the input format is
400
+ * usually runtime-detected rather than known statically at the `parseOffice()` call site, so this
401
+ * mainly exists for internal typing/extensibility rather than compile-time narrowing per call.
402
+ */
403
+ export type ParserSpecificConfig<F extends string> = F extends "html" | "epub" ? {
404
+ htmlParserConfig?: HtmlParserConfig;
405
+ } : Partial<{
406
+ htmlParserConfig: HtmlParserConfig;
407
+ }>;
408
+ /**
409
+ * Configuration options for the OfficeParser.
410
+ */
411
+ export type OfficeParserConfig<F extends string = string> = CommonOfficeParserConfig & ParserSpecificConfig<F>;
366
412
  /**
367
413
  * Limits applied to ZIP archive decompression.
368
414
  */
@@ -379,6 +425,28 @@ export interface DecompressionLimits {
379
425
  * Default is 10000.
380
426
  */
381
427
  maxZipEntries?: number;
428
+ /**
429
+ * Maximum number of table cells materialized from a single document.
430
+ *
431
+ * ODF encodes runs of identical cells and rows with `table:number-columns-repeated` and
432
+ * `table:number-rows-repeated` rather than repeating the markup, so a few hundred bytes of XML
433
+ * can ask the parser to build an arbitrary number of nodes - and because the two multiply, a
434
+ * row repeat times a column repeat compounds it. The ZIP limits above cannot catch this: the
435
+ * XML is tiny before decompression and the expansion happens afterwards, while building the
436
+ * AST.
437
+ *
438
+ * Real documents are nowhere near this. The repeat counts LibreOffice writes are large
439
+ * (`number-rows-repeated="1048566"` is routine) but they sit on *empty* trailing runs, which
440
+ * are skipped for spreadsheets; the bundled fixtures top out around 350 cells.
441
+ *
442
+ * On reaching the limit the parser stops materializing further cells, emits a
443
+ * `TABLE_CELL_LIMIT_EXCEEDED` warning, and returns what it has rather than throwing, so a
444
+ * genuinely enormous sheet still yields usable output. Raise it if you routinely process
445
+ * spreadsheets larger than this; note the memory cost scales with it.
446
+ *
447
+ * Default is 1000000.
448
+ */
449
+ maxTableCells?: number;
382
450
  }
383
451
  /**
384
452
  * Represents a single issue (warning, error, or info) generated during document processing.
@@ -398,7 +466,7 @@ export interface OfficeIssue {
398
466
  /**
399
467
  * The result of a document conversion operation.
400
468
  */
401
- export type ConversionValue<D extends UniversalGeneratorFormat> = D extends "pdf" ? Uint8Array | string : D extends "chunks" ? OfficeChunk[] : D extends "csv" ? string | Uint8Array : string;
469
+ export type ConversionValue<D extends UniversalGeneratorFormat> = D extends "pdf" ? Uint8Array | string : D extends "chunks" ? OfficeChunk[] : D extends "csv" ? string | Uint8Array : D extends "epub" ? Uint8Array : string;
402
470
  export interface ConversionResult<D extends UniversalGeneratorFormat> {
403
471
  /** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
404
472
  value: ConversionValue<D>;
@@ -408,7 +476,7 @@ export interface ConversionResult<D extends UniversalGeneratorFormat> {
408
476
  /**
409
477
  * Universal formats supported by all source types for generation.
410
478
  */
411
- export type UniversalGeneratorFormat = "text" | "md" | "html" | "pdf" | "csv" | "rtf" | "chunks";
479
+ export type UniversalGeneratorFormat = "text" | "md" | "html" | "pdf" | "csv" | "rtf" | "chunks" | "epub";
412
480
  /**
413
481
  * Allowed destination formats for a given source type.
414
482
  * Currently, all generators are universal across all source formats.
@@ -420,6 +488,56 @@ export type SupportedDestination<_T extends SupportedFileType = SupportedFileTyp
420
488
  /**
421
489
  * Common configuration options for all generators.
422
490
  */
491
+ /**
492
+ * Per-field overrides for the metadata written into generated output.
493
+ *
494
+ * Field names mirror `OfficeMetadata` so the same vocabulary describes what was parsed and what
495
+ * gets written. Only the fields generators can actually represent are listed; arbitrary
496
+ * caller-defined entries go in `custom`.
497
+ *
498
+ * **Not every format can represent every field.** HTML (`<meta>`) and Markdown (YAML frontmatter)
499
+ * accept anything; EPUB's OPF is a closed Dublin Core vocabulary and RTF's `\info` group has a
500
+ * fixed set of control words, so a `custom` entry has nowhere to go in those. Rather than
501
+ * silently dropping it, generators report the loss through `onWarning`
502
+ * (`OfficeWarningType.MetadataNotRepresentable`) and continue.
503
+ */
504
+ export interface MetadataOverrides {
505
+ /** Document title. */
506
+ title?: string;
507
+ /** Document author. */
508
+ author?: string;
509
+ /** Description/comments. */
510
+ description?: string;
511
+ /** Subject/topic. */
512
+ subject?: string;
513
+ /** Keywords. */
514
+ keywords?: string;
515
+ /** User who last modified the document. */
516
+ lastModifiedBy?: string;
517
+ /** Creation date. */
518
+ created?: Date;
519
+ /**
520
+ * Last modification date.
521
+ *
522
+ * EPUB writes it as the required `dcterms:modified` property and as the mtime on every zip
523
+ * entry. When unset, the source document's own `metadata.modified` is used, falling back to
524
+ * the current time only if the document has none.
525
+ */
526
+ modified?: Date;
527
+ /**
528
+ * Language tag (e.g. `'en'`, `'de-DE'`). Written as EPUB `dc:language` and HTML `lang`.
529
+ */
530
+ language?: string;
531
+ /**
532
+ * Arbitrary caller-defined key/value pairs, kept in their own bucket rather than mixed in
533
+ * beside the named fields above: with a bare index signature a typo like `titel` would
534
+ * silently become a custom entry instead of a compile error.
535
+ *
536
+ * Written where the format allows it (HTML `<meta name="custom:KEY">`, Markdown frontmatter);
537
+ * reported via `onWarning` where it does not (EPUB, RTF).
538
+ */
539
+ custom?: Record<string, string | number | boolean | Date>;
540
+ }
423
541
  export interface CommonGeneratorConfig {
424
542
  /**
425
543
  * Callback called for every node during generation.
@@ -488,6 +606,29 @@ export interface CommonGeneratorConfig {
488
606
  * Defaults to false.
489
607
  */
490
608
  renderMetadata?: boolean;
609
+ /**
610
+ * Overrides for the document metadata written into the generated output, applied on top of
611
+ * `ast.metadata`.
612
+ *
613
+ * Merged **per field**, so setting only `modified` leaves the parsed title, author, and
614
+ * everything else intact. Every field is optional; an omitted field keeps the source
615
+ * document's value.
616
+ *
617
+ * These are output overrides only - `ast.metadata` itself is never mutated, so the same AST
618
+ * can be generated repeatedly with different metadata.
619
+ *
620
+ * @example Set the modification date written into the output
621
+ * ```typescript
622
+ * await ast.to('epub', { metadataOverrides: { modified: new Date('2024-01-01T00:00:00Z') } });
623
+ * ```
624
+ * @example Rebrand the output without touching the parsed document
625
+ * ```typescript
626
+ * await ast.to('html', {
627
+ * metadataOverrides: { title: 'Q4 Report', author: 'Acme Inc', custom: { department: 'Finance' } },
628
+ * });
629
+ * ```
630
+ */
631
+ metadataOverrides?: MetadataOverrides;
491
632
  /**
492
633
  * Whether to ignore the built-in default style mappings (e.g. "Heading 1" -> h1).
493
634
  * Set to true if you want full control over style mapping.
@@ -603,15 +744,64 @@ export interface HtmlInjectionConfig {
603
744
  /** Raw HTML injected immediately before the closing </body> tag */
604
745
  bodyEnd?: string;
605
746
  }
747
+ /**
748
+ * Granular control over which parts of the full HTML "document envelope" are emitted.
749
+ * Shorthand: `standalone: true` == every part on (a complete document); `standalone: false` ==
750
+ * every part off (a bare content fragment). When an object is passed, any field you omit
751
+ * defaults to its "on" (standalone) value.
752
+ */
753
+ export interface StandaloneConfig {
754
+ /**
755
+ * Wrap the output in `<!DOCTYPE html><html><head>…</head><body>…</body></html>`.
756
+ * When false, only the inner content fragment is emitted. Defaults to true.
757
+ */
758
+ document?: boolean;
759
+ /**
760
+ * Emit `<title>` and `<meta>` tags (author, description, dates, custom properties) in the head.
761
+ * Only meaningful when `document` is true. Defaults to true.
762
+ */
763
+ metaTags?: boolean;
764
+ /**
765
+ * How the library's built-in CSS is delivered:
766
+ * - `'full'` — the complete premium stylesheet using global selectors (`body`, `h1`, `table`, …).
767
+ * This is what `standalone: true` has always emitted.
768
+ * - `'scoped'` — the same styling, scoped under the fragment's container via CSS `@scope` so it
769
+ * cannot leak into a host page's own styles. Requires a modern browser engine (Chrome 118+,
770
+ * Safari 17.4+, Firefox 128+).
771
+ * - `'none'` — no stylesheet is emitted; the host page (or EPUB reader, or rich-text editor)
772
+ * supplies its own styling.
773
+ * The boolean shorthand for `standalone` maps `true` → `'full'`, `false` → `'none'`.
774
+ * Defaults to `'full'`.
775
+ */
776
+ styles?: "full" | "scoped" | "none";
777
+ /**
778
+ * Emit injected `<script>` tags: the Chart.js loader (when `includeCharts` is true and charts
779
+ * are present) and the spreadsheet interactivity script. Defaults to true.
780
+ */
781
+ scripts?: boolean;
782
+ /**
783
+ * Apply `injections.headStart` / `injections.headEnd`. Only meaningful when `document` is true
784
+ * (there is no `<head>` to inject into otherwise). Defaults to true.
785
+ */
786
+ headInjections?: boolean;
787
+ /**
788
+ * Apply `injections.bodyStart` / `injections.bodyEnd`. Applies even when generating a bare
789
+ * fragment (`document: false`), since these wrap body *content*, not the document shell.
790
+ * Defaults to true.
791
+ */
792
+ bodyInjections?: boolean;
793
+ }
606
794
  /**
607
795
  * Configuration options for HTML generation.
608
796
  */
609
797
  export interface HtmlGeneratorConfig {
610
798
  /**
611
799
  * Whether to wrap the output in a full HTML document structure (e.g., <html>, <head>, etc.).
800
+ * Pass an object instead of a boolean for granular control over individual parts of the
801
+ * envelope (document shell, meta tags, styles, scripts, injections) - see `StandaloneConfig`.
612
802
  * Defaults to true.
613
803
  */
614
- standalone?: boolean;
804
+ standalone?: boolean | StandaloneConfig;
615
805
  /**
616
806
  * URL for the Chart.js library to use when 'includeCharts' is true.
617
807
  * Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
@@ -786,12 +976,81 @@ export interface CsvGeneratorConfig {
786
976
  */
787
977
  columnDelimiter?: string;
788
978
  }
979
+ /**
980
+ * Named Markdown dialect presets for `MarkdownDialectConfig`/`MdGeneratorConfig.dialect`.
981
+ * `'extended'` is officeParser's own kitchen-sink default and reproduces this library's
982
+ * historical output exactly (every feature on, GitHub-style admonitions).
983
+ */
984
+ export type MarkdownDialectPreset = "extended" | "github" | "gitlab" | "obsidian" | "pandoc" | "commonmark";
985
+ /**
986
+ * Granular control over which native Markdown syntax the generator emits for constructs that
987
+ * differ across real-world dialects (e.g. GitHub's `> [!NOTE]` vs GitLab's `:::note` vs Pandoc's
988
+ * `::: {.note}` admonitions). Shorthand: pass a `MarkdownDialectPreset` string for a named target;
989
+ * pass an object for granular control. Any field you omit from the object form falls back to the
990
+ * preset named by `extends` (default `'extended'`) - **not** to whatever preset may have been
991
+ * ambient before, since config merging replaces the whole field rather than layering on top of it.
992
+ */
993
+ export interface MarkdownDialectConfig {
994
+ /** Base preset any omitted field inherits from. Defaults to 'extended'. */
995
+ extends?: MarkdownDialectPreset;
996
+ /** Admonition syntax: GitHub `> [!NOTE]`, GitLab `:::note`, Pandoc `::: {.note}`, or `'none'`
997
+ * to degrade to a plain bold-labeled blockquote with no special marker. */
998
+ admonitions?: "github" | "gitlab" | "pandoc" | "none";
999
+ /** Markdown Extra/Pandoc-style `Term\n: Description` definition lists. */
1000
+ definitionLists?: boolean;
1001
+ /** `[^id]` footnote references/definitions. When false, note content is inlined as a
1002
+ * parenthetical right at the reference point instead of using footnote syntax. */
1003
+ footnotes?: boolean;
1004
+ /** Pandoc-style `[@citekey]` citations. When false, emits `[citekey]` (brackets, no `@`). */
1005
+ citations?: boolean;
1006
+ /** Obsidian-style `[[Page]]`/`[[Page|Alias]]` wikilinks. When false, falls back to a plain
1007
+ * `[text](url)` link using the same target. */
1008
+ wikilinks?: boolean;
1009
+ /** Inline `$...$`/block `$$...$$` math delimiters, or `'none'` for bare LaTeX text. */
1010
+ math?: "dollar" | "none";
1011
+ /** Pandoc-style `{width=50% .centered}` attribute lists after images/tables. */
1012
+ attributeLists?: boolean;
1013
+ /** GFM `~~text~~` strikethrough (not part of base CommonMark). */
1014
+ strikethrough?: boolean;
1015
+ /** Unordered list bullet character. */
1016
+ bulletListMarker?: "-" | "*" | "+";
1017
+ /** Ordered list marker punctuation. */
1018
+ orderedListMarker?: "." | ")";
1019
+ /** Emphasis delimiter style for bold/italic. */
1020
+ emphasisMarker?: "asterisk" | "underscore";
1021
+ /** Table syntax: native GFM pipe tables, or forced HTML `<table>` (required for strict
1022
+ * CommonMark, which has no table syntax of its own). */
1023
+ tables?: "native" | "html";
1024
+ }
1025
+ /**
1026
+ * Granular control over when the Markdown generator falls back to raw HTML tags for features
1027
+ * standard Markdown can't express natively. Shorthand: `true`/`false` (via
1028
+ * `MdGeneratorConfig.fallbackToHtml`) turns every part on/off at once; pass an object instead to
1029
+ * control them independently. Omitted object fields default to on, matching the boolean shorthand.
1030
+ */
1031
+ export interface FallbackToHtmlConfig {
1032
+ /** Underline/subscript/superscript via `<u>`/`<sub>`/`<sup>`. */
1033
+ textFormatting?: boolean;
1034
+ /** Heading/paragraph text alignment via `<div style="text-align:...">`. */
1035
+ alignment?: boolean;
1036
+ /** Internal-link/heading `<a id>`/`<a name>` anchor tags. */
1037
+ anchors?: boolean;
1038
+ /** Nested-table and merged-cell (colspan/rowspan) HTML `<table>` fallback. */
1039
+ tables?: boolean;
1040
+ /** YouTube embed `<div data-youtube-video>` vs. a plain link. */
1041
+ embeds?: boolean;
1042
+ /** Multi-line table cell content joined with `<br>` instead of a space. */
1043
+ cellLineBreaks?: boolean;
1044
+ }
789
1045
  /**
790
1046
  * Configuration options for Markdown generation.
791
1047
  */
792
1048
  export interface MdGeneratorConfig {
793
1049
  /**
794
1050
  * Whether to fallback to HTML tags for features not supported by standard Markdown.
1051
+ * Pass an object instead of a boolean for granular control over individual parts (text
1052
+ * formatting, alignment, anchors, tables, embeds, cell line breaks) - see
1053
+ * `FallbackToHtmlConfig`. Omitted object fields default to on, matching `true`.
795
1054
  *
796
1055
  * Markdown has limited support for complex document structures. This flag controls how
797
1056
  * the generator handles features that cannot be represented in pure Markdown:
@@ -810,7 +1069,14 @@ export interface MdGeneratorConfig {
810
1069
  *
811
1070
  * Defaults to true.
812
1071
  */
813
- fallbackToHtml?: boolean;
1072
+ fallbackToHtml?: boolean | FallbackToHtmlConfig;
1073
+ /**
1074
+ * Target Markdown dialect for generation - which native syntax to emit for constructs that
1075
+ * differ across real-world targets (GitHub/GitLab/Obsidian/Pandoc/strict CommonMark). See
1076
+ * `MarkdownDialectConfig` for the full per-feature field list. Defaults to `'extended'`
1077
+ * (officeParser's own historical kitchen-sink behavior, unchanged from prior versions).
1078
+ */
1079
+ dialect?: MarkdownDialectPreset | MarkdownDialectConfig;
814
1080
  }
815
1081
  /**
816
1082
  * Configuration options for plain text generation.
@@ -823,11 +1089,24 @@ export interface TextGeneratorConfig {
823
1089
  newlineDelimiter?: string;
824
1090
  /**
825
1091
  * Whether to attempt to preserve the original document layout.
826
- * If true, tables will be rendered with separators and aligned columns.
827
- * If false, output will be a flat stream of text nodes.
828
- * Defaults to false.
1092
+ * If true, tables are rendered with separators and aligned columns, and list items get their
1093
+ * markers and indentation.
1094
+ * If false, output is a flat stream of text nodes (cells are tab-separated).
1095
+ * Defaults to **true**.
829
1096
  */
830
1097
  preserveLayout?: boolean;
1098
+ /**
1099
+ * Whether to append the collected footnotes/endnotes as a trailing `--- Notes ---` section.
1100
+ * Set false to omit it when you want only the document body; the notes are still parsed and
1101
+ * remain available on the AST, they are simply not rendered into the text output.
1102
+ *
1103
+ * Note this differs from the parser's `ignoreNotes`, which discards notes at parse time so they
1104
+ * never reach the AST at all. Use this when you want the AST to keep them but the text output
1105
+ * to leave them out.
1106
+ *
1107
+ * Defaults to true.
1108
+ */
1109
+ renderNotes?: boolean;
831
1110
  }
832
1111
  /**
833
1112
  * The strategy used for chunking a document for RAG pipelines.
@@ -1032,11 +1311,11 @@ export interface OfficeChunk {
1032
1311
  /**
1033
1312
  * Supported file types for parsing.
1034
1313
  */
1035
- export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods" | "pdf" | "rtf" | "md" | "html" | "csv";
1314
+ export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods" | "pdf" | "rtf" | "md" | "html" | "csv" | "epub";
1036
1315
  /**
1037
1316
  * Types of content nodes in the AST.
1038
1317
  */
1039
- export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break" | "code" | "comment" | "header" | "footer" | "slideMaster";
1318
+ export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break" | "code" | "comment" | "header" | "footer" | "slideMaster" | "embed" | "admonition" | "definitionList" | "definitionTerm" | "definitionDescription";
1040
1319
  /**
1041
1320
  * Supported MIME types for attachments.
1042
1321
  */
@@ -1230,6 +1509,10 @@ export interface ListMetadata {
1230
1509
  style?: string;
1231
1510
  /** Unique anchor IDs for internal linking. */
1232
1511
  anchorIds?: string[];
1512
+ /** True when this list item is a GFM task-list item (checkbox), regardless of checked state. */
1513
+ isTask?: boolean;
1514
+ /** Checked state for a task-list item. Only meaningful when isTask is true. */
1515
+ checked?: boolean;
1233
1516
  }
1234
1517
  /**
1235
1518
  * Metadata for a table cell (primarily used in Excel/spreadsheet parsing).
@@ -1269,6 +1552,11 @@ export interface CellMetadata {
1269
1552
  export interface TableMetadata {
1270
1553
  /** Unique anchor IDs for internal linking. */
1271
1554
  anchorIds?: string[];
1555
+ /**
1556
+ * Layout alignment of the table on the page (e.g. inscript-editor's `CustomTable`).
1557
+ * @example 'center'
1558
+ */
1559
+ align?: "left" | "center" | "right";
1272
1560
  }
1273
1561
  /**
1274
1562
  * Metadata for a chart node in the document.
@@ -1309,6 +1597,45 @@ export interface ImageMetadata {
1309
1597
  url?: string;
1310
1598
  /** Unique anchor IDs for internal linking. */
1311
1599
  anchorIds?: string[];
1600
+ /**
1601
+ * Display width of the image (e.g. inscript-editor's `CustomImage`), as a CSS length or percentage.
1602
+ * @example "50%"
1603
+ */
1604
+ width?: string;
1605
+ /**
1606
+ * Layout alignment of the image (e.g. inscript-editor's `CustomImage`).
1607
+ * @example 'center'
1608
+ */
1609
+ align?: "left" | "center" | "right";
1610
+ }
1611
+ /**
1612
+ * Metadata for an embedded external media node (e.g. a YouTube video).
1613
+ * Markdown has no native syntax for this - see `MarkdownGenerator`'s `embed` case.
1614
+ */
1615
+ export interface EmbedMetadata {
1616
+ /** The kind of embed. Only 'youtube' is supported today; the shape is generic for future providers. */
1617
+ embedType: "youtube";
1618
+ /** The provider-specific video ID (e.g. the 11-character YouTube video ID). */
1619
+ videoId: string;
1620
+ /** The original/canonical URL of the embedded media, if known. */
1621
+ url?: string;
1622
+ /** Display width, as a CSS length or percentage. */
1623
+ width?: string;
1624
+ /** Layout alignment of the embed. */
1625
+ align?: "left" | "center" | "right";
1626
+ }
1627
+ /**
1628
+ * Metadata for an admonition/alert node (e.g. GitHub's `> [!NOTE]` or GLFM's `:::note`).
1629
+ * `MarkdownParser` accepts both syntaxes (and generates either, plus Pandoc's `::: {.note}`,
1630
+ * depending on `MdGeneratorConfig.dialect`). Children are block content (paragraphs) wrapped by
1631
+ * the admonition.
1632
+ */
1633
+ export interface AdmonitionMetadata {
1634
+ admonitionType: "note" | "tip" | "important" | "warning" | "caution";
1635
+ /** Optional custom title; falls back to the type label. */
1636
+ title?: string;
1637
+ /** Which concrete input syntax produced this node. Always populated by the parser. */
1638
+ sourceSyntax?: "github" | "gitlab";
1312
1639
  }
1313
1640
  /**
1314
1641
  * Metadata for PDF page nodes.
@@ -1339,6 +1666,25 @@ export interface TextMetadata {
1339
1666
  * - 'external': Link to an external URL
1340
1667
  */
1341
1668
  linkType?: "internal" | "external";
1669
+ /**
1670
+ * When set, this text is an abbreviation and this is its full-form expansion,
1671
+ * rendered as `<abbr title="...">`. Populated from Markdown Extra's
1672
+ * `*[HTML]: Hypertext Markup Language` syntax or an HTML `<abbr>` tag.
1673
+ */
1674
+ abbreviationTitle?: string;
1675
+ /**
1676
+ * When set, this text is a Pandoc/MultiMarkdown-style citation reference
1677
+ * (`[@citekey]`), and this is the bare citekey (e.g. "smith2024"). Bibliography
1678
+ * resolution (author/year display, .bib management) is left to the consuming app.
1679
+ */
1680
+ citationKey?: string;
1681
+ /**
1682
+ * True when this is an Obsidian-style wikilink (`[[page]]` / `[[page|alias]]`).
1683
+ * `link` holds the bare page name and `linkType` is always 'internal'; the
1684
+ * per-workspace enable/disable toggle lives in markdownwriter, not here -
1685
+ * officeParser always parses/generates the syntax.
1686
+ */
1687
+ wikilink?: boolean;
1342
1688
  }
1343
1689
  /**
1344
1690
  * Metadata for note nodes (footnotes/endnotes).
@@ -1392,6 +1738,12 @@ export interface CodeMetadata {
1392
1738
  language?: string;
1393
1739
  /** Unique anchor IDs for internal linking. */
1394
1740
  anchorIds?: string[];
1741
+ /**
1742
+ * When set, this node is a LaTeX math expression rather than a code block. `node.text`
1743
+ * holds the bare LaTeX (delimiters excluded); 'inline' round-trips as `$...$`,
1744
+ * 'block' as `$$...$$`. Matches inscript-editor's math node (Roadmap Step 11.5).
1745
+ */
1746
+ math?: "inline" | "block";
1395
1747
  }
1396
1748
  /**
1397
1749
  * Metadata for a comment/annotation.
@@ -1411,7 +1763,7 @@ export interface HeaderFooterMetadata {
1411
1763
  /**
1412
1764
  * Union type for content metadata.
1413
1765
  */
1414
- export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | CommentMetadata | HeaderFooterMetadata | TableMetadata | undefined;
1766
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | CommentMetadata | HeaderFooterMetadata | TableMetadata | EmbedMetadata | AdmonitionMetadata | undefined;
1415
1767
  /**
1416
1768
  * Represents a node in the document content tree.
1417
1769
  * This is the core building block of the parsed document structure.
@@ -1486,6 +1838,23 @@ export interface BaseContentNode {
1486
1838
  * @example "<w:p><w:r><w:t>Hello</w:t></w:r></w:p>" for DOCX
1487
1839
  */
1488
1840
  rawContent?: string;
1841
+ /**
1842
+ * Source HTML attributes that no typed metadata field consumed, preserved for round-trip
1843
+ * fidelity (e.g. a `data-*` attribute an editor round-trips through officeParser).
1844
+ *
1845
+ * Only populated by the HTML/XHTML parser, only for elements it recognises, and only when
1846
+ * `htmlParserConfig.preserveAttributes` is enabled - so by default this is always absent.
1847
+ *
1848
+ * Sanitized on both legs, since an AST can also be constructed programmatically rather than
1849
+ * parsed: event handlers (`on*`) and `srcdoc` are never carried, URL-bearing attributes go
1850
+ * through the same URL sanitizer as typed fields, and every value is escaped on output. A
1851
+ * typed field always wins over a same-named entry here.
1852
+ *
1853
+ * Ignored by the non-HTML generators (Markdown, RTF, CSV, text, chunking) by design - these
1854
+ * are HTML attributes and have no meaning in those targets.
1855
+ * @example { 'data-tracking-id': 'abc123', 'class': 'lead' }
1856
+ */
1857
+ htmlAttributes?: Record<string, string>;
1489
1858
  }
1490
1859
  /**
1491
1860
  * Represents a node in the document content tree.
@@ -1573,6 +1942,21 @@ export type OfficeContentNode = BaseContentNode & ({
1573
1942
  } | {
1574
1943
  type: "slideMaster";
1575
1944
  metadata?: SlideMetadata;
1945
+ } | {
1946
+ type: "embed";
1947
+ metadata?: EmbedMetadata;
1948
+ } | {
1949
+ type: "admonition";
1950
+ metadata?: AdmonitionMetadata;
1951
+ } | {
1952
+ type: "definitionList";
1953
+ metadata?: undefined;
1954
+ } | {
1955
+ type: "definitionTerm";
1956
+ metadata?: undefined;
1957
+ } | {
1958
+ type: "definitionDescription";
1959
+ metadata?: undefined;
1576
1960
  });
1577
1961
  /**
1578
1962
  * Structured information extracted from a chart.
@@ -1815,14 +2199,36 @@ export interface OfficeParserAST {
1815
2199
  /** Any warnings or non-fatal issues encountered during parsing. */
1816
2200
  warnings: OfficeIssue[];
1817
2201
  /**
1818
- * @deprecated Use `.to('text')` instead.
1819
- * Note: This method is synchronous, while the new `.to()` method is asynchronous.
2202
+ * @deprecated Use `.to('text')` instead. This method is the older renderer and takes no
2203
+ * configuration; `.to('text')` produces the same content and lets you configure the rest.
1820
2204
  *
1821
- * Converts the entire AST to plain text.
1822
- * This method flattens the document structure and returns just the text content,
1823
- * stripping out all formatting, metadata, and structure.
2205
+ * Converts the entire AST to plain text, flattening the document structure and stripping all
2206
+ * formatting, metadata, and structure. Text is joined using `config.newlineDelimiter`
2207
+ * (default: `'\n'`).
2208
+ *
2209
+ * **Migrating.** `.to('text')` is asynchronous and configurable. At its defaults it emits
2210
+ * everything this method emits, verified across every bundled fixture in all 12 supported
2211
+ * formats in both layout modes: no word produced here is missing there. It also renders merged
2212
+ * table cells correctly, where this method glues them (`OneThree` vs `One Three`).
2213
+ *
2214
+ * Where the two differ is configuration, not capability. Notes and image placeholders are
2215
+ * emitted by default but are switchable; this method emits neither and offers no way to ask
2216
+ * for them. Layout is likewise a knob rather than a fixed behavior:
2217
+ *
2218
+ * ```typescript
2219
+ * // Default: aligned table grids, list markers, notes, image placeholders.
2220
+ * const { value } = await ast.to('text');
2221
+ *
2222
+ * // Deliberate opt-out - closest to this method's shape.
2223
+ * const { value } = await ast.to('text', {
2224
+ * includeImages: false,
2225
+ * textConfig: { preserveLayout: false, renderNotes: false },
2226
+ * });
2227
+ * ```
1824
2228
  *
1825
- * The text is concatenated using the delimiter specified in `config.newlineDelimiter` (default: '\n').
2229
+ * Spreadsheets (CSV/ODS/XLSX) are unaffected by `preserveLayout`, since it governs
2230
+ * `table`/`list` nodes rather than `sheet`/`row`/`cell`; there the default aligned grid is the
2231
+ * most faithful rendering.
1826
2232
  *
1827
2233
  * @returns A plain text representation of the document
1828
2234
  * @example
@@ -1877,6 +2283,7 @@ export declare class OfficeParser {
1877
2283
  * - `.csv` → CsvParser
1878
2284
  * - `.md` → MarkdownParser
1879
2285
  * - `.html` → HtmlParser
2286
+ * - `.epub` → EpubParser
1880
2287
  *
1881
2288
  * @param file - File path (string), Buffer, or ArrayBuffer containing the document
1882
2289
  * @param config - Optional configuration object (defaults applied for all omitted options)
@@ -1978,7 +2385,7 @@ export declare class OfficeConverter {
1978
2385
  * });
1979
2386
  * ```
1980
2387
  */
1981
- static convert<F extends string | Buffer | ArrayBuffer | Uint8Array, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
2388
+ static convert<F extends string | Buffer | ArrayBuffer | Uint8Array, T extends SupportedFileType = InferFileTypeFromPath<F>, D extends SupportedDestination<T> = SupportedDestination<T>>(file: F, destination: D, config?: OfficeConverterConfig<D, T>): Promise<ConversionResult<D>>;
1982
2389
  }
1983
2390
  export declare const parseOffice: typeof OfficeParser.parseOffice;
1984
2391
  export declare const terminateOcr: typeof OfficeParser.terminateOcr;