officeparser 7.2.3 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +277 -17
  2. package/dist/OfficeConverter.d.ts +1 -1
  3. package/dist/OfficeConverter.js +3 -0
  4. package/dist/OfficeGenerator.js +4 -0
  5. package/dist/OfficeParser.d.ts +2 -0
  6. package/dist/OfficeParser.js +6 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +5 -2
  9. package/dist/defaults.js +12 -0
  10. package/dist/generators/BaseGenerator.d.ts +34 -1
  11. package/dist/generators/BaseGenerator.js +98 -0
  12. package/dist/generators/CsvGenerator.d.ts +9 -1
  13. package/dist/generators/CsvGenerator.js +28 -16
  14. package/dist/generators/EpubGenerator.d.ts +43 -0
  15. package/dist/generators/EpubGenerator.js +312 -0
  16. package/dist/generators/HtmlGenerator.d.ts +12 -0
  17. package/dist/generators/HtmlGenerator.js +378 -61
  18. package/dist/generators/MarkdownGenerator.d.ts +28 -5
  19. package/dist/generators/MarkdownGenerator.js +432 -51
  20. package/dist/generators/PdfGenerator.js +32 -0
  21. package/dist/generators/RtfGenerator.js +47 -22
  22. package/dist/generators/TextGenerator.js +98 -11
  23. package/dist/index.d.ts +1 -0
  24. package/dist/index.js +1 -0
  25. package/dist/officeparser.browser.d.ts +427 -20
  26. package/dist/officeparser.browser.iife.js +338 -206
  27. package/dist/officeparser.browser.mjs +346 -214
  28. package/dist/officeparser.browser.slim.d.ts +427 -20
  29. package/dist/officeparser.browser.slim.iife.js +346 -214
  30. package/dist/officeparser.browser.slim.mjs +346 -214
  31. package/dist/parsers/EpubParser.d.ts +8 -0
  32. package/dist/parsers/EpubParser.js +217 -0
  33. package/dist/parsers/ExcelParser.js +2 -0
  34. package/dist/parsers/HtmlParser.js +507 -48
  35. package/dist/parsers/MarkdownParser.js +704 -92
  36. package/dist/parsers/OpenOfficeParser.js +128 -20
  37. package/dist/parsers/PdfParser.js +4 -1
  38. package/dist/parsers/PowerPointParser.js +1 -0
  39. package/dist/parsers/WordParser.js +1 -0
  40. package/dist/sbom.cdx.json +1695 -0
  41. package/dist/types.d.ts +427 -20
  42. package/dist/types.js +8 -0
  43. package/dist/utils/configUtils.js +53 -4
  44. package/dist/utils/errorUtils.js +7 -3
  45. package/dist/utils/sanitize.d.ts +139 -0
  46. package/dist/utils/sanitize.js +318 -0
  47. package/dist/utils/xmlUtils.js +2 -2
  48. package/dist/utils/zipUtils.js +76 -26
  49. package/package.json +16 -12
package/dist/types.d.ts CHANGED
@@ -39,6 +39,8 @@ export declare enum OfficeErrorType {
39
39
  ZIP_ENTRY_INVALID_SIZE = "ZIP_ENTRY_INVALID_SIZE",
40
40
  /** ZIP uncompressed size limit exceeded */
41
41
  ZIP_SIZE_LIMIT_EXCEEDED = "ZIP_SIZE_LIMIT_EXCEEDED",
42
+ /** Document element/structure nesting exceeded the safe recursion depth */
43
+ MAX_NESTING_DEPTH_EXCEEDED = "MAX_NESTING_DEPTH_EXCEEDED",
42
44
  /** Embedding call timed out */
43
45
  EMBEDDING_TIMEOUT = "EMBEDDING_TIMEOUT"
44
46
  }
@@ -80,7 +82,13 @@ export declare enum OfficeWarningType {
80
82
  /** A node was skipped because it only contained whitespace */
81
83
  WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED",
82
84
  /** The HTML generator containerWidth option is invalid */
83
- INVALID_CONTAINER_WIDTH = "INVALID_CONTAINER_WIDTH"
85
+ INVALID_CONTAINER_WIDTH = "INVALID_CONTAINER_WIDTH",
86
+ /** A document's repeated-cell expansion hit the configured cell limit and was truncated */
87
+ TABLE_CELL_LIMIT_EXCEEDED = "TABLE_CELL_LIMIT_EXCEEDED",
88
+ /** A metadata override could not be represented in the destination format's vocabulary */
89
+ METADATA_NOT_REPRESENTABLE = "METADATA_NOT_REPRESENTABLE",
90
+ /** A styleMap output.tag was not an allowed element name and was ignored */
91
+ INVALID_STYLE_MAP_TAG = "INVALID_STYLE_MAP_TAG"
84
92
  }
85
93
  /**
86
94
  * Consolidated timeout settings for OCR operations.
@@ -207,9 +215,9 @@ export interface OcrConfig {
207
215
  abortSignal?: AbortSignal | null;
208
216
  }
209
217
  /**
210
- * Configuration options for the OfficeParser.
218
+ * Configuration options shared across every input format.
211
219
  */
212
- export interface OfficeParserConfig {
220
+ export interface CommonOfficeParserConfig {
213
221
  /**
214
222
  * @deprecated Use `onWarning` instead.
215
223
  * Flag to show all the logs to console in case of an error irrespective of your own handling.
@@ -361,6 +369,44 @@ export interface OfficeParserConfig {
361
369
  */
362
370
  decompressionLimits?: DecompressionLimits;
363
371
  }
372
+ /**
373
+ * Format-specific options for HTML (and XHTML/EPUB, which parse through the same code path).
374
+ *
375
+ * Note there is deliberately no `MdParserConfig`: the Markdown parser populates its
376
+ * dialect-provenance metadata (e.g. `AdmonitionMetadata.sourceSyntax`) unconditionally because
377
+ * doing so costs nothing and changes no existing field's value, so it has nothing to configure.
378
+ * An empty placeholder interface would be worse than useless here - `interface X {}` accepts any
379
+ * non-nullish value in TypeScript, so `mdParserConfig: 5` would type-check.
380
+ */
381
+ export interface HtmlParserConfig {
382
+ /**
383
+ * Preserve source HTML attributes that no typed metadata field consumed, on
384
+ * `OfficeContentNode.htmlAttributes`, so they can be replayed on generation.
385
+ *
386
+ * Off by default: with it off nothing is populated, so the AST is byte-identical to previous
387
+ * releases, and the attribute-replay surface stays something a consumer opts into rather than
388
+ * something switched on for every existing caller. Captured values are sanitized on the way in
389
+ * *and* on the way out - see `BaseContentNode.htmlAttributes`.
390
+ *
391
+ * Defaults to false.
392
+ */
393
+ preserveAttributes?: boolean;
394
+ }
395
+ /**
396
+ * Maps an input format string to its corresponding format-specific parser configuration, mirroring
397
+ * `GeneratorSpecificConfig<D>` on the generator side. Unlike the generator side, the input format is
398
+ * usually runtime-detected rather than known statically at the `parseOffice()` call site, so this
399
+ * mainly exists for internal typing/extensibility rather than compile-time narrowing per call.
400
+ */
401
+ type ParserSpecificConfig<F extends string> = F extends 'html' | 'epub' ? {
402
+ htmlParserConfig?: HtmlParserConfig;
403
+ } : Partial<{
404
+ htmlParserConfig: HtmlParserConfig;
405
+ }>;
406
+ /**
407
+ * Configuration options for the OfficeParser.
408
+ */
409
+ export type OfficeParserConfig<F extends string = string> = CommonOfficeParserConfig & ParserSpecificConfig<F>;
364
410
  /**
365
411
  * Limits applied to ZIP archive decompression.
366
412
  */
@@ -377,6 +423,28 @@ export interface DecompressionLimits {
377
423
  * Default is 10000.
378
424
  */
379
425
  maxZipEntries?: number;
426
+ /**
427
+ * Maximum number of table cells materialized from a single document.
428
+ *
429
+ * ODF encodes runs of identical cells and rows with `table:number-columns-repeated` and
430
+ * `table:number-rows-repeated` rather than repeating the markup, so a few hundred bytes of XML
431
+ * can ask the parser to build an arbitrary number of nodes - and because the two multiply, a
432
+ * row repeat times a column repeat compounds it. The ZIP limits above cannot catch this: the
433
+ * XML is tiny before decompression and the expansion happens afterwards, while building the
434
+ * AST.
435
+ *
436
+ * Real documents are nowhere near this. The repeat counts LibreOffice writes are large
437
+ * (`number-rows-repeated="1048566"` is routine) but they sit on *empty* trailing runs, which
438
+ * are skipped for spreadsheets; the bundled fixtures top out around 350 cells.
439
+ *
440
+ * On reaching the limit the parser stops materializing further cells, emits a
441
+ * `TABLE_CELL_LIMIT_EXCEEDED` warning, and returns what it has rather than throwing, so a
442
+ * genuinely enormous sheet still yields usable output. Raise it if you routinely process
443
+ * spreadsheets larger than this; note the memory cost scales with it.
444
+ *
445
+ * Default is 1000000.
446
+ */
447
+ maxTableCells?: number;
380
448
  }
381
449
  /**
382
450
  * A fully-populated parser configuration containing all options.
@@ -401,7 +469,7 @@ export interface OfficeIssue {
401
469
  /**
402
470
  * The result of a document conversion operation.
403
471
  */
404
- type ConversionValue<D extends UniversalGeneratorFormat> = D extends 'pdf' ? Uint8Array | string : D extends 'chunks' ? OfficeChunk[] : D extends 'csv' ? string | Uint8Array : string;
472
+ type ConversionValue<D extends UniversalGeneratorFormat> = D extends 'pdf' ? Uint8Array | string : D extends 'chunks' ? OfficeChunk[] : D extends 'csv' ? string | Uint8Array : D extends 'epub' ? Uint8Array : string;
405
473
  export interface ConversionResult<D extends UniversalGeneratorFormat> {
406
474
  /** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
407
475
  value: ConversionValue<D>;
@@ -411,7 +479,7 @@ export interface ConversionResult<D extends UniversalGeneratorFormat> {
411
479
  /**
412
480
  * Universal formats supported by all source types for generation.
413
481
  */
414
- export type UniversalGeneratorFormat = 'text' | 'md' | 'html' | 'pdf' | 'csv' | 'rtf' | 'chunks';
482
+ export type UniversalGeneratorFormat = 'text' | 'md' | 'html' | 'pdf' | 'csv' | 'rtf' | 'chunks' | 'epub';
415
483
  /**
416
484
  * Allowed destination formats for a given source type.
417
485
  * Currently, all generators are universal across all source formats.
@@ -423,6 +491,56 @@ export type SupportedDestination<_T extends SupportedFileType = SupportedFileTyp
423
491
  /**
424
492
  * Common configuration options for all generators.
425
493
  */
494
+ /**
495
+ * Per-field overrides for the metadata written into generated output.
496
+ *
497
+ * Field names mirror `OfficeMetadata` so the same vocabulary describes what was parsed and what
498
+ * gets written. Only the fields generators can actually represent are listed; arbitrary
499
+ * caller-defined entries go in `custom`.
500
+ *
501
+ * **Not every format can represent every field.** HTML (`<meta>`) and Markdown (YAML frontmatter)
502
+ * accept anything; EPUB's OPF is a closed Dublin Core vocabulary and RTF's `\info` group has a
503
+ * fixed set of control words, so a `custom` entry has nowhere to go in those. Rather than
504
+ * silently dropping it, generators report the loss through `onWarning`
505
+ * (`OfficeWarningType.MetadataNotRepresentable`) and continue.
506
+ */
507
+ export interface MetadataOverrides {
508
+ /** Document title. */
509
+ title?: string;
510
+ /** Document author. */
511
+ author?: string;
512
+ /** Description/comments. */
513
+ description?: string;
514
+ /** Subject/topic. */
515
+ subject?: string;
516
+ /** Keywords. */
517
+ keywords?: string;
518
+ /** User who last modified the document. */
519
+ lastModifiedBy?: string;
520
+ /** Creation date. */
521
+ created?: Date;
522
+ /**
523
+ * Last modification date.
524
+ *
525
+ * EPUB writes it as the required `dcterms:modified` property and as the mtime on every zip
526
+ * entry. When unset, the source document's own `metadata.modified` is used, falling back to
527
+ * the current time only if the document has none.
528
+ */
529
+ modified?: Date;
530
+ /**
531
+ * Language tag (e.g. `'en'`, `'de-DE'`). Written as EPUB `dc:language` and HTML `lang`.
532
+ */
533
+ language?: string;
534
+ /**
535
+ * Arbitrary caller-defined key/value pairs, kept in their own bucket rather than mixed in
536
+ * beside the named fields above: with a bare index signature a typo like `titel` would
537
+ * silently become a custom entry instead of a compile error.
538
+ *
539
+ * Written where the format allows it (HTML `<meta name="custom:KEY">`, Markdown frontmatter);
540
+ * reported via `onWarning` where it does not (EPUB, RTF).
541
+ */
542
+ custom?: Record<string, string | number | boolean | Date>;
543
+ }
426
544
  export interface CommonGeneratorConfig {
427
545
  /**
428
546
  * Callback called for every node during generation.
@@ -491,6 +609,29 @@ export interface CommonGeneratorConfig {
491
609
  * Defaults to false.
492
610
  */
493
611
  renderMetadata?: boolean;
612
+ /**
613
+ * Overrides for the document metadata written into the generated output, applied on top of
614
+ * `ast.metadata`.
615
+ *
616
+ * Merged **per field**, so setting only `modified` leaves the parsed title, author, and
617
+ * everything else intact. Every field is optional; an omitted field keeps the source
618
+ * document's value.
619
+ *
620
+ * These are output overrides only - `ast.metadata` itself is never mutated, so the same AST
621
+ * can be generated repeatedly with different metadata.
622
+ *
623
+ * @example Set the modification date written into the output
624
+ * ```typescript
625
+ * await ast.to('epub', { metadataOverrides: { modified: new Date('2024-01-01T00:00:00Z') } });
626
+ * ```
627
+ * @example Rebrand the output without touching the parsed document
628
+ * ```typescript
629
+ * await ast.to('html', {
630
+ * metadataOverrides: { title: 'Q4 Report', author: 'Acme Inc', custom: { department: 'Finance' } },
631
+ * });
632
+ * ```
633
+ */
634
+ metadataOverrides?: MetadataOverrides;
494
635
  /**
495
636
  * Whether to ignore the built-in default style mappings (e.g. "Heading 1" -> h1).
496
637
  * Set to true if you want full control over style mapping.
@@ -605,7 +746,7 @@ export type DeepRequired<T> = T extends Function | Date | Buffer | RegExp ? T :
605
746
  * `chunksConfig` is typed as `ChunkingConfig` directly (not DeepRequired) because
606
747
  * it is a discriminated union whose members cannot be uniformly deep-required.
607
748
  */
608
- export type FullGeneratorConfig = DeepRequired<CommonGeneratorConfig & {
749
+ export type FullGeneratorConfig = DeepRequired<Omit<CommonGeneratorConfig, 'metadataOverrides'> & {
609
750
  htmlConfig: HtmlGeneratorConfig;
610
751
  mdConfig: MdGeneratorConfig;
611
752
  pdfConfig: PdfGeneratorConfig;
@@ -614,6 +755,7 @@ export type FullGeneratorConfig = DeepRequired<CommonGeneratorConfig & {
614
755
  rtfConfig: RtfGeneratorConfig;
615
756
  }> & {
616
757
  chunksConfig: ChunkingConfig;
758
+ metadataOverrides: MetadataOverrides;
617
759
  };
618
760
  /**
619
761
  * Configuration options for granular raw HTML injections.
@@ -628,15 +770,64 @@ export interface HtmlInjectionConfig {
628
770
  /** Raw HTML injected immediately before the closing </body> tag */
629
771
  bodyEnd?: string;
630
772
  }
773
+ /**
774
+ * Granular control over which parts of the full HTML "document envelope" are emitted.
775
+ * Shorthand: `standalone: true` == every part on (a complete document); `standalone: false` ==
776
+ * every part off (a bare content fragment). When an object is passed, any field you omit
777
+ * defaults to its "on" (standalone) value.
778
+ */
779
+ export interface StandaloneConfig {
780
+ /**
781
+ * Wrap the output in `<!DOCTYPE html><html><head>…</head><body>…</body></html>`.
782
+ * When false, only the inner content fragment is emitted. Defaults to true.
783
+ */
784
+ document?: boolean;
785
+ /**
786
+ * Emit `<title>` and `<meta>` tags (author, description, dates, custom properties) in the head.
787
+ * Only meaningful when `document` is true. Defaults to true.
788
+ */
789
+ metaTags?: boolean;
790
+ /**
791
+ * How the library's built-in CSS is delivered:
792
+ * - `'full'` — the complete premium stylesheet using global selectors (`body`, `h1`, `table`, …).
793
+ * This is what `standalone: true` has always emitted.
794
+ * - `'scoped'` — the same styling, scoped under the fragment's container via CSS `@scope` so it
795
+ * cannot leak into a host page's own styles. Requires a modern browser engine (Chrome 118+,
796
+ * Safari 17.4+, Firefox 128+).
797
+ * - `'none'` — no stylesheet is emitted; the host page (or EPUB reader, or rich-text editor)
798
+ * supplies its own styling.
799
+ * The boolean shorthand for `standalone` maps `true` → `'full'`, `false` → `'none'`.
800
+ * Defaults to `'full'`.
801
+ */
802
+ styles?: 'full' | 'scoped' | 'none';
803
+ /**
804
+ * Emit injected `<script>` tags: the Chart.js loader (when `includeCharts` is true and charts
805
+ * are present) and the spreadsheet interactivity script. Defaults to true.
806
+ */
807
+ scripts?: boolean;
808
+ /**
809
+ * Apply `injections.headStart` / `injections.headEnd`. Only meaningful when `document` is true
810
+ * (there is no `<head>` to inject into otherwise). Defaults to true.
811
+ */
812
+ headInjections?: boolean;
813
+ /**
814
+ * Apply `injections.bodyStart` / `injections.bodyEnd`. Applies even when generating a bare
815
+ * fragment (`document: false`), since these wrap body *content*, not the document shell.
816
+ * Defaults to true.
817
+ */
818
+ bodyInjections?: boolean;
819
+ }
631
820
  /**
632
821
  * Configuration options for HTML generation.
633
822
  */
634
823
  export interface HtmlGeneratorConfig {
635
824
  /**
636
825
  * Whether to wrap the output in a full HTML document structure (e.g., <html>, <head>, etc.).
826
+ * Pass an object instead of a boolean for granular control over individual parts of the
827
+ * envelope (document shell, meta tags, styles, scripts, injections) - see `StandaloneConfig`.
637
828
  * Defaults to true.
638
829
  */
639
- standalone?: boolean;
830
+ standalone?: boolean | StandaloneConfig;
640
831
  /**
641
832
  * URL for the Chart.js library to use when 'includeCharts' is true.
642
833
  * Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
@@ -811,12 +1002,81 @@ export interface CsvGeneratorConfig {
811
1002
  */
812
1003
  columnDelimiter?: string;
813
1004
  }
1005
+ /**
1006
+ * Named Markdown dialect presets for `MarkdownDialectConfig`/`MdGeneratorConfig.dialect`.
1007
+ * `'extended'` is officeParser's own kitchen-sink default and reproduces this library's
1008
+ * historical output exactly (every feature on, GitHub-style admonitions).
1009
+ */
1010
+ export type MarkdownDialectPreset = 'extended' | 'github' | 'gitlab' | 'obsidian' | 'pandoc' | 'commonmark';
1011
+ /**
1012
+ * Granular control over which native Markdown syntax the generator emits for constructs that
1013
+ * differ across real-world dialects (e.g. GitHub's `> [!NOTE]` vs GitLab's `:::note` vs Pandoc's
1014
+ * `::: {.note}` admonitions). Shorthand: pass a `MarkdownDialectPreset` string for a named target;
1015
+ * pass an object for granular control. Any field you omit from the object form falls back to the
1016
+ * preset named by `extends` (default `'extended'`) - **not** to whatever preset may have been
1017
+ * ambient before, since config merging replaces the whole field rather than layering on top of it.
1018
+ */
1019
+ export interface MarkdownDialectConfig {
1020
+ /** Base preset any omitted field inherits from. Defaults to 'extended'. */
1021
+ extends?: MarkdownDialectPreset;
1022
+ /** Admonition syntax: GitHub `> [!NOTE]`, GitLab `:::note`, Pandoc `::: {.note}`, or `'none'`
1023
+ * to degrade to a plain bold-labeled blockquote with no special marker. */
1024
+ admonitions?: 'github' | 'gitlab' | 'pandoc' | 'none';
1025
+ /** Markdown Extra/Pandoc-style `Term\n: Description` definition lists. */
1026
+ definitionLists?: boolean;
1027
+ /** `[^id]` footnote references/definitions. When false, note content is inlined as a
1028
+ * parenthetical right at the reference point instead of using footnote syntax. */
1029
+ footnotes?: boolean;
1030
+ /** Pandoc-style `[@citekey]` citations. When false, emits `[citekey]` (brackets, no `@`). */
1031
+ citations?: boolean;
1032
+ /** Obsidian-style `[[Page]]`/`[[Page|Alias]]` wikilinks. When false, falls back to a plain
1033
+ * `[text](url)` link using the same target. */
1034
+ wikilinks?: boolean;
1035
+ /** Inline `$...$`/block `$$...$$` math delimiters, or `'none'` for bare LaTeX text. */
1036
+ math?: 'dollar' | 'none';
1037
+ /** Pandoc-style `{width=50% .centered}` attribute lists after images/tables. */
1038
+ attributeLists?: boolean;
1039
+ /** GFM `~~text~~` strikethrough (not part of base CommonMark). */
1040
+ strikethrough?: boolean;
1041
+ /** Unordered list bullet character. */
1042
+ bulletListMarker?: '-' | '*' | '+';
1043
+ /** Ordered list marker punctuation. */
1044
+ orderedListMarker?: '.' | ')';
1045
+ /** Emphasis delimiter style for bold/italic. */
1046
+ emphasisMarker?: 'asterisk' | 'underscore';
1047
+ /** Table syntax: native GFM pipe tables, or forced HTML `<table>` (required for strict
1048
+ * CommonMark, which has no table syntax of its own). */
1049
+ tables?: 'native' | 'html';
1050
+ }
1051
+ /**
1052
+ * Granular control over when the Markdown generator falls back to raw HTML tags for features
1053
+ * standard Markdown can't express natively. Shorthand: `true`/`false` (via
1054
+ * `MdGeneratorConfig.fallbackToHtml`) turns every part on/off at once; pass an object instead to
1055
+ * control them independently. Omitted object fields default to on, matching the boolean shorthand.
1056
+ */
1057
+ export interface FallbackToHtmlConfig {
1058
+ /** Underline/subscript/superscript via `<u>`/`<sub>`/`<sup>`. */
1059
+ textFormatting?: boolean;
1060
+ /** Heading/paragraph text alignment via `<div style="text-align:...">`. */
1061
+ alignment?: boolean;
1062
+ /** Internal-link/heading `<a id>`/`<a name>` anchor tags. */
1063
+ anchors?: boolean;
1064
+ /** Nested-table and merged-cell (colspan/rowspan) HTML `<table>` fallback. */
1065
+ tables?: boolean;
1066
+ /** YouTube embed `<div data-youtube-video>` vs. a plain link. */
1067
+ embeds?: boolean;
1068
+ /** Multi-line table cell content joined with `<br>` instead of a space. */
1069
+ cellLineBreaks?: boolean;
1070
+ }
814
1071
  /**
815
1072
  * Configuration options for Markdown generation.
816
1073
  */
817
1074
  export interface MdGeneratorConfig {
818
1075
  /**
819
1076
  * Whether to fallback to HTML tags for features not supported by standard Markdown.
1077
+ * Pass an object instead of a boolean for granular control over individual parts (text
1078
+ * formatting, alignment, anchors, tables, embeds, cell line breaks) - see
1079
+ * `FallbackToHtmlConfig`. Omitted object fields default to on, matching `true`.
820
1080
  *
821
1081
  * Markdown has limited support for complex document structures. This flag controls how
822
1082
  * the generator handles features that cannot be represented in pure Markdown:
@@ -835,7 +1095,14 @@ export interface MdGeneratorConfig {
835
1095
  *
836
1096
  * Defaults to true.
837
1097
  */
838
- fallbackToHtml?: boolean;
1098
+ fallbackToHtml?: boolean | FallbackToHtmlConfig;
1099
+ /**
1100
+ * Target Markdown dialect for generation - which native syntax to emit for constructs that
1101
+ * differ across real-world targets (GitHub/GitLab/Obsidian/Pandoc/strict CommonMark). See
1102
+ * `MarkdownDialectConfig` for the full per-feature field list. Defaults to `'extended'`
1103
+ * (officeParser's own historical kitchen-sink behavior, unchanged from prior versions).
1104
+ */
1105
+ dialect?: MarkdownDialectPreset | MarkdownDialectConfig;
839
1106
  }
840
1107
  /**
841
1108
  * Configuration options for plain text generation.
@@ -848,11 +1115,24 @@ export interface TextGeneratorConfig {
848
1115
  newlineDelimiter?: string;
849
1116
  /**
850
1117
  * Whether to attempt to preserve the original document layout.
851
- * If true, tables will be rendered with separators and aligned columns.
852
- * If false, output will be a flat stream of text nodes.
853
- * Defaults to false.
1118
+ * If true, tables are rendered with separators and aligned columns, and list items get their
1119
+ * markers and indentation.
1120
+ * If false, output is a flat stream of text nodes (cells are tab-separated).
1121
+ * Defaults to **true**.
854
1122
  */
855
1123
  preserveLayout?: boolean;
1124
+ /**
1125
+ * Whether to append the collected footnotes/endnotes as a trailing `--- Notes ---` section.
1126
+ * Set false to omit it when you want only the document body; the notes are still parsed and
1127
+ * remain available on the AST, they are simply not rendered into the text output.
1128
+ *
1129
+ * Note this differs from the parser's `ignoreNotes`, which discards notes at parse time so they
1130
+ * never reach the AST at all. Use this when you want the AST to keep them but the text output
1131
+ * to leave them out.
1132
+ *
1133
+ * Defaults to true.
1134
+ */
1135
+ renderNotes?: boolean;
856
1136
  }
857
1137
  /**
858
1138
  * The strategy used for chunking a document for RAG pipelines.
@@ -1057,11 +1337,11 @@ export interface OfficeChunk {
1057
1337
  /**
1058
1338
  * Supported file types for parsing.
1059
1339
  */
1060
- export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods' | 'pdf' | 'rtf' | 'md' | 'html' | 'csv';
1340
+ export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods' | 'pdf' | 'rtf' | 'md' | 'html' | 'csv' | 'epub';
1061
1341
  /**
1062
1342
  * Types of content nodes in the AST.
1063
1343
  */
1064
- export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break' | 'code' | 'comment' | 'header' | 'footer' | 'slideMaster';
1344
+ export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break' | 'code' | 'comment' | 'header' | 'footer' | 'slideMaster' | 'embed' | 'admonition' | 'definitionList' | 'definitionTerm' | 'definitionDescription';
1065
1345
  /**
1066
1346
  * Supported MIME types for attachments.
1067
1347
  */
@@ -1255,6 +1535,10 @@ export interface ListMetadata {
1255
1535
  style?: string;
1256
1536
  /** Unique anchor IDs for internal linking. */
1257
1537
  anchorIds?: string[];
1538
+ /** True when this list item is a GFM task-list item (checkbox), regardless of checked state. */
1539
+ isTask?: boolean;
1540
+ /** Checked state for a task-list item. Only meaningful when isTask is true. */
1541
+ checked?: boolean;
1258
1542
  }
1259
1543
  /**
1260
1544
  * Metadata for a table cell (primarily used in Excel/spreadsheet parsing).
@@ -1294,6 +1578,11 @@ export interface CellMetadata {
1294
1578
  export interface TableMetadata {
1295
1579
  /** Unique anchor IDs for internal linking. */
1296
1580
  anchorIds?: string[];
1581
+ /**
1582
+ * Layout alignment of the table on the page (e.g. inscript-editor's `CustomTable`).
1583
+ * @example 'center'
1584
+ */
1585
+ align?: 'left' | 'center' | 'right';
1297
1586
  }
1298
1587
  /**
1299
1588
  * Metadata for a chart node in the document.
@@ -1334,6 +1623,45 @@ export interface ImageMetadata {
1334
1623
  url?: string;
1335
1624
  /** Unique anchor IDs for internal linking. */
1336
1625
  anchorIds?: string[];
1626
+ /**
1627
+ * Display width of the image (e.g. inscript-editor's `CustomImage`), as a CSS length or percentage.
1628
+ * @example "50%"
1629
+ */
1630
+ width?: string;
1631
+ /**
1632
+ * Layout alignment of the image (e.g. inscript-editor's `CustomImage`).
1633
+ * @example 'center'
1634
+ */
1635
+ align?: 'left' | 'center' | 'right';
1636
+ }
1637
+ /**
1638
+ * Metadata for an embedded external media node (e.g. a YouTube video).
1639
+ * Markdown has no native syntax for this - see `MarkdownGenerator`'s `embed` case.
1640
+ */
1641
+ export interface EmbedMetadata {
1642
+ /** The kind of embed. Only 'youtube' is supported today; the shape is generic for future providers. */
1643
+ embedType: 'youtube';
1644
+ /** The provider-specific video ID (e.g. the 11-character YouTube video ID). */
1645
+ videoId: string;
1646
+ /** The original/canonical URL of the embedded media, if known. */
1647
+ url?: string;
1648
+ /** Display width, as a CSS length or percentage. */
1649
+ width?: string;
1650
+ /** Layout alignment of the embed. */
1651
+ align?: 'left' | 'center' | 'right';
1652
+ }
1653
+ /**
1654
+ * Metadata for an admonition/alert node (e.g. GitHub's `> [!NOTE]` or GLFM's `:::note`).
1655
+ * `MarkdownParser` accepts both syntaxes (and generates either, plus Pandoc's `::: {.note}`,
1656
+ * depending on `MdGeneratorConfig.dialect`). Children are block content (paragraphs) wrapped by
1657
+ * the admonition.
1658
+ */
1659
+ export interface AdmonitionMetadata {
1660
+ admonitionType: 'note' | 'tip' | 'important' | 'warning' | 'caution';
1661
+ /** Optional custom title; falls back to the type label. */
1662
+ title?: string;
1663
+ /** Which concrete input syntax produced this node. Always populated by the parser. */
1664
+ sourceSyntax?: 'github' | 'gitlab';
1337
1665
  }
1338
1666
  /**
1339
1667
  * Metadata for PDF page nodes.
@@ -1364,6 +1692,25 @@ export interface TextMetadata {
1364
1692
  * - 'external': Link to an external URL
1365
1693
  */
1366
1694
  linkType?: 'internal' | 'external';
1695
+ /**
1696
+ * When set, this text is an abbreviation and this is its full-form expansion,
1697
+ * rendered as `<abbr title="...">`. Populated from Markdown Extra's
1698
+ * `*[HTML]: Hypertext Markup Language` syntax or an HTML `<abbr>` tag.
1699
+ */
1700
+ abbreviationTitle?: string;
1701
+ /**
1702
+ * When set, this text is a Pandoc/MultiMarkdown-style citation reference
1703
+ * (`[@citekey]`), and this is the bare citekey (e.g. "smith2024"). Bibliography
1704
+ * resolution (author/year display, .bib management) is left to the consuming app.
1705
+ */
1706
+ citationKey?: string;
1707
+ /**
1708
+ * True when this is an Obsidian-style wikilink (`[[page]]` / `[[page|alias]]`).
1709
+ * `link` holds the bare page name and `linkType` is always 'internal'; the
1710
+ * per-workspace enable/disable toggle lives in markdownwriter, not here -
1711
+ * officeParser always parses/generates the syntax.
1712
+ */
1713
+ wikilink?: boolean;
1367
1714
  }
1368
1715
  /**
1369
1716
  * Metadata for note nodes (footnotes/endnotes).
@@ -1417,6 +1764,12 @@ export interface CodeMetadata {
1417
1764
  language?: string;
1418
1765
  /** Unique anchor IDs for internal linking. */
1419
1766
  anchorIds?: string[];
1767
+ /**
1768
+ * When set, this node is a LaTeX math expression rather than a code block. `node.text`
1769
+ * holds the bare LaTeX (delimiters excluded); 'inline' round-trips as `$...$`,
1770
+ * 'block' as `$$...$$`. Matches inscript-editor's math node (Roadmap Step 11.5).
1771
+ */
1772
+ math?: 'inline' | 'block';
1420
1773
  }
1421
1774
  /**
1422
1775
  * Metadata for a comment/annotation.
@@ -1436,7 +1789,7 @@ export interface HeaderFooterMetadata {
1436
1789
  /**
1437
1790
  * Union type for content metadata.
1438
1791
  */
1439
- export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | CommentMetadata | HeaderFooterMetadata | TableMetadata | undefined;
1792
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | CommentMetadata | HeaderFooterMetadata | TableMetadata | EmbedMetadata | AdmonitionMetadata | undefined;
1440
1793
  /**
1441
1794
  * Represents a node in the document content tree.
1442
1795
  * This is the core building block of the parsed document structure.
@@ -1511,6 +1864,23 @@ export interface BaseContentNode {
1511
1864
  * @example "<w:p><w:r><w:t>Hello</w:t></w:r></w:p>" for DOCX
1512
1865
  */
1513
1866
  rawContent?: string;
1867
+ /**
1868
+ * Source HTML attributes that no typed metadata field consumed, preserved for round-trip
1869
+ * fidelity (e.g. a `data-*` attribute an editor round-trips through officeParser).
1870
+ *
1871
+ * Only populated by the HTML/XHTML parser, only for elements it recognises, and only when
1872
+ * `htmlParserConfig.preserveAttributes` is enabled - so by default this is always absent.
1873
+ *
1874
+ * Sanitized on both legs, since an AST can also be constructed programmatically rather than
1875
+ * parsed: event handlers (`on*`) and `srcdoc` are never carried, URL-bearing attributes go
1876
+ * through the same URL sanitizer as typed fields, and every value is escaped on output. A
1877
+ * typed field always wins over a same-named entry here.
1878
+ *
1879
+ * Ignored by the non-HTML generators (Markdown, RTF, CSV, text, chunking) by design - these
1880
+ * are HTML attributes and have no meaning in those targets.
1881
+ * @example { 'data-tracking-id': 'abc123', 'class': 'lead' }
1882
+ */
1883
+ htmlAttributes?: Record<string, string>;
1514
1884
  }
1515
1885
  /**
1516
1886
  * Represents a node in the document content tree.
@@ -1598,6 +1968,21 @@ export type OfficeContentNode = BaseContentNode & ({
1598
1968
  } | {
1599
1969
  type: 'slideMaster';
1600
1970
  metadata?: SlideMetadata;
1971
+ } | {
1972
+ type: 'embed';
1973
+ metadata?: EmbedMetadata;
1974
+ } | {
1975
+ type: 'admonition';
1976
+ metadata?: AdmonitionMetadata;
1977
+ } | {
1978
+ type: 'definitionList';
1979
+ metadata?: undefined;
1980
+ } | {
1981
+ type: 'definitionTerm';
1982
+ metadata?: undefined;
1983
+ } | {
1984
+ type: 'definitionDescription';
1985
+ metadata?: undefined;
1601
1986
  });
1602
1987
  /**
1603
1988
  * Structured information extracted from a chart.
@@ -1840,14 +2225,36 @@ export interface OfficeParserAST {
1840
2225
  /** Any warnings or non-fatal issues encountered during parsing. */
1841
2226
  warnings: OfficeIssue[];
1842
2227
  /**
1843
- * @deprecated Use `.to('text')` instead.
1844
- * Note: This method is synchronous, while the new `.to()` method is asynchronous.
2228
+ * @deprecated Use `.to('text')` instead. This method is the older renderer and takes no
2229
+ * configuration; `.to('text')` produces the same content and lets you configure the rest.
2230
+ *
2231
+ * Converts the entire AST to plain text, flattening the document structure and stripping all
2232
+ * formatting, metadata, and structure. Text is joined using `config.newlineDelimiter`
2233
+ * (default: `'\n'`).
2234
+ *
2235
+ * **Migrating.** `.to('text')` is asynchronous and configurable. At its defaults it emits
2236
+ * everything this method emits, verified across every bundled fixture in all 12 supported
2237
+ * formats in both layout modes: no word produced here is missing there. It also renders merged
2238
+ * table cells correctly, where this method glues them (`OneThree` vs `One Three`).
2239
+ *
2240
+ * Where the two differ is configuration, not capability. Notes and image placeholders are
2241
+ * emitted by default but are switchable; this method emits neither and offers no way to ask
2242
+ * for them. Layout is likewise a knob rather than a fixed behavior:
1845
2243
  *
1846
- * Converts the entire AST to plain text.
1847
- * This method flattens the document structure and returns just the text content,
1848
- * stripping out all formatting, metadata, and structure.
2244
+ * ```typescript
2245
+ * // Default: aligned table grids, list markers, notes, image placeholders.
2246
+ * const { value } = await ast.to('text');
2247
+ *
2248
+ * // Deliberate opt-out - closest to this method's shape.
2249
+ * const { value } = await ast.to('text', {
2250
+ * includeImages: false,
2251
+ * textConfig: { preserveLayout: false, renderNotes: false },
2252
+ * });
2253
+ * ```
1849
2254
  *
1850
- * The text is concatenated using the delimiter specified in `config.newlineDelimiter` (default: '\n').
2255
+ * Spreadsheets (CSV/ODS/XLSX) are unaffected by `preserveLayout`, since it governs
2256
+ * `table`/`list` nodes rather than `sheet`/`row`/`cell`; there the default aligned grid is the
2257
+ * most faithful rendering.
1851
2258
  *
1852
2259
  * @returns A plain text representation of the document
1853
2260
  * @example
package/dist/types.js CHANGED
@@ -43,6 +43,8 @@ var OfficeErrorType;
43
43
  OfficeErrorType["ZIP_ENTRY_INVALID_SIZE"] = "ZIP_ENTRY_INVALID_SIZE";
44
44
  /** ZIP uncompressed size limit exceeded */
45
45
  OfficeErrorType["ZIP_SIZE_LIMIT_EXCEEDED"] = "ZIP_SIZE_LIMIT_EXCEEDED";
46
+ /** Document element/structure nesting exceeded the safe recursion depth */
47
+ OfficeErrorType["MAX_NESTING_DEPTH_EXCEEDED"] = "MAX_NESTING_DEPTH_EXCEEDED";
46
48
  /** Embedding call timed out */
47
49
  OfficeErrorType["EMBEDDING_TIMEOUT"] = "EMBEDDING_TIMEOUT";
48
50
  })(OfficeErrorType || (exports.OfficeErrorType = OfficeErrorType = {}));
@@ -86,4 +88,10 @@ var OfficeWarningType;
86
88
  OfficeWarningType["WHITESPACE_NODE_SKIPPED"] = "WHITESPACE_NODE_SKIPPED";
87
89
  /** The HTML generator containerWidth option is invalid */
88
90
  OfficeWarningType["INVALID_CONTAINER_WIDTH"] = "INVALID_CONTAINER_WIDTH";
91
+ /** A document's repeated-cell expansion hit the configured cell limit and was truncated */
92
+ OfficeWarningType["TABLE_CELL_LIMIT_EXCEEDED"] = "TABLE_CELL_LIMIT_EXCEEDED";
93
+ /** A metadata override could not be represented in the destination format's vocabulary */
94
+ OfficeWarningType["METADATA_NOT_REPRESENTABLE"] = "METADATA_NOT_REPRESENTABLE";
95
+ /** A styleMap output.tag was not an allowed element name and was ignored */
96
+ OfficeWarningType["INVALID_STYLE_MAP_TAG"] = "INVALID_STYLE_MAP_TAG";
89
97
  })(OfficeWarningType || (exports.OfficeWarningType = OfficeWarningType = {}));