officeparser 7.2.3 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +277 -17
  2. package/dist/OfficeConverter.d.ts +1 -1
  3. package/dist/OfficeConverter.js +3 -0
  4. package/dist/OfficeGenerator.js +4 -0
  5. package/dist/OfficeParser.d.ts +2 -0
  6. package/dist/OfficeParser.js +6 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +5 -2
  9. package/dist/defaults.js +12 -0
  10. package/dist/generators/BaseGenerator.d.ts +34 -1
  11. package/dist/generators/BaseGenerator.js +98 -0
  12. package/dist/generators/CsvGenerator.d.ts +9 -1
  13. package/dist/generators/CsvGenerator.js +28 -16
  14. package/dist/generators/EpubGenerator.d.ts +43 -0
  15. package/dist/generators/EpubGenerator.js +312 -0
  16. package/dist/generators/HtmlGenerator.d.ts +12 -0
  17. package/dist/generators/HtmlGenerator.js +378 -61
  18. package/dist/generators/MarkdownGenerator.d.ts +28 -5
  19. package/dist/generators/MarkdownGenerator.js +432 -51
  20. package/dist/generators/PdfGenerator.js +32 -0
  21. package/dist/generators/RtfGenerator.js +47 -22
  22. package/dist/generators/TextGenerator.js +98 -11
  23. package/dist/index.d.ts +1 -0
  24. package/dist/index.js +1 -0
  25. package/dist/officeparser.browser.d.ts +427 -20
  26. package/dist/officeparser.browser.iife.js +338 -206
  27. package/dist/officeparser.browser.mjs +346 -214
  28. package/dist/officeparser.browser.slim.d.ts +427 -20
  29. package/dist/officeparser.browser.slim.iife.js +346 -214
  30. package/dist/officeparser.browser.slim.mjs +346 -214
  31. package/dist/parsers/EpubParser.d.ts +8 -0
  32. package/dist/parsers/EpubParser.js +217 -0
  33. package/dist/parsers/ExcelParser.js +2 -0
  34. package/dist/parsers/HtmlParser.js +507 -48
  35. package/dist/parsers/MarkdownParser.js +704 -92
  36. package/dist/parsers/OpenOfficeParser.js +128 -20
  37. package/dist/parsers/PdfParser.js +4 -1
  38. package/dist/parsers/PowerPointParser.js +1 -0
  39. package/dist/parsers/WordParser.js +1 -0
  40. package/dist/sbom.cdx.json +1695 -0
  41. package/dist/types.d.ts +427 -20
  42. package/dist/types.js +8 -0
  43. package/dist/utils/configUtils.js +53 -4
  44. package/dist/utils/errorUtils.js +7 -3
  45. package/dist/utils/sanitize.d.ts +139 -0
  46. package/dist/utils/sanitize.js +318 -0
  47. package/dist/utils/xmlUtils.js +2 -2
  48. package/dist/utils/zipUtils.js +76 -26
  49. package/package.json +16 -12
package/README.md CHANGED
@@ -2,9 +2,9 @@
2
2
 
3
3
  A robust, strictly-typed **Node.js and Browser** library for parsing office files into a rich **Abstract Syntax Tree (AST)** and generating high-fidelity output in multiple formats.
4
4
 
5
- **Parses:** [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`odt`](https://en.wikipedia.org/wiki/OpenDocument) · [`odp`](https://en.wikipedia.org/wiki/OpenDocument) · [`ods`](https://en.wikipedia.org/wiki/OpenDocument) · [`pdf`](https://en.wikipedia.org/wiki/PDF) · [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format) · [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values) · [`md`](https://en.wikipedia.org/wiki/Markdown) · [`html`](https://en.wikipedia.org/wiki/HTML)
5
+ **Parses:** [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`odt`](https://en.wikipedia.org/wiki/OpenDocument) · [`odp`](https://en.wikipedia.org/wiki/OpenDocument) · [`ods`](https://en.wikipedia.org/wiki/OpenDocument) · [`pdf`](https://en.wikipedia.org/wiki/PDF) · [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format) · [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values) · [`md`](https://en.wikipedia.org/wiki/Markdown) · [`html`](https://en.wikipedia.org/wiki/HTML) · [`epub`](https://en.wikipedia.org/wiki/EPUB)
6
6
 
7
- **Generates:** `Markdown` · `HTML` · `CSV` · `RTF` · `PDF` · `Plain Text` · `RAG Chunks`
7
+ **Generates:** `Markdown` · `HTML` · `CSV` · `RTF` · `PDF` · `EPUB` · `Plain Text` · `RAG Chunks`
8
8
 
9
9
  [![npm version](https://badge.fury.io/js/officeparser.svg)](https://badge.fury.io/js/officeparser)
10
10
  [![Total Downloads](https://img.shields.io/npm/dt/officeparser.svg)](https://www.npmjs.com/package/officeparser)
@@ -42,6 +42,8 @@ A robust, strictly-typed **Node.js and Browser** library for parsing office file
42
42
  - [Native RAG Chunking](#native-rag-chunking)
43
43
  - [The AST Structure](#the-ast-structure)
44
44
  - [Deep Dive: Document Components](#deep-dive-document-components)
45
+ - [Markdown Dialect Support](#markdown-dialect-support)
46
+ - [EPUB Support](#epub-support)
45
47
  - [Performance Highlights](#performance-highlights)
46
48
  - [Advanced AST Usage](#advanced-ast-usage)
47
49
  - [Configuration Reference](#configuration-reference)
@@ -54,12 +56,14 @@ A robust, strictly-typed **Node.js and Browser** library for parsing office file
54
56
  - [PdfGeneratorConfig](#pdfgeneratorconfig)
55
57
  - [CsvGeneratorConfig](#csvgeneratorconfig)
56
58
  - [TextGeneratorConfig](#textgeneratorconfig)
59
+ - [metadataOverrides](#metadataoverrides)
57
60
  - [OfficeConverterConfig](#officeconverterconfig)
58
61
  - [ChunkingConfig](#chunkingconfig)
59
62
  - [OCR Scheduler & Resource Management](#ocr-scheduler--resource-management)
60
63
  - [Browser Usage](#browser-usage)
61
64
  - [Troubleshooting & Common Issues](#troubleshooting--common-issues)
62
65
  - [Known Limitations](#known-limitations)
66
+ - [Security & Trust Boundary](#security--trust-boundary)
63
67
  - [Contributing](#contributing)
64
68
 
65
69
  ---
@@ -93,6 +97,9 @@ npx officeparser data.xlsx --to=csv --csvDelimiter=";"
93
97
  # Generate RAG chunks
94
98
  npx officeparser document.pdf --to=chunks
95
99
 
100
+ # Convert DOCX to EPUB (--extractAttachments is required to embed images)
101
+ npx officeparser book.docx --extractAttachments --to=epub --output=book.epub
102
+
96
103
  # Overriding file extension mapping
97
104
  npx officeparser my_document --fileType=docx --to=json
98
105
  ```
@@ -106,9 +113,9 @@ npx officeparser my_document --fileType=docx --to=json
106
113
 
107
114
  | Flag | Values | Default | Description |
108
115
  |------|--------|---------|-------------|
109
- | `--to` | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | Output format |
116
+ | `--to` | `json\|text\|md\|html\|csv\|rtf\|pdf\|epub\|chunks` | `json` | Output format |
110
117
  | `--output` | path | — | Write output to a file |
111
- | `--fileType` | `docx\|xlsx\|pptx\|odt\|odp\|ods\|pdf\|rtf\|csv\|md\|html` | — | Explicitly override input file type detection |
118
+ | `--fileType` | `docx\|xlsx\|pptx\|odt\|odp\|ods\|pdf\|rtf\|csv\|md\|html\|epub` | — | Explicitly override input file type detection |
112
119
  | `--ocr` | boolean | `false` | Enable OCR for images |
113
120
  | `--extractAttachments` | boolean | `false` | Extract images/charts as Base64 |
114
121
  | `--ignoreNotes` | boolean | `false` | Ignore footnotes/endnotes/speaker notes |
@@ -126,8 +133,8 @@ npx officeparser my_document --fileType=docx --to=json
126
133
  | `--includeFormatting` | boolean | `true` | Include formatting style map matching |
127
134
  | `--renderMetadata` | boolean | `false` | Render metadata as visible content in the generated output |
128
135
  | `--htmlConfig.containerWidth` | string \| number | `auto` | HTML output container width (e.g. `900px`, `100%`) |
129
- | ~~`--format`~~ | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | **Deprecated.** Use `--to` |
130
- | ~~`--toText`~~ | `true\|false` | `false` | **Deprecated.** Use `--to=text` |
136
+ | ~~`--format`~~ | `json\|text\|md\|html\|csv\|rtf\|pdf\|epub\|chunks` | `json` | **Deprecated.** Use `--to` |
137
+ | ~~`--toText`~~ | `true\|false` | `false` | **Deprecated.** Use `--to=text`, which keeps footnote text and image placeholders by default (both switchable); this flag drops them unconditionally |
131
138
  | ~~`--ocrLanguage`~~ | string | `eng` | **Deprecated.** Use `--ocrConfig.language` |
132
139
  | ~~`--putNotesAtLast`~~ | `true\|false` | `false` | **Deprecated and ignored.** Notes are attached structurally to their nodes. |
133
140
  | ~~`--outputErrorToConsole`~~ | `true\|false` | `false` | **Deprecated.** Use `--verbose` |
@@ -269,14 +276,57 @@ const { value: pdfBytes } = await ast.to('pdf'); // Uint8Array
269
276
 
270
277
  ### `ast.toText()`: Quick Text Extraction
271
278
 
272
- > [!NOTE]
273
- > `toText()` is **synchronous** and deprecated in favour of the async `ast.to('text')`.
274
- > It remains available for backward compatibility.
279
+ > [!WARNING]
280
+ > `toText()` is **synchronous** and deprecated in favour of the async `ast.to('text')`. It remains
281
+ > available for backward compatibility, but it is the older, less capable renderer: it has no
282
+ > configuration at all, so footnote/endnote text and image placeholders are **unconditionally
283
+ > dropped** rather than being something you can ask for. Prefer `.to('text')` for new code.
275
284
 
276
285
  ```js
277
286
  const text = ast.toText(); // synchronous, returns plain string
278
287
  ```
279
288
 
289
+ #### Migrating to `.to('text')`
290
+
291
+ `.to('text')` is asynchronous and configurable. Its defaults render tables as aligned grids, lists
292
+ with markers/indentation, and include notes and image placeholders:
293
+
294
+ ```js
295
+ // Default: aligned table grids, list markers, notes, image placeholders
296
+ const { value } = await ast.to('text');
297
+
298
+ // Deliberate opt-out: the combination closest to toText()'s shape
299
+ const { value } = await ast.to('text', {
300
+ includeImages: false,
301
+ textConfig: { preserveLayout: false, renderNotes: false },
302
+ });
303
+ ```
304
+
305
+ **At its default configuration, `.to('text')` emits everything `toText()` emits.** Verified across
306
+ every bundled fixture in all 12 supported formats, in both layout modes: no word `toText()` produces
307
+ is missing from `.to('text')`. It additionally emits notes and image placeholders, which `toText()`
308
+ never produces, and it renders merged table cells correctly (`toText()` glues a two-cell row into
309
+ `OneThree`, where `.to('text')` gives `One Three`).
310
+
311
+ Notes and images are **configuration, not intrinsic behavior**. They are on by default and you can
312
+ turn them off. The real difference from `toText()` is that they are a choice at all:
313
+
314
+ | | `toText()` | `.to('text')` | governed by |
315
+ |---|---|---|---|
316
+ | Tables | one cell per line | aligned grid, or tab-separated | `textConfig.preserveLayout` (default `true`) |
317
+ | Lists | plain text | markers + indentation, or plain | `textConfig.preserveLayout` (default `true`) |
318
+ | Footnotes/endnotes | never emitted | emitted by default | `textConfig.renderNotes` (default `true`) |
319
+ | Image placeholders | never emitted | emitted by default | `includeImages` (default `true`) |
320
+ | Chart data series | emitted | emitted | n/a |
321
+
322
+ Nothing about `.to('text')` forces the richer output on you. The defaults simply start from the more
323
+ complete document, and the opt-out above gets you back to `toText()`'s shape deliberately rather
324
+ than by having no alternative.
325
+
326
+ Spreadsheets (CSV/ODS/XLSX) are unaffected by `preserveLayout`: it governs `table`/`list` nodes,
327
+ while spreadsheet content is `sheet`/`row`/`cell`. There the default aligned grid is the most
328
+ faithful rendering.
329
+
280
330
  ---
281
331
 
282
332
  ## OfficeGenerator
@@ -306,13 +356,16 @@ const { value: html } = await OfficeGenerator.generate(ast, 'html', {
306
356
  const { value: csv } = await OfficeGenerator.generate(ast, 'csv');
307
357
  ```
308
358
 
309
- **Supported destinations:** `'text'` · `'md'` · `'html'` · `'csv'` · `'rtf'` · `'pdf'` · `'chunks'`
359
+ **Supported destinations:** `'text'` · `'md'` · `'html'` · `'csv'` · `'rtf'` · `'pdf'` · `'epub'` · `'chunks'`
310
360
 
311
361
  > [!NOTE]
312
362
  > **PDF generation** requires the optional `puppeteer` peer dependency:
313
363
  > ```bash
314
364
  > npm install puppeteer
315
365
  > ```
366
+ >
367
+ > **EPUB generation with images** requires `extractAttachments: true` on the parse step that
368
+ > produced the AST — see [EPUB Support](#epub-support).
316
369
 
317
370
  ---
318
371
 
@@ -440,10 +493,10 @@ interface OfficeChunk {
440
493
 
441
494
  ```text
442
495
  OfficeParserAST
443
- ├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | ... (11 formats)
496
+ ├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | 'epub' | ... (12 formats)
444
497
  ├── metadata: { author, title, created, modified, keywords, customProperties, nativeProperties, styleMap, ... }
445
498
  ├── content: [ OfficeContentNode ]
446
- │ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | ...
499
+ │ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | 'admonition' | 'embed' | 'definitionList' | ...
447
500
  │ ├── text: string (concatenated text of node + all descendants)
448
501
  │ ├── children: [ OfficeContentNode ] (recursive structural children)
449
502
  │ ├── notes: [ OfficeContentNode ] (footnotes/endnotes/slide notes attached to this node)
@@ -463,7 +516,7 @@ OfficeParserAST
463
516
  │ └── chartData?: { title, dataSets, labels }
464
517
  ├── warnings: OfficeIssue[] (non-fatal issues from the parsing phase)
465
518
  ├── to(format, config?) (format: 'html'|'md'|'text'|'csv'|'rtf'|'pdf'|'chunks', returns { value, messages })
466
- └── ~~toText()~~ (Deprecated: use .to('text') instead)
519
+ └── ~~toText()~~ (Deprecated: use .to('text'); drops footnotes + image placeholders)
467
520
  ```
468
521
 
469
522
  ### `OfficeIssue`: Warning / Error Object
@@ -596,6 +649,94 @@ console.log(ast.metadata.nativeProperties);
596
649
  // PDF: { Title: 'Report', XMP: { ... } }
597
650
  ```
598
651
 
652
+ ### 8. Admonitions, Embeds & Definition Lists
653
+
654
+ ```text
655
+ Admonition Node (type: 'admonition')
656
+ ├── metadata: { admonitionType: 'note' | 'tip' | 'important' | 'warning' | 'caution', title?: string }
657
+ └── children: [ Paragraph | List | ... ] (block content)
658
+
659
+ Embed Node (type: 'embed')
660
+ └── metadata: { embedType: 'youtube', videoId: string, url?: string, width?: string, align?: string }
661
+
662
+ Definition List Node (type: 'definitionList')
663
+ └── children:
664
+ ├── Definition Term (type: 'definitionTerm')
665
+ └── Definition Description (type: 'definitionDescription')
666
+ ```
667
+
668
+ - `admonition` round-trips through both Markdown (`> [!NOTE]` / `:::note ... :::`) and HTML (`<div class="admonition admonition-note" data-type="note">`)
669
+ - `embed` currently models YouTube videos; HTML round-trips via `<div data-youtube-video="ID">`, Markdown falls back to a raw HTML block or a plain link
670
+ - Abbreviations (`*[HTML]: Hypertext Markup Language`) are stored as `TextMetadata.abbreviationTitle` on the abbreviated text node rather than as a separate node type
671
+
672
+ ---
673
+
674
+ ## Markdown Dialect Support
675
+
676
+ Beyond CommonMark/GFM basics, `MarkdownParser`/`MarkdownGenerator` support an extended dialect aimed at
677
+ full-fidelity round-tripping with rich Markdown editors. Every construct below parses to a first-class
678
+ AST node/metadata field and regenerates back to the canonical syntax shown, so `.md → AST → .md` is
679
+ idempotent and `.md → AST → HTML → AST → .md` survives unchanged.
680
+
681
+ | Feature | Markdown syntax | AST representation |
682
+ |---|---|---|
683
+ | Task lists (GFM) | `- [x] Done` / `- [ ] Todo` | `ListMetadata.isTask` / `.checked` |
684
+ | Admonitions | `> [!NOTE]` (also accepts GLFM `:::note ... :::` on import) | `type: 'admonition'`, `AdmonitionMetadata` |
685
+ | Footnotes | `Text[^1]` + `[^1]: Definition` | `type: 'note'`, keyed by footnote id |
686
+ | Definition lists | `Term\n: Definition` | `type: 'definitionList'` / `'definitionTerm'` / `'definitionDescription'` |
687
+ | Abbreviations | `*[HTML]: Hypertext Markup Language` | `TextMetadata.abbreviationTitle` |
688
+ | Attribute lists | `![alt](img.png){width=50% .centered}` | `ImageMetadata.width` / `.align`, `TableMetadata.align` |
689
+ | Citations | `[@smith2024]` | `TextMetadata.citationKey` |
690
+ | Wikilinks | `[[Page]]` / `[[Page\|Alias]]` | `TextMetadata.wikilink`, `.link`, `.linkType` |
691
+ | Inline/block math | `$E=mc^2$` / `` $$...$$ `` | `TextMetadata.math` (`'inline' \| 'block'`) |
692
+ | Frontmatter arrays | `tags: [a, b]` or `tags: ["a","b"]` | Real array in `metadata.customProperties`/`nativeProperties` |
693
+ | MDX components (import-only) | `<Component prop="x">...</Component>` | Stripped; inner Markdown is kept. Never generated back. |
694
+
695
+ > [!NOTE]
696
+ > MDX/JSX stripping is one-directional (parse-only) — officeParser never authors JSX back into Markdown.
697
+ > Wikilink enable/disable and citekey→bibliography resolution are application-level concerns; officeParser
698
+ > always parses/generates the syntax itself.
699
+
700
+ The same round-trip fidelity extends to HTML, so content saved from a rich-text editor survives a
701
+ save→reload cycle:
702
+
703
+ | HTML attribute | AST field | Notes |
704
+ |---|---|---|
705
+ | `data-width` / `data-align` / inline `style="width:…"` on `<img>` | `ImageMetadata.width` / `.align` | |
706
+ | `data-align` on `<table>` | `TableMetadata.align` | |
707
+ | `colspan` / `rowspan` on `<td>`/`<th>` | `CellMetadata.colSpan` / `.rowSpan` | Previously dropped on HTML import — merged cells now survive a save→reload cycle |
708
+ | `<div data-youtube-video="ID">` / `<iframe src="...youtube.com...">` | `type: 'embed'` | |
709
+ | `<ul data-type="taskList">` / `<li data-checked>` | `ListMetadata.isTask` / `.checked` | |
710
+
711
+ ---
712
+
713
+ ## EPUB Support
714
+
715
+ EPUB files are ZIP archives of XHTML content plus an OPF manifest — `EpubParser` unzips the archive,
716
+ resolves the spine's reading order from `content.opf`, and parses each XHTML document through the
717
+ existing `HtmlParser`, so EPUB content shares the same AST shape (and the same Markdown-dialect
718
+ fidelity above) as every other format. Dublin Core metadata (`dc:title`, `dc:creator`, `dc:description`,
719
+ `dc:subject`, `dc:date`, `dc:publisher`, `dc:language`, `dc:identifier`) maps into `ast.metadata` /
720
+ `ast.metadata.nativeProperties`, and cover art is exposed via `metadata.customProperties.coverImageName`.
721
+
722
+ `EpubGenerator` renders the AST through `HtmlGenerator` and packages the result as a minimal, valid
723
+ EPUB 3 (`mimetype`, `META-INF/container.xml`, an OPF manifest, a nav document, and one XHTML chapter).
724
+
725
+ > [!IMPORTANT]
726
+ > **Pass `extractAttachments: true` when converting to or from EPUB if the document has images.**
727
+ > Without it, the parser never pulls embedded image bytes out of the source document, so there is
728
+ > nothing for the EPUB generator to package — images silently disappear even though everything else
729
+ > converts correctly. Images are packaged as real zip entries (`OEBPS/images/...`) declared in the OPF
730
+ > manifest, not `data:` URIs — most EPUB reading systems do not render `data:` URIs in image `src`.
731
+ >
732
+ > This only matters for the two-step `OfficeParser.parseOffice()` → `OfficeGenerator.generate()` API
733
+ > and the CLI. [`OfficeConverter.convert()`](#officeconverter-one-step-api) enables `extractAttachments`
734
+ > automatically unless you explicitly set `generatorConfig.includeImages: false`.
735
+ >
736
+ > ```bash
737
+ > npx officeparser book.docx --extractAttachments --to=epub --output=book.epub
738
+ > ```
739
+
599
740
  ---
600
741
 
601
742
  ## Performance Highlights
@@ -768,6 +909,7 @@ Options shared by all generator formats. Pass to `OfficeGenerator.generate(ast,
768
909
  | `includeFormatting` | `boolean` | `true` | Include bold/italic/colors/sizes in output |
769
910
  | `generateIds` | `boolean` | `true` | Add slug-based `id` attributes to headings |
770
911
  | `renderMetadata` | `boolean` | `false` | Render title/author as visible header block |
912
+ | `metadataOverrides` | `MetadataOverrides` | `{}` | Override the metadata embedded in the output, merged per field over `ast.metadata` |
771
913
  | `includeImages` | `boolean` | `true` | Include image nodes in output |
772
914
  | `includeCharts` | `boolean` | `true` | Include interactive charts (HTML only) |
773
915
  | `ignoreInternalLinks` | `boolean` | `false` | Strip bookmarks and internal anchors from output |
@@ -852,7 +994,7 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
852
994
 
853
995
  | Option | Type | Default | Description |
854
996
  |--------|------|---------|-------------|
855
- | `standalone` | `boolean` | `true` | Wrap output in a full `<html>` document with CSS |
997
+ | `standalone` | `boolean \| StandaloneConfig` | `true` | Controls the HTML "document envelope" — see below |
856
998
  | `chartJsSrc` | `string` | jsDelivr CDN | URL for the Chart.js library |
857
999
  | `containerWidth` | `string \| number` | `'auto'` | Max width of the content container. Positive number (px), CSS length string (`'900px'`, `'100%'`, `'60vw'`), or `'auto'`. Invalid values fall back to `'auto'` with an `INVALID_CONTAINER_WIDTH` warning |
858
1000
  | `customCss` | `string` | `''` | Raw CSS injected into the `<style>` block; use this to override built-in styles |
@@ -861,6 +1003,49 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
861
1003
  | `injections.bodyStart` | `string` | `''` | Raw HTML injected after `<body>` |
862
1004
  | `injections.bodyEnd` | `string` | `''` | Raw HTML injected before `</body>` |
863
1005
 
1006
+ #### `standalone`: granular envelope control
1007
+
1008
+ `standalone` conflates several independent decisions: whether to emit the `<!doctype>/<html>/<head>/
1009
+ <body>` shell, how CSS is delivered, and whether to inject scripts/meta tags/injections. The boolean
1010
+ shorthand still works — **`true`/omitted turns every part on** (a complete document); **`false` turns
1011
+ every part off** (a bare content fragment, safe to drop into a page you don't control). Pass an
1012
+ object instead for granular control; any field you omit defaults to its "on" (standalone) value:
1013
+
1014
+ | `StandaloneConfig` field | Type | Default | Description |
1015
+ |--------|------|---------|-------------|
1016
+ | `document` | `boolean` | `true` | Wrap in `<!DOCTYPE html><html><head>…</head><body>…</body></html>` |
1017
+ | `metaTags` | `boolean` | `true` | Emit `<title>`/`<meta>` tags. Only meaningful when `document` is true |
1018
+ | `styles` | `'full' \| 'scoped' \| 'none'` | `'full'` | See below |
1019
+ | `scripts` | `boolean` | `true` | Emit the Chart.js CDN loader and spreadsheet-interactivity `<script>` tags |
1020
+ | `headInjections` | `boolean` | `true` | Apply `injections.headStart`/`headEnd`. Only meaningful when `document` is true |
1021
+ | `bodyInjections` | `boolean` | `true` | Apply `injections.bodyStart`/`bodyEnd` — applies even to a bare fragment |
1022
+
1023
+ `styles` controls how the built-in stylesheet is delivered:
1024
+ - **`'full'`** — the complete stylesheet using global selectors (`body`, `h1`, `table`, …). This is
1025
+ what `standalone: true` has always emitted.
1026
+ - **`'scoped'`** — the same styling, scoped under the fragment's own wrapper via CSS `@scope` so it
1027
+ cannot leak onto a host page's elements. Requires a modern engine (Chrome 118+, Safari 17.4+,
1028
+ Firefox 128+); for universal support use `'none'` (bring your own CSS) or `'full'`.
1029
+ - **`'none'`** — no stylesheet at all; the host page (or rich-text editor, or EPUB reader) supplies
1030
+ its own styling.
1031
+
1032
+ ```js
1033
+ // A styled fragment to embed in your own page, without a document shell:
1034
+ await ast.to('html', { htmlConfig: { standalone: { document: false } } });
1035
+
1036
+ // The same, but with styles scoped so they can't leak onto your page's own elements:
1037
+ await ast.to('html', { htmlConfig: { standalone: { document: false, styles: 'scoped' } } });
1038
+
1039
+ // A completely bare fragment (no shell, no styles, no scripts) — e.g. for a rich-text editor:
1040
+ await ast.to('html', { htmlConfig: { standalone: false } });
1041
+ ```
1042
+
1043
+ > [!NOTE]
1044
+ > **Behavior change from `standalone: false`:** previously this emitted a fragment with a *global,
1045
+ > unscoped* `<style>` block. It now emits a genuinely bare fragment (no `<style>` at all), matching
1046
+ > "every part off." If you relied on the old styled-fragment behavior, pass
1047
+ > `{ document: false }` (or `{ document: false, styles: 'full' }`) instead.
1048
+
864
1049
  ### MdGeneratorConfig
865
1050
 
866
1051
  Pass as `mdConfig` inside `GeneratorConfig`.
@@ -906,6 +1091,52 @@ Pass as `textConfig` inside `GeneratorConfig`.
906
1091
  |--------|------|---------|-------------|
907
1092
  | `newlineDelimiter` | `string` | `'\n'` | String inserted between structural blocks |
908
1093
  | `preserveLayout` | `boolean` | `true` | Render tables with aligned columns using whitespace |
1094
+ | `renderNotes` | `boolean` | `true` | Append the collected footnote/endnote section |
1095
+
1096
+ ### metadataOverrides
1097
+
1098
+ Part of the common `GeneratorConfig` (not format-specific). Overrides the metadata embedded in
1099
+ generated output, applied **per field** on top of `ast.metadata`, so setting one field leaves the
1100
+ rest of the parsed metadata intact. `ast.metadata` itself is never mutated, so the same AST can be
1101
+ generated repeatedly with different metadata.
1102
+
1103
+ | Field | Type | Written as |
1104
+ |-------|------|-----------|
1105
+ | `title` | `string` | HTML `<title>`/`<meta>`, EPUB `dc:title`, Markdown frontmatter, RTF `\title` |
1106
+ | `author` | `string` | HTML `<meta name="author">`, EPUB `dc:creator`, frontmatter, RTF `\author` |
1107
+ | `description` | `string` | HTML `<meta name="description">`, EPUB `dc:description`, frontmatter |
1108
+ | `subject` / `keywords` / `lastModifiedBy` | `string` | Where the destination format has a slot |
1109
+ | `created` / `modified` | `Date` | HTML `dcterms.*`, EPUB `dcterms:modified`, frontmatter |
1110
+ | `language` | `string` | EPUB `dc:language` |
1111
+ | `custom` | `Record<string, string \| number \| boolean \| Date>` | HTML `<meta name="custom:KEY">`, Markdown frontmatter |
1112
+
1113
+ ```js
1114
+ // Rebrand the output without touching the parsed document
1115
+ const { value } = await ast.to('html', {
1116
+ metadataOverrides: { title: 'Q4 Report', author: 'Acme Inc', custom: { department: 'Finance' } },
1117
+ });
1118
+ ```
1119
+
1120
+ #### Dates
1121
+
1122
+ `created` and `modified` take a `Date`. When `modified` is unset, officeParser uses the source
1123
+ document's own `ast.metadata.modified`, falling back to the current time only if the document has
1124
+ none. Dates outside the 1980-2099 range representable in a ZIP timestamp are clamped where EPUB
1125
+ writes them onto zip entries.
1126
+
1127
+ ```js
1128
+ const { value } = await ast.to('epub', {
1129
+ metadataOverrides: { modified: new Date('2024-01-01T00:00:00Z') },
1130
+ });
1131
+ ```
1132
+
1133
+ #### Not every format can represent every field
1134
+
1135
+ HTML `<meta>` tags and Markdown frontmatter are open vocabularies and accept anything. EPUB's OPF
1136
+ is a closed Dublin Core vocabulary and RTF's `\info` group has a fixed set of control words, so a
1137
+ `custom` entry has nowhere to go in either. Rather than dropping it silently, those generators
1138
+ report it through `onWarning` (`OfficeWarningType.METADATA_NOT_REPRESENTABLE`) and continue; the
1139
+ named fields still apply.
909
1140
 
910
1141
  ---
911
1142
 
@@ -1028,7 +1259,7 @@ const handleFile = async (event) => {
1028
1259
  const file = event.target.files[0];
1029
1260
  const buffer = await file.arrayBuffer();
1030
1261
  const ast = await OfficeParser.parseOffice(new Uint8Array(buffer));
1031
- console.log(ast.toText());
1262
+ console.log((await ast.to('text')).value);
1032
1263
  };
1033
1264
  ```
1034
1265
 
@@ -1041,7 +1272,7 @@ const handleFile = async (event) => {
1041
1272
  const file = event.target.files[0];
1042
1273
  const buffer = await file.arrayBuffer();
1043
1274
  const ast = await officeParser.parseOffice(new Uint8Array(buffer));
1044
- console.log(ast.toText());
1275
+ console.log((await ast.to('text')).value);
1045
1276
  }
1046
1277
  </script>
1047
1278
  ```
@@ -1076,7 +1307,7 @@ const ast = await officeParser.parseOffice(pdfArrayBuffer, {
1076
1307
  | Node.js process stays alive after finishing | Call `await officeParser.terminateOcr()` at end of script when OCR was used |
1077
1308
  | `"Worker not found"` in browser for PDF | Verify `pdfWorkerSrc` points to `pdf.worker.min.mjs` matching version `6.1.200` |
1078
1309
  | Low OCR accuracy | Verify `ocrConfig.language` matches the document language; quality depends on image resolution |
1079
- | Out of memory on large Excel files | Call `ast.toText()` early and discard the AST object to allow garbage collection |
1310
+ | Out of memory on large Excel files | Call `await ast.to('text')` early and discard the AST object to allow garbage collection |
1080
1311
  | `md`/`html`/`csv` buffer not detected | Add `fileType: 'md'` (or `'html'`, `'csv'`) to config (these formats have no magic bytes) |
1081
1312
  | `IMPROPER_BUFFERS` error | Usually means no file extension and no `fileType` hint was provided for a buffer input |
1082
1313
  | PDF generation fails | Install the optional peer dependency: `npm install puppeteer` |
@@ -1092,6 +1323,35 @@ For a full debugging guide, visit the [Live Documentation](https://harshankur.gi
1092
1323
 
1093
1324
  ---
1094
1325
 
1326
+ ## Security & Trust Boundary
1327
+
1328
+ `officeParser` is a **parsing, generation, and conversion** library. Like any parser, its whole job
1329
+ is to open and interpret files it is handed, and those files may come from an untrusted source (a
1330
+ user upload, an email attachment, a scraped document). A parser that accepts arbitrary documents
1331
+ has a large and inherently open attack surface.
1332
+
1333
+ I do sanitize output and apply hardening where I can: injection escaping across the
1334
+ HTML/CSS/URL/script/CSV/RTF/Markdown sinks, decompression limits, some resource and recursion
1335
+ bounds, and SSRF precautions during PDF rendering. I fix issues as I learn of them (see
1336
+ [CHANGELOG.md](CHANGELOG.md)). But this is **best-effort, not a guarantee.** A document parser of
1337
+ this size will have attack vectors I have not found or have not yet addressed, and no amount of
1338
+ internal hardening makes it safe to feed fully untrusted input without your own precautions.
1339
+
1340
+ **Treat this as garbage in, garbage out.** The library does its best with what you give it, but
1341
+ responsibility for what you feed it, and for the effect a malicious file has on your system, rests
1342
+ with you. If you process files from untrusted sources, sanitize and validate them at your own
1343
+ boundary, and run the parsing in isolation appropriate to your threat model: sandboxing or
1344
+ containerization, memory and time limits, a low-privilege process, and the `abortSignal` and
1345
+ `decompressionLimits` options this library exposes. Do not rely on any single library's hardening
1346
+ as a complete defense.
1347
+
1348
+ I am the sole maintainer, with no security team behind me. I take legitimate reports seriously and
1349
+ will fix what I reasonably can, but I cannot commit to a response or resolution timeline. The
1350
+ software is provided "AS IS" without warranty of any kind (see [LICENSE](LICENSE)). To report an
1351
+ issue privately, see [SECURITY.md](SECURITY.md).
1352
+
1353
+ ---
1354
+
1095
1355
  **npm**: [https://npmjs.com/package/officeparser](https://npmjs.com/package/officeparser)
1096
1356
 
1097
1357
  **github**: [https://github.com/harshankur/officeParser](https://github.com/harshankur/officeParser)
@@ -42,6 +42,6 @@ export declare class OfficeConverter {
42
42
  * });
43
43
  * ```
44
44
  */
45
- static convert<F extends string | Buffer | ArrayBuffer | Uint8Array, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
45
+ static convert<F extends string | Buffer | ArrayBuffer | Uint8Array, T extends SupportedFileType = InferFileTypeFromPath<F>, D extends SupportedDestination<T> = SupportedDestination<T>>(file: F, destination: D, config?: OfficeConverterConfig<D, T>): Promise<ConversionResult<D>>;
46
46
  }
47
47
  export {};
@@ -65,6 +65,9 @@ class OfficeConverter {
65
65
  ...config?.generatorConfig,
66
66
  onWarning: config?.onWarning || config?.generatorConfig?.onWarning,
67
67
  };
68
+ // The spread widens the object to an inferred literal; it is a `GeneratorConfig<D>` by
69
+ // construction (config.generatorConfig is already typed for D, and onWarning is common to
70
+ // every destination), so the cast restates what the types otherwise lose.
68
71
  const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, destination, generatorConfig);
69
72
  result.messages = [...(ast.warnings || []), ...result.messages];
70
73
  return result;
@@ -3,6 +3,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.OfficeGenerator = void 0;
4
4
  const ChunkingGenerator_js_1 = require("./generators/ChunkingGenerator.js");
5
5
  const CsvGenerator_js_1 = require("./generators/CsvGenerator.js");
6
+ const EpubGenerator_js_1 = require("./generators/EpubGenerator.js");
6
7
  const HtmlGenerator_js_1 = require("./generators/HtmlGenerator.js");
7
8
  const MarkdownGenerator_js_1 = require("./generators/MarkdownGenerator.js");
8
9
  const PdfGenerator_js_1 = require("./generators/PdfGenerator.js");
@@ -60,6 +61,9 @@ class OfficeGenerator {
60
61
  case 'chunks':
61
62
  generator = new ChunkingGenerator_js_1.ChunkingGenerator(ast, config);
62
63
  break;
64
+ case 'epub':
65
+ generator = new EpubGenerator_js_1.EpubGenerator(ast, config);
66
+ break;
63
67
  default:
64
68
  throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED, undefined, destination);
65
69
  }
@@ -14,6 +14,7 @@
14
14
  * - CSV (Comma-Separated Values)
15
15
  * - MD (Markdown)
16
16
  * - HTML (HyperText Markup Language)
17
+ * - EPUB (E-book format)
17
18
  *
18
19
  * **Usage:**
19
20
  * ```typescript
@@ -66,6 +67,7 @@ export declare class OfficeParser {
66
67
  * - `.csv` → CsvParser
67
68
  * - `.md` → MarkdownParser
68
69
  * - `.html` → HtmlParser
70
+ * - `.epub` → EpubParser
69
71
  *
70
72
  * @param file - File path (string), Buffer, or ArrayBuffer containing the document
71
73
  * @param config - Optional configuration object (defaults applied for all omitted options)
@@ -15,6 +15,7 @@
15
15
  * - CSV (Comma-Separated Values)
16
16
  * - MD (Markdown)
17
17
  * - HTML (HyperText Markup Language)
18
+ * - EPUB (E-book format)
18
19
  *
19
20
  * **Usage:**
20
21
  * ```typescript
@@ -39,6 +40,7 @@
39
40
  Object.defineProperty(exports, "__esModule", { value: true });
40
41
  exports.OfficeParser = void 0;
41
42
  const CsvParser_js_1 = require("./parsers/CsvParser.js");
43
+ const EpubParser_js_1 = require("./parsers/EpubParser.js");
42
44
  const ExcelParser_js_1 = require("./parsers/ExcelParser.js");
43
45
  const HtmlParser_js_1 = require("./parsers/HtmlParser.js");
44
46
  const MarkdownParser_js_1 = require("./parsers/MarkdownParser.js");
@@ -83,6 +85,7 @@ class OfficeParser {
83
85
  * - `.csv` → CsvParser
84
86
  * - `.md` → MarkdownParser
85
87
  * - `.html` → HtmlParser
88
+ * - `.epub` → EpubParser
86
89
  *
87
90
  * @param file - File path (string), Buffer, or ArrayBuffer containing the document
88
91
  * @param config - Optional configuration object (defaults applied for all omitted options)
@@ -232,6 +235,9 @@ class OfficeParser {
232
235
  case 'md':
233
236
  result = await (0, MarkdownParser_js_1.parseMarkdown)(buffer, internalConfig);
234
237
  break;
238
+ case 'epub':
239
+ result = await (0, EpubParser_js_1.parseEpub)(buffer, internalConfig);
240
+ break;
235
241
  default:
236
242
  throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
237
243
  }
package/dist/cli.d.ts CHANGED
@@ -8,7 +8,7 @@
8
8
  * officeparser file.docx --ocr --extractAttachments
9
9
  *
10
10
  * Options (--key=value, --key value, or bare flags):
11
- * --to=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format (default: json)
11
+ * --to=json|text|md|html|csv|rtf|pdf|epub|chunks Convert AST to specified format (default: json)
12
12
  * --output=path Save result to a file
13
13
  * --fileType=docx|xlsx|... Override file type detection
14
14
  * --ocr Enable OCR for images (default: false)
package/dist/cli.js CHANGED
@@ -9,7 +9,7 @@
9
9
  * officeparser file.docx --ocr --extractAttachments
10
10
  *
11
11
  * Options (--key=value, --key value, or bare flags):
12
- * --to=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format (default: json)
12
+ * --to=json|text|md|html|csv|rtf|pdf|epub|chunks Convert AST to specified format (default: json)
13
13
  * --output=path Save result to a file
14
14
  * --fileType=docx|xlsx|... Override file type detection
15
15
  * --ocr Enable OCR for images (default: false)
@@ -333,7 +333,7 @@ else {
333
333
  console.log('Usage: officeparser <file> [options]');
334
334
  console.log('');
335
335
  console.log('Options:');
336
- console.log(' --to=json|text|md|html|pdf|csv|rtf|chunks Target conversion format (default: json)');
336
+ console.log(' --to=json|text|md|html|pdf|csv|rtf|epub|chunks Target conversion format (default: json)');
337
337
  console.log(' --output=file.ext Save output to file instead of stdout');
338
338
  console.log(' --fileType=docx|xlsx|pptx|odt|... Explicitly override input file type detection');
339
339
  console.log(' --ocr Enable OCR for images (default: false)');
@@ -360,6 +360,8 @@ else {
360
360
  console.log('Advanced Nested Config Examples:');
361
361
  console.log(' --pdfConfig.format=Letter Configure Puppeteer PDF format (A4 | Letter | Legal etc.)');
362
362
  console.log(' --chunksConfig.strategy=fixed-size Chunking strategy (fixed-size | document-structure | semantic)');
363
+ console.log(' --mdConfig.dialect=github Markdown dialect (extended | github | gitlab | obsidian | pandoc | commonmark)');
364
+ console.log(' --mdConfig.fallbackToHtml=false Disable HTML fallback for unsupported Markdown features (default: true)');
363
365
  console.log('');
364
366
  console.log('Format Syntax:');
365
367
  console.log(' Flags can be written as --flag (presence implies true), --no-flag (negation),');
@@ -371,5 +373,6 @@ else {
371
373
  console.log(' officeparser document.docx --to md');
372
374
  console.log(' officeparser report.pdf --ocr --ocrConfig.language eng --to text');
373
375
  console.log(' officeparser data.xlsx --to csv --output data.csv --csvDelimiter ";"');
376
+ console.log(' officeparser document.docx --extractAttachments --to epub --output document.epub');
374
377
  console.log(' officeparser image_doc --fileType docx --to json');
375
378
  }
package/dist/defaults.js CHANGED
@@ -34,6 +34,13 @@ const DEFAULT_OCR_CONFIG = {
34
34
  autoTerminateTimeout: DEFAULT_OCR_TIMEOUT.autoTerminate,
35
35
  abortSignal: null,
36
36
  };
37
+ /**
38
+ * Default configuration for HTML/XHTML parsing. `preserveAttributes` is off so that the AST is
39
+ * byte-identical to previous releases unless a caller opts in - see `HtmlParserConfig`.
40
+ */
41
+ const DEFAULT_HTML_PARSER_CONFIG = {
42
+ preserveAttributes: false,
43
+ };
37
44
  /**
38
45
  * Default configuration for the OfficeParser.
39
46
  */
@@ -62,7 +69,9 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
62
69
  decompressionLimits: {
63
70
  maxUncompressedBytes: 512 * 1024 * 1024,
64
71
  maxZipEntries: 10000,
72
+ maxTableCells: 1000000,
65
73
  },
74
+ htmlParserConfig: DEFAULT_HTML_PARSER_CONFIG,
66
75
  };
67
76
  /**
68
77
  * Default configuration for HTML generation.
@@ -117,6 +126,7 @@ const DEFAULT_CSV_GENERATOR_CONFIG = {
117
126
  */
118
127
  const DEFAULT_MD_GENERATOR_CONFIG = {
119
128
  fallbackToHtml: true,
129
+ dialect: 'extended',
120
130
  };
121
131
  /**
122
132
  * Default configuration for plain text generation.
@@ -124,6 +134,7 @@ const DEFAULT_MD_GENERATOR_CONFIG = {
124
134
  const DEFAULT_TEXT_GENERATOR_CONFIG = {
125
135
  newlineDelimiter: '\n',
126
136
  preserveLayout: true,
137
+ renderNotes: true,
127
138
  };
128
139
  /**
129
140
  * Default configuration for Fixed-Size chunking.
@@ -187,6 +198,7 @@ exports.DEFAULT_GENERATOR_CONFIG = {
187
198
  includeFormatting: true,
188
199
  generateIds: true,
189
200
  renderMetadata: false,
201
+ metadataOverrides: {},
190
202
  ignoreDefaultStyleMap: false,
191
203
  includeImages: true,
192
204
  includeCharts: true,