officeparser 7.2.2 → 7.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/README.md +161 -17
  2. package/dist/OfficeGenerator.js +4 -0
  3. package/dist/OfficeParser.d.ts +2 -0
  4. package/dist/OfficeParser.js +6 -0
  5. package/dist/cli.d.ts +1 -1
  6. package/dist/cli.js +3 -2
  7. package/dist/defaults.js +3 -3
  8. package/dist/generators/BaseGenerator.d.ts +11 -0
  9. package/dist/generators/BaseGenerator.js +29 -0
  10. package/dist/generators/CsvGenerator.d.ts +9 -1
  11. package/dist/generators/CsvGenerator.js +24 -14
  12. package/dist/generators/EpubGenerator.d.ts +18 -0
  13. package/dist/generators/EpubGenerator.js +242 -0
  14. package/dist/generators/HtmlGenerator.d.ts +12 -0
  15. package/dist/generators/HtmlGenerator.js +266 -51
  16. package/dist/generators/MarkdownGenerator.d.ts +16 -0
  17. package/dist/generators/MarkdownGenerator.js +173 -24
  18. package/dist/generators/PdfGenerator.js +32 -0
  19. package/dist/generators/RtfGenerator.js +12 -15
  20. package/dist/generators/TextGenerator.js +11 -0
  21. package/dist/index.d.ts +1 -0
  22. package/dist/index.js +1 -0
  23. package/dist/officeparser.browser.d.ts +144 -7
  24. package/dist/officeparser.browser.iife.js +289 -193
  25. package/dist/officeparser.browser.mjs +289 -193
  26. package/dist/officeparser.browser.slim.d.ts +2129 -0
  27. package/dist/officeparser.browser.slim.iife.js +1278 -0
  28. package/dist/officeparser.browser.slim.mjs +1277 -0
  29. package/dist/parsers/EpubParser.d.ts +8 -0
  30. package/dist/parsers/EpubParser.js +217 -0
  31. package/dist/parsers/HtmlParser.js +284 -20
  32. package/dist/parsers/MarkdownParser.js +424 -33
  33. package/dist/parsers/OpenOfficeParser.js +241 -54
  34. package/dist/parsers/PdfParser.js +4 -1
  35. package/dist/parsers/WordParser.js +2 -2
  36. package/dist/sbom.cdx.json +111 -223
  37. package/dist/types.d.ts +146 -7
  38. package/dist/types.js +2 -0
  39. package/dist/utils/errorUtils.js +3 -2
  40. package/dist/utils/sanitize.d.ts +99 -0
  41. package/dist/utils/sanitize.js +228 -0
  42. package/dist/utils/zipUtils.js +76 -26
  43. package/package.json +19 -10
package/README.md CHANGED
@@ -2,9 +2,9 @@
2
2
 
3
3
  A robust, strictly-typed **Node.js and Browser** library for parsing office files into a rich **Abstract Syntax Tree (AST)** and generating high-fidelity output in multiple formats.
4
4
 
5
- **Parses:** [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`odt`](https://en.wikipedia.org/wiki/OpenDocument) · [`odp`](https://en.wikipedia.org/wiki/OpenDocument) · [`ods`](https://en.wikipedia.org/wiki/OpenDocument) · [`pdf`](https://en.wikipedia.org/wiki/PDF) · [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format) · [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values) · [`md`](https://en.wikipedia.org/wiki/Markdown) · [`html`](https://en.wikipedia.org/wiki/HTML)
5
+ **Parses:** [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`odt`](https://en.wikipedia.org/wiki/OpenDocument) · [`odp`](https://en.wikipedia.org/wiki/OpenDocument) · [`ods`](https://en.wikipedia.org/wiki/OpenDocument) · [`pdf`](https://en.wikipedia.org/wiki/PDF) · [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format) · [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values) · [`md`](https://en.wikipedia.org/wiki/Markdown) · [`html`](https://en.wikipedia.org/wiki/HTML) · [`epub`](https://en.wikipedia.org/wiki/EPUB)
6
6
 
7
- **Generates:** `Markdown` · `HTML` · `CSV` · `RTF` · `PDF` · `Plain Text` · `RAG Chunks`
7
+ **Generates:** `Markdown` · `HTML` · `CSV` · `RTF` · `PDF` · `EPUB` · `Plain Text` · `RAG Chunks`
8
8
 
9
9
  [![npm version](https://badge.fury.io/js/officeparser.svg)](https://badge.fury.io/js/officeparser)
10
10
  [![Total Downloads](https://img.shields.io/npm/dt/officeparser.svg)](https://www.npmjs.com/package/officeparser)
@@ -42,6 +42,8 @@ A robust, strictly-typed **Node.js and Browser** library for parsing office file
42
42
  - [Native RAG Chunking](#native-rag-chunking)
43
43
  - [The AST Structure](#the-ast-structure)
44
44
  - [Deep Dive: Document Components](#deep-dive-document-components)
45
+ - [Markdown Dialect Support](#markdown-dialect-support)
46
+ - [EPUB Support](#epub-support)
45
47
  - [Performance Highlights](#performance-highlights)
46
48
  - [Advanced AST Usage](#advanced-ast-usage)
47
49
  - [Configuration Reference](#configuration-reference)
@@ -93,6 +95,9 @@ npx officeparser data.xlsx --to=csv --csvDelimiter=";"
93
95
  # Generate RAG chunks
94
96
  npx officeparser document.pdf --to=chunks
95
97
 
98
+ # Convert DOCX to EPUB (--extractAttachments is required to embed images)
99
+ npx officeparser book.docx --extractAttachments --to=epub --output=book.epub
100
+
96
101
  # Overriding file extension mapping
97
102
  npx officeparser my_document --fileType=docx --to=json
98
103
  ```
@@ -106,9 +111,9 @@ npx officeparser my_document --fileType=docx --to=json
106
111
 
107
112
  | Flag | Values | Default | Description |
108
113
  |------|--------|---------|-------------|
109
- | `--to` | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | Output format |
114
+ | `--to` | `json\|text\|md\|html\|csv\|rtf\|pdf\|epub\|chunks` | `json` | Output format |
110
115
  | `--output` | path | — | Write output to a file |
111
- | `--fileType` | `docx\|xlsx\|pptx\|odt\|odp\|ods\|pdf\|rtf\|csv\|md\|html` | — | Explicitly override input file type detection |
116
+ | `--fileType` | `docx\|xlsx\|pptx\|odt\|odp\|ods\|pdf\|rtf\|csv\|md\|html\|epub` | — | Explicitly override input file type detection |
112
117
  | `--ocr` | boolean | `false` | Enable OCR for images |
113
118
  | `--extractAttachments` | boolean | `false` | Extract images/charts as Base64 |
114
119
  | `--ignoreNotes` | boolean | `false` | Ignore footnotes/endnotes/speaker notes |
@@ -126,7 +131,7 @@ npx officeparser my_document --fileType=docx --to=json
126
131
  | `--includeFormatting` | boolean | `true` | Include formatting style map matching |
127
132
  | `--renderMetadata` | boolean | `false` | Render metadata as visible content in the generated output |
128
133
  | `--htmlConfig.containerWidth` | string \| number | `auto` | HTML output container width (e.g. `900px`, `100%`) |
129
- | ~~`--format`~~ | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | **Deprecated.** Use `--to` |
134
+ | ~~`--format`~~ | `json\|text\|md\|html\|csv\|rtf\|pdf\|epub\|chunks` | `json` | **Deprecated.** Use `--to` |
130
135
  | ~~`--toText`~~ | `true\|false` | `false` | **Deprecated.** Use `--to=text` |
131
136
  | ~~`--ocrLanguage`~~ | string | `eng` | **Deprecated.** Use `--ocrConfig.language` |
132
137
  | ~~`--putNotesAtLast`~~ | `true\|false` | `false` | **Deprecated and ignored.** Notes are attached structurally to their nodes. |
@@ -306,13 +311,16 @@ const { value: html } = await OfficeGenerator.generate(ast, 'html', {
306
311
  const { value: csv } = await OfficeGenerator.generate(ast, 'csv');
307
312
  ```
308
313
 
309
- **Supported destinations:** `'text'` · `'md'` · `'html'` · `'csv'` · `'rtf'` · `'pdf'` · `'chunks'`
314
+ **Supported destinations:** `'text'` · `'md'` · `'html'` · `'csv'` · `'rtf'` · `'pdf'` · `'epub'` · `'chunks'`
310
315
 
311
316
  > [!NOTE]
312
317
  > **PDF generation** requires the optional `puppeteer` peer dependency:
313
318
  > ```bash
314
319
  > npm install puppeteer
315
320
  > ```
321
+ >
322
+ > **EPUB generation with images** requires `extractAttachments: true` on the parse step that
323
+ > produced the AST — see [EPUB Support](#epub-support).
316
324
 
317
325
  ---
318
326
 
@@ -440,10 +448,10 @@ interface OfficeChunk {
440
448
 
441
449
  ```text
442
450
  OfficeParserAST
443
- ├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | ... (11 formats)
451
+ ├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | 'epub' | ... (12 formats)
444
452
  ├── metadata: { author, title, created, modified, keywords, customProperties, nativeProperties, styleMap, ... }
445
453
  ├── content: [ OfficeContentNode ]
446
- │ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | ...
454
+ │ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | 'admonition' | 'embed' | 'definitionList' | ...
447
455
  │ ├── text: string (concatenated text of node + all descendants)
448
456
  │ ├── children: [ OfficeContentNode ] (recursive structural children)
449
457
  │ ├── notes: [ OfficeContentNode ] (footnotes/endnotes/slide notes attached to this node)
@@ -596,6 +604,94 @@ console.log(ast.metadata.nativeProperties);
596
604
  // PDF: { Title: 'Report', XMP: { ... } }
597
605
  ```
598
606
 
607
+ ### 8. Admonitions, Embeds & Definition Lists
608
+
609
+ ```text
610
+ Admonition Node (type: 'admonition')
611
+ ├── metadata: { admonitionType: 'note' | 'tip' | 'important' | 'warning' | 'caution', title?: string }
612
+ └── children: [ Paragraph | List | ... ] (block content)
613
+
614
+ Embed Node (type: 'embed')
615
+ └── metadata: { embedType: 'youtube', videoId: string, url?: string, width?: string, align?: string }
616
+
617
+ Definition List Node (type: 'definitionList')
618
+ └── children:
619
+ ├── Definition Term (type: 'definitionTerm')
620
+ └── Definition Description (type: 'definitionDescription')
621
+ ```
622
+
623
+ - `admonition` round-trips through both Markdown (`> [!NOTE]` / `:::note ... :::`) and HTML (`<div class="admonition admonition-note" data-type="note">`)
624
+ - `embed` currently models YouTube videos; HTML round-trips via `<div data-youtube-video="ID">`, Markdown falls back to a raw HTML block or a plain link
625
+ - Abbreviations (`*[HTML]: Hypertext Markup Language`) are stored as `TextMetadata.abbreviationTitle` on the abbreviated text node rather than as a separate node type
626
+
627
+ ---
628
+
629
+ ## Markdown Dialect Support
630
+
631
+ Beyond CommonMark/GFM basics, `MarkdownParser`/`MarkdownGenerator` support an extended dialect aimed at
632
+ full-fidelity round-tripping with rich Markdown editors. Every construct below parses to a first-class
633
+ AST node/metadata field and regenerates back to the canonical syntax shown, so `.md → AST → .md` is
634
+ idempotent and `.md → AST → HTML → AST → .md` survives unchanged.
635
+
636
+ | Feature | Markdown syntax | AST representation |
637
+ |---|---|---|
638
+ | Task lists (GFM) | `- [x] Done` / `- [ ] Todo` | `ListMetadata.isTask` / `.checked` |
639
+ | Admonitions | `> [!NOTE]` (also accepts GLFM `:::note ... :::` on import) | `type: 'admonition'`, `AdmonitionMetadata` |
640
+ | Footnotes | `Text[^1]` + `[^1]: Definition` | `type: 'note'`, keyed by footnote id |
641
+ | Definition lists | `Term\n: Definition` | `type: 'definitionList'` / `'definitionTerm'` / `'definitionDescription'` |
642
+ | Abbreviations | `*[HTML]: Hypertext Markup Language` | `TextMetadata.abbreviationTitle` |
643
+ | Attribute lists | `![alt](img.png){width=50% .centered}` | `ImageMetadata.width` / `.align`, `TableMetadata.align` |
644
+ | Citations | `[@smith2024]` | `TextMetadata.citationKey` |
645
+ | Wikilinks | `[[Page]]` / `[[Page\|Alias]]` | `TextMetadata.wikilink`, `.link`, `.linkType` |
646
+ | Inline/block math | `$E=mc^2$` / `` $$...$$ `` | `TextMetadata.math` (`'inline' \| 'block'`) |
647
+ | Frontmatter arrays | `tags: [a, b]` or `tags: ["a","b"]` | Real array in `metadata.customProperties`/`nativeProperties` |
648
+ | MDX components (import-only) | `<Component prop="x">...</Component>` | Stripped; inner Markdown is kept. Never generated back. |
649
+
650
+ > [!NOTE]
651
+ > MDX/JSX stripping is one-directional (parse-only) — officeParser never authors JSX back into Markdown.
652
+ > Wikilink enable/disable and citekey→bibliography resolution are application-level concerns; officeParser
653
+ > always parses/generates the syntax itself.
654
+
655
+ The same round-trip fidelity extends to HTML, so content saved from a rich-text editor survives a
656
+ save→reload cycle:
657
+
658
+ | HTML attribute | AST field | Notes |
659
+ |---|---|---|
660
+ | `data-width` / `data-align` / inline `style="width:…"` on `<img>` | `ImageMetadata.width` / `.align` | |
661
+ | `data-align` on `<table>` | `TableMetadata.align` | |
662
+ | `colspan` / `rowspan` on `<td>`/`<th>` | `CellMetadata.colSpan` / `.rowSpan` | Previously dropped on HTML import — merged cells now survive a save→reload cycle |
663
+ | `<div data-youtube-video="ID">` / `<iframe src="...youtube.com...">` | `type: 'embed'` | |
664
+ | `<ul data-type="taskList">` / `<li data-checked>` | `ListMetadata.isTask` / `.checked` | |
665
+
666
+ ---
667
+
668
+ ## EPUB Support
669
+
670
+ EPUB files are ZIP archives of XHTML content plus an OPF manifest — `EpubParser` unzips the archive,
671
+ resolves the spine's reading order from `content.opf`, and parses each XHTML document through the
672
+ existing `HtmlParser`, so EPUB content shares the same AST shape (and the same Markdown-dialect
673
+ fidelity above) as every other format. Dublin Core metadata (`dc:title`, `dc:creator`, `dc:description`,
674
+ `dc:subject`, `dc:date`, `dc:publisher`, `dc:language`, `dc:identifier`) maps into `ast.metadata` /
675
+ `ast.metadata.nativeProperties`, and cover art is exposed via `metadata.customProperties.coverImageName`.
676
+
677
+ `EpubGenerator` renders the AST through `HtmlGenerator` and packages the result as a minimal, valid
678
+ EPUB 3 (`mimetype`, `META-INF/container.xml`, an OPF manifest, a nav document, and one XHTML chapter).
679
+
680
+ > [!IMPORTANT]
681
+ > **Pass `extractAttachments: true` when converting to or from EPUB if the document has images.**
682
+ > Without it, the parser never pulls embedded image bytes out of the source document, so there is
683
+ > nothing for the EPUB generator to package — images silently disappear even though everything else
684
+ > converts correctly. Images are packaged as real zip entries (`OEBPS/images/...`) declared in the OPF
685
+ > manifest, not `data:` URIs — most EPUB reading systems do not render `data:` URIs in image `src`.
686
+ >
687
+ > This only matters for the two-step `OfficeParser.parseOffice()` → `OfficeGenerator.generate()` API
688
+ > and the CLI. [`OfficeConverter.convert()`](#officeconverter-one-step-api) enables `extractAttachments`
689
+ > automatically unless you explicitly set `generatorConfig.includeImages: false`.
690
+ >
691
+ > ```bash
692
+ > npx officeparser book.docx --extractAttachments --to=epub --output=book.epub
693
+ > ```
694
+
599
695
  ---
600
696
 
601
697
  ## Performance Highlights
@@ -852,7 +948,7 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
852
948
 
853
949
  | Option | Type | Default | Description |
854
950
  |--------|------|---------|-------------|
855
- | `standalone` | `boolean` | `true` | Wrap output in a full `<html>` document with CSS |
951
+ | `standalone` | `boolean \| StandaloneConfig` | `true` | Controls the HTML "document envelope" — see below |
856
952
  | `chartJsSrc` | `string` | jsDelivr CDN | URL for the Chart.js library |
857
953
  | `containerWidth` | `string \| number` | `'auto'` | Max width of the content container. Positive number (px), CSS length string (`'900px'`, `'100%'`, `'60vw'`), or `'auto'`. Invalid values fall back to `'auto'` with an `INVALID_CONTAINER_WIDTH` warning |
858
954
  | `customCss` | `string` | `''` | Raw CSS injected into the `<style>` block; use this to override built-in styles |
@@ -861,6 +957,49 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
861
957
  | `injections.bodyStart` | `string` | `''` | Raw HTML injected after `<body>` |
862
958
  | `injections.bodyEnd` | `string` | `''` | Raw HTML injected before `</body>` |
863
959
 
960
+ #### `standalone`: granular envelope control
961
+
962
+ `standalone` conflates several independent decisions: whether to emit the `<!doctype>/<html>/<head>/
963
+ <body>` shell, how CSS is delivered, and whether to inject scripts/meta tags/injections. The boolean
964
+ shorthand still works — **`true`/omitted turns every part on** (a complete document); **`false` turns
965
+ every part off** (a bare content fragment, safe to drop into a page you don't control). Pass an
966
+ object instead for granular control; any field you omit defaults to its "on" (standalone) value:
967
+
968
+ | `StandaloneConfig` field | Type | Default | Description |
969
+ |--------|------|---------|-------------|
970
+ | `document` | `boolean` | `true` | Wrap in `<!DOCTYPE html><html><head>…</head><body>…</body></html>` |
971
+ | `metaTags` | `boolean` | `true` | Emit `<title>`/`<meta>` tags. Only meaningful when `document` is true |
972
+ | `styles` | `'full' \| 'scoped' \| 'none'` | `'full'` | See below |
973
+ | `scripts` | `boolean` | `true` | Emit the Chart.js CDN loader and spreadsheet-interactivity `<script>` tags |
974
+ | `headInjections` | `boolean` | `true` | Apply `injections.headStart`/`headEnd`. Only meaningful when `document` is true |
975
+ | `bodyInjections` | `boolean` | `true` | Apply `injections.bodyStart`/`bodyEnd` — applies even to a bare fragment |
976
+
977
+ `styles` controls how the built-in stylesheet is delivered:
978
+ - **`'full'`** — the complete stylesheet using global selectors (`body`, `h1`, `table`, …). This is
979
+ what `standalone: true` has always emitted.
980
+ - **`'scoped'`** — the same styling, scoped under the fragment's own wrapper via CSS `@scope` so it
981
+ cannot leak onto a host page's elements. Requires a modern engine (Chrome 118+, Safari 17.4+,
982
+ Firefox 128+); for universal support use `'none'` (bring your own CSS) or `'full'`.
983
+ - **`'none'`** — no stylesheet at all; the host page (or rich-text editor, or EPUB reader) supplies
984
+ its own styling.
985
+
986
+ ```js
987
+ // A styled fragment to embed in your own page, without a document shell:
988
+ await ast.to('html', { htmlConfig: { standalone: { document: false } } });
989
+
990
+ // The same, but with styles scoped so they can't leak onto your page's own elements:
991
+ await ast.to('html', { htmlConfig: { standalone: { document: false, styles: 'scoped' } } });
992
+
993
+ // A completely bare fragment (no shell, no styles, no scripts) — e.g. for a rich-text editor:
994
+ await ast.to('html', { htmlConfig: { standalone: false } });
995
+ ```
996
+
997
+ > [!NOTE]
998
+ > **Behavior change from `standalone: false`:** previously this emitted a fragment with a *global,
999
+ > unscoped* `<style>` block. It now emits a genuinely bare fragment (no `<style>` at all), matching
1000
+ > "every part off." If you relied on the old styled-fragment behavior, pass
1001
+ > `{ document: false }` (or `{ document: false, styles: 'full' }`) instead.
1002
+
864
1003
  ### MdGeneratorConfig
865
1004
 
866
1005
  Pass as `mdConfig` inside `GeneratorConfig`.
@@ -1007,12 +1146,17 @@ await officeParser.terminateOcr(); // immediate exit
1007
1146
 
1008
1147
  ## Browser Usage
1009
1148
 
1010
- Two bundles are available in the `dist/` directory:
1149
+ Four bundles are available in the `dist/` directory:
1150
+
1151
+ | Bundle | Type | Description |
1152
+ |--------|------|-------------|
1153
+ | `officeparser.browser.mjs` | ESM | Standard ESM bundle for modern bundlers (Vite, Webpack, Next.js). |
1154
+ | `officeparser.browser.iife.js` | IIFE | Standard UMD bundle for direct `<script>` inclusion (exposes global `officeParser`). |
1155
+ | `officeparser.browser.slim.mjs` | ESM | Slim ESM bundle with Tesseract.js (OCR) stubbed out and remote CDN URLs removed. |
1156
+ | `officeparser.browser.slim.iife.js` | IIFE | Slim UMD bundle with Tesseract.js (OCR) stubbed out and remote CDN URLs removed. |
1011
1157
 
1012
- | Bundle | Usage |
1013
- |--------|-------|
1014
- | `officeparser.browser.mjs` | ESM, use with `import` statements or modern bundlers (Vite, Webpack, Next.js) |
1015
- | `officeparser.browser.iife.js` | IIFE, use with a `<script>` tag; exposes the global `officeParser` object |
1158
+ ### Manifest V3 & Extension Compliance (Slim Bundles)
1159
+ For strict browser environments like **Chrome/Edge Manifest V3 extensions**, remotely hosted code is forbidden. Use the **slim** bundles (`officeparser.browser.slim.mjs` or `officeparser.browser.slim.iife.js`) as they do not include default remote CDN urls or the Tesseract OCR engine.
1016
1160
 
1017
1161
  ### ESM (Vite / Webpack / Next.js)
1018
1162
 
@@ -1055,12 +1199,12 @@ const ast = await officeParser.parseOffice(pdfArrayBuffer);
1055
1199
 
1056
1200
  // Or specify your own:
1057
1201
  const ast = await officeParser.parseOffice(pdfArrayBuffer, {
1058
- pdfWorkerSrc: 'https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs'
1202
+ pdfWorkerSrc: 'https://cdn.jsdelivr.net/npm/pdfjs-dist@6.1.200/build/pdf.worker.min.mjs'
1059
1203
  });
1060
1204
  ```
1061
1205
 
1062
1206
  > [!NOTE]
1063
- > The `pdfjs-dist` worker version must match the version bundled with `officeparser` (currently **5.6.205**).
1207
+ > The `pdfjs-dist` worker version must match the version bundled with `officeparser` (currently **6.1.200**).
1064
1208
 
1065
1209
  ---
1066
1210
 
@@ -1069,7 +1213,7 @@ const ast = await officeParser.parseOffice(pdfArrayBuffer, {
1069
1213
  | Symptom | Fix |
1070
1214
  |---------|-----|
1071
1215
  | Node.js process stays alive after finishing | Call `await officeParser.terminateOcr()` at end of script when OCR was used |
1072
- | `"Worker not found"` in browser for PDF | Verify `pdfWorkerSrc` points to `pdf.worker.min.mjs` matching version `5.6.205` |
1216
+ | `"Worker not found"` in browser for PDF | Verify `pdfWorkerSrc` points to `pdf.worker.min.mjs` matching version `6.1.200` |
1073
1217
  | Low OCR accuracy | Verify `ocrConfig.language` matches the document language; quality depends on image resolution |
1074
1218
  | Out of memory on large Excel files | Call `ast.toText()` early and discard the AST object to allow garbage collection |
1075
1219
  | `md`/`html`/`csv` buffer not detected | Add `fileType: 'md'` (or `'html'`, `'csv'`) to config (these formats have no magic bytes) |
@@ -3,6 +3,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.OfficeGenerator = void 0;
4
4
  const ChunkingGenerator_js_1 = require("./generators/ChunkingGenerator.js");
5
5
  const CsvGenerator_js_1 = require("./generators/CsvGenerator.js");
6
+ const EpubGenerator_js_1 = require("./generators/EpubGenerator.js");
6
7
  const HtmlGenerator_js_1 = require("./generators/HtmlGenerator.js");
7
8
  const MarkdownGenerator_js_1 = require("./generators/MarkdownGenerator.js");
8
9
  const PdfGenerator_js_1 = require("./generators/PdfGenerator.js");
@@ -60,6 +61,9 @@ class OfficeGenerator {
60
61
  case 'chunks':
61
62
  generator = new ChunkingGenerator_js_1.ChunkingGenerator(ast, config);
62
63
  break;
64
+ case 'epub':
65
+ generator = new EpubGenerator_js_1.EpubGenerator(ast, config);
66
+ break;
63
67
  default:
64
68
  throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED, undefined, destination);
65
69
  }
@@ -14,6 +14,7 @@
14
14
  * - CSV (Comma-Separated Values)
15
15
  * - MD (Markdown)
16
16
  * - HTML (HyperText Markup Language)
17
+ * - EPUB (E-book format)
17
18
  *
18
19
  * **Usage:**
19
20
  * ```typescript
@@ -66,6 +67,7 @@ export declare class OfficeParser {
66
67
  * - `.csv` → CsvParser
67
68
  * - `.md` → MarkdownParser
68
69
  * - `.html` → HtmlParser
70
+ * - `.epub` → EpubParser
69
71
  *
70
72
  * @param file - File path (string), Buffer, or ArrayBuffer containing the document
71
73
  * @param config - Optional configuration object (defaults applied for all omitted options)
@@ -15,6 +15,7 @@
15
15
  * - CSV (Comma-Separated Values)
16
16
  * - MD (Markdown)
17
17
  * - HTML (HyperText Markup Language)
18
+ * - EPUB (E-book format)
18
19
  *
19
20
  * **Usage:**
20
21
  * ```typescript
@@ -39,6 +40,7 @@
39
40
  Object.defineProperty(exports, "__esModule", { value: true });
40
41
  exports.OfficeParser = void 0;
41
42
  const CsvParser_js_1 = require("./parsers/CsvParser.js");
43
+ const EpubParser_js_1 = require("./parsers/EpubParser.js");
42
44
  const ExcelParser_js_1 = require("./parsers/ExcelParser.js");
43
45
  const HtmlParser_js_1 = require("./parsers/HtmlParser.js");
44
46
  const MarkdownParser_js_1 = require("./parsers/MarkdownParser.js");
@@ -83,6 +85,7 @@ class OfficeParser {
83
85
  * - `.csv` → CsvParser
84
86
  * - `.md` → MarkdownParser
85
87
  * - `.html` → HtmlParser
88
+ * - `.epub` → EpubParser
86
89
  *
87
90
  * @param file - File path (string), Buffer, or ArrayBuffer containing the document
88
91
  * @param config - Optional configuration object (defaults applied for all omitted options)
@@ -232,6 +235,9 @@ class OfficeParser {
232
235
  case 'md':
233
236
  result = await (0, MarkdownParser_js_1.parseMarkdown)(buffer, internalConfig);
234
237
  break;
238
+ case 'epub':
239
+ result = await (0, EpubParser_js_1.parseEpub)(buffer, internalConfig);
240
+ break;
235
241
  default:
236
242
  throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
237
243
  }
package/dist/cli.d.ts CHANGED
@@ -8,7 +8,7 @@
8
8
  * officeparser file.docx --ocr --extractAttachments
9
9
  *
10
10
  * Options (--key=value, --key value, or bare flags):
11
- * --to=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format (default: json)
11
+ * --to=json|text|md|html|csv|rtf|pdf|epub|chunks Convert AST to specified format (default: json)
12
12
  * --output=path Save result to a file
13
13
  * --fileType=docx|xlsx|... Override file type detection
14
14
  * --ocr Enable OCR for images (default: false)
package/dist/cli.js CHANGED
@@ -9,7 +9,7 @@
9
9
  * officeparser file.docx --ocr --extractAttachments
10
10
  *
11
11
  * Options (--key=value, --key value, or bare flags):
12
- * --to=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format (default: json)
12
+ * --to=json|text|md|html|csv|rtf|pdf|epub|chunks Convert AST to specified format (default: json)
13
13
  * --output=path Save result to a file
14
14
  * --fileType=docx|xlsx|... Override file type detection
15
15
  * --ocr Enable OCR for images (default: false)
@@ -333,7 +333,7 @@ else {
333
333
  console.log('Usage: officeparser <file> [options]');
334
334
  console.log('');
335
335
  console.log('Options:');
336
- console.log(' --to=json|text|md|html|pdf|csv|rtf|chunks Target conversion format (default: json)');
336
+ console.log(' --to=json|text|md|html|pdf|csv|rtf|epub|chunks Target conversion format (default: json)');
337
337
  console.log(' --output=file.ext Save output to file instead of stdout');
338
338
  console.log(' --fileType=docx|xlsx|pptx|odt|... Explicitly override input file type detection');
339
339
  console.log(' --ocr Enable OCR for images (default: false)');
@@ -371,5 +371,6 @@ else {
371
371
  console.log(' officeparser document.docx --to md');
372
372
  console.log(' officeparser report.pdf --ocr --ocrConfig.language eng --to text');
373
373
  console.log(' officeparser data.xlsx --to csv --output data.csv --csvDelimiter ";"');
374
+ console.log(' officeparser document.docx --extractAttachments --to epub --output document.epub');
374
375
  console.log(' officeparser image_doc --fileType docx --to json');
375
376
  }
package/dist/defaults.js CHANGED
@@ -1,8 +1,8 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.DEFAULT_GENERATOR_CONFIG = exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG = exports.DEFAULT_FIXED_SIZE_CHUNKING_CONFIG = exports.DEFAULT_OFFICE_PARSER_CONFIG = exports.DEFAULT_ABBREVIATIONS = exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = void 0;
4
- const PDFJS_VERSION = '5.6.205';
5
- const DEFAULT_PDF_WORKER_SRC = `https://cdn.jsdelivr.net/npm/pdfjs-dist@${PDFJS_VERSION}/build/pdf.worker.min.mjs`;
4
+ const PDFJS_VERSION = '6.1.200';
5
+ const DEFAULT_PDF_WORKER_SRC = typeof __SLIM__ !== 'undefined' && __SLIM__ ? '' : `https://cdn.jsdelivr.net/npm/pdfjs-dist@${PDFJS_VERSION}/build/pdf.worker.min.mjs`;
6
6
  /**
7
7
  * The default regex used for identifying sentence boundaries.
8
8
  * When this default is used, the generator employs a high-fidelity "robust"
@@ -69,7 +69,7 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
69
69
  */
70
70
  const DEFAULT_HTML_GENERATOR_CONFIG = {
71
71
  standalone: true,
72
- chartJsSrc: 'https://cdn.jsdelivr.net/npm/chart.js',
72
+ chartJsSrc: typeof __SLIM__ !== 'undefined' && __SLIM__ ? '' : 'https://cdn.jsdelivr.net/npm/chart.js',
73
73
  containerWidth: 'auto',
74
74
  customCss: '',
75
75
  injections: {
@@ -48,6 +48,17 @@ export declare abstract class BaseGenerator<D extends UniversalGeneratorFormat =
48
48
  * Helper to generate a unique ID (slug) from text.
49
49
  */
50
50
  protected slugify(text: string): string;
51
+ private noteFootnoteKeys;
52
+ private usedFootnoteKeys;
53
+ private footnoteKeyCounter;
54
+ /**
55
+ * Assigns a stable, unique reference key to a footnote/endnote node, reused for both
56
+ * its inline reference marker and its collected definition. Source ids aren't reliably
57
+ * unique across a document - DOCX/ODT number footnotes and endnotes in separate
58
+ * sequences, so both can carry noteId "1" - so a source id is only reused when it
59
+ * hasn't already been claimed; otherwise a sequential counter guarantees uniqueness.
60
+ */
61
+ protected getFootnoteKey(note: OfficeContentNode): string;
51
62
  /**
52
63
  * Recursively extracts plain text from a node and its children.
53
64
  */
@@ -89,6 +89,35 @@ class BaseGenerator {
89
89
  .replace(/[\s_-]+/g, '-')
90
90
  .replace(/^-+|-+$/g, '');
91
91
  }
92
+ noteFootnoteKeys = new Map();
93
+ usedFootnoteKeys = new Set();
94
+ footnoteKeyCounter = 0;
95
+ /**
96
+ * Assigns a stable, unique reference key to a footnote/endnote node, reused for both
97
+ * its inline reference marker and its collected definition. Source ids aren't reliably
98
+ * unique across a document - DOCX/ODT number footnotes and endnotes in separate
99
+ * sequences, so both can carry noteId "1" - so a source id is only reused when it
100
+ * hasn't already been claimed; otherwise a sequential counter guarantees uniqueness.
101
+ */
102
+ getFootnoteKey(note) {
103
+ const cached = this.noteFootnoteKeys.get(note);
104
+ if (cached)
105
+ return cached;
106
+ const preferred = note.metadata?.noteId;
107
+ let key;
108
+ if (preferred && !this.usedFootnoteKeys.has(preferred)) {
109
+ key = preferred;
110
+ }
111
+ else {
112
+ do {
113
+ this.footnoteKeyCounter++;
114
+ key = String(this.footnoteKeyCounter);
115
+ } while (this.usedFootnoteKeys.has(key));
116
+ }
117
+ this.usedFootnoteKeys.add(key);
118
+ this.noteFootnoteKeys.set(note, key);
119
+ return key;
120
+ }
92
121
  /**
93
122
  * Recursively extracts plain text from a node and its children.
94
123
  */
@@ -20,9 +20,17 @@ export declare class CsvGenerator extends BaseGenerator<'csv'> {
20
20
  */
21
21
  private renderNodeToRows;
22
22
  /**
23
- * Escapes a value for CSV formatting.
23
+ * Escapes a value for CSV formatting: RFC 4180 quoting plus a spreadsheet
24
+ * formula-injection guard (see csvSafeCell in ../utils/sanitize.js).
24
25
  */
25
26
  private escapeCsvValue;
27
+ /** Sanitizes a value for a `#` comment line (document metadata or sheet name).
28
+ * Comment lines are free text prefixed with `#`, not RFC-4180 cells, so they
29
+ * can't be quoted; instead we (a) collapse newlines so the value can't break out
30
+ * and inject a new row, and (b) replace the column delimiter with a space so a
31
+ * value like `x,=1+1` can't split into a second cell that a spreadsheet would
32
+ * evaluate as a formula (CSV formula/DDE injection). */
33
+ private sanitizeComment;
26
34
  /**
27
35
  * Renders metadata as comments.
28
36
  */
@@ -4,6 +4,7 @@ exports.CsvGenerator = void 0;
4
4
  const fflate_1 = require("fflate");
5
5
  const types_js_1 = require("../types.js");
6
6
  const sheetUtils_js_1 = require("../utils/sheetUtils.js");
7
+ const sanitize_js_1 = require("../utils/sanitize.js");
7
8
  const BaseGenerator_js_1 = require("./BaseGenerator.js");
8
9
  /**
9
10
  * Generates CSV files from an AST.
@@ -26,7 +27,7 @@ class CsvGenerator extends BaseGenerator_js_1.BaseGenerator {
26
27
  if (sheetNodes.length === 0) {
27
28
  return { value: '', messages: this.messages };
28
29
  }
29
- const metadataHeader = this.config.renderMetadata ? this.renderMetadata(this.ast) : '';
30
+ const metadataHeader = this.config.renderMetadata ? this.renderMetadata(this.ast, delimiter) : '';
30
31
  // 2. Filter sheets based on range
31
32
  let selectedNodes = sheetNodes;
32
33
  if (csvConfig.sheets) {
@@ -59,7 +60,7 @@ class CsvGenerator extends BaseGenerator_js_1.BaseGenerator {
59
60
  mergedLines.push(metadataHeader.trim());
60
61
  for (const sheet of sheetData) {
61
62
  if (sheetData.length > 1)
62
- mergedLines.push(`# Sheet: ${sheet.name}`);
63
+ mergedLines.push(`# Sheet: ${this.sanitizeComment(sheet.name, delimiter)}`);
63
64
  for (const row of sheet.rows) {
64
65
  const paddedRow = [...row];
65
66
  // Don't pad comments
@@ -197,34 +198,43 @@ class CsvGenerator extends BaseGenerator_js_1.BaseGenerator {
197
198
  return rows;
198
199
  }
199
200
  /**
200
- * Escapes a value for CSV formatting.
201
+ * Escapes a value for CSV formatting: RFC 4180 quoting plus a spreadsheet
202
+ * formula-injection guard (see csvSafeCell in ../utils/sanitize.js).
201
203
  */
202
204
  escapeCsvValue(val, delimiter) {
203
- const needsQuotes = val.includes(delimiter) || val.includes('"') || val.includes('\n') || val.includes('\r');
204
- if (!needsQuotes)
205
- return val;
206
- // Double up existing quotes and wrap in quotes
207
- return `"${val.replace(/"/g, '""')}"`;
205
+ return (0, sanitize_js_1.csvSafeCell)(val, delimiter);
206
+ }
207
+ /** Sanitizes a value for a `#` comment line (document metadata or sheet name).
208
+ * Comment lines are free text prefixed with `#`, not RFC-4180 cells, so they
209
+ * can't be quoted; instead we (a) collapse newlines so the value can't break out
210
+ * and inject a new row, and (b) replace the column delimiter with a space so a
211
+ * value like `x,=1+1` can't split into a second cell that a spreadsheet would
212
+ * evaluate as a formula (CSV formula/DDE injection). */
213
+ sanitizeComment(val, delimiter) {
214
+ let s = String(val ?? '').replace(/[\r\n]+/g, ' ');
215
+ if (delimiter)
216
+ s = s.split(delimiter).join(' ');
217
+ return s;
208
218
  }
209
219
  /**
210
220
  * Renders metadata as comments.
211
221
  */
212
- renderMetadata(ast) {
222
+ renderMetadata(ast, delimiter) {
213
223
  if (!ast.metadata)
214
224
  return '';
215
225
  const m = ast.metadata;
216
226
  let output = '';
217
227
  if (m.title)
218
- output += `# Title: ${m.title}\n`;
228
+ output += `# Title: ${this.sanitizeComment(m.title, delimiter)}\n`;
219
229
  if (m.author)
220
- output += `# Author: ${m.author}\n`;
230
+ output += `# Author: ${this.sanitizeComment(m.author, delimiter)}\n`;
221
231
  if (m.created)
222
- output += `# Created: ${new Date(m.created).toLocaleString()}\n`;
232
+ output += `# Created: ${this.sanitizeComment(new Date(m.created).toLocaleString(), delimiter)}\n`;
223
233
  if (m.modified)
224
- output += `# Modified: ${new Date(m.modified).toLocaleString()}\n`;
234
+ output += `# Modified: ${this.sanitizeComment(new Date(m.modified).toLocaleString(), delimiter)}\n`;
225
235
  if (m.customProperties) {
226
236
  for (const [k, v] of Object.entries(m.customProperties)) {
227
- output += `# ${k}: ${v}\n`;
237
+ output += `# ${this.sanitizeComment(k, delimiter)}: ${this.sanitizeComment(String(v), delimiter)}\n`;
228
238
  }
229
239
  }
230
240
  return output ? output + '\n' : '';
@@ -0,0 +1,18 @@
1
+ import { ConversionResult, GeneratorConfig, OfficeParserAST } from '../types.js';
2
+ import { BaseGenerator } from './BaseGenerator.js';
3
+ /**
4
+ * Generates a minimal, valid EPUB 3 file from an AST.
5
+ *
6
+ * Every AST node is rendered as a single XHTML content document (reusing `HtmlGenerator`
7
+ * for the actual markup, since EPUB content documents are XHTML) and packaged with the
8
+ * required `mimetype`, `META-INF/container.xml`, OPF manifest, and navigation document.
9
+ *
10
+ * `HtmlGenerator` embeds images as base64 `data:` URIs, but EPUB reading systems do not
11
+ * render `data:` URIs - images must be packaged as separate resources referenced by a
12
+ * relative path. So each data-URI image is extracted into `OEBPS/images/`, declared in
13
+ * the manifest, and its `<img src>` rewritten to point at the packaged file.
14
+ */
15
+ export declare class EpubGenerator extends BaseGenerator<'epub'> {
16
+ constructor(ast: OfficeParserAST, config?: GeneratorConfig<'epub'>);
17
+ generate(): Promise<ConversionResult<'epub'>>;
18
+ }