officeparser 7.2.3 → 7.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +277 -17
- package/dist/OfficeConverter.d.ts +1 -1
- package/dist/OfficeConverter.js +3 -0
- package/dist/OfficeGenerator.js +4 -0
- package/dist/OfficeParser.d.ts +2 -0
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +5 -2
- package/dist/defaults.js +12 -0
- package/dist/generators/BaseGenerator.d.ts +34 -1
- package/dist/generators/BaseGenerator.js +98 -0
- package/dist/generators/CsvGenerator.d.ts +9 -1
- package/dist/generators/CsvGenerator.js +28 -16
- package/dist/generators/EpubGenerator.d.ts +43 -0
- package/dist/generators/EpubGenerator.js +312 -0
- package/dist/generators/HtmlGenerator.d.ts +12 -0
- package/dist/generators/HtmlGenerator.js +378 -61
- package/dist/generators/MarkdownGenerator.d.ts +28 -5
- package/dist/generators/MarkdownGenerator.js +432 -51
- package/dist/generators/PdfGenerator.js +32 -0
- package/dist/generators/RtfGenerator.js +47 -22
- package/dist/generators/TextGenerator.js +98 -11
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/officeparser.browser.d.ts +427 -20
- package/dist/officeparser.browser.iife.js +338 -206
- package/dist/officeparser.browser.mjs +346 -214
- package/dist/officeparser.browser.slim.d.ts +427 -20
- package/dist/officeparser.browser.slim.iife.js +346 -214
- package/dist/officeparser.browser.slim.mjs +346 -214
- package/dist/parsers/EpubParser.d.ts +8 -0
- package/dist/parsers/EpubParser.js +217 -0
- package/dist/parsers/ExcelParser.js +2 -0
- package/dist/parsers/HtmlParser.js +507 -48
- package/dist/parsers/MarkdownParser.js +704 -92
- package/dist/parsers/OpenOfficeParser.js +128 -20
- package/dist/parsers/PdfParser.js +4 -1
- package/dist/parsers/PowerPointParser.js +1 -0
- package/dist/parsers/WordParser.js +1 -0
- package/dist/sbom.cdx.json +1695 -0
- package/dist/types.d.ts +427 -20
- package/dist/types.js +8 -0
- package/dist/utils/configUtils.js +53 -4
- package/dist/utils/errorUtils.js +7 -3
- package/dist/utils/sanitize.d.ts +139 -0
- package/dist/utils/sanitize.js +318 -0
- package/dist/utils/xmlUtils.js +2 -2
- package/dist/utils/zipUtils.js +76 -26
- package/package.json +16 -12
package/README.md
CHANGED
|
@@ -2,9 +2,9 @@
|
|
|
2
2
|
|
|
3
3
|
A robust, strictly-typed **Node.js and Browser** library for parsing office files into a rich **Abstract Syntax Tree (AST)** and generating high-fidelity output in multiple formats.
|
|
4
4
|
|
|
5
|
-
**Parses:** [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`odt`](https://en.wikipedia.org/wiki/OpenDocument) · [`odp`](https://en.wikipedia.org/wiki/OpenDocument) · [`ods`](https://en.wikipedia.org/wiki/OpenDocument) · [`pdf`](https://en.wikipedia.org/wiki/PDF) · [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format) · [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values) · [`md`](https://en.wikipedia.org/wiki/Markdown) · [`html`](https://en.wikipedia.org/wiki/HTML)
|
|
5
|
+
**Parses:** [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML) · [`odt`](https://en.wikipedia.org/wiki/OpenDocument) · [`odp`](https://en.wikipedia.org/wiki/OpenDocument) · [`ods`](https://en.wikipedia.org/wiki/OpenDocument) · [`pdf`](https://en.wikipedia.org/wiki/PDF) · [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format) · [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values) · [`md`](https://en.wikipedia.org/wiki/Markdown) · [`html`](https://en.wikipedia.org/wiki/HTML) · [`epub`](https://en.wikipedia.org/wiki/EPUB)
|
|
6
6
|
|
|
7
|
-
**Generates:** `Markdown` · `HTML` · `CSV` · `RTF` · `PDF` · `Plain Text` · `RAG Chunks`
|
|
7
|
+
**Generates:** `Markdown` · `HTML` · `CSV` · `RTF` · `PDF` · `EPUB` · `Plain Text` · `RAG Chunks`
|
|
8
8
|
|
|
9
9
|
[](https://badge.fury.io/js/officeparser)
|
|
10
10
|
[](https://www.npmjs.com/package/officeparser)
|
|
@@ -42,6 +42,8 @@ A robust, strictly-typed **Node.js and Browser** library for parsing office file
|
|
|
42
42
|
- [Native RAG Chunking](#native-rag-chunking)
|
|
43
43
|
- [The AST Structure](#the-ast-structure)
|
|
44
44
|
- [Deep Dive: Document Components](#deep-dive-document-components)
|
|
45
|
+
- [Markdown Dialect Support](#markdown-dialect-support)
|
|
46
|
+
- [EPUB Support](#epub-support)
|
|
45
47
|
- [Performance Highlights](#performance-highlights)
|
|
46
48
|
- [Advanced AST Usage](#advanced-ast-usage)
|
|
47
49
|
- [Configuration Reference](#configuration-reference)
|
|
@@ -54,12 +56,14 @@ A robust, strictly-typed **Node.js and Browser** library for parsing office file
|
|
|
54
56
|
- [PdfGeneratorConfig](#pdfgeneratorconfig)
|
|
55
57
|
- [CsvGeneratorConfig](#csvgeneratorconfig)
|
|
56
58
|
- [TextGeneratorConfig](#textgeneratorconfig)
|
|
59
|
+
- [metadataOverrides](#metadataoverrides)
|
|
57
60
|
- [OfficeConverterConfig](#officeconverterconfig)
|
|
58
61
|
- [ChunkingConfig](#chunkingconfig)
|
|
59
62
|
- [OCR Scheduler & Resource Management](#ocr-scheduler--resource-management)
|
|
60
63
|
- [Browser Usage](#browser-usage)
|
|
61
64
|
- [Troubleshooting & Common Issues](#troubleshooting--common-issues)
|
|
62
65
|
- [Known Limitations](#known-limitations)
|
|
66
|
+
- [Security & Trust Boundary](#security--trust-boundary)
|
|
63
67
|
- [Contributing](#contributing)
|
|
64
68
|
|
|
65
69
|
---
|
|
@@ -93,6 +97,9 @@ npx officeparser data.xlsx --to=csv --csvDelimiter=";"
|
|
|
93
97
|
# Generate RAG chunks
|
|
94
98
|
npx officeparser document.pdf --to=chunks
|
|
95
99
|
|
|
100
|
+
# Convert DOCX to EPUB (--extractAttachments is required to embed images)
|
|
101
|
+
npx officeparser book.docx --extractAttachments --to=epub --output=book.epub
|
|
102
|
+
|
|
96
103
|
# Overriding file extension mapping
|
|
97
104
|
npx officeparser my_document --fileType=docx --to=json
|
|
98
105
|
```
|
|
@@ -106,9 +113,9 @@ npx officeparser my_document --fileType=docx --to=json
|
|
|
106
113
|
|
|
107
114
|
| Flag | Values | Default | Description |
|
|
108
115
|
|------|--------|---------|-------------|
|
|
109
|
-
| `--to` | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | Output format |
|
|
116
|
+
| `--to` | `json\|text\|md\|html\|csv\|rtf\|pdf\|epub\|chunks` | `json` | Output format |
|
|
110
117
|
| `--output` | path | — | Write output to a file |
|
|
111
|
-
| `--fileType` | `docx\|xlsx\|pptx\|odt\|odp\|ods\|pdf\|rtf\|csv\|md\|html` | — | Explicitly override input file type detection |
|
|
118
|
+
| `--fileType` | `docx\|xlsx\|pptx\|odt\|odp\|ods\|pdf\|rtf\|csv\|md\|html\|epub` | — | Explicitly override input file type detection |
|
|
112
119
|
| `--ocr` | boolean | `false` | Enable OCR for images |
|
|
113
120
|
| `--extractAttachments` | boolean | `false` | Extract images/charts as Base64 |
|
|
114
121
|
| `--ignoreNotes` | boolean | `false` | Ignore footnotes/endnotes/speaker notes |
|
|
@@ -126,8 +133,8 @@ npx officeparser my_document --fileType=docx --to=json
|
|
|
126
133
|
| `--includeFormatting` | boolean | `true` | Include formatting style map matching |
|
|
127
134
|
| `--renderMetadata` | boolean | `false` | Render metadata as visible content in the generated output |
|
|
128
135
|
| `--htmlConfig.containerWidth` | string \| number | `auto` | HTML output container width (e.g. `900px`, `100%`) |
|
|
129
|
-
| ~~`--format`~~ | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | **Deprecated.** Use `--to` |
|
|
130
|
-
| ~~`--toText`~~ | `true\|false` | `false` | **Deprecated.** Use `--to=text
|
|
136
|
+
| ~~`--format`~~ | `json\|text\|md\|html\|csv\|rtf\|pdf\|epub\|chunks` | `json` | **Deprecated.** Use `--to` |
|
|
137
|
+
| ~~`--toText`~~ | `true\|false` | `false` | **Deprecated.** Use `--to=text`, which keeps footnote text and image placeholders by default (both switchable); this flag drops them unconditionally |
|
|
131
138
|
| ~~`--ocrLanguage`~~ | string | `eng` | **Deprecated.** Use `--ocrConfig.language` |
|
|
132
139
|
| ~~`--putNotesAtLast`~~ | `true\|false` | `false` | **Deprecated and ignored.** Notes are attached structurally to their nodes. |
|
|
133
140
|
| ~~`--outputErrorToConsole`~~ | `true\|false` | `false` | **Deprecated.** Use `--verbose` |
|
|
@@ -269,14 +276,57 @@ const { value: pdfBytes } = await ast.to('pdf'); // Uint8Array
|
|
|
269
276
|
|
|
270
277
|
### `ast.toText()`: Quick Text Extraction
|
|
271
278
|
|
|
272
|
-
> [!
|
|
273
|
-
> `toText()` is **synchronous** and deprecated in favour of the async `ast.to('text')`.
|
|
274
|
-
>
|
|
279
|
+
> [!WARNING]
|
|
280
|
+
> `toText()` is **synchronous** and deprecated in favour of the async `ast.to('text')`. It remains
|
|
281
|
+
> available for backward compatibility, but it is the older, less capable renderer: it has no
|
|
282
|
+
> configuration at all, so footnote/endnote text and image placeholders are **unconditionally
|
|
283
|
+
> dropped** rather than being something you can ask for. Prefer `.to('text')` for new code.
|
|
275
284
|
|
|
276
285
|
```js
|
|
277
286
|
const text = ast.toText(); // synchronous, returns plain string
|
|
278
287
|
```
|
|
279
288
|
|
|
289
|
+
#### Migrating to `.to('text')`
|
|
290
|
+
|
|
291
|
+
`.to('text')` is asynchronous and configurable. Its defaults render tables as aligned grids, lists
|
|
292
|
+
with markers/indentation, and include notes and image placeholders:
|
|
293
|
+
|
|
294
|
+
```js
|
|
295
|
+
// Default: aligned table grids, list markers, notes, image placeholders
|
|
296
|
+
const { value } = await ast.to('text');
|
|
297
|
+
|
|
298
|
+
// Deliberate opt-out: the combination closest to toText()'s shape
|
|
299
|
+
const { value } = await ast.to('text', {
|
|
300
|
+
includeImages: false,
|
|
301
|
+
textConfig: { preserveLayout: false, renderNotes: false },
|
|
302
|
+
});
|
|
303
|
+
```
|
|
304
|
+
|
|
305
|
+
**At its default configuration, `.to('text')` emits everything `toText()` emits.** Verified across
|
|
306
|
+
every bundled fixture in all 12 supported formats, in both layout modes: no word `toText()` produces
|
|
307
|
+
is missing from `.to('text')`. It additionally emits notes and image placeholders, which `toText()`
|
|
308
|
+
never produces, and it renders merged table cells correctly (`toText()` glues a two-cell row into
|
|
309
|
+
`OneThree`, where `.to('text')` gives `One Three`).
|
|
310
|
+
|
|
311
|
+
Notes and images are **configuration, not intrinsic behavior**. They are on by default and you can
|
|
312
|
+
turn them off. The real difference from `toText()` is that they are a choice at all:
|
|
313
|
+
|
|
314
|
+
| | `toText()` | `.to('text')` | governed by |
|
|
315
|
+
|---|---|---|---|
|
|
316
|
+
| Tables | one cell per line | aligned grid, or tab-separated | `textConfig.preserveLayout` (default `true`) |
|
|
317
|
+
| Lists | plain text | markers + indentation, or plain | `textConfig.preserveLayout` (default `true`) |
|
|
318
|
+
| Footnotes/endnotes | never emitted | emitted by default | `textConfig.renderNotes` (default `true`) |
|
|
319
|
+
| Image placeholders | never emitted | emitted by default | `includeImages` (default `true`) |
|
|
320
|
+
| Chart data series | emitted | emitted | n/a |
|
|
321
|
+
|
|
322
|
+
Nothing about `.to('text')` forces the richer output on you. The defaults simply start from the more
|
|
323
|
+
complete document, and the opt-out above gets you back to `toText()`'s shape deliberately rather
|
|
324
|
+
than by having no alternative.
|
|
325
|
+
|
|
326
|
+
Spreadsheets (CSV/ODS/XLSX) are unaffected by `preserveLayout`: it governs `table`/`list` nodes,
|
|
327
|
+
while spreadsheet content is `sheet`/`row`/`cell`. There the default aligned grid is the most
|
|
328
|
+
faithful rendering.
|
|
329
|
+
|
|
280
330
|
---
|
|
281
331
|
|
|
282
332
|
## OfficeGenerator
|
|
@@ -306,13 +356,16 @@ const { value: html } = await OfficeGenerator.generate(ast, 'html', {
|
|
|
306
356
|
const { value: csv } = await OfficeGenerator.generate(ast, 'csv');
|
|
307
357
|
```
|
|
308
358
|
|
|
309
|
-
**Supported destinations:** `'text'` · `'md'` · `'html'` · `'csv'` · `'rtf'` · `'pdf'` · `'chunks'`
|
|
359
|
+
**Supported destinations:** `'text'` · `'md'` · `'html'` · `'csv'` · `'rtf'` · `'pdf'` · `'epub'` · `'chunks'`
|
|
310
360
|
|
|
311
361
|
> [!NOTE]
|
|
312
362
|
> **PDF generation** requires the optional `puppeteer` peer dependency:
|
|
313
363
|
> ```bash
|
|
314
364
|
> npm install puppeteer
|
|
315
365
|
> ```
|
|
366
|
+
>
|
|
367
|
+
> **EPUB generation with images** requires `extractAttachments: true` on the parse step that
|
|
368
|
+
> produced the AST — see [EPUB Support](#epub-support).
|
|
316
369
|
|
|
317
370
|
---
|
|
318
371
|
|
|
@@ -440,10 +493,10 @@ interface OfficeChunk {
|
|
|
440
493
|
|
|
441
494
|
```text
|
|
442
495
|
OfficeParserAST
|
|
443
|
-
├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | ... (
|
|
496
|
+
├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | 'epub' | ... (12 formats)
|
|
444
497
|
├── metadata: { author, title, created, modified, keywords, customProperties, nativeProperties, styleMap, ... }
|
|
445
498
|
├── content: [ OfficeContentNode ]
|
|
446
|
-
│ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | ...
|
|
499
|
+
│ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | 'admonition' | 'embed' | 'definitionList' | ...
|
|
447
500
|
│ ├── text: string (concatenated text of node + all descendants)
|
|
448
501
|
│ ├── children: [ OfficeContentNode ] (recursive structural children)
|
|
449
502
|
│ ├── notes: [ OfficeContentNode ] (footnotes/endnotes/slide notes attached to this node)
|
|
@@ -463,7 +516,7 @@ OfficeParserAST
|
|
|
463
516
|
│ └── chartData?: { title, dataSets, labels }
|
|
464
517
|
├── warnings: OfficeIssue[] (non-fatal issues from the parsing phase)
|
|
465
518
|
├── to(format, config?) (format: 'html'|'md'|'text'|'csv'|'rtf'|'pdf'|'chunks', returns { value, messages })
|
|
466
|
-
└── ~~toText()~~ (Deprecated: use .to('text')
|
|
519
|
+
└── ~~toText()~~ (Deprecated: use .to('text'); drops footnotes + image placeholders)
|
|
467
520
|
```
|
|
468
521
|
|
|
469
522
|
### `OfficeIssue`: Warning / Error Object
|
|
@@ -596,6 +649,94 @@ console.log(ast.metadata.nativeProperties);
|
|
|
596
649
|
// PDF: { Title: 'Report', XMP: { ... } }
|
|
597
650
|
```
|
|
598
651
|
|
|
652
|
+
### 8. Admonitions, Embeds & Definition Lists
|
|
653
|
+
|
|
654
|
+
```text
|
|
655
|
+
Admonition Node (type: 'admonition')
|
|
656
|
+
├── metadata: { admonitionType: 'note' | 'tip' | 'important' | 'warning' | 'caution', title?: string }
|
|
657
|
+
└── children: [ Paragraph | List | ... ] (block content)
|
|
658
|
+
|
|
659
|
+
Embed Node (type: 'embed')
|
|
660
|
+
└── metadata: { embedType: 'youtube', videoId: string, url?: string, width?: string, align?: string }
|
|
661
|
+
|
|
662
|
+
Definition List Node (type: 'definitionList')
|
|
663
|
+
└── children:
|
|
664
|
+
├── Definition Term (type: 'definitionTerm')
|
|
665
|
+
└── Definition Description (type: 'definitionDescription')
|
|
666
|
+
```
|
|
667
|
+
|
|
668
|
+
- `admonition` round-trips through both Markdown (`> [!NOTE]` / `:::note ... :::`) and HTML (`<div class="admonition admonition-note" data-type="note">`)
|
|
669
|
+
- `embed` currently models YouTube videos; HTML round-trips via `<div data-youtube-video="ID">`, Markdown falls back to a raw HTML block or a plain link
|
|
670
|
+
- Abbreviations (`*[HTML]: Hypertext Markup Language`) are stored as `TextMetadata.abbreviationTitle` on the abbreviated text node rather than as a separate node type
|
|
671
|
+
|
|
672
|
+
---
|
|
673
|
+
|
|
674
|
+
## Markdown Dialect Support
|
|
675
|
+
|
|
676
|
+
Beyond CommonMark/GFM basics, `MarkdownParser`/`MarkdownGenerator` support an extended dialect aimed at
|
|
677
|
+
full-fidelity round-tripping with rich Markdown editors. Every construct below parses to a first-class
|
|
678
|
+
AST node/metadata field and regenerates back to the canonical syntax shown, so `.md → AST → .md` is
|
|
679
|
+
idempotent and `.md → AST → HTML → AST → .md` survives unchanged.
|
|
680
|
+
|
|
681
|
+
| Feature | Markdown syntax | AST representation |
|
|
682
|
+
|---|---|---|
|
|
683
|
+
| Task lists (GFM) | `- [x] Done` / `- [ ] Todo` | `ListMetadata.isTask` / `.checked` |
|
|
684
|
+
| Admonitions | `> [!NOTE]` (also accepts GLFM `:::note ... :::` on import) | `type: 'admonition'`, `AdmonitionMetadata` |
|
|
685
|
+
| Footnotes | `Text[^1]` + `[^1]: Definition` | `type: 'note'`, keyed by footnote id |
|
|
686
|
+
| Definition lists | `Term\n: Definition` | `type: 'definitionList'` / `'definitionTerm'` / `'definitionDescription'` |
|
|
687
|
+
| Abbreviations | `*[HTML]: Hypertext Markup Language` | `TextMetadata.abbreviationTitle` |
|
|
688
|
+
| Attribute lists | `{width=50% .centered}` | `ImageMetadata.width` / `.align`, `TableMetadata.align` |
|
|
689
|
+
| Citations | `[@smith2024]` | `TextMetadata.citationKey` |
|
|
690
|
+
| Wikilinks | `[[Page]]` / `[[Page\|Alias]]` | `TextMetadata.wikilink`, `.link`, `.linkType` |
|
|
691
|
+
| Inline/block math | `$E=mc^2$` / `` $$...$$ `` | `TextMetadata.math` (`'inline' \| 'block'`) |
|
|
692
|
+
| Frontmatter arrays | `tags: [a, b]` or `tags: ["a","b"]` | Real array in `metadata.customProperties`/`nativeProperties` |
|
|
693
|
+
| MDX components (import-only) | `<Component prop="x">...</Component>` | Stripped; inner Markdown is kept. Never generated back. |
|
|
694
|
+
|
|
695
|
+
> [!NOTE]
|
|
696
|
+
> MDX/JSX stripping is one-directional (parse-only) — officeParser never authors JSX back into Markdown.
|
|
697
|
+
> Wikilink enable/disable and citekey→bibliography resolution are application-level concerns; officeParser
|
|
698
|
+
> always parses/generates the syntax itself.
|
|
699
|
+
|
|
700
|
+
The same round-trip fidelity extends to HTML, so content saved from a rich-text editor survives a
|
|
701
|
+
save→reload cycle:
|
|
702
|
+
|
|
703
|
+
| HTML attribute | AST field | Notes |
|
|
704
|
+
|---|---|---|
|
|
705
|
+
| `data-width` / `data-align` / inline `style="width:…"` on `<img>` | `ImageMetadata.width` / `.align` | |
|
|
706
|
+
| `data-align` on `<table>` | `TableMetadata.align` | |
|
|
707
|
+
| `colspan` / `rowspan` on `<td>`/`<th>` | `CellMetadata.colSpan` / `.rowSpan` | Previously dropped on HTML import — merged cells now survive a save→reload cycle |
|
|
708
|
+
| `<div data-youtube-video="ID">` / `<iframe src="...youtube.com...">` | `type: 'embed'` | |
|
|
709
|
+
| `<ul data-type="taskList">` / `<li data-checked>` | `ListMetadata.isTask` / `.checked` | |
|
|
710
|
+
|
|
711
|
+
---
|
|
712
|
+
|
|
713
|
+
## EPUB Support
|
|
714
|
+
|
|
715
|
+
EPUB files are ZIP archives of XHTML content plus an OPF manifest — `EpubParser` unzips the archive,
|
|
716
|
+
resolves the spine's reading order from `content.opf`, and parses each XHTML document through the
|
|
717
|
+
existing `HtmlParser`, so EPUB content shares the same AST shape (and the same Markdown-dialect
|
|
718
|
+
fidelity above) as every other format. Dublin Core metadata (`dc:title`, `dc:creator`, `dc:description`,
|
|
719
|
+
`dc:subject`, `dc:date`, `dc:publisher`, `dc:language`, `dc:identifier`) maps into `ast.metadata` /
|
|
720
|
+
`ast.metadata.nativeProperties`, and cover art is exposed via `metadata.customProperties.coverImageName`.
|
|
721
|
+
|
|
722
|
+
`EpubGenerator` renders the AST through `HtmlGenerator` and packages the result as a minimal, valid
|
|
723
|
+
EPUB 3 (`mimetype`, `META-INF/container.xml`, an OPF manifest, a nav document, and one XHTML chapter).
|
|
724
|
+
|
|
725
|
+
> [!IMPORTANT]
|
|
726
|
+
> **Pass `extractAttachments: true` when converting to or from EPUB if the document has images.**
|
|
727
|
+
> Without it, the parser never pulls embedded image bytes out of the source document, so there is
|
|
728
|
+
> nothing for the EPUB generator to package — images silently disappear even though everything else
|
|
729
|
+
> converts correctly. Images are packaged as real zip entries (`OEBPS/images/...`) declared in the OPF
|
|
730
|
+
> manifest, not `data:` URIs — most EPUB reading systems do not render `data:` URIs in image `src`.
|
|
731
|
+
>
|
|
732
|
+
> This only matters for the two-step `OfficeParser.parseOffice()` → `OfficeGenerator.generate()` API
|
|
733
|
+
> and the CLI. [`OfficeConverter.convert()`](#officeconverter-one-step-api) enables `extractAttachments`
|
|
734
|
+
> automatically unless you explicitly set `generatorConfig.includeImages: false`.
|
|
735
|
+
>
|
|
736
|
+
> ```bash
|
|
737
|
+
> npx officeparser book.docx --extractAttachments --to=epub --output=book.epub
|
|
738
|
+
> ```
|
|
739
|
+
|
|
599
740
|
---
|
|
600
741
|
|
|
601
742
|
## Performance Highlights
|
|
@@ -768,6 +909,7 @@ Options shared by all generator formats. Pass to `OfficeGenerator.generate(ast,
|
|
|
768
909
|
| `includeFormatting` | `boolean` | `true` | Include bold/italic/colors/sizes in output |
|
|
769
910
|
| `generateIds` | `boolean` | `true` | Add slug-based `id` attributes to headings |
|
|
770
911
|
| `renderMetadata` | `boolean` | `false` | Render title/author as visible header block |
|
|
912
|
+
| `metadataOverrides` | `MetadataOverrides` | `{}` | Override the metadata embedded in the output, merged per field over `ast.metadata` |
|
|
771
913
|
| `includeImages` | `boolean` | `true` | Include image nodes in output |
|
|
772
914
|
| `includeCharts` | `boolean` | `true` | Include interactive charts (HTML only) |
|
|
773
915
|
| `ignoreInternalLinks` | `boolean` | `false` | Strip bookmarks and internal anchors from output |
|
|
@@ -852,7 +994,7 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
|
|
|
852
994
|
|
|
853
995
|
| Option | Type | Default | Description |
|
|
854
996
|
|--------|------|---------|-------------|
|
|
855
|
-
| `standalone` | `boolean` | `true` |
|
|
997
|
+
| `standalone` | `boolean \| StandaloneConfig` | `true` | Controls the HTML "document envelope" — see below |
|
|
856
998
|
| `chartJsSrc` | `string` | jsDelivr CDN | URL for the Chart.js library |
|
|
857
999
|
| `containerWidth` | `string \| number` | `'auto'` | Max width of the content container. Positive number (px), CSS length string (`'900px'`, `'100%'`, `'60vw'`), or `'auto'`. Invalid values fall back to `'auto'` with an `INVALID_CONTAINER_WIDTH` warning |
|
|
858
1000
|
| `customCss` | `string` | `''` | Raw CSS injected into the `<style>` block; use this to override built-in styles |
|
|
@@ -861,6 +1003,49 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
|
|
|
861
1003
|
| `injections.bodyStart` | `string` | `''` | Raw HTML injected after `<body>` |
|
|
862
1004
|
| `injections.bodyEnd` | `string` | `''` | Raw HTML injected before `</body>` |
|
|
863
1005
|
|
|
1006
|
+
#### `standalone`: granular envelope control
|
|
1007
|
+
|
|
1008
|
+
`standalone` conflates several independent decisions: whether to emit the `<!doctype>/<html>/<head>/
|
|
1009
|
+
<body>` shell, how CSS is delivered, and whether to inject scripts/meta tags/injections. The boolean
|
|
1010
|
+
shorthand still works — **`true`/omitted turns every part on** (a complete document); **`false` turns
|
|
1011
|
+
every part off** (a bare content fragment, safe to drop into a page you don't control). Pass an
|
|
1012
|
+
object instead for granular control; any field you omit defaults to its "on" (standalone) value:
|
|
1013
|
+
|
|
1014
|
+
| `StandaloneConfig` field | Type | Default | Description |
|
|
1015
|
+
|--------|------|---------|-------------|
|
|
1016
|
+
| `document` | `boolean` | `true` | Wrap in `<!DOCTYPE html><html><head>…</head><body>…</body></html>` |
|
|
1017
|
+
| `metaTags` | `boolean` | `true` | Emit `<title>`/`<meta>` tags. Only meaningful when `document` is true |
|
|
1018
|
+
| `styles` | `'full' \| 'scoped' \| 'none'` | `'full'` | See below |
|
|
1019
|
+
| `scripts` | `boolean` | `true` | Emit the Chart.js CDN loader and spreadsheet-interactivity `<script>` tags |
|
|
1020
|
+
| `headInjections` | `boolean` | `true` | Apply `injections.headStart`/`headEnd`. Only meaningful when `document` is true |
|
|
1021
|
+
| `bodyInjections` | `boolean` | `true` | Apply `injections.bodyStart`/`bodyEnd` — applies even to a bare fragment |
|
|
1022
|
+
|
|
1023
|
+
`styles` controls how the built-in stylesheet is delivered:
|
|
1024
|
+
- **`'full'`** — the complete stylesheet using global selectors (`body`, `h1`, `table`, …). This is
|
|
1025
|
+
what `standalone: true` has always emitted.
|
|
1026
|
+
- **`'scoped'`** — the same styling, scoped under the fragment's own wrapper via CSS `@scope` so it
|
|
1027
|
+
cannot leak onto a host page's elements. Requires a modern engine (Chrome 118+, Safari 17.4+,
|
|
1028
|
+
Firefox 128+); for universal support use `'none'` (bring your own CSS) or `'full'`.
|
|
1029
|
+
- **`'none'`** — no stylesheet at all; the host page (or rich-text editor, or EPUB reader) supplies
|
|
1030
|
+
its own styling.
|
|
1031
|
+
|
|
1032
|
+
```js
|
|
1033
|
+
// A styled fragment to embed in your own page, without a document shell:
|
|
1034
|
+
await ast.to('html', { htmlConfig: { standalone: { document: false } } });
|
|
1035
|
+
|
|
1036
|
+
// The same, but with styles scoped so they can't leak onto your page's own elements:
|
|
1037
|
+
await ast.to('html', { htmlConfig: { standalone: { document: false, styles: 'scoped' } } });
|
|
1038
|
+
|
|
1039
|
+
// A completely bare fragment (no shell, no styles, no scripts) — e.g. for a rich-text editor:
|
|
1040
|
+
await ast.to('html', { htmlConfig: { standalone: false } });
|
|
1041
|
+
```
|
|
1042
|
+
|
|
1043
|
+
> [!NOTE]
|
|
1044
|
+
> **Behavior change from `standalone: false`:** previously this emitted a fragment with a *global,
|
|
1045
|
+
> unscoped* `<style>` block. It now emits a genuinely bare fragment (no `<style>` at all), matching
|
|
1046
|
+
> "every part off." If you relied on the old styled-fragment behavior, pass
|
|
1047
|
+
> `{ document: false }` (or `{ document: false, styles: 'full' }`) instead.
|
|
1048
|
+
|
|
864
1049
|
### MdGeneratorConfig
|
|
865
1050
|
|
|
866
1051
|
Pass as `mdConfig` inside `GeneratorConfig`.
|
|
@@ -906,6 +1091,52 @@ Pass as `textConfig` inside `GeneratorConfig`.
|
|
|
906
1091
|
|--------|------|---------|-------------|
|
|
907
1092
|
| `newlineDelimiter` | `string` | `'\n'` | String inserted between structural blocks |
|
|
908
1093
|
| `preserveLayout` | `boolean` | `true` | Render tables with aligned columns using whitespace |
|
|
1094
|
+
| `renderNotes` | `boolean` | `true` | Append the collected footnote/endnote section |
|
|
1095
|
+
|
|
1096
|
+
### metadataOverrides
|
|
1097
|
+
|
|
1098
|
+
Part of the common `GeneratorConfig` (not format-specific). Overrides the metadata embedded in
|
|
1099
|
+
generated output, applied **per field** on top of `ast.metadata`, so setting one field leaves the
|
|
1100
|
+
rest of the parsed metadata intact. `ast.metadata` itself is never mutated, so the same AST can be
|
|
1101
|
+
generated repeatedly with different metadata.
|
|
1102
|
+
|
|
1103
|
+
| Field | Type | Written as |
|
|
1104
|
+
|-------|------|-----------|
|
|
1105
|
+
| `title` | `string` | HTML `<title>`/`<meta>`, EPUB `dc:title`, Markdown frontmatter, RTF `\title` |
|
|
1106
|
+
| `author` | `string` | HTML `<meta name="author">`, EPUB `dc:creator`, frontmatter, RTF `\author` |
|
|
1107
|
+
| `description` | `string` | HTML `<meta name="description">`, EPUB `dc:description`, frontmatter |
|
|
1108
|
+
| `subject` / `keywords` / `lastModifiedBy` | `string` | Where the destination format has a slot |
|
|
1109
|
+
| `created` / `modified` | `Date` | HTML `dcterms.*`, EPUB `dcterms:modified`, frontmatter |
|
|
1110
|
+
| `language` | `string` | EPUB `dc:language` |
|
|
1111
|
+
| `custom` | `Record<string, string \| number \| boolean \| Date>` | HTML `<meta name="custom:KEY">`, Markdown frontmatter |
|
|
1112
|
+
|
|
1113
|
+
```js
|
|
1114
|
+
// Rebrand the output without touching the parsed document
|
|
1115
|
+
const { value } = await ast.to('html', {
|
|
1116
|
+
metadataOverrides: { title: 'Q4 Report', author: 'Acme Inc', custom: { department: 'Finance' } },
|
|
1117
|
+
});
|
|
1118
|
+
```
|
|
1119
|
+
|
|
1120
|
+
#### Dates
|
|
1121
|
+
|
|
1122
|
+
`created` and `modified` take a `Date`. When `modified` is unset, officeParser uses the source
|
|
1123
|
+
document's own `ast.metadata.modified`, falling back to the current time only if the document has
|
|
1124
|
+
none. Dates outside the 1980-2099 range representable in a ZIP timestamp are clamped where EPUB
|
|
1125
|
+
writes them onto zip entries.
|
|
1126
|
+
|
|
1127
|
+
```js
|
|
1128
|
+
const { value } = await ast.to('epub', {
|
|
1129
|
+
metadataOverrides: { modified: new Date('2024-01-01T00:00:00Z') },
|
|
1130
|
+
});
|
|
1131
|
+
```
|
|
1132
|
+
|
|
1133
|
+
#### Not every format can represent every field
|
|
1134
|
+
|
|
1135
|
+
HTML `<meta>` tags and Markdown frontmatter are open vocabularies and accept anything. EPUB's OPF
|
|
1136
|
+
is a closed Dublin Core vocabulary and RTF's `\info` group has a fixed set of control words, so a
|
|
1137
|
+
`custom` entry has nowhere to go in either. Rather than dropping it silently, those generators
|
|
1138
|
+
report it through `onWarning` (`OfficeWarningType.METADATA_NOT_REPRESENTABLE`) and continue; the
|
|
1139
|
+
named fields still apply.
|
|
909
1140
|
|
|
910
1141
|
---
|
|
911
1142
|
|
|
@@ -1028,7 +1259,7 @@ const handleFile = async (event) => {
|
|
|
1028
1259
|
const file = event.target.files[0];
|
|
1029
1260
|
const buffer = await file.arrayBuffer();
|
|
1030
1261
|
const ast = await OfficeParser.parseOffice(new Uint8Array(buffer));
|
|
1031
|
-
console.log(ast.
|
|
1262
|
+
console.log((await ast.to('text')).value);
|
|
1032
1263
|
};
|
|
1033
1264
|
```
|
|
1034
1265
|
|
|
@@ -1041,7 +1272,7 @@ const handleFile = async (event) => {
|
|
|
1041
1272
|
const file = event.target.files[0];
|
|
1042
1273
|
const buffer = await file.arrayBuffer();
|
|
1043
1274
|
const ast = await officeParser.parseOffice(new Uint8Array(buffer));
|
|
1044
|
-
console.log(ast.
|
|
1275
|
+
console.log((await ast.to('text')).value);
|
|
1045
1276
|
}
|
|
1046
1277
|
</script>
|
|
1047
1278
|
```
|
|
@@ -1076,7 +1307,7 @@ const ast = await officeParser.parseOffice(pdfArrayBuffer, {
|
|
|
1076
1307
|
| Node.js process stays alive after finishing | Call `await officeParser.terminateOcr()` at end of script when OCR was used |
|
|
1077
1308
|
| `"Worker not found"` in browser for PDF | Verify `pdfWorkerSrc` points to `pdf.worker.min.mjs` matching version `6.1.200` |
|
|
1078
1309
|
| Low OCR accuracy | Verify `ocrConfig.language` matches the document language; quality depends on image resolution |
|
|
1079
|
-
| Out of memory on large Excel files | Call `ast.
|
|
1310
|
+
| Out of memory on large Excel files | Call `await ast.to('text')` early and discard the AST object to allow garbage collection |
|
|
1080
1311
|
| `md`/`html`/`csv` buffer not detected | Add `fileType: 'md'` (or `'html'`, `'csv'`) to config (these formats have no magic bytes) |
|
|
1081
1312
|
| `IMPROPER_BUFFERS` error | Usually means no file extension and no `fileType` hint was provided for a buffer input |
|
|
1082
1313
|
| PDF generation fails | Install the optional peer dependency: `npm install puppeteer` |
|
|
@@ -1092,6 +1323,35 @@ For a full debugging guide, visit the [Live Documentation](https://harshankur.gi
|
|
|
1092
1323
|
|
|
1093
1324
|
---
|
|
1094
1325
|
|
|
1326
|
+
## Security & Trust Boundary
|
|
1327
|
+
|
|
1328
|
+
`officeParser` is a **parsing, generation, and conversion** library. Like any parser, its whole job
|
|
1329
|
+
is to open and interpret files it is handed, and those files may come from an untrusted source (a
|
|
1330
|
+
user upload, an email attachment, a scraped document). A parser that accepts arbitrary documents
|
|
1331
|
+
has a large and inherently open attack surface.
|
|
1332
|
+
|
|
1333
|
+
I do sanitize output and apply hardening where I can: injection escaping across the
|
|
1334
|
+
HTML/CSS/URL/script/CSV/RTF/Markdown sinks, decompression limits, some resource and recursion
|
|
1335
|
+
bounds, and SSRF precautions during PDF rendering. I fix issues as I learn of them (see
|
|
1336
|
+
[CHANGELOG.md](CHANGELOG.md)). But this is **best-effort, not a guarantee.** A document parser of
|
|
1337
|
+
this size will have attack vectors I have not found or have not yet addressed, and no amount of
|
|
1338
|
+
internal hardening makes it safe to feed fully untrusted input without your own precautions.
|
|
1339
|
+
|
|
1340
|
+
**Treat this as garbage in, garbage out.** The library does its best with what you give it, but
|
|
1341
|
+
responsibility for what you feed it, and for the effect a malicious file has on your system, rests
|
|
1342
|
+
with you. If you process files from untrusted sources, sanitize and validate them at your own
|
|
1343
|
+
boundary, and run the parsing in isolation appropriate to your threat model: sandboxing or
|
|
1344
|
+
containerization, memory and time limits, a low-privilege process, and the `abortSignal` and
|
|
1345
|
+
`decompressionLimits` options this library exposes. Do not rely on any single library's hardening
|
|
1346
|
+
as a complete defense.
|
|
1347
|
+
|
|
1348
|
+
I am the sole maintainer, with no security team behind me. I take legitimate reports seriously and
|
|
1349
|
+
will fix what I reasonably can, but I cannot commit to a response or resolution timeline. The
|
|
1350
|
+
software is provided "AS IS" without warranty of any kind (see [LICENSE](LICENSE)). To report an
|
|
1351
|
+
issue privately, see [SECURITY.md](SECURITY.md).
|
|
1352
|
+
|
|
1353
|
+
---
|
|
1354
|
+
|
|
1095
1355
|
**npm**: [https://npmjs.com/package/officeparser](https://npmjs.com/package/officeparser)
|
|
1096
1356
|
|
|
1097
1357
|
**github**: [https://github.com/harshankur/officeParser](https://github.com/harshankur/officeParser)
|
|
@@ -42,6 +42,6 @@ export declare class OfficeConverter {
|
|
|
42
42
|
* });
|
|
43
43
|
* ```
|
|
44
44
|
*/
|
|
45
|
-
static convert<F extends string | Buffer | ArrayBuffer | Uint8Array, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination:
|
|
45
|
+
static convert<F extends string | Buffer | ArrayBuffer | Uint8Array, T extends SupportedFileType = InferFileTypeFromPath<F>, D extends SupportedDestination<T> = SupportedDestination<T>>(file: F, destination: D, config?: OfficeConverterConfig<D, T>): Promise<ConversionResult<D>>;
|
|
46
46
|
}
|
|
47
47
|
export {};
|
package/dist/OfficeConverter.js
CHANGED
|
@@ -65,6 +65,9 @@ class OfficeConverter {
|
|
|
65
65
|
...config?.generatorConfig,
|
|
66
66
|
onWarning: config?.onWarning || config?.generatorConfig?.onWarning,
|
|
67
67
|
};
|
|
68
|
+
// The spread widens the object to an inferred literal; it is a `GeneratorConfig<D>` by
|
|
69
|
+
// construction (config.generatorConfig is already typed for D, and onWarning is common to
|
|
70
|
+
// every destination), so the cast restates what the types otherwise lose.
|
|
68
71
|
const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, destination, generatorConfig);
|
|
69
72
|
result.messages = [...(ast.warnings || []), ...result.messages];
|
|
70
73
|
return result;
|
package/dist/OfficeGenerator.js
CHANGED
|
@@ -3,6 +3,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
|
|
|
3
3
|
exports.OfficeGenerator = void 0;
|
|
4
4
|
const ChunkingGenerator_js_1 = require("./generators/ChunkingGenerator.js");
|
|
5
5
|
const CsvGenerator_js_1 = require("./generators/CsvGenerator.js");
|
|
6
|
+
const EpubGenerator_js_1 = require("./generators/EpubGenerator.js");
|
|
6
7
|
const HtmlGenerator_js_1 = require("./generators/HtmlGenerator.js");
|
|
7
8
|
const MarkdownGenerator_js_1 = require("./generators/MarkdownGenerator.js");
|
|
8
9
|
const PdfGenerator_js_1 = require("./generators/PdfGenerator.js");
|
|
@@ -60,6 +61,9 @@ class OfficeGenerator {
|
|
|
60
61
|
case 'chunks':
|
|
61
62
|
generator = new ChunkingGenerator_js_1.ChunkingGenerator(ast, config);
|
|
62
63
|
break;
|
|
64
|
+
case 'epub':
|
|
65
|
+
generator = new EpubGenerator_js_1.EpubGenerator(ast, config);
|
|
66
|
+
break;
|
|
63
67
|
default:
|
|
64
68
|
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED, undefined, destination);
|
|
65
69
|
}
|
package/dist/OfficeParser.d.ts
CHANGED
|
@@ -14,6 +14,7 @@
|
|
|
14
14
|
* - CSV (Comma-Separated Values)
|
|
15
15
|
* - MD (Markdown)
|
|
16
16
|
* - HTML (HyperText Markup Language)
|
|
17
|
+
* - EPUB (E-book format)
|
|
17
18
|
*
|
|
18
19
|
* **Usage:**
|
|
19
20
|
* ```typescript
|
|
@@ -66,6 +67,7 @@ export declare class OfficeParser {
|
|
|
66
67
|
* - `.csv` → CsvParser
|
|
67
68
|
* - `.md` → MarkdownParser
|
|
68
69
|
* - `.html` → HtmlParser
|
|
70
|
+
* - `.epub` → EpubParser
|
|
69
71
|
*
|
|
70
72
|
* @param file - File path (string), Buffer, or ArrayBuffer containing the document
|
|
71
73
|
* @param config - Optional configuration object (defaults applied for all omitted options)
|
package/dist/OfficeParser.js
CHANGED
|
@@ -15,6 +15,7 @@
|
|
|
15
15
|
* - CSV (Comma-Separated Values)
|
|
16
16
|
* - MD (Markdown)
|
|
17
17
|
* - HTML (HyperText Markup Language)
|
|
18
|
+
* - EPUB (E-book format)
|
|
18
19
|
*
|
|
19
20
|
* **Usage:**
|
|
20
21
|
* ```typescript
|
|
@@ -39,6 +40,7 @@
|
|
|
39
40
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
40
41
|
exports.OfficeParser = void 0;
|
|
41
42
|
const CsvParser_js_1 = require("./parsers/CsvParser.js");
|
|
43
|
+
const EpubParser_js_1 = require("./parsers/EpubParser.js");
|
|
42
44
|
const ExcelParser_js_1 = require("./parsers/ExcelParser.js");
|
|
43
45
|
const HtmlParser_js_1 = require("./parsers/HtmlParser.js");
|
|
44
46
|
const MarkdownParser_js_1 = require("./parsers/MarkdownParser.js");
|
|
@@ -83,6 +85,7 @@ class OfficeParser {
|
|
|
83
85
|
* - `.csv` → CsvParser
|
|
84
86
|
* - `.md` → MarkdownParser
|
|
85
87
|
* - `.html` → HtmlParser
|
|
88
|
+
* - `.epub` → EpubParser
|
|
86
89
|
*
|
|
87
90
|
* @param file - File path (string), Buffer, or ArrayBuffer containing the document
|
|
88
91
|
* @param config - Optional configuration object (defaults applied for all omitted options)
|
|
@@ -232,6 +235,9 @@ class OfficeParser {
|
|
|
232
235
|
case 'md':
|
|
233
236
|
result = await (0, MarkdownParser_js_1.parseMarkdown)(buffer, internalConfig);
|
|
234
237
|
break;
|
|
238
|
+
case 'epub':
|
|
239
|
+
result = await (0, EpubParser_js_1.parseEpub)(buffer, internalConfig);
|
|
240
|
+
break;
|
|
235
241
|
default:
|
|
236
242
|
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
|
|
237
243
|
}
|
package/dist/cli.d.ts
CHANGED
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
* officeparser file.docx --ocr --extractAttachments
|
|
9
9
|
*
|
|
10
10
|
* Options (--key=value, --key value, or bare flags):
|
|
11
|
-
* --to=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format (default: json)
|
|
11
|
+
* --to=json|text|md|html|csv|rtf|pdf|epub|chunks Convert AST to specified format (default: json)
|
|
12
12
|
* --output=path Save result to a file
|
|
13
13
|
* --fileType=docx|xlsx|... Override file type detection
|
|
14
14
|
* --ocr Enable OCR for images (default: false)
|
package/dist/cli.js
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
* officeparser file.docx --ocr --extractAttachments
|
|
10
10
|
*
|
|
11
11
|
* Options (--key=value, --key value, or bare flags):
|
|
12
|
-
* --to=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format (default: json)
|
|
12
|
+
* --to=json|text|md|html|csv|rtf|pdf|epub|chunks Convert AST to specified format (default: json)
|
|
13
13
|
* --output=path Save result to a file
|
|
14
14
|
* --fileType=docx|xlsx|... Override file type detection
|
|
15
15
|
* --ocr Enable OCR for images (default: false)
|
|
@@ -333,7 +333,7 @@ else {
|
|
|
333
333
|
console.log('Usage: officeparser <file> [options]');
|
|
334
334
|
console.log('');
|
|
335
335
|
console.log('Options:');
|
|
336
|
-
console.log(' --to=json|text|md|html|pdf|csv|rtf|chunks
|
|
336
|
+
console.log(' --to=json|text|md|html|pdf|csv|rtf|epub|chunks Target conversion format (default: json)');
|
|
337
337
|
console.log(' --output=file.ext Save output to file instead of stdout');
|
|
338
338
|
console.log(' --fileType=docx|xlsx|pptx|odt|... Explicitly override input file type detection');
|
|
339
339
|
console.log(' --ocr Enable OCR for images (default: false)');
|
|
@@ -360,6 +360,8 @@ else {
|
|
|
360
360
|
console.log('Advanced Nested Config Examples:');
|
|
361
361
|
console.log(' --pdfConfig.format=Letter Configure Puppeteer PDF format (A4 | Letter | Legal etc.)');
|
|
362
362
|
console.log(' --chunksConfig.strategy=fixed-size Chunking strategy (fixed-size | document-structure | semantic)');
|
|
363
|
+
console.log(' --mdConfig.dialect=github Markdown dialect (extended | github | gitlab | obsidian | pandoc | commonmark)');
|
|
364
|
+
console.log(' --mdConfig.fallbackToHtml=false Disable HTML fallback for unsupported Markdown features (default: true)');
|
|
363
365
|
console.log('');
|
|
364
366
|
console.log('Format Syntax:');
|
|
365
367
|
console.log(' Flags can be written as --flag (presence implies true), --no-flag (negation),');
|
|
@@ -371,5 +373,6 @@ else {
|
|
|
371
373
|
console.log(' officeparser document.docx --to md');
|
|
372
374
|
console.log(' officeparser report.pdf --ocr --ocrConfig.language eng --to text');
|
|
373
375
|
console.log(' officeparser data.xlsx --to csv --output data.csv --csvDelimiter ";"');
|
|
376
|
+
console.log(' officeparser document.docx --extractAttachments --to epub --output document.epub');
|
|
374
377
|
console.log(' officeparser image_doc --fileType docx --to json');
|
|
375
378
|
}
|
package/dist/defaults.js
CHANGED
|
@@ -34,6 +34,13 @@ const DEFAULT_OCR_CONFIG = {
|
|
|
34
34
|
autoTerminateTimeout: DEFAULT_OCR_TIMEOUT.autoTerminate,
|
|
35
35
|
abortSignal: null,
|
|
36
36
|
};
|
|
37
|
+
/**
|
|
38
|
+
* Default configuration for HTML/XHTML parsing. `preserveAttributes` is off so that the AST is
|
|
39
|
+
* byte-identical to previous releases unless a caller opts in - see `HtmlParserConfig`.
|
|
40
|
+
*/
|
|
41
|
+
const DEFAULT_HTML_PARSER_CONFIG = {
|
|
42
|
+
preserveAttributes: false,
|
|
43
|
+
};
|
|
37
44
|
/**
|
|
38
45
|
* Default configuration for the OfficeParser.
|
|
39
46
|
*/
|
|
@@ -62,7 +69,9 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
|
|
|
62
69
|
decompressionLimits: {
|
|
63
70
|
maxUncompressedBytes: 512 * 1024 * 1024,
|
|
64
71
|
maxZipEntries: 10000,
|
|
72
|
+
maxTableCells: 1000000,
|
|
65
73
|
},
|
|
74
|
+
htmlParserConfig: DEFAULT_HTML_PARSER_CONFIG,
|
|
66
75
|
};
|
|
67
76
|
/**
|
|
68
77
|
* Default configuration for HTML generation.
|
|
@@ -117,6 +126,7 @@ const DEFAULT_CSV_GENERATOR_CONFIG = {
|
|
|
117
126
|
*/
|
|
118
127
|
const DEFAULT_MD_GENERATOR_CONFIG = {
|
|
119
128
|
fallbackToHtml: true,
|
|
129
|
+
dialect: 'extended',
|
|
120
130
|
};
|
|
121
131
|
/**
|
|
122
132
|
* Default configuration for plain text generation.
|
|
@@ -124,6 +134,7 @@ const DEFAULT_MD_GENERATOR_CONFIG = {
|
|
|
124
134
|
const DEFAULT_TEXT_GENERATOR_CONFIG = {
|
|
125
135
|
newlineDelimiter: '\n',
|
|
126
136
|
preserveLayout: true,
|
|
137
|
+
renderNotes: true,
|
|
127
138
|
};
|
|
128
139
|
/**
|
|
129
140
|
* Default configuration for Fixed-Size chunking.
|
|
@@ -187,6 +198,7 @@ exports.DEFAULT_GENERATOR_CONFIG = {
|
|
|
187
198
|
includeFormatting: true,
|
|
188
199
|
generateIds: true,
|
|
189
200
|
renderMetadata: false,
|
|
201
|
+
metadataOverrides: {},
|
|
190
202
|
ignoreDefaultStyleMap: false,
|
|
191
203
|
includeImages: true,
|
|
192
204
|
includeCharts: true,
|