officeparser 7.1.0 → 7.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +152 -56
- package/dist/OfficeGenerator.d.ts +6 -2
- package/dist/OfficeGenerator.js +30 -9
- package/dist/OfficeParser.d.ts +1 -1
- package/dist/OfficeParser.js +1 -1
- package/dist/cli.d.ts +18 -12
- package/dist/cli.js +255 -81
- package/dist/defaults.js +12 -1
- package/dist/generators/BaseGenerator.d.ts +4 -3
- package/dist/generators/BaseGenerator.js +13 -1
- package/dist/generators/ChunkingGenerator.js +32 -5
- package/dist/generators/CsvGenerator.d.ts +1 -1
- package/dist/generators/HtmlGenerator.d.ts +2 -1
- package/dist/generators/HtmlGenerator.js +481 -42
- package/dist/generators/MarkdownGenerator.d.ts +1 -1
- package/dist/generators/MarkdownGenerator.js +35 -2
- package/dist/generators/PdfGenerator.d.ts +1 -1
- package/dist/generators/PdfGenerator.js +0 -6
- package/dist/generators/RtfGenerator.d.ts +2 -1
- package/dist/generators/RtfGenerator.js +49 -6
- package/dist/generators/TextGenerator.d.ts +1 -1
- package/dist/generators/TextGenerator.js +6 -0
- package/dist/officeparser.browser.d.ts +267 -54
- package/dist/officeparser.browser.iife.js +599 -187
- package/dist/officeparser.browser.mjs +599 -187
- package/dist/parsers/CsvParser.js +1 -1
- package/dist/parsers/ExcelParser.js +63 -19
- package/dist/parsers/HtmlParser.js +10 -1
- package/dist/parsers/MarkdownParser.js +13 -10
- package/dist/parsers/OpenOfficeParser.js +57 -34
- package/dist/parsers/PdfParser.js +28 -3
- package/dist/parsers/PowerPointParser.js +164 -40
- package/dist/parsers/RtfParser.js +28 -24
- package/dist/parsers/WordParser.js +154 -11
- package/dist/sbom.cdx.json +100 -100
- package/dist/types.d.ts +268 -53
- package/dist/types.js +4 -0
- package/dist/utils/astUtils.d.ts +2 -2
- package/dist/utils/astUtils.js +2 -1
- package/dist/utils/configUtils.d.ts +5 -0
- package/dist/utils/configUtils.js +55 -1
- package/dist/utils/errorUtils.js +3 -1
- package/dist/utils/moduleLoader.js +55 -11
- package/dist/utils/xmlUtils.d.ts +9 -0
- package/dist/utils/xmlUtils.js +53 -1
- package/package.json +6 -3
package/README.md
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
# officeParser
|
|
1
|
+
# officeParser: Universal Office Document Parser & Generator
|
|
2
2
|
|
|
3
3
|
A robust, strictly-typed **Node.js and Browser** library for parsing office files into a rich **Abstract Syntax Tree (AST)** and generating high-fidelity output in multiple formats.
|
|
4
4
|
|
|
@@ -14,7 +14,7 @@ A robust, strictly-typed **Node.js and Browser** library for parsing office file
|
|
|
14
14
|
---
|
|
15
15
|
|
|
16
16
|
### 🌟 [Live Interactive AST Visualizer & Documentation](https://harshankur.github.io/officeParser/) 🌟
|
|
17
|
-
*Upload any office file in your browser
|
|
17
|
+
*Upload any office file in your browser: inspect the AST, tweak config, and preview generated output in real-time.*
|
|
18
18
|
|
|
19
19
|
- **AST Visualizer**: Inspect the hierarchical node tree, metadata, and raw content
|
|
20
20
|
- **Config Configurator**: Tweak options (`ignoreNotes`, `ocr`, `newlineDelimiter`) and see results instantly
|
|
@@ -35,10 +35,10 @@ A robust, strictly-typed **Node.js and Browser** library for parsing office file
|
|
|
35
35
|
- [Async/Await](#asyncawait)
|
|
36
36
|
- [Callback (Backward Compat)](#callback-backward-compat)
|
|
37
37
|
- [File Buffers & ArrayBuffers](#file-buffers--arraybuffers)
|
|
38
|
-
- [`ast.to()
|
|
39
|
-
- [`ast.toText()
|
|
38
|
+
- [`ast.to()`: Generate from AST](#astto-generate-from-ast)
|
|
39
|
+
- [`ast.toText()`: Quick Text Extraction](#asttotext-quick-text-extraction)
|
|
40
40
|
- [OfficeGenerator](#officegenerator)
|
|
41
|
-
- [OfficeConverter
|
|
41
|
+
- [OfficeConverter: One-Step API](#officeconverter-one-step-api)
|
|
42
42
|
- [Native RAG Chunking](#native-rag-chunking)
|
|
43
43
|
- [The AST Structure](#the-ast-structure)
|
|
44
44
|
- [Deep Dive: Document Components](#deep-dive-document-components)
|
|
@@ -47,8 +47,8 @@ A robust, strictly-typed **Node.js and Browser** library for parsing office file
|
|
|
47
47
|
- [Configuration Reference](#configuration-reference)
|
|
48
48
|
- [OfficeParserConfig](#officeparserconfig)
|
|
49
49
|
- [GeneratorConfig (Common)](#generatorconfig-common)
|
|
50
|
-
- [onNode Callback](#onnode-callback
|
|
51
|
-
- [styleMap
|
|
50
|
+
- [onNode Callback](#onnode-callback-advanced-node-manipulation)
|
|
51
|
+
- [styleMap: Semantic Style Mapping](#stylemap-semantic-style-mapping)
|
|
52
52
|
- [HtmlGeneratorConfig](#htmlgeneratorconfig)
|
|
53
53
|
- [MdGeneratorConfig](#mdgeneratorconfig)
|
|
54
54
|
- [PdfGeneratorConfig](#pdfgeneratorconfig)
|
|
@@ -79,37 +79,58 @@ npm i officeparser
|
|
|
79
79
|
npx officeparser /path/to/file.docx
|
|
80
80
|
|
|
81
81
|
# Plain text output
|
|
82
|
-
npx officeparser /path/to/file.docx --
|
|
82
|
+
npx officeparser /path/to/file.docx --to=text
|
|
83
83
|
|
|
84
84
|
# Convert DOCX to Markdown and save
|
|
85
|
-
npx officeparser report.docx --
|
|
85
|
+
npx officeparser report.docx --to=md --output=report.md
|
|
86
86
|
|
|
87
|
-
# Convert PPTX to HTML
|
|
88
|
-
npx officeparser presentation.pptx --
|
|
87
|
+
# Convert PPTX to HTML (using a bare flag for ocr)
|
|
88
|
+
npx officeparser presentation.pptx --to=html --output=preview.html --ocr
|
|
89
89
|
|
|
90
|
-
# Convert XLSX to CSV
|
|
91
|
-
npx officeparser data.xlsx --
|
|
90
|
+
# Convert XLSX to CSV with a custom delimiter
|
|
91
|
+
npx officeparser data.xlsx --to=csv --csvDelimiter=";"
|
|
92
92
|
|
|
93
93
|
# Generate RAG chunks
|
|
94
|
-
npx officeparser document.pdf --
|
|
94
|
+
npx officeparser document.pdf --to=chunks
|
|
95
|
+
|
|
96
|
+
# Overriding file extension mapping
|
|
97
|
+
npx officeparser my_document --fileType=docx --to=json
|
|
95
98
|
```
|
|
96
99
|
|
|
100
|
+
### CLI Syntax
|
|
101
|
+
- **Values:** Flags can be passed as `--flag=value` or `--flag value`.
|
|
102
|
+
- **Booleans:** Bare flags imply `true` (e.g. `--ocr` is equivalent to `--ocr=true`). Negation flags start with `no-` (e.g. `--no-ocr` is equivalent to `--ocr=false`).
|
|
103
|
+
- **Nested Objects:** You can pass nested properties directly using JSON dot-notation (e.g. `--ocrConfig.language=fra` or `--htmlConfig.containerWidth=900px`).
|
|
104
|
+
|
|
97
105
|
### CLI Options
|
|
98
106
|
|
|
99
107
|
| Flag | Values | Default | Description |
|
|
100
108
|
|------|--------|---------|-------------|
|
|
101
|
-
| `--
|
|
109
|
+
| `--to` | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | Output format |
|
|
102
110
|
| `--output` | path | — | Write output to a file |
|
|
103
|
-
| `--
|
|
104
|
-
| `--
|
|
105
|
-
| `--
|
|
106
|
-
| `--
|
|
107
|
-
| `--
|
|
108
|
-
| `--
|
|
109
|
-
| `--
|
|
110
|
-
| `--
|
|
111
|
-
| `--
|
|
112
|
-
| `--
|
|
111
|
+
| `--fileType` | `docx\|xlsx\|pptx\|odt\|odp\|ods\|pdf\|rtf\|csv\|md\|html` | — | Explicitly override input file type detection |
|
|
112
|
+
| `--ocr` | boolean | `false` | Enable OCR for images |
|
|
113
|
+
| `--extractAttachments` | boolean | `false` | Extract images/charts as Base64 |
|
|
114
|
+
| `--ignoreNotes` | boolean | `false` | Ignore footnotes/endnotes/speaker notes |
|
|
115
|
+
| `--ignoreComments` | boolean | `false` | Ignore inline comments |
|
|
116
|
+
| `--ignoreHeadersAndFooters` | boolean | `false` | Ignore headers and footers |
|
|
117
|
+
| `--ignoreSlideMasters` | boolean | `false` | Ignore slide masters |
|
|
118
|
+
| `--ignoreInternalLinks` | boolean | `false` | Ignore internal links |
|
|
119
|
+
| `--newlineDelimiter` | string | `\n` | Delimiter between lines/blocks in plaintext outputs |
|
|
120
|
+
| `--csvDelimiter` | string | `,` | Custom delimiter for CSV files |
|
|
121
|
+
| `--includeRawContent` | boolean | `false` | Include raw XML/RTF in nodes |
|
|
122
|
+
| `--serializeRawContent` | boolean | `true` | Include stringified XML in metadata |
|
|
123
|
+
| `--preserveXmlWhitespace` | boolean | `false` | Keep raw formatting space |
|
|
124
|
+
| `--includeBreakNodes` | boolean | `false` | Include break nodes (DOCX only) |
|
|
125
|
+
| `--verbose` | boolean | `false` | Show full error stack traces and warning logs |
|
|
126
|
+
| `--includeFormatting` | boolean | `true` | Include formatting style map matching |
|
|
127
|
+
| `--renderMetadata` | boolean | `false` | Render metadata as visible content in the generated output |
|
|
128
|
+
| `--htmlConfig.containerWidth` | string \| number | `auto` | HTML output container width (e.g. `900px`, `100%`) |
|
|
129
|
+
| ~~`--format`~~ | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | **Deprecated.** Use `--to` |
|
|
130
|
+
| ~~`--toText`~~ | `true\|false` | `false` | **Deprecated.** Use `--to=text` |
|
|
131
|
+
| ~~`--ocrLanguage`~~ | string | `eng` | **Deprecated.** Use `--ocrConfig.language` |
|
|
132
|
+
| ~~`--putNotesAtLast`~~ | `true\|false` | `false` | **Deprecated and ignored.** Notes are attached structurally to their nodes. |
|
|
133
|
+
| ~~`--outputErrorToConsole`~~ | `true\|false` | `false` | **Deprecated.** Use `--verbose` |
|
|
113
134
|
|
|
114
135
|
---
|
|
115
136
|
|
|
@@ -232,7 +253,7 @@ const ast = await officeParser.parseOffice('scanned_document.pdf', {
|
|
|
232
253
|
> **Non-Fatal Timeout Recovery**
|
|
233
254
|
> If `workerLoad` or `recognition` timeouts are exceeded, the parser will log a warning in `ast.warnings` and **continue parsing the rest of the document**. The overall promise resolves successfully with the text extracted from the document layers (rather than failing the entire parse).
|
|
234
255
|
|
|
235
|
-
### `ast.to()
|
|
256
|
+
### `ast.to()`: Generate from AST
|
|
236
257
|
|
|
237
258
|
The preferred way to convert a parsed AST to another format. Returns a `ConversionResult`.
|
|
238
259
|
|
|
@@ -246,7 +267,7 @@ const { value: chunks } = await ast.to('chunks', { strategy: 'fixed-
|
|
|
246
267
|
const { value: pdfBytes } = await ast.to('pdf'); // Uint8Array
|
|
247
268
|
```
|
|
248
269
|
|
|
249
|
-
### `ast.toText()
|
|
270
|
+
### `ast.toText()`: Quick Text Extraction
|
|
250
271
|
|
|
251
272
|
> [!NOTE]
|
|
252
273
|
> `toText()` is **synchronous** and deprecated in favour of the async `ast.to('text')`.
|
|
@@ -295,7 +316,7 @@ const { value: csv } = await OfficeGenerator.generate(ast, 'csv');
|
|
|
295
316
|
|
|
296
317
|
---
|
|
297
318
|
|
|
298
|
-
## OfficeConverter
|
|
319
|
+
## OfficeConverter: One-Step API
|
|
299
320
|
|
|
300
321
|
`OfficeConverter.convert()` combines parsing and generation in a single call. It automatically syncs parser options from generator config (e.g., enables `extractAttachments` when images are requested).
|
|
301
322
|
|
|
@@ -326,7 +347,7 @@ const { value: html, messages } = await OfficeConverter.convert('data.xlsx', 'ht
|
|
|
326
347
|
|
|
327
348
|
> [!IMPORTANT]
|
|
328
349
|
> The `OfficeConverterConfig` shape uses **nested** `parseConfig` and `generatorConfig` sub-objects.
|
|
329
|
-
> Do **not** put parser or generator options at the top level
|
|
350
|
+
> Do **not** put parser or generator options at the top level; only `onWarning` lives there.
|
|
330
351
|
|
|
331
352
|
---
|
|
332
353
|
|
|
@@ -344,7 +365,7 @@ const { value: chunks } = await OfficeConverter.convert('report.docx', 'chunks',
|
|
|
344
365
|
strategy: 'document-structure',
|
|
345
366
|
splitBy: 'heading', // 'paragraph' | 'heading' | 'page' | 'slide' | 'sheet'
|
|
346
367
|
maxChunkSize: 1500,
|
|
347
|
-
tableSplitStrategy: 'row', // repeats header row in every chunk
|
|
368
|
+
tableSplitStrategy: 'row', // repeats header row in every chunk, ideal for RAG
|
|
348
369
|
}
|
|
349
370
|
}
|
|
350
371
|
});
|
|
@@ -420,13 +441,19 @@ interface OfficeChunk {
|
|
|
420
441
|
```text
|
|
421
442
|
OfficeParserAST
|
|
422
443
|
├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | ... (11 formats)
|
|
423
|
-
├── metadata: { author, title, created, modified, customProperties, styleMap, ... }
|
|
444
|
+
├── metadata: { author, title, created, modified, keywords, customProperties, nativeProperties, styleMap, ... }
|
|
424
445
|
├── content: [ OfficeContentNode ]
|
|
425
|
-
│ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | ...
|
|
446
|
+
│ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | ...
|
|
426
447
|
│ ├── text: string (concatenated text of node + all descendants)
|
|
427
|
-
│ ├── children: [ OfficeContentNode ] (recursive)
|
|
448
|
+
│ ├── children: [ OfficeContentNode ] (recursive structural children)
|
|
449
|
+
│ ├── notes: [ OfficeContentNode ] (footnotes/endnotes/slide notes attached to this node)
|
|
450
|
+
│ ├── comments: [ OfficeContentNode ] (inline comments attached to this node)
|
|
428
451
|
│ ├── formatting: { bold, italic, underline, color, size, font, alignment, ... }
|
|
429
|
-
│ └── metadata: { level, listId, row, col, rowSpan, colSpan, style, ... }
|
|
452
|
+
│ └── metadata: { level, listId, row, col, rowSpan, colSpan, backgroundColor, style, ... }
|
|
453
|
+
├── auxiliary?: OfficeAuxiliaryContent (out-of-band layout elements)
|
|
454
|
+
│ ├── headers?: OfficeContentNode[] (DOCX headers)
|
|
455
|
+
│ ├── footers?: OfficeContentNode[] (DOCX footers)
|
|
456
|
+
│ └── slideMasters?: OfficeContentNode[] (PPTX slide masters)
|
|
430
457
|
├── attachments: [ OfficeAttachment ] (populated when extractAttachments: true)
|
|
431
458
|
│ ├── type: 'image' | 'chart'
|
|
432
459
|
│ ├── name: string
|
|
@@ -436,10 +463,10 @@ OfficeParserAST
|
|
|
436
463
|
│ └── chartData?: { title, dataSets, labels }
|
|
437
464
|
├── warnings: OfficeIssue[] (non-fatal issues from the parsing phase)
|
|
438
465
|
├── to(format, config?) (format: 'html'|'md'|'text'|'csv'|'rtf'|'pdf'|'chunks', returns { value, messages })
|
|
439
|
-
└── toText() (Deprecated: use .to('text') instead)
|
|
466
|
+
└── ~~toText()~~ (Deprecated: use .to('text') instead)
|
|
440
467
|
```
|
|
441
468
|
|
|
442
|
-
### `OfficeIssue
|
|
469
|
+
### `OfficeIssue`: Warning / Error Object
|
|
443
470
|
|
|
444
471
|
All warnings and errors (from both parsing and generation) use this shape:
|
|
445
472
|
|
|
@@ -552,17 +579,21 @@ ast.metadata = {
|
|
|
552
579
|
created?: Date
|
|
553
580
|
modified?: Date
|
|
554
581
|
description?: string
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
582
|
+
keywords?: string // NEW: Keywords from document properties
|
|
583
|
+
customProperties?: Record<string, any> // User-defined metadata from the document
|
|
584
|
+
nativeProperties?: Record<string, any> // NEW: All format-specific raw metadata
|
|
585
|
+
styleMap?: Record<string, TextFormatting> // Named styles → formatting definitions
|
|
586
|
+
formatting?: TextFormatting // Document-wide defaults
|
|
558
587
|
}
|
|
559
588
|
```
|
|
560
589
|
|
|
561
|
-
**Accessing
|
|
590
|
+
**Accessing native properties (format-specific metadata):**
|
|
562
591
|
```js
|
|
563
592
|
const ast = await officeParser.parseOffice('contract.docx');
|
|
564
|
-
console.log(ast.metadata.
|
|
565
|
-
// {
|
|
593
|
+
console.log(ast.metadata.nativeProperties);
|
|
594
|
+
// DOCX: { Pages: 5, Application: 'Microsoft Word' }
|
|
595
|
+
// HTML: { description: 'My page', 'og:title': 'Title' }
|
|
596
|
+
// PDF: { Title: 'Report', XMP: { ... } }
|
|
566
597
|
```
|
|
567
598
|
|
|
568
599
|
---
|
|
@@ -586,6 +617,61 @@ const headings = ast.content.filter(n => n.type === 'heading' && n.metadata?.lev
|
|
|
586
617
|
console.log(headings.map(h => h.text));
|
|
587
618
|
```
|
|
588
619
|
|
|
620
|
+
### Extract comments
|
|
621
|
+
```ts
|
|
622
|
+
// Comments can be attached to any nested node, so we must traverse recursively
|
|
623
|
+
const printComments = (nodes: OfficeContentNode[]) => {
|
|
624
|
+
nodes.forEach(node => {
|
|
625
|
+
if (node.comments) {
|
|
626
|
+
node.comments.forEach(c => {
|
|
627
|
+
console.log(`Comment by ${c.metadata?.author}: ${c.text}`);
|
|
628
|
+
});
|
|
629
|
+
}
|
|
630
|
+
if (node.children) {
|
|
631
|
+
printComments(node.children);
|
|
632
|
+
}
|
|
633
|
+
});
|
|
634
|
+
};
|
|
635
|
+
|
|
636
|
+
printComments(ast.content);
|
|
637
|
+
```
|
|
638
|
+
|
|
639
|
+
Set `ignoreComments: true` to skip extraction.
|
|
640
|
+
|
|
641
|
+
### Extract footnotes, endnotes & slide notes
|
|
642
|
+
```ts
|
|
643
|
+
// Slide speaker notes (PPTX) live on the slide node itself
|
|
644
|
+
const slide = ast.content.find(n => n.type === 'slide');
|
|
645
|
+
console.log(slide?.notes?.map(n => n.text));
|
|
646
|
+
|
|
647
|
+
// Footnotes and endnotes (DOCX/RTF) can be deeply nested, so we traverse recursively:
|
|
648
|
+
const printNotes = (nodes: OfficeContentNode[]) => {
|
|
649
|
+
nodes.forEach(node => {
|
|
650
|
+
if (node.notes) {
|
|
651
|
+
node.notes.forEach(note => console.log(note.text));
|
|
652
|
+
}
|
|
653
|
+
if (node.children) {
|
|
654
|
+
printNotes(node.children);
|
|
655
|
+
}
|
|
656
|
+
});
|
|
657
|
+
};
|
|
658
|
+
|
|
659
|
+
printNotes(ast.content);
|
|
660
|
+
```
|
|
661
|
+
|
|
662
|
+
> [!IMPORTANT]
|
|
663
|
+
> `putNotesAtLast` is **deprecated**. Notes are always attached via `node.notes`; this flag has no effect and will be removed in a future major version.
|
|
664
|
+
|
|
665
|
+
### Access headers, footers & slide masters
|
|
666
|
+
```ts
|
|
667
|
+
// These are NOT in ast.content; use ast.auxiliary
|
|
668
|
+
console.log(ast.auxiliary?.headers?.map(h => h.text)); // DOCX headers
|
|
669
|
+
console.log(ast.auxiliary?.footers?.map(f => f.text)); // DOCX footers
|
|
670
|
+
console.log(ast.auxiliary?.slideMasters?.length); // PPTX slide masters
|
|
671
|
+
```
|
|
672
|
+
|
|
673
|
+
Set `ignoreHeadersAndFooters: true` or `ignoreSlideMasters: true` to skip extraction.
|
|
674
|
+
|
|
589
675
|
### Extract images with OCR text
|
|
590
676
|
```js
|
|
591
677
|
const ast = await officeParser.parseOffice('report.docx', { extractAttachments: true, ocr: true });
|
|
@@ -650,11 +736,14 @@ Pass as the second argument to `parseOffice(file, config)`.
|
|
|
650
736
|
| Option | Type | Default | Description |
|
|
651
737
|
|--------|------|---------|-------------|
|
|
652
738
|
| `newlineDelimiter` | `string` | `'\n'` | Delimiter inserted between lines in text output |
|
|
653
|
-
| `ignoreNotes` | `boolean` | `false` | Ignore speaker notes (PPTX/ODP) |
|
|
654
|
-
| `
|
|
739
|
+
| `ignoreNotes` | `boolean` | `false` | Ignore footnotes/endnotes (DOCX, RTF) and speaker notes (PPTX/ODP) |
|
|
740
|
+
| `ignoreComments` | `boolean` | `false` | **New**: Ignore inline comments/annotations (DOCX, XLSX, PPTX), attached by default via `node.comments[]` |
|
|
741
|
+
| `ignoreHeadersAndFooters` | `boolean` | `false` | **New**: Skip DOCX headers & footers (populated in `ast.auxiliary.headers/footers` by default) |
|
|
742
|
+
| `ignoreSlideMasters` | `boolean` | `false` | **New**: Skip PPTX slide masters (populated in `ast.auxiliary.slideMasters` by default) |
|
|
743
|
+
| ~~`putNotesAtLast`~~ | `boolean` | `false` | **Deprecated**: Notes are now attached via `node.notes[]`. This flag has no effect |
|
|
655
744
|
| `extractAttachments` | `boolean` | `false` | Populate `ast.attachments` with Base64 images/charts |
|
|
656
745
|
| `ocr` | `boolean` | `false` | Run Tesseract OCR on images (requires `extractAttachments: true`) |
|
|
657
|
-
| `ocrConfig` | `OcrConfig` | `{}` | OCR worker pool settings
|
|
746
|
+
| `ocrConfig` | `OcrConfig` | `{}` | OCR worker pool settings (see [OCR section](#ocr-scheduler--resource-management)) |
|
|
658
747
|
| `includeRawContent` | `boolean` | `false` | Attach raw XML/RTF source to each node |
|
|
659
748
|
| `serializeRawContent` | `boolean` | `true` | Re-serialize XML to clean strings (only if `includeRawContent: true`) |
|
|
660
749
|
| `preserveXmlWhitespace` | `boolean` | `false` | Preserve original XML whitespace during serialization |
|
|
@@ -665,7 +754,7 @@ Pass as the second argument to `parseOffice(file, config)`.
|
|
|
665
754
|
| `pdfWorkerSrc` | `string` | CDN (jsDelivr) | Path/URL to `pdf.worker.min.mjs` (required in browser) |
|
|
666
755
|
| `onWarning` | `(issue: OfficeIssue) => void` | — | Callback for non-fatal parsing issues |
|
|
667
756
|
| `abortSignal` | `AbortSignal \| null` | `null` | Optional signal to cancel parsing (rejects with AbortError) |
|
|
668
|
-
|
|
|
757
|
+
| ~~`outputErrorToConsole`~~ | `boolean` | `false` | **Deprecated.** Use `onWarning` instead |
|
|
669
758
|
|
|
670
759
|
---
|
|
671
760
|
|
|
@@ -689,7 +778,7 @@ Options shared by all generator formats. Pass to `OfficeGenerator.generate(ast,
|
|
|
689
778
|
|
|
690
779
|
---
|
|
691
780
|
|
|
692
|
-
### `onNode` Callback
|
|
781
|
+
### `onNode` Callback: Advanced Node Manipulation
|
|
693
782
|
|
|
694
783
|
Called for **every node** in the AST during generation. Can be `async`.
|
|
695
784
|
|
|
@@ -720,7 +809,7 @@ const { value: md } = await ast.to('md', {
|
|
|
720
809
|
|
|
721
810
|
---
|
|
722
811
|
|
|
723
|
-
### `styleMap
|
|
812
|
+
### `styleMap`: Semantic Style Mapping
|
|
724
813
|
|
|
725
814
|
Maps document style names to semantic output elements. Two formats supported:
|
|
726
815
|
|
|
@@ -764,6 +853,12 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
|
|
|
764
853
|
|--------|------|---------|-------------|
|
|
765
854
|
| `standalone` | `boolean` | `true` | Wrap output in a full `<html>` document with CSS |
|
|
766
855
|
| `chartJsSrc` | `string` | jsDelivr CDN | URL for the Chart.js library |
|
|
856
|
+
| `containerWidth` | `string \| number` | `'auto'` | Max width of the content container. Positive number (px), CSS length string (`'900px'`, `'100%'`, `'60vw'`), or `'auto'`. Invalid values fall back to `'auto'` with an `INVALID_CONTAINER_WIDTH` warning |
|
|
857
|
+
| `customCss` | `string` | `''` | Raw CSS injected into the `<style>` block; use this to override built-in styles |
|
|
858
|
+
| `injections.headStart` | `string` | `''` | Raw HTML injected after `<head>` |
|
|
859
|
+
| `injections.headEnd` | `string` | `''` | Raw HTML injected before `</head>` |
|
|
860
|
+
| `injections.bodyStart` | `string` | `''` | Raw HTML injected after `<body>` |
|
|
861
|
+
| `injections.bodyEnd` | `string` | `''` | Raw HTML injected before `</body>` |
|
|
767
862
|
|
|
768
863
|
### MdGeneratorConfig
|
|
769
864
|
|
|
@@ -780,6 +875,8 @@ Pass as `pdfConfig` inside `GeneratorConfig`. Requires the optional `puppeteer`
|
|
|
780
875
|
| Option | Type | Default | Description |
|
|
781
876
|
|--------|------|---------|-------------|
|
|
782
877
|
| `format` | `string` | `'A4'` | Paper format (`'A4'`, `'Letter'`, `'Legal'`, etc.) |
|
|
878
|
+
| `width` | `string \| number` | `''` | Paper width (e.g., `'5in'`, `'3cm'`) or pixels |
|
|
879
|
+
| `height` | `string \| number` | `''` | Paper height (e.g., `'5in'`, `'3cm'`) or pixels |
|
|
783
880
|
| `landscape` | `boolean` | `false` | Landscape page orientation |
|
|
784
881
|
| `printBackground` | `boolean` | `true` | Print background graphics |
|
|
785
882
|
| `margin` | `object` | `{0,0,0,0}` | Page margins (`top`, `right`, `bottom`, `left`) |
|
|
@@ -807,7 +904,7 @@ Pass as `textConfig` inside `GeneratorConfig`.
|
|
|
807
904
|
| Option | Type | Default | Description |
|
|
808
905
|
|--------|------|---------|-------------|
|
|
809
906
|
| `newlineDelimiter` | `string` | `'\n'` | String inserted between structural blocks |
|
|
810
|
-
| `preserveLayout` | `boolean` | `
|
|
907
|
+
| `preserveLayout` | `boolean` | `true` | Render tables with aligned columns using whitespace |
|
|
811
908
|
|
|
812
909
|
---
|
|
813
910
|
|
|
@@ -825,7 +922,7 @@ Configuration for `OfficeConverter.convert(file, format, config)`.
|
|
|
825
922
|
|
|
826
923
|
### ChunkingConfig
|
|
827
924
|
|
|
828
|
-
`ChunkingConfig` is a **discriminated union
|
|
925
|
+
`ChunkingConfig` is a **discriminated union**: the available options depend on the `strategy` field.
|
|
829
926
|
|
|
830
927
|
#### Common Options (all strategies)
|
|
831
928
|
|
|
@@ -885,7 +982,7 @@ When `ocr: true` is set, `officeParser` maintains an intelligent **Smart Worker
|
|
|
885
982
|
| `corePath` | `string` | `''` | Custom path to Tesseract core script |
|
|
886
983
|
| `langPath` | `string` | `''` | Custom path for language data files |
|
|
887
984
|
| `timeout` | `OcrTimeoutConfig` | `{}` | Consolidated timeouts: `autoTerminate`, `workerLoad`, `recognition` |
|
|
888
|
-
|
|
|
985
|
+
| ~~`autoTerminateTimeout`~~ | `number` | `10000` | **Deprecated.** Use `timeout.autoTerminate` instead |
|
|
889
986
|
|
|
890
987
|
See all language codes at [tesseract-ocr.github.io](https://tesseract-ocr.github.io/tessdoc/Data-Files).
|
|
891
988
|
|
|
@@ -913,8 +1010,8 @@ Two bundles are available in the `dist/` directory:
|
|
|
913
1010
|
|
|
914
1011
|
| Bundle | Usage |
|
|
915
1012
|
|--------|-------|
|
|
916
|
-
| `officeparser.browser.mjs` | ESM
|
|
917
|
-
| `officeparser.browser.iife.js` | IIFE
|
|
1013
|
+
| `officeparser.browser.mjs` | ESM, use with `import` statements or modern bundlers (Vite, Webpack, Next.js) |
|
|
1014
|
+
| `officeparser.browser.iife.js` | IIFE, use with a `<script>` tag; exposes the global `officeParser` object |
|
|
918
1015
|
|
|
919
1016
|
### ESM (Vite / Webpack / Next.js)
|
|
920
1017
|
|
|
@@ -974,7 +1071,7 @@ const ast = await officeParser.parseOffice(pdfArrayBuffer, {
|
|
|
974
1071
|
| `"Worker not found"` in browser for PDF | Verify `pdfWorkerSrc` points to `pdf.worker.min.mjs` matching version `5.6.205` |
|
|
975
1072
|
| Low OCR accuracy | Verify `ocrConfig.language` matches the document language; quality depends on image resolution |
|
|
976
1073
|
| Out of memory on large Excel files | Call `ast.toText()` early and discard the AST object to allow garbage collection |
|
|
977
|
-
| `md`/`html`/`csv` buffer not detected | Add `fileType: 'md'` (or `'html'`, `'csv'`) to config
|
|
1074
|
+
| `md`/`html`/`csv` buffer not detected | Add `fileType: 'md'` (or `'html'`, `'csv'`) to config (these formats have no magic bytes) |
|
|
978
1075
|
| `IMPROPER_BUFFERS` error | Usually means no file extension and no `fileType` hint was provided for a buffer input |
|
|
979
1076
|
| PDF generation fails | Install the optional peer dependency: `npm install puppeteer` |
|
|
980
1077
|
|
|
@@ -986,7 +1083,6 @@ For a full debugging guide, visit the [Live Documentation](https://harshankur.gi
|
|
|
986
1083
|
|
|
987
1084
|
1. **ODT/ODS Charts**: May show inaccurate data when the chart references external cell ranges or uses complex layout-based data.
|
|
988
1085
|
2. **PDF Images (Browser)**: Extracted as BMP files for cross-platform compatibility. Conversion is automatic.
|
|
989
|
-
3. **RTF Notes**: `putNotesAtLast` has no effect for RTF files; footnotes and endnotes are always appended at the end.
|
|
990
1086
|
|
|
991
1087
|
---
|
|
992
1088
|
|
|
@@ -1011,4 +1107,4 @@ Contributions are welcome! Please see [CONTRIBUTING.md](CONTRIBUTING.md) for det
|
|
|
1011
1107
|
|
|
1012
1108
|
## License
|
|
1013
1109
|
|
|
1014
|
-
This project is licensed under the MIT License
|
|
1110
|
+
This project is licensed under the MIT License; see the [LICENSE](LICENSE) file for details.
|
|
@@ -1,8 +1,12 @@
|
|
|
1
|
-
import { ConversionResult, GeneratorConfig, OfficeParserAST, SupportedDestination, SupportedFileType } from './types.js';
|
|
1
|
+
import { ConversionResult, GeneratorConfig, OfficeParserAST, SupportedDestination, SupportedFileType, UniversalGeneratorFormat } from './types.js';
|
|
2
2
|
/**
|
|
3
3
|
* Main generator class providing document conversion functionality.
|
|
4
4
|
*/
|
|
5
5
|
export declare class OfficeGenerator {
|
|
6
|
+
/**
|
|
7
|
+
* Normalizes format aliases (e.g., 'txt' to 'text', 'markdown' to 'md') to standard internal formats.
|
|
8
|
+
*/
|
|
9
|
+
static normalizeDestination(dest: string): UniversalGeneratorFormat;
|
|
6
10
|
/**
|
|
7
11
|
* Generates a file of the specified type from an AST.
|
|
8
12
|
* This is the single source of truth for generation logic.
|
|
@@ -15,5 +19,5 @@ export declare class OfficeGenerator {
|
|
|
15
19
|
*/
|
|
16
20
|
static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
|
|
17
21
|
type: T;
|
|
18
|
-
}, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult
|
|
22
|
+
}, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
|
|
19
23
|
}
|
package/dist/OfficeGenerator.js
CHANGED
|
@@ -14,6 +14,17 @@ const errorUtils_js_1 = require("./utils/errorUtils.js");
|
|
|
14
14
|
* Main generator class providing document conversion functionality.
|
|
15
15
|
*/
|
|
16
16
|
class OfficeGenerator {
|
|
17
|
+
/**
|
|
18
|
+
* Normalizes format aliases (e.g., 'txt' to 'text', 'markdown' to 'md') to standard internal formats.
|
|
19
|
+
*/
|
|
20
|
+
static normalizeDestination(dest) {
|
|
21
|
+
const d = dest?.toLowerCase();
|
|
22
|
+
if (d === 'txt')
|
|
23
|
+
return 'text';
|
|
24
|
+
if (d === 'markdown')
|
|
25
|
+
return 'md';
|
|
26
|
+
return d;
|
|
27
|
+
}
|
|
17
28
|
/**
|
|
18
29
|
* Generates a file of the specified type from an AST.
|
|
19
30
|
* This is the single source of truth for generation logic.
|
|
@@ -25,24 +36,34 @@ class OfficeGenerator {
|
|
|
25
36
|
* @throws {Error} If the destination format is unsupported
|
|
26
37
|
*/
|
|
27
38
|
static async generate(ast, destination, config) {
|
|
28
|
-
|
|
39
|
+
let generator;
|
|
40
|
+
const normalizedDestination = OfficeGenerator.normalizeDestination(destination);
|
|
41
|
+
switch (normalizedDestination) {
|
|
29
42
|
case 'text':
|
|
30
|
-
|
|
43
|
+
generator = new TextGenerator_js_1.TextGenerator(ast, config);
|
|
44
|
+
break;
|
|
31
45
|
case 'md':
|
|
32
|
-
|
|
46
|
+
generator = new MarkdownGenerator_js_1.MarkdownGenerator(ast, config);
|
|
47
|
+
break;
|
|
33
48
|
case 'html':
|
|
34
|
-
|
|
49
|
+
generator = new HtmlGenerator_js_1.HtmlGenerator(ast, config);
|
|
50
|
+
break;
|
|
35
51
|
case 'pdf':
|
|
36
|
-
|
|
52
|
+
generator = new PdfGenerator_js_1.PdfGenerator(ast, config);
|
|
53
|
+
break;
|
|
37
54
|
case 'csv':
|
|
38
|
-
|
|
55
|
+
generator = new CsvGenerator_js_1.CsvGenerator(ast, config);
|
|
56
|
+
break;
|
|
39
57
|
case 'rtf':
|
|
40
|
-
|
|
58
|
+
generator = new RtfGenerator_js_1.RtfGenerator(ast, config);
|
|
59
|
+
break;
|
|
41
60
|
case 'chunks':
|
|
42
|
-
|
|
61
|
+
generator = new ChunkingGenerator_js_1.ChunkingGenerator(ast, config);
|
|
62
|
+
break;
|
|
43
63
|
default:
|
|
44
|
-
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.
|
|
64
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED, undefined, destination);
|
|
45
65
|
}
|
|
66
|
+
return generator.generate();
|
|
46
67
|
}
|
|
47
68
|
}
|
|
48
69
|
exports.OfficeGenerator = OfficeGenerator;
|
package/dist/OfficeParser.d.ts
CHANGED
|
@@ -81,7 +81,7 @@ export declare class OfficeParser {
|
|
|
81
81
|
* });
|
|
82
82
|
*
|
|
83
83
|
* // Parse a Buffer with OCR enabled
|
|
84
|
-
* const buffer = await
|
|
84
|
+
* const buffer = await retrieveData('document.pdf').then(r => r.arrayBuffer());
|
|
85
85
|
* const ast = await OfficeParser.parseOffice(buffer, {
|
|
86
86
|
* ocr: true,
|
|
87
87
|
* ocrLanguage: 'eng+fra'
|
package/dist/OfficeParser.js
CHANGED
|
@@ -98,7 +98,7 @@ class OfficeParser {
|
|
|
98
98
|
* });
|
|
99
99
|
*
|
|
100
100
|
* // Parse a Buffer with OCR enabled
|
|
101
|
-
* const buffer = await
|
|
101
|
+
* const buffer = await retrieveData('document.pdf').then(r => r.arrayBuffer());
|
|
102
102
|
* const ast = await OfficeParser.parseOffice(buffer, {
|
|
103
103
|
* ocr: true,
|
|
104
104
|
* ocrLanguage: 'eng+fra'
|
package/dist/cli.d.ts
CHANGED
|
@@ -4,19 +4,25 @@
|
|
|
4
4
|
*
|
|
5
5
|
* Allows running officeparser from the command line:
|
|
6
6
|
* npx officeparser file.docx
|
|
7
|
-
* officeparser file.docx --
|
|
8
|
-
* officeparser file.docx --ocr
|
|
7
|
+
* officeparser file.docx --to=text
|
|
8
|
+
* officeparser file.docx --ocr --extractAttachments
|
|
9
9
|
*
|
|
10
|
-
* Options (--key=value):
|
|
11
|
-
* --
|
|
10
|
+
* Options (--key=value, --key value, or bare flags):
|
|
11
|
+
* --to=json|text|md|html|csv|rtf|pdf|chunks Convert AST to specified format (default: json)
|
|
12
12
|
* --output=path Save result to a file
|
|
13
|
-
* --
|
|
14
|
-
* --ocr
|
|
15
|
-
* --
|
|
16
|
-
* --extractAttachments
|
|
17
|
-
* --ignoreNotes
|
|
18
|
-
* --
|
|
19
|
-
* --
|
|
20
|
-
* --
|
|
13
|
+
* --fileType=docx|xlsx|... Override file type detection
|
|
14
|
+
* --ocr Enable OCR for images (default: false)
|
|
15
|
+
* --ocrConfig.language=eng OCR language (default: eng)
|
|
16
|
+
* --extractAttachments Extract embedded attachments (default: false)
|
|
17
|
+
* --ignoreNotes Ignore footnotes/endnotes/speaker notes (default: false)
|
|
18
|
+
* --ignoreComments Ignore inline comments (default: false)
|
|
19
|
+
* --ignoreHeadersAndFooters Ignore headers and footers (default: false)
|
|
20
|
+
* --ignoreSlideMasters Ignore slide masters (default: false)
|
|
21
|
+
* --ignoreInternalLinks Ignore internal links (default: false)
|
|
22
|
+
* --includeRawContent Include raw content in AST (default: false)
|
|
23
|
+
* --serializeRawContent Include stringified XML in metadata (default: true)
|
|
24
|
+
* --preserveXmlWhitespace Keep raw formatting space (default: false)
|
|
25
|
+
* --includeBreakNodes Include break nodes (DOCX only, default: false)
|
|
26
|
+
* --verbose Show full error stack traces and warning logs
|
|
21
27
|
*/
|
|
22
28
|
export {};
|