officeparser 6.1.1 → 7.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +301 -26
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +74 -31
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +106 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +828 -5
- package/dist/officeparser.browser.iife.js +703 -52
- package/dist/officeparser.browser.mjs +703 -52
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +140 -79
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +20 -23
- package/dist/parsers/RtfParser.d.ts +2 -2
- package/dist/parsers/RtfParser.js +1291 -1240
- package/dist/parsers/WordParser.d.ts +2 -2
- package/dist/parsers/WordParser.js +232 -97
- package/dist/sbom.cdx.json +99 -99
- package/dist/types.d.ts +783 -5
- package/dist/types.js +73 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.d.ts +8 -3
- package/dist/utils/envUtils.js +117 -34
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +110 -52
- package/dist/utils/moduleLoader.js +19 -11
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +26 -7
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
# officeParser 📄🚀
|
|
1
|
+
# officeParser 📄🚀 - The Most Versatile Office Parser & Generator
|
|
2
2
|
|
|
3
|
-
A robust, strictly-typed Node.js and Browser library for parsing office files
|
|
3
|
+
A robust, strictly-typed Node.js and Browser library for parsing and generating office files. It not only extracts content from [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format), [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values), [`md`](https://en.wikipedia.org/wiki/Markdown), and [`html`](https://en.wikipedia.org/wiki/HTML) into a rich Abstract Syntax Tree (AST), but also provides a powerful generation engine to convert that AST into formats like **Markdown**, **HTML**, **CSV**, **RTF**, **Text**, **PDF**, and **JSON**, including native **RAG-focused chunking** support.
|
|
4
4
|
|
|
5
5
|
[](https://badge.fury.io/js/officeparser)
|
|
6
6
|
[](https://www.npmjs.com/package/officeparser)
|
|
@@ -18,8 +18,6 @@ A robust, strictly-typed Node.js and Browser library for parsing office files ([
|
|
|
18
18
|
- **Debugging**: Use the visualizer to debug parsing issues by inspecting exactly how nodes are interpreted.
|
|
19
19
|
- **Format Specs**: Read detailed specifications for the AST structure and configuration options.
|
|
20
20
|
|
|
21
|
-
*(Legacy Visualizer: If you prefer the [old simple visualizer](https://harshankur.github.io/officeParser/visualizer_old.html), it is still available.)*
|
|
22
|
-
|
|
23
21
|
---
|
|
24
22
|
|
|
25
23
|
|
|
@@ -30,6 +28,30 @@ A robust, strictly-typed Node.js and Browser library for parsing office files ([
|
|
|
30
28
|
|
|
31
29
|
---
|
|
32
30
|
|
|
31
|
+
## Table of Contents
|
|
32
|
+
- [Install via npm](#install-via-npm)
|
|
33
|
+
- [Command Line Usage](#command-line-usage)
|
|
34
|
+
- [Library Usage](#library-usage)
|
|
35
|
+
- [Using the OfficeGenerator](#using-the-officegenerator)
|
|
36
|
+
- [The New "One-Step" API: OfficeConverter](#the-new-one-step-api-officeconverter)
|
|
37
|
+
- [Native RAG Chunking](#native-rag-chunking)
|
|
38
|
+
- [The AST Structure](#the-ast-structure)
|
|
39
|
+
- [Deep Dive: Document Components](#deep-dive-document-components)
|
|
40
|
+
- [Performance & Fidelity Highlights (v7.0.0)](#performance--fidelity-highlights-v700)
|
|
41
|
+
- [Advanced AST Usage](#advanced-ast-usage)
|
|
42
|
+
- [Configuration Object: OfficeParserConfig](#configuration-object-officeparserconfig)
|
|
43
|
+
- [Generator Configuration: GeneratorConfig](#generator-configuration-generatorconfig)
|
|
44
|
+
- [Format-Specific Generator Configuration](#format-specific-generator-configuration)
|
|
45
|
+
- [One-Step Conversion: OfficeConverterConfig](#one-step-conversion-officeconverterconfig)
|
|
46
|
+
- [Chunking Configuration: ChunkingConfig](#chunking-configuration-chunkingconfig)
|
|
47
|
+
- [OCR Scheduler & Resource Management](#ocr-scheduler--resource-management)
|
|
48
|
+
- [Examples](#examples)
|
|
49
|
+
- [Browser Usage](#browser-usage)
|
|
50
|
+
- [Troubleshooting & Common Issues](#troubleshooting--common-issues)
|
|
51
|
+
- [Known Limitations](#known-limitations)
|
|
52
|
+
|
|
53
|
+
---
|
|
54
|
+
|
|
33
55
|
## Install via npm
|
|
34
56
|
|
|
35
57
|
```bash
|
|
@@ -37,7 +59,7 @@ npm i officeparser
|
|
|
37
59
|
```
|
|
38
60
|
|
|
39
61
|
## Command Line usage
|
|
40
|
-
You can use `officeparser` directly from the terminal to
|
|
62
|
+
You can use `officeparser` directly from the terminal to extract content as JSON AST, plain text, or generate new formats like Markdown and HTML.
|
|
41
63
|
|
|
42
64
|
```bash
|
|
43
65
|
# Get full AST as JSON (default)
|
|
@@ -46,25 +68,33 @@ npx officeparser /path/to/officeFile.docx
|
|
|
46
68
|
# Get plain text only
|
|
47
69
|
npx officeparser /path/to/officeFile.docx --toText=true
|
|
48
70
|
|
|
49
|
-
#
|
|
50
|
-
npx officeparser
|
|
71
|
+
# Generate Markdown file
|
|
72
|
+
npx officeparser report.docx --format=md --output=report.md
|
|
73
|
+
|
|
74
|
+
# Generate HTML file with specific output
|
|
75
|
+
npx officeparser presentation.pptx --format=html --output=preview.html
|
|
76
|
+
|
|
77
|
+
# Convert spreadsheet to CSV
|
|
78
|
+
npx officeparser data.xlsx --format=csv
|
|
51
79
|
```
|
|
52
80
|
|
|
53
81
|
### Config Options:
|
|
54
|
-
- `--
|
|
82
|
+
- `--format=[json|text|md|html|csv|rtf|pdf|chunks]` The output format. Default is `json`.
|
|
83
|
+
- `--output=[path]` Optional file path to write the output to.
|
|
84
|
+
- `--toText=[true|false]` Legacy flag to output only plain text. Use `--format=text` instead.
|
|
55
85
|
- `--ignoreNotes=[true|false]` Flag to ignore notes from files like PowerPoint. Default is false.
|
|
56
86
|
- `--newlineDelimiter=[delimiter]` The delimiter to use for new lines. Default is `\n`.
|
|
57
87
|
- `--putNotesAtLast=[true|false]` Flag to collect notes at the end of files like PowerPoint. Default is false.
|
|
58
|
-
- `--outputErrorToConsole=[true|false]` Flag to output errors to the console.
|
|
88
|
+
- `--outputErrorToConsole=[true|false]` **(Deprecated)** Flag to output errors to the console. Use `onWarning` callback in library usage.
|
|
59
89
|
- `--extractAttachments=[true|false]` Flag to extract images/charts as Base64. Default is false.
|
|
60
90
|
- `--ocr=[true|false]` Flag to enable OCR for extracted images. Default is false.
|
|
61
91
|
- `--includeRawContent=[true|false]` Flag to include raw XML/RTF content in nodes. Default is false.
|
|
62
|
-
- `--includeBreakNodes=[true|false]` Flag to include break nodes. Currently only available for DOCX documents
|
|
92
|
+
- `--includeBreakNodes=[true|false]` Flag to include break nodes. Currently only available for DOCX documents.
|
|
63
93
|
- `--verbose=[true|false]` Show full error stack traces.
|
|
64
94
|
|
|
65
95
|
|
|
66
96
|
## Library Usage
|
|
67
|
-
In **
|
|
97
|
+
In **v7.0.0**, the library has evolved into a dual-purpose **Parser** and **Generator**. You can first parse any office file into a structured AST and then use the `OfficeGenerator` to transform that AST into various formats or chunks.
|
|
68
98
|
|
|
69
99
|
### Getting Started (Async/Await)
|
|
70
100
|
```js
|
|
@@ -99,6 +129,101 @@ const text = await getText("/path/to/officeFile.docx");
|
|
|
99
129
|
console.log(text);
|
|
100
130
|
```
|
|
101
131
|
|
|
132
|
+
## Using the OfficeGenerator
|
|
133
|
+
The `OfficeGenerator` is a powerful tool to convert your AST into human-readable formats or structured data.
|
|
134
|
+
|
|
135
|
+
```typescript
|
|
136
|
+
import { OfficeParser, OfficeGenerator } from 'officeparser';
|
|
137
|
+
|
|
138
|
+
const ast = await OfficeParser.parseOffice('report.docx');
|
|
139
|
+
|
|
140
|
+
// 1. Convert to Markdown
|
|
141
|
+
const md = await OfficeGenerator.generate(ast, 'md');
|
|
142
|
+
console.log(md.value);
|
|
143
|
+
|
|
144
|
+
// 2. Convert to HTML with structured style mapping (Recommended)
|
|
145
|
+
const html = await OfficeGenerator.generate(ast, 'html', {
|
|
146
|
+
includeFormatting: true,
|
|
147
|
+
styleMap: [
|
|
148
|
+
{
|
|
149
|
+
selector: { nodeType: 'paragraph', attributes: { style: 'Heading 1' } },
|
|
150
|
+
output: { tag: 'h1', classes: ['main-title'] }
|
|
151
|
+
}
|
|
152
|
+
]
|
|
153
|
+
});
|
|
154
|
+
console.log(html.value);
|
|
155
|
+
|
|
156
|
+
// 3. Convert to CSV (for spreadsheets)
|
|
157
|
+
const csv = await OfficeGenerator.generate(ast, 'csv');
|
|
158
|
+
console.log(csv.value);
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
## The New "One-Step" API: `OfficeConverter`
|
|
162
|
+
In **v7.0.0**, we introduced the `OfficeConverter.convert` method. This is the new high-level API designed for one-step transformations where you don't need to manually interact with the AST. It automatically handles parser and generator configuration synchronization.
|
|
163
|
+
|
|
164
|
+
```typescript
|
|
165
|
+
import { OfficeConverter } from 'officeparser';
|
|
166
|
+
|
|
167
|
+
// One-step conversion from DOCX to Markdown
|
|
168
|
+
const result = await OfficeConverter.convert('report.docx', 'md');
|
|
169
|
+
console.log(result.value); // The generated Markdown string
|
|
170
|
+
console.log(result.messages); // Array of warnings/info (e.g., "Skipped unsupported drawing")
|
|
171
|
+
|
|
172
|
+
// Complex conversion with nested configuration
|
|
173
|
+
const htmlResult = await OfficeConverter.convert('data.xlsx', 'html', {
|
|
174
|
+
parseConfig: {
|
|
175
|
+
ignoreNotes: true
|
|
176
|
+
},
|
|
177
|
+
generatorConfig: {
|
|
178
|
+
includeFormatting: true,
|
|
179
|
+
styleMap: [
|
|
180
|
+
{
|
|
181
|
+
selector: { attributes: { style: { value: 'Header', operator: '~=' } } },
|
|
182
|
+
output: { tag: 'h2', classes: ['data-header'] }
|
|
183
|
+
}
|
|
184
|
+
]
|
|
185
|
+
},
|
|
186
|
+
onWarning: (msg) => console.warn("Conversion Warning:", msg)
|
|
187
|
+
});
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
## Native RAG Chunking
|
|
191
|
+
`officeParser` provides native support for document chunking, specifically designed for Retrieval-Augmented Generation (RAG) workflows. It offers three distinct strategies to split your documents while maintaining context and metadata.
|
|
192
|
+
|
|
193
|
+
### 1. Fixed-Size Strategy (Recursive)
|
|
194
|
+
Splits text into chunks based on character count with a specified overlap. It uses smart boundary detection to avoid cutting in the middle of sentences or paragraphs.
|
|
195
|
+
|
|
196
|
+
### 2. Document Structure Strategy
|
|
197
|
+
Splits the document at natural structural boundaries like pages (PDF/Word), slides (PPTX), or high-level headings. This preserves the logical flow of the document.
|
|
198
|
+
|
|
199
|
+
### 3. Semantic Strategy
|
|
200
|
+
Uses cosine similarity between sentence embeddings to identify coherent topic boundaries. This ensures that each chunk contains semantically related content (requires an embedding function).
|
|
201
|
+
|
|
202
|
+
### The `OfficeChunk` Interface
|
|
203
|
+
Every chunk produced contains not just text, but rich metadata to help your RAG pipeline:
|
|
204
|
+
```typescript
|
|
205
|
+
{
|
|
206
|
+
text: string; // The chunk content
|
|
207
|
+
metadata: {
|
|
208
|
+
sourceType: string; // e.g., "docx", "pdf"
|
|
209
|
+
pageNumber?: number; // Current page
|
|
210
|
+
slideNumber?: number; // Current slide
|
|
211
|
+
closestHeading?: string; // The heading this chunk belongs to
|
|
212
|
+
chunkIndex: number; // Sequential index
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
#### Example: Generating Chunks
|
|
218
|
+
```typescript
|
|
219
|
+
const chunks = await OfficeGenerator.generate(ast, 'chunks', {
|
|
220
|
+
strategy: 'fixed-size',
|
|
221
|
+
maxChunkSize: 1000,
|
|
222
|
+
chunkOverlap: 200
|
|
223
|
+
});
|
|
224
|
+
console.log(`Generated ${chunks.value.length} chunks`);
|
|
225
|
+
```
|
|
226
|
+
|
|
102
227
|
### Using Callbacks (Backward Compatibility Support)
|
|
103
228
|
Callbacks are still supported for those preferred, but the data returned is now the AST object.
|
|
104
229
|
```js
|
|
@@ -134,7 +259,7 @@ The `OfficeParserAST` provides a format-agnostic representation of your document
|
|
|
134
259
|
|
|
135
260
|
```text
|
|
136
261
|
OfficeParserAST
|
|
137
|
-
├── type: "docx" | "
|
|
262
|
+
├── type: "docx" | "pdf" | "xlsx" | "csv" | "md" | ... (11 formats supported)
|
|
138
263
|
├── metadata: { author, title, created, modified, ..., customProperties }
|
|
139
264
|
├── content: [ OfficeContentNode ]
|
|
140
265
|
│ ├── type: "paragraph" | "heading" | "table" | "list" | ...
|
|
@@ -313,6 +438,14 @@ console.log("Custom Metadata:", ast.metadata.customProperties);
|
|
|
313
438
|
// Output: { "ProjectID": "ABC-123", "InternalReview": true }
|
|
314
439
|
```
|
|
315
440
|
|
|
441
|
+
## Performance & Fidelity Highlights (v7.0.0)
|
|
442
|
+
The v7.0.0 release brings significant internal optimizations and fidelity improvements:
|
|
443
|
+
- **OpenOffice Speedups**: Up to **23x faster** parsing for ODP presentations thanks to optimized XML caching.
|
|
444
|
+
- **Excel Memory Efficiency**: Resolved $O(n)$ memory overhead issues for large spreadsheets (#91) by switching to iterative stream-based parsing.
|
|
445
|
+
- **RTF Performance**: Rewritten core loop to resolve $O(n^2)$ bottlenecks during string accumulation.
|
|
446
|
+
- **Advanced Table Fidelity**: Native support for **vertical cell merging** (`vMerge`) and **horizontal spanning** (`gridSpan`) in DOCX, ensuring complex tables look exactly as they do in Word.
|
|
447
|
+
- **Parser Extensions**: You can now parse `CSV`, `Markdown`, and `HTML` files *into* the unified Office AST, allowing you to use the `OfficeGenerator` on them just like any other format.
|
|
448
|
+
|
|
316
449
|
### Advanced AST Usage
|
|
317
450
|
Beyond using `ast.toText()`, you can interact with the structural data directly:
|
|
318
451
|
|
|
@@ -407,24 +540,166 @@ Pass an optional config object as the second argument to `parseOffice`.
|
|
|
407
540
|
|
|
408
541
|
| Flag | DataType | Default | Explanation |
|
|
409
542
|
|------|----------|---------|-------------|
|
|
410
|
-
| `outputErrorToConsole` | boolean | `false` | Show logs to console in case of an error. |
|
|
543
|
+
| `outputErrorToConsole` | boolean | `false` | **Deprecated**: Use `onWarning` instead. Show logs to console in case of an error. |
|
|
411
544
|
| `newlineDelimiter` | string | `\n` | Delimiter for new lines in text output. |
|
|
412
545
|
| `ignoreNotes` | boolean | `false` | Ignore notes in files like PowerPoint/ODP. |
|
|
413
|
-
| `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document.
|
|
546
|
+
| `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. |
|
|
414
547
|
| `extractAttachments` | boolean | `false` | Extract images and charts as Base64. |
|
|
415
548
|
| `includeRawContent` | boolean | `false` | Include raw XML/RTF markup in the nodes. |
|
|
416
|
-
| `serializeRawContent` | boolean | `true` |
|
|
417
|
-
| `preserveXmlWhitespace` | boolean | `false` |
|
|
549
|
+
| `serializeRawContent` | boolean | `true` | Re-serializes raw XML to clean strings. |
|
|
550
|
+
| `preserveXmlWhitespace` | boolean | `false` | Preserves original XML whitespace. |
|
|
418
551
|
| `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
|
|
419
|
-
| `
|
|
420
|
-
| `
|
|
421
|
-
| `
|
|
422
|
-
| `
|
|
423
|
-
| `
|
|
424
|
-
| `
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
552
|
+
| `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. |
|
|
553
|
+
| `ocrConfig` | object | `{}` | OCR Scheduler configuration. |
|
|
554
|
+
| `includeBreakNodes` | boolean | `false` | Include `w:br`, `w:cr` nodes (DOCX only).|
|
|
555
|
+
| `ignoreInternalLinks` | boolean | `false` | Remove all bookmarks and internal jumps. |
|
|
556
|
+
| `csvDelimiter` | string | `,` | Custom delimiter for parsing CSV files. |
|
|
557
|
+
| `fileType` | string | `null` | Manual format override (authoritative). |
|
|
558
|
+
|
|
559
|
+
## Generator Configuration: GeneratorConfig
|
|
560
|
+
Configuration options for `OfficeGenerator.generate`.
|
|
561
|
+
|
|
562
|
+
| Flag | DataType | Default | Explanation |
|
|
563
|
+
|------|----------|---------|-------------|
|
|
564
|
+
| `includeFormatting` | boolean | `true` | Whether to include semantic styles (bold, italic, colors, sizes) in output. |
|
|
565
|
+
| `generateIds` | boolean | `true` | Automatically generates unique slug-based IDs for heading nodes. |
|
|
566
|
+
| `renderMetadata` | boolean | `false` | Renders document metadata (Title, Author) as a visible header block. |
|
|
567
|
+
| `includeImages` | boolean | `true` | Whether to include image nodes in the generated output. |
|
|
568
|
+
| `includeCharts` | boolean | `true` | Whether to include interactive charts (HTML only). |
|
|
569
|
+
| `ignoreInternalLinks`| boolean | `false` | Suppresses all internal bookmarks and anchor references. |
|
|
570
|
+
| `ignoreDefaultStyleMap`| boolean | `false` | Ignore the library's default style mappings. |
|
|
571
|
+
| `styleMap` | string[] \| array | `[]` | Array of style mappings (DSL strings or structured objects). |
|
|
572
|
+
| `onNode` | function | `undefined` | Callback to filter, override, or mutate any node during generation. |
|
|
573
|
+
| `onWarning` | function | `undefined` | Callback for generation-phase warnings. |
|
|
574
|
+
| `htmlConfig` | object | `{}` | Format-specific settings for HTML generation. |
|
|
575
|
+
| `mdConfig` | object | `{}` | Format-specific settings for Markdown generation. |
|
|
576
|
+
| `pdfConfig` | object | `{}` | Format-specific settings for PDF generation. |
|
|
577
|
+
| `csvConfig` | object | `{}` | Format-specific settings for CSV generation. |
|
|
578
|
+
| `textConfig` | object | `{}` | Format-specific settings for Plain Text generation. |
|
|
579
|
+
| `rtfConfig` | object | `{}` | Format-specific settings for RTF generation. |
|
|
580
|
+
| `chunksConfig` | object | `(doc-struct)` | Settings for RAG chunking strategies. |
|
|
581
|
+
|
|
582
|
+
### 🛠️ Advanced Node Manipulation (Pro Users)
|
|
583
|
+
The `onNode` callback is a powerful tool that gives you complete control over the generation process. It is called for **every single node** in the AST before it is rendered.
|
|
584
|
+
|
|
585
|
+
#### Callback Capabilities:
|
|
586
|
+
1. **Filter/Remove Nodes**: Return `false` to skip a node and all its children.
|
|
587
|
+
2. **Override Rendering**: Return a `string` to use that exact text as the output, bypassing default logic and recursion.
|
|
588
|
+
3. **Mutate Nodes**: Modify the `node` object directly (e.g., changing `node.text`) and return `void` to let the generator proceed with your changes.
|
|
589
|
+
4. **Async Support**: The callback can be `async`, allowing you to fetch external data or perform complex logic during generation.
|
|
590
|
+
|
|
591
|
+
#### Pro Example:
|
|
592
|
+
```typescript
|
|
593
|
+
const result = await ast.to('md', {
|
|
594
|
+
onNode: async (node) => {
|
|
595
|
+
// 1. Skip all images
|
|
596
|
+
if (node.type === 'image') return false;
|
|
597
|
+
|
|
598
|
+
// 2. Redact sensitive info by mutating the node
|
|
599
|
+
if (node.text?.includes('SECRET_KEY')) {
|
|
600
|
+
node.text = node.text.replace(/SECRET_KEY: \w+/, 'SECRET_KEY: [REDACTED]');
|
|
601
|
+
}
|
|
602
|
+
|
|
603
|
+
// 3. Custom rendering for specific styles
|
|
604
|
+
if (node.metadata?.style === 'Callout') {
|
|
605
|
+
return `> [!INFO]\n> ${node.text}`;
|
|
606
|
+
}
|
|
607
|
+
|
|
608
|
+
// 4. Proceed with default rendering (implicitly returns void)
|
|
609
|
+
}
|
|
610
|
+
});
|
|
611
|
+
```
|
|
612
|
+
|
|
613
|
+
### Advanced Style Mapping (Semantic Translation)
|
|
614
|
+
The `styleMap` configuration is the primary way to define the "semantic meaning" of document styles. We recommend using **Structured Style Mappings** for full type safety and power.
|
|
615
|
+
|
|
616
|
+
#### 1. Structured Style Mappings (Recommended)
|
|
617
|
+
Use structured objects to match nodes based on type and attributes, and specify detailed output properties like classes and custom attributes.
|
|
618
|
+
|
|
619
|
+
```typescript
|
|
620
|
+
styleMap: [
|
|
621
|
+
{
|
|
622
|
+
selector: {
|
|
623
|
+
nodeType: 'paragraph',
|
|
624
|
+
attributes: { style: 'Heading 1' }
|
|
625
|
+
},
|
|
626
|
+
output: {
|
|
627
|
+
tag: 'h1',
|
|
628
|
+
classes: ['main-title'],
|
|
629
|
+
attributes: { id: 'top' }
|
|
630
|
+
}
|
|
631
|
+
},
|
|
632
|
+
{
|
|
633
|
+
// Use operators like '~=' for partial matches
|
|
634
|
+
selector: { attributes: { style: { value: 'Quote', operator: '~=' } } },
|
|
635
|
+
output: { tag: 'blockquote' }
|
|
636
|
+
}
|
|
637
|
+
]
|
|
638
|
+
```
|
|
639
|
+
|
|
640
|
+
#### 2. Legacy String DSL
|
|
641
|
+
The library also maintains support for a simple string-based DSL, highly compatible with `mammoth.js`.
|
|
642
|
+
|
|
643
|
+
- **Literal Matching**: `"p[style-name='Heading 1'] => h1"`
|
|
644
|
+
- **Regex-like Matching**: `"p[style~='Title'] => h2"`
|
|
645
|
+
- **Attribute Filters**: `"p[style-name='Quote'][lang='en'] => blockquote"`
|
|
646
|
+
|
|
647
|
+
## Format-Specific Generator Configuration
|
|
648
|
+
Each destination format has its own specialized sub-configuration object nested within the main `GeneratorConfig`.
|
|
649
|
+
|
|
650
|
+
### 1. HtmlGeneratorConfig (`htmlConfig`)
|
|
651
|
+
| Flag | DataType | Default | Explanation |
|
|
652
|
+
|------|----------|---------|-------------|
|
|
653
|
+
| `standalone` | boolean | `true` | Wraps output in a full `<html>` document with CSS and metadata. |
|
|
654
|
+
| `chartJsSrc` | string | `(CDN)` | URL for the Chart.js library used for interactive charts. |
|
|
655
|
+
|
|
656
|
+
### 2. MdGeneratorConfig (`mdConfig`)
|
|
657
|
+
| Flag | DataType | Default | Explanation |
|
|
658
|
+
|------|----------|---------|-------------|
|
|
659
|
+
| `fallbackToHtml` | boolean | `true` | Uses HTML tags for features not supported by Markdown (underlines, complex tables). |
|
|
660
|
+
|
|
661
|
+
### 3. PdfGeneratorConfig (`pdfConfig`)
|
|
662
|
+
| Flag | DataType | Default | Explanation |
|
|
663
|
+
|------|----------|---------|-------------|
|
|
664
|
+
| `format` | string | `'A4'` | Paper format (e.g., 'Letter', 'A4', 'Legal'). |
|
|
665
|
+
| `landscape` | boolean | `false` | Page orientation. |
|
|
666
|
+
| `margin` | object | `{0,0,0,0}` | Top, right, bottom, left margins. |
|
|
667
|
+
| `displayHeaderFooter`| boolean | `false` | Whether to display print headers and footers. |
|
|
668
|
+
| `headerTemplate` | string | `''` | HTML template for the print header. |
|
|
669
|
+
| `footerTemplate` | string | `''` | HTML template for the print footer. |
|
|
670
|
+
|
|
671
|
+
### 4. CsvGeneratorConfig (`csvConfig`)
|
|
672
|
+
| Flag | DataType | Default | Explanation |
|
|
673
|
+
|------|----------|---------|-------------|
|
|
674
|
+
| `sheets` | string | `''` | Range of sheets to export (e.g., "1", "1-3", "1,3"). |
|
|
675
|
+
| `mergeSheets` | boolean | `true` | Merges all sheets into one CSV string. If false, returns a ZIP. |
|
|
676
|
+
| `columnDelimiter` | string | `','` | Custom delimiter for the CSV output. |
|
|
677
|
+
|
|
678
|
+
### 5. TextGeneratorConfig (`textConfig`)
|
|
679
|
+
| Flag | DataType | Default | Explanation |
|
|
680
|
+
|------|----------|---------|-------------|
|
|
681
|
+
| `newlineDelimiter` | string | `\n` | String inserted between structural blocks. |
|
|
682
|
+
| `preserveLayout` | boolean | `false` | Attempts to maintain table structures using whitespace. |
|
|
683
|
+
|
|
684
|
+
## One-Step Conversion: OfficeConverterConfig
|
|
685
|
+
Configuration for the `OfficeConverter.convert()` API.
|
|
686
|
+
|
|
687
|
+
| Flag | DataType | Default | Explanation |
|
|
688
|
+
|------|----------|---------|-------------|
|
|
689
|
+
| `parseConfig` | object | `{}` | Settings for the `OfficeParser` phase. |
|
|
690
|
+
| `generatorConfig` | object | `{}` | Settings for the `OfficeGenerator` phase. |
|
|
691
|
+
| `onWarning` | function | `undefined` | Global callback for issues in either phase. Overrides phase-specific callbacks. |
|
|
692
|
+
|
|
693
|
+
## Chunking Configuration: ChunkingConfig
|
|
694
|
+
Specific options when using `format: 'chunks'`.
|
|
695
|
+
|
|
696
|
+
| Flag | DataType | Default | Explanation |
|
|
697
|
+
|------|----------|---------|-------------|
|
|
698
|
+
| `strategy` | string | `'fixed-size'`| The chunking strategy (`fixed-size`, `document-structure`, `semantic`). |
|
|
699
|
+
| `maxChunkSize` | number | `1000` | Maximum characters per chunk. |
|
|
700
|
+
| `chunkOverlap` | number | `200` | Overlap between consecutive chunks. |
|
|
701
|
+
| `similarityThreshold`| number | `0.5` | Threshold for semantic splitting (0.0 to 1.0). |
|
|
702
|
+
| `embedBatchSize` | number | `50` | Batch size for embedding requests. |
|
|
428
703
|
|
|
429
704
|
### OCR Scheduler & Resource Management
|
|
430
705
|
If your application uses OCR, `officeParser` utilizes an intelligent **Smart Worker Pool** to maintain a background worker pool and optimize repeated parse requests.
|
|
@@ -555,7 +830,7 @@ const ast = await officeParser.parseOffice(file);
|
|
|
555
830
|
|
|
556
831
|
// Or override it with your own path or a different version:
|
|
557
832
|
const ast2 = await officeParser.parseOffice(file, {
|
|
558
|
-
pdfWorkerSrc: "https://
|
|
833
|
+
pdfWorkerSrc: "https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
|
|
559
834
|
});
|
|
560
835
|
```
|
|
561
836
|
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { ConversionResult, OfficeConverterConfig, SupportedDestination, SupportedFileType } from './types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Utility type to infer the file type from a file path string literal.
|
|
4
|
+
*/
|
|
5
|
+
type InferFileTypeFromPath<T> = T extends `${string}.${infer E}` ? (Lowercase<E> extends SupportedFileType ? Lowercase<E> : SupportedFileType) : SupportedFileType;
|
|
6
|
+
/**
|
|
7
|
+
* Main converter class providing a streamlined one-step API for document conversion.
|
|
8
|
+
*
|
|
9
|
+
* This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
|
|
10
|
+
* documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
|
|
11
|
+
*/
|
|
12
|
+
export declare class OfficeConverter {
|
|
13
|
+
/**
|
|
14
|
+
* Converts an office document from its source format to a specified destination format.
|
|
15
|
+
*
|
|
16
|
+
* This method:
|
|
17
|
+
* 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
|
|
18
|
+
* 2. Automatically configures the parser based on the generator requirements (e.g., enabling
|
|
19
|
+
* attachment extraction if images are requested in the output).
|
|
20
|
+
* 3. Generates the destination document from the AST using `OfficeGenerator`.
|
|
21
|
+
*
|
|
22
|
+
* @template F The inferred type of the input file (path string or buffer).
|
|
23
|
+
* @template T The authoritative source file type (inferred from path or config).
|
|
24
|
+
*
|
|
25
|
+
* @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
|
|
26
|
+
* @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
|
|
27
|
+
* @param config - Optional unified configuration for both the parser and generator phases.
|
|
28
|
+
*
|
|
29
|
+
* @returns A promise resolving to the ConversionResult containing the value and messages.
|
|
30
|
+
* @throws {Error} If the source format is unsupported or parsing/generation fails.
|
|
31
|
+
*
|
|
32
|
+
* @example
|
|
33
|
+
* ```typescript
|
|
34
|
+
* // Convert Word to Markdown with a single call
|
|
35
|
+
* const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
|
|
36
|
+
*
|
|
37
|
+
* // Convert PDF to HTML with OCR enabled for images
|
|
38
|
+
* const { value: html } = await OfficeConverter.convert(buffer, 'html', {
|
|
39
|
+
* ocr: true,
|
|
40
|
+
* includeImages: true
|
|
41
|
+
* });
|
|
42
|
+
* ```
|
|
43
|
+
*/
|
|
44
|
+
static convert<F extends string | Buffer | ArrayBuffer, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
|
|
45
|
+
}
|
|
46
|
+
export {};
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.OfficeConverter = void 0;
|
|
4
|
+
const OfficeGenerator_js_1 = require("./OfficeGenerator.js");
|
|
5
|
+
const OfficeParser_js_1 = require("./OfficeParser.js");
|
|
6
|
+
/**
|
|
7
|
+
* Main converter class providing a streamlined one-step API for document conversion.
|
|
8
|
+
*
|
|
9
|
+
* This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
|
|
10
|
+
* documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
|
|
11
|
+
*/
|
|
12
|
+
class OfficeConverter {
|
|
13
|
+
/**
|
|
14
|
+
* Converts an office document from its source format to a specified destination format.
|
|
15
|
+
*
|
|
16
|
+
* This method:
|
|
17
|
+
* 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
|
|
18
|
+
* 2. Automatically configures the parser based on the generator requirements (e.g., enabling
|
|
19
|
+
* attachment extraction if images are requested in the output).
|
|
20
|
+
* 3. Generates the destination document from the AST using `OfficeGenerator`.
|
|
21
|
+
*
|
|
22
|
+
* @template F The inferred type of the input file (path string or buffer).
|
|
23
|
+
* @template T The authoritative source file type (inferred from path or config).
|
|
24
|
+
*
|
|
25
|
+
* @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
|
|
26
|
+
* @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
|
|
27
|
+
* @param config - Optional unified configuration for both the parser and generator phases.
|
|
28
|
+
*
|
|
29
|
+
* @returns A promise resolving to the ConversionResult containing the value and messages.
|
|
30
|
+
* @throws {Error} If the source format is unsupported or parsing/generation fails.
|
|
31
|
+
*
|
|
32
|
+
* @example
|
|
33
|
+
* ```typescript
|
|
34
|
+
* // Convert Word to Markdown with a single call
|
|
35
|
+
* const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
|
|
36
|
+
*
|
|
37
|
+
* // Convert PDF to HTML with OCR enabled for images
|
|
38
|
+
* const { value: html } = await OfficeConverter.convert(buffer, 'html', {
|
|
39
|
+
* ocr: true,
|
|
40
|
+
* includeImages: true
|
|
41
|
+
* });
|
|
42
|
+
* ```
|
|
43
|
+
*/
|
|
44
|
+
static async convert(file, destination, config) {
|
|
45
|
+
// 1. Prepare Parser Configuration
|
|
46
|
+
// We prioritize the top-level onWarning if provided.
|
|
47
|
+
const parserConfig = {
|
|
48
|
+
...config?.parseConfig,
|
|
49
|
+
onWarning: config?.onWarning || config?.parseConfig?.onWarning,
|
|
50
|
+
};
|
|
51
|
+
// Remove OCR settings for the streamlined converter as requested
|
|
52
|
+
parserConfig.ocr = false;
|
|
53
|
+
// Remove undefined keys to prevent overwriting defaults in resolveParserConfig
|
|
54
|
+
Object.keys(parserConfig).forEach((key) => parserConfig[key] === undefined && delete parserConfig[key]);
|
|
55
|
+
/**
|
|
56
|
+
* AUTOMATIC CONFIGURATION SYNC
|
|
57
|
+
* We sync extractAttachments from the generator configuration.
|
|
58
|
+
*/
|
|
59
|
+
parserConfig.extractAttachments = (config?.generatorConfig?.includeImages !== false) || (config?.generatorConfig?.includeCharts !== false);
|
|
60
|
+
// 2. Parse the source document into the universal AST
|
|
61
|
+
const ast = await OfficeParser_js_1.OfficeParser.parseOffice(file, parserConfig);
|
|
62
|
+
// 3. Generate the destination document from the AST
|
|
63
|
+
const generatorConfig = {
|
|
64
|
+
...config?.generatorConfig,
|
|
65
|
+
onWarning: config?.onWarning || config?.generatorConfig?.onWarning,
|
|
66
|
+
};
|
|
67
|
+
const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, destination, generatorConfig);
|
|
68
|
+
result.messages = [...(ast.warnings || []), ...result.messages];
|
|
69
|
+
return result;
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
exports.OfficeConverter = OfficeConverter;
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
import { ConversionResult, GeneratorConfig, OfficeParserAST, SupportedDestination, SupportedFileType } from './types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Main generator class providing document conversion functionality.
|
|
4
|
+
*/
|
|
5
|
+
export declare class OfficeGenerator {
|
|
6
|
+
/**
|
|
7
|
+
* Generates a file of the specified type from an AST.
|
|
8
|
+
* This is the single source of truth for generation logic.
|
|
9
|
+
*
|
|
10
|
+
* @param ast - The OfficeParserAST to generate from
|
|
11
|
+
* @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
|
|
12
|
+
* @param config - Optional configuration for the generator
|
|
13
|
+
* @returns A promise resolving to the ConversionResult containing the value and messages
|
|
14
|
+
* @throws {Error} If the destination format is unsupported
|
|
15
|
+
*/
|
|
16
|
+
static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
|
|
17
|
+
type: T;
|
|
18
|
+
}, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
|
|
19
|
+
}
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.OfficeGenerator = void 0;
|
|
4
|
+
const ChunkingGenerator_js_1 = require("./generators/ChunkingGenerator.js");
|
|
5
|
+
const CsvGenerator_js_1 = require("./generators/CsvGenerator.js");
|
|
6
|
+
const HtmlGenerator_js_1 = require("./generators/HtmlGenerator.js");
|
|
7
|
+
const MarkdownGenerator_js_1 = require("./generators/MarkdownGenerator.js");
|
|
8
|
+
const PdfGenerator_js_1 = require("./generators/PdfGenerator.js");
|
|
9
|
+
const RtfGenerator_js_1 = require("./generators/RtfGenerator.js");
|
|
10
|
+
const TextGenerator_js_1 = require("./generators/TextGenerator.js");
|
|
11
|
+
const types_js_1 = require("./types.js");
|
|
12
|
+
const errorUtils_js_1 = require("./utils/errorUtils.js");
|
|
13
|
+
/**
|
|
14
|
+
* Main generator class providing document conversion functionality.
|
|
15
|
+
*/
|
|
16
|
+
class OfficeGenerator {
|
|
17
|
+
/**
|
|
18
|
+
* Generates a file of the specified type from an AST.
|
|
19
|
+
* This is the single source of truth for generation logic.
|
|
20
|
+
*
|
|
21
|
+
* @param ast - The OfficeParserAST to generate from
|
|
22
|
+
* @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
|
|
23
|
+
* @param config - Optional configuration for the generator
|
|
24
|
+
* @returns A promise resolving to the ConversionResult containing the value and messages
|
|
25
|
+
* @throws {Error} If the destination format is unsupported
|
|
26
|
+
*/
|
|
27
|
+
static async generate(ast, destination, config) {
|
|
28
|
+
switch (destination.toLowerCase()) {
|
|
29
|
+
case 'text':
|
|
30
|
+
return new TextGenerator_js_1.TextGenerator(ast, config).generate();
|
|
31
|
+
case 'md':
|
|
32
|
+
return new MarkdownGenerator_js_1.MarkdownGenerator(ast, config).generate();
|
|
33
|
+
case 'html':
|
|
34
|
+
return new HtmlGenerator_js_1.HtmlGenerator(ast, config).generate();
|
|
35
|
+
case 'pdf':
|
|
36
|
+
return new PdfGenerator_js_1.PdfGenerator(ast, config).generate();
|
|
37
|
+
case 'csv':
|
|
38
|
+
return new CsvGenerator_js_1.CsvGenerator(ast, config).generate();
|
|
39
|
+
case 'rtf':
|
|
40
|
+
return new RtfGenerator_js_1.RtfGenerator(ast, config).generate();
|
|
41
|
+
case 'chunks':
|
|
42
|
+
return new ChunkingGenerator_js_1.ChunkingGenerator(ast, config).generate();
|
|
43
|
+
default:
|
|
44
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, undefined, destination);
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
exports.OfficeGenerator = OfficeGenerator;
|
package/dist/OfficeParser.d.ts
CHANGED
|
@@ -11,6 +11,9 @@
|
|
|
11
11
|
* - ODT, ODP, ODS (OpenDocument formats)
|
|
12
12
|
* - PDF (Portable Document Format)
|
|
13
13
|
* - RTF (Rich Text Format)
|
|
14
|
+
* - CSV (Comma-Separated Values)
|
|
15
|
+
* - MD (Markdown)
|
|
16
|
+
* - HTML (HyperText Markup Language)
|
|
14
17
|
*
|
|
15
18
|
* **Usage:**
|
|
16
19
|
* ```typescript
|
|
@@ -60,6 +63,9 @@ export declare class OfficeParser {
|
|
|
60
63
|
* - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
|
|
61
64
|
* - `.pdf` → PdfParser (PDF.js)
|
|
62
65
|
* - `.rtf` → RtfParser (custom RTF parser)
|
|
66
|
+
* - `.csv` → CsvParser
|
|
67
|
+
* - `.md` → MarkdownParser
|
|
68
|
+
* - `.html` → HtmlParser
|
|
63
69
|
*
|
|
64
70
|
* @param file - File path (string), Buffer, or ArrayBuffer containing the document
|
|
65
71
|
* @param config - Optional configuration object (defaults applied for all omitted options)
|