officeparser 6.1.0 → 7.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +284 -86
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +55 -28
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +107 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +878 -5
- package/dist/officeparser.browser.iife.js +703 -49
- package/dist/officeparser.browser.mjs +703 -49
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +237 -128
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +132 -123
- package/dist/parsers/RtfParser.d.ts +22 -2
- package/dist/parsers/RtfParser.js +1398 -1282
- package/dist/parsers/WordParser.d.ts +3 -2
- package/dist/parsers/WordParser.js +333 -115
- package/dist/sbom.cdx.json +103 -103
- package/dist/types.d.ts +833 -5
- package/dist/types.js +71 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.js +56 -2
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +109 -52
- package/dist/utils/moduleLoader.js +15 -9
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +28 -9
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
# officeParser 📄🚀
|
|
1
|
+
# officeParser 📄🚀 - The Most Versatile Office Parser & Generator
|
|
2
2
|
|
|
3
|
-
A robust, strictly-typed Node.js and Browser library for parsing office files
|
|
3
|
+
A robust, strictly-typed Node.js and Browser library for parsing and generating office files. It not only extracts content from [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format), [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values), [`md`](https://en.wikipedia.org/wiki/Markdown), and [`html`](https://en.wikipedia.org/wiki/HTML) into a rich Abstract Syntax Tree (AST), but also provides a powerful generation engine to convert that AST into formats like **Markdown**, **HTML**, **CSV**, **RTF**, **Text**, **PDF**, and **JSON**, including native **RAG-focused chunking** support.
|
|
4
4
|
|
|
5
5
|
[](https://badge.fury.io/js/officeparser)
|
|
6
6
|
[](https://www.npmjs.com/package/officeparser)
|
|
@@ -18,49 +18,15 @@ A robust, strictly-typed Node.js and Browser library for parsing office files ([
|
|
|
18
18
|
- **Debugging**: Use the visualizer to debug parsing issues by inspecting exactly how nodes are interpreted.
|
|
19
19
|
- **Format Specs**: Read detailed specifications for the AST structure and configuration options.
|
|
20
20
|
|
|
21
|
-
|
|
21
|
+
---
|
|
22
|
+
|
|
22
23
|
|
|
23
24
|
---
|
|
24
25
|
|
|
26
|
+
### 📝 [Changelog](CHANGELOG.md)
|
|
27
|
+
*Detailed release notes and the full history of updates are available in the project changelog.*
|
|
25
28
|
|
|
26
|
-
|
|
27
|
-
* 2026-04-14 - **v6.1.0 Release**: Major Infrastructure & Resource Stability. (Incremental since v6.0.0)
|
|
28
|
-
- **OCR Scheduler**: Intelligent worker pool that optimizes Tesseract lifecycle across parallel requests. **Note**: By default, Node.js processes stay active for 10s after OCR to keep workers warm (configurable via `ocrConfig.autoTerminateTimeout`); use `terminateOcr()` for immediate CLI/script exit.
|
|
29
|
-
- **Core Engine**: Replaced legacy zip extraction with `fflate` for significant performance gains and robust browser/edge compatibility.
|
|
30
|
-
- **Module System**: Full native ESM support with `Node16` resolution and verified browser bundles (Vite/Angular compatible).
|
|
31
|
-
- **Format Refinements**: Hierarchical PDF coordinate alignment and ODT/RTF list parsing stability.
|
|
32
|
-
- **Custom Properties**: Added support for extracting custom document metadata across OOXML, ODF, and PDF formats.
|
|
33
|
-
- **Sponsorship**: Integrated `funding.json` manifest and GitHub Sponsors support.
|
|
34
|
-
* 2025/12/29 - **v6.0.0 Release**: Major overhaul of the library. Transitioned from simple text extraction to a rich **Abstract Syntax Tree (AST)** output.
|
|
35
|
-
- Simplified API: Use `parseOffice` for all parsing needs (returns a Promise).
|
|
36
|
-
- Structured Output: Access hierarchical document structure (paragraphs, headings, tables, lists, etc.).
|
|
37
|
-
- Rich Metadata: Extracted document properties (author, title, creation date).
|
|
38
|
-
- Enhanced Formatting: Support for bold, italic, colors, fonts, alignment, etc.
|
|
39
|
-
- Attachment Handling: Extract images, charts, and embedded files as Base64.
|
|
40
|
-
- OCR Integration: Optional OCR for images using Tesseract.js.
|
|
41
|
-
- RTF Support: Added full support for Rich Text Format files.
|
|
42
|
-
- Improved Type Definitions: Full TypeScript support with detailed interfaces.
|
|
43
|
-
* 2024/11/12 - Added ArrayBuffer as a type of file input. Generating bundle files now which exposes namespace officeParser to be able to access parseOffice directly on the browser.
|
|
44
|
-
* 2024/10/21 - Replaced extracting zip files from decompress to yauzl. This means that we now extract files in memory and we no longer need to write them to disk. Removed config flags related to extracted files. Added flags for CLI execution.
|
|
45
|
-
* 2024/10/15 - Fixed erroring out while deleting temp files when multiple worker threads make parallel executions resulting in same file name for multiple files. Fixed erroring out when multiple executions are made without waiting for the previous execution to finish which resulted in deleting the file from other execution. Upgraded dependencies.
|
|
46
|
-
* 2024/10/13 - Fixed parsing text from xlsx files which contain no shared strings file and files which have inlineStr based strings.
|
|
47
|
-
* 2024/05/06 - Replaced pdf parsing support from pdf-parse library to natively building it using pdf.js library from Mozilla by analyzing its output. Added pdfjs-dist build as a local library.
|
|
48
|
-
* 2023/11/25 - Fixed error catching when an error occurs within the parsing of a file, especially after decompressing it. Also fixed the problem with parallel parsing of files as we were using only timestamp in file names.
|
|
49
|
-
* 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
|
|
50
|
-
* 2023/04/26 - Added support for file buffers as argument for filepath for parseOffice and parseOfficeAsync
|
|
51
|
-
* 2023/04/07 - Added typings to methods to help with Typescript projects.
|
|
52
|
-
* 2022/12/28 - Added command line method to use officeParser with or without installing it and instantly get parsed content on the console.
|
|
53
|
-
* 2022/12/10 - Fixed memory leak issues, bugs related to parsing open document files and improved error handling.
|
|
54
|
-
* 2021/11/21 - Added promise way to existing callback functions.
|
|
55
|
-
* 2020/06/01 - Added error handling and console.log enable/disable methods. Default is set at enabled. Everything backward compatible.
|
|
56
|
-
* 2019/06/17 - Added method to change location for decompressing office files in places with restricted write access.
|
|
57
|
-
* 2019/04/30 - Removed case sensitive file extension bug. File names with capital lettered extensions now supported.
|
|
58
|
-
* 2019/04/23 - Added support for open office files *.odt, *.odp, *.ods through parseOffice function. Created a new method parseOpenOffice for those who prefer targetted functions.
|
|
59
|
-
* 2019/04/23 - Added feature to delete the generated dist folder after function callback.
|
|
60
|
-
* 2019/04/22 - Added parseOffice method to avoid confusion between type of file and their extension.
|
|
61
|
-
* 2019/04/22 - Added file extension validations. Removed errors for excel files with no drawing elements.
|
|
62
|
-
* 2019/04/19 - Support added for *.xlsx files.
|
|
63
|
-
* 2019/04/18 - Support added for *.pptx files.
|
|
29
|
+
---
|
|
64
30
|
|
|
65
31
|
## Install via npm
|
|
66
32
|
|
|
@@ -69,7 +35,7 @@ npm i officeparser
|
|
|
69
35
|
```
|
|
70
36
|
|
|
71
37
|
## Command Line usage
|
|
72
|
-
You can use `officeparser` directly from the terminal to
|
|
38
|
+
You can use `officeparser` directly from the terminal to extract content as JSON AST, plain text, or generate new formats like Markdown and HTML.
|
|
73
39
|
|
|
74
40
|
```bash
|
|
75
41
|
# Get full AST as JSON (default)
|
|
@@ -78,24 +44,33 @@ npx officeparser /path/to/officeFile.docx
|
|
|
78
44
|
# Get plain text only
|
|
79
45
|
npx officeparser /path/to/officeFile.docx --toText=true
|
|
80
46
|
|
|
81
|
-
#
|
|
82
|
-
npx officeparser
|
|
47
|
+
# Generate Markdown file
|
|
48
|
+
npx officeparser report.docx --format=md --output=report.md
|
|
49
|
+
|
|
50
|
+
# Generate HTML file with specific output
|
|
51
|
+
npx officeparser presentation.pptx --format=html --output=preview.html
|
|
52
|
+
|
|
53
|
+
# Convert spreadsheet to CSV
|
|
54
|
+
npx officeparser data.xlsx --format=csv
|
|
83
55
|
```
|
|
84
56
|
|
|
85
57
|
### Config Options:
|
|
86
|
-
- `--
|
|
58
|
+
- `--format=[json|text|md|html|csv|rtf|pdf|chunks]` The output format. Default is `json`.
|
|
59
|
+
- `--output=[path]` Optional file path to write the output to.
|
|
60
|
+
- `--toText=[true|false]` Legacy flag to output only plain text. Use `--format=text` instead.
|
|
87
61
|
- `--ignoreNotes=[true|false]` Flag to ignore notes from files like PowerPoint. Default is false.
|
|
88
62
|
- `--newlineDelimiter=[delimiter]` The delimiter to use for new lines. Default is `\n`.
|
|
89
63
|
- `--putNotesAtLast=[true|false]` Flag to collect notes at the end of files like PowerPoint. Default is false.
|
|
90
|
-
- `--outputErrorToConsole=[true|false]` Flag to output errors to the console.
|
|
64
|
+
- `--outputErrorToConsole=[true|false]` **(Deprecated)** Flag to output errors to the console. Use `onWarning` callback in library usage.
|
|
91
65
|
- `--extractAttachments=[true|false]` Flag to extract images/charts as Base64. Default is false.
|
|
92
66
|
- `--ocr=[true|false]` Flag to enable OCR for extracted images. Default is false.
|
|
93
67
|
- `--includeRawContent=[true|false]` Flag to include raw XML/RTF content in nodes. Default is false.
|
|
68
|
+
- `--includeBreakNodes=[true|false]` Flag to include break nodes. Currently only available for DOCX documents.
|
|
94
69
|
- `--verbose=[true|false]` Show full error stack traces.
|
|
95
70
|
|
|
96
71
|
|
|
97
72
|
## Library Usage
|
|
98
|
-
In **
|
|
73
|
+
In **v7.0.0**, the library has evolved into a dual-purpose **Parser** and **Generator**. You can first parse any office file into a structured AST and then use the `OfficeGenerator` to transform that AST into various formats or chunks.
|
|
99
74
|
|
|
100
75
|
### Getting Started (Async/Await)
|
|
101
76
|
```js
|
|
@@ -130,8 +105,103 @@ const text = await getText("/path/to/officeFile.docx");
|
|
|
130
105
|
console.log(text);
|
|
131
106
|
```
|
|
132
107
|
|
|
108
|
+
## Using the OfficeGenerator
|
|
109
|
+
The `OfficeGenerator` is a powerful tool to convert your AST into human-readable formats or structured data.
|
|
110
|
+
|
|
111
|
+
```typescript
|
|
112
|
+
import { OfficeParser, OfficeGenerator } from 'officeparser';
|
|
113
|
+
|
|
114
|
+
const ast = await OfficeParser.parseOffice('report.docx');
|
|
115
|
+
|
|
116
|
+
// 1. Convert to Markdown
|
|
117
|
+
const md = await OfficeGenerator.generate(ast, 'md');
|
|
118
|
+
console.log(md.value);
|
|
119
|
+
|
|
120
|
+
// 2. Convert to HTML with structured style mapping (Recommended)
|
|
121
|
+
const html = await OfficeGenerator.generate(ast, 'html', {
|
|
122
|
+
includeFormatting: true,
|
|
123
|
+
styleMap: [
|
|
124
|
+
{
|
|
125
|
+
selector: { nodeType: 'paragraph', attributes: { style: 'Heading 1' } },
|
|
126
|
+
output: { tag: 'h1', classes: ['main-title'] }
|
|
127
|
+
}
|
|
128
|
+
]
|
|
129
|
+
});
|
|
130
|
+
console.log(html.value);
|
|
131
|
+
|
|
132
|
+
// 3. Convert to CSV (for spreadsheets)
|
|
133
|
+
const csv = await OfficeGenerator.generate(ast, 'csv');
|
|
134
|
+
console.log(csv.value);
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
## The New "One-Step" API: `OfficeConverter`
|
|
138
|
+
In **v7.0.0**, we introduced the `OfficeConverter.convert` method. This is the new high-level API designed for one-step transformations where you don't need to manually interact with the AST. It automatically handles parser and generator configuration synchronization.
|
|
139
|
+
|
|
140
|
+
```typescript
|
|
141
|
+
import { OfficeConverter } from 'officeparser';
|
|
142
|
+
|
|
143
|
+
// One-step conversion from DOCX to Markdown
|
|
144
|
+
const result = await OfficeConverter.convert('report.docx', 'md');
|
|
145
|
+
console.log(result.value); // The generated Markdown string
|
|
146
|
+
console.log(result.messages); // Array of warnings/info (e.g., "Skipped unsupported drawing")
|
|
147
|
+
|
|
148
|
+
// Complex conversion with nested configuration
|
|
149
|
+
const htmlResult = await OfficeConverter.convert('data.xlsx', 'html', {
|
|
150
|
+
parseConfig: {
|
|
151
|
+
ignoreNotes: true
|
|
152
|
+
},
|
|
153
|
+
generatorConfig: {
|
|
154
|
+
includeFormatting: true,
|
|
155
|
+
styleMap: [
|
|
156
|
+
{
|
|
157
|
+
selector: { attributes: { style: { value: 'Header', operator: '~=' } } },
|
|
158
|
+
output: { tag: 'h2', classes: ['data-header'] }
|
|
159
|
+
}
|
|
160
|
+
]
|
|
161
|
+
},
|
|
162
|
+
onWarning: (msg) => console.warn("Conversion Warning:", msg)
|
|
163
|
+
});
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
## Native RAG Chunking
|
|
167
|
+
`officeParser` provides native support for document chunking, specifically designed for Retrieval-Augmented Generation (RAG) workflows. It offers three distinct strategies to split your documents while maintaining context and metadata.
|
|
168
|
+
|
|
169
|
+
### 1. Fixed-Size Strategy (Recursive)
|
|
170
|
+
Splits text into chunks based on character count with a specified overlap. It uses smart boundary detection to avoid cutting in the middle of sentences or paragraphs.
|
|
171
|
+
|
|
172
|
+
### 2. Document Structure Strategy
|
|
173
|
+
Splits the document at natural structural boundaries like pages (PDF/Word), slides (PPTX), or high-level headings. This preserves the logical flow of the document.
|
|
174
|
+
|
|
175
|
+
### 3. Semantic Strategy
|
|
176
|
+
Uses cosine similarity between sentence embeddings to identify coherent topic boundaries. This ensures that each chunk contains semantically related content (requires an embedding function).
|
|
177
|
+
|
|
178
|
+
### The `OfficeChunk` Interface
|
|
179
|
+
Every chunk produced contains not just text, but rich metadata to help your RAG pipeline:
|
|
180
|
+
```typescript
|
|
181
|
+
{
|
|
182
|
+
text: string; // The chunk content
|
|
183
|
+
metadata: {
|
|
184
|
+
sourceType: string; // e.g., "docx", "pdf"
|
|
185
|
+
pageNumber?: number; // Current page
|
|
186
|
+
slideNumber?: number; // Current slide
|
|
187
|
+
closestHeading?: string; // The heading this chunk belongs to
|
|
188
|
+
chunkIndex: number; // Sequential index
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
#### Example: Generating Chunks
|
|
194
|
+
```typescript
|
|
195
|
+
const chunks = await OfficeGenerator.generate(ast, 'chunks', {
|
|
196
|
+
strategy: 'fixed-size',
|
|
197
|
+
maxChunkSize: 1000,
|
|
198
|
+
chunkOverlap: 200
|
|
199
|
+
});
|
|
200
|
+
console.log(`Generated ${chunks.value.length} chunks`);
|
|
201
|
+
```
|
|
202
|
+
|
|
133
203
|
### Using Callbacks (Backward Compatibility Support)
|
|
134
|
-
|
|
204
|
+
Callbacks are still supported for those preferred, but the data returned is now the AST object.
|
|
135
205
|
```js
|
|
136
206
|
const officeParser = require('officeparser');
|
|
137
207
|
|
|
@@ -165,14 +235,14 @@ The `OfficeParserAST` provides a format-agnostic representation of your document
|
|
|
165
235
|
|
|
166
236
|
```text
|
|
167
237
|
OfficeParserAST
|
|
168
|
-
├── type: "docx" | "
|
|
238
|
+
├── type: "docx" | "pdf" | "xlsx" | "csv" | "md" | ... (11 formats supported)
|
|
169
239
|
├── metadata: { author, title, created, modified, ..., customProperties }
|
|
170
240
|
├── content: [ OfficeContentNode ]
|
|
171
241
|
│ ├── type: "paragraph" | "heading" | "table" | "list" | ...
|
|
172
242
|
│ ├── text: "Concatenated text of this node and all children"
|
|
173
243
|
│ ├── children: [ OfficeContentNode ] (recursive)
|
|
174
244
|
│ ├── formatting: { bold, italic, color, size, font, ... }
|
|
175
|
-
│ ├── metadata: { level, listId, row, col, ... }
|
|
245
|
+
│ ├── metadata: { level, listId, paragraphIndentation, row, col, ... }
|
|
176
246
|
│ └── rawContent: "<xml>...</xml>" (if enabled)
|
|
177
247
|
├── attachments: [ OfficeAttachment ]
|
|
178
248
|
│ ├── type: "image" | "chart"
|
|
@@ -224,13 +294,15 @@ List Node
|
|
|
224
294
|
listId: "1",
|
|
225
295
|
listType: "ordered",
|
|
226
296
|
indentation: 0,
|
|
297
|
+
paragraphIndentation: { left: 720, hanging: 360 },
|
|
227
298
|
itemIndex: 0
|
|
228
299
|
}
|
|
229
300
|
└── children: [ Text Content... ]
|
|
230
301
|
```
|
|
231
302
|
|
|
232
303
|
- **`listId`**: A unique identifier for the list definition. Multiple items with the same `listId` belong to the same logical list.
|
|
233
|
-
- **`indentation`**: The nesting level (0-based).
|
|
304
|
+
- **`indentation`**: The structural nesting level (0-based).
|
|
305
|
+
- **`paragraphIndentation`**: The physical indentation formatting in twentieths of a point (twips) (e.g., `left`, `right`, `firstLine`, `hanging`).
|
|
234
306
|
- **`itemIndex`**: The sequential position within that list level.
|
|
235
307
|
- **`listType`**: Either `ordered` (numbered) or `unordered` (bulleted).
|
|
236
308
|
|
|
@@ -309,13 +381,31 @@ Formatting can be found at two levels:
|
|
|
309
381
|
1. **Node Level**: Applied directly to a text run or paragraph.
|
|
310
382
|
2. **Document Level**: Found in `ast.metadata.formatting` (defaults) or `ast.metadata.styleMap` (named styles).
|
|
311
383
|
|
|
312
|
-
### 6.
|
|
384
|
+
### 6. Breaks
|
|
385
|
+
Breaks are currently only supported when parsing DOCX-documents. Breaks are added as a node of type `break` and carry metadata of the type `BreakMetadata`. When `includeRawContent` is enabled, they also include the `rawContent` string from the original XML.
|
|
386
|
+
|
|
387
|
+
```text
|
|
388
|
+
Break Node
|
|
389
|
+
├── type: "break"
|
|
390
|
+
└── metadata: {
|
|
391
|
+
breakType: "textWrapping" | "page" | "column" | "lastRenderedPage" | "carriageReturn",
|
|
392
|
+
clear?: "all" | "left" | "none" | "right"
|
|
393
|
+
}
|
|
394
|
+
```
|
|
395
|
+
|
|
396
|
+
- `breakType`: Type of break. `textWrapping` (default) is a standard line break, `page` is a page break, `column` is a break to the next column, `lastRenderedPage` is a soft break inserted by Word, and `carriageReturn` is an explicit carriage return (`w:cr`).
|
|
397
|
+
- `clear`: Relevant for `textWrapping`. Indicates if text should wrap around floating objects.
|
|
398
|
+
|
|
399
|
+
> [!NOTE]
|
|
400
|
+
> Even though break nodes don't have a `text` property, the `ast.toText()` method will automatically convert them to newlines (`\n`) or the configured delimiter in the final string output.
|
|
401
|
+
|
|
402
|
+
### 7. Advanced Metadata
|
|
313
403
|
The `ast.metadata` object provides document-wide context:
|
|
314
404
|
- **`styleMap`**: A dictionary of style names to their `TextFormatting` definitions found in the document.
|
|
315
405
|
- **`formatting`**: Document-wide default settings (e.g., default font or font size).
|
|
316
406
|
- **`customProperties`**: A dictionary of user-defined metadata embedded in the document (OOXML `custom.xml`, ODF `meta:user-defined`, or PDF Info dictionary).
|
|
317
407
|
|
|
318
|
-
###
|
|
408
|
+
### 8. Custom Properties
|
|
319
409
|
You can access custom user-defined metadata that might be embedded in the document:
|
|
320
410
|
|
|
321
411
|
```javascript
|
|
@@ -324,6 +414,14 @@ console.log("Custom Metadata:", ast.metadata.customProperties);
|
|
|
324
414
|
// Output: { "ProjectID": "ABC-123", "InternalReview": true }
|
|
325
415
|
```
|
|
326
416
|
|
|
417
|
+
## Performance & Fidelity Highlights (v7.0.0)
|
|
418
|
+
The v7.0.0 release brings significant internal optimizations and fidelity improvements:
|
|
419
|
+
- **OpenOffice Speedups**: Up to **23x faster** parsing for ODP presentations thanks to optimized XML caching.
|
|
420
|
+
- **Excel Memory Efficiency**: Resolved $O(n)$ memory overhead issues for large spreadsheets (#91) by switching to iterative stream-based parsing.
|
|
421
|
+
- **RTF Performance**: Rewritten core loop to resolve $O(n^2)$ bottlenecks during string accumulation.
|
|
422
|
+
- **Advanced Table Fidelity**: Native support for **vertical cell merging** (`vMerge`) and **horizontal spanning** (`gridSpan`) in DOCX, ensuring complex tables look exactly as they do in Word.
|
|
423
|
+
- **Parser Extensions**: You can now parse `CSV`, `Markdown`, and `HTML` files *into* the unified Office AST, allowing you to use the `OfficeGenerator` on them just like any other format.
|
|
424
|
+
|
|
327
425
|
### Advanced AST Usage
|
|
328
426
|
Beyond using `ast.toText()`, you can interact with the structural data directly:
|
|
329
427
|
|
|
@@ -418,29 +516,114 @@ Pass an optional config object as the second argument to `parseOffice`.
|
|
|
418
516
|
|
|
419
517
|
| Flag | DataType | Default | Explanation |
|
|
420
518
|
|------|----------|---------|-------------|
|
|
421
|
-
| `outputErrorToConsole` | boolean | `false` | Show logs to console in case of an error. |
|
|
519
|
+
| `outputErrorToConsole` | boolean | `false` | **Deprecated**: Use `onWarning` instead. Show logs to console in case of an error. |
|
|
422
520
|
| `newlineDelimiter` | string | `\n` | Delimiter for new lines in text output. |
|
|
423
521
|
| `ignoreNotes` | boolean | `false` | Ignore notes in files like PowerPoint/ODP. |
|
|
424
|
-
| `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document.
|
|
522
|
+
| `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. |
|
|
425
523
|
| `extractAttachments` | boolean | `false` | Extract images and charts as Base64. |
|
|
426
524
|
| `includeRawContent` | boolean | `false` | Include raw XML/RTF markup in the nodes. |
|
|
427
|
-
| `serializeRawContent` | boolean | `true` |
|
|
428
|
-
| `preserveXmlWhitespace` | boolean | `false` |
|
|
525
|
+
| `serializeRawContent` | boolean | `true` | Re-serializes raw XML to clean strings. |
|
|
526
|
+
| `preserveXmlWhitespace` | boolean | `false` | Preserves original XML whitespace. |
|
|
429
527
|
| `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
|
|
430
|
-
| `
|
|
431
|
-
| `
|
|
432
|
-
| `
|
|
433
|
-
| `
|
|
434
|
-
| `
|
|
435
|
-
| `
|
|
436
|
-
|
|
437
|
-
|
|
528
|
+
| `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. |
|
|
529
|
+
| `ocrConfig` | object | `{}` | OCR Scheduler configuration. |
|
|
530
|
+
| `includeBreakNodes` | boolean | `false` | Include `w:br`, `w:cr` nodes (DOCX only).|
|
|
531
|
+
| `ignoreInternalLinks` | boolean | `false` | Remove all bookmarks and internal jumps. |
|
|
532
|
+
| `csvDelimiter` | string | `,` | Custom delimiter for parsing CSV files. |
|
|
533
|
+
| `fileType` | string | `null` | Manual format override (authoritative). |
|
|
534
|
+
|
|
535
|
+
## Generator Configuration: GeneratorConfig
|
|
536
|
+
Configuration options for `OfficeGenerator.generate`.
|
|
537
|
+
|
|
538
|
+
| Flag | DataType | Default | Explanation |
|
|
539
|
+
|------|----------|---------|-------------|
|
|
540
|
+
| `includeFormatting` | boolean | `false` | Whether to include semantic styles (bold, italic) in output. |
|
|
541
|
+
| `styleMap` | string[] \| array | `[]` | Array of style mappings (DSL strings or structured objects). |
|
|
542
|
+
| `ignoreDefaultStyleMap`| boolean | `false` | Ignore the library's default style mappings. |
|
|
543
|
+
| `includeMetadata` | boolean | `false` | Include document metadata in the output (e.g., as frontmatter). |
|
|
544
|
+
| `onNode` | function | `undefined` | Callback to intercept/modify any node during generation. |
|
|
545
|
+
|
|
546
|
+
### 🛠️ Advanced Node Manipulation (Pro Users)
|
|
547
|
+
The `onNode` callback is a powerful tool that gives you complete control over the generation process. It is called for **every single node** in the AST before it is rendered.
|
|
548
|
+
|
|
549
|
+
#### Callback Capabilities:
|
|
550
|
+
1. **Filter/Remove Nodes**: Return `false` to skip a node and all its children.
|
|
551
|
+
2. **Override Rendering**: Return a `string` to use that exact text as the output, bypassing default logic and recursion.
|
|
552
|
+
3. **Mutate Nodes**: Modify the `node` object directly (e.g., changing `node.text`) and return `void` to let the generator proceed with your changes.
|
|
553
|
+
4. **Async Support**: The callback can be `async`, allowing you to fetch external data or perform complex logic during generation.
|
|
554
|
+
|
|
555
|
+
#### Pro Example:
|
|
556
|
+
```typescript
|
|
557
|
+
const result = await ast.to('md', {
|
|
558
|
+
onNode: async (node) => {
|
|
559
|
+
// 1. Skip all images
|
|
560
|
+
if (node.type === 'image') return false;
|
|
561
|
+
|
|
562
|
+
// 2. Redact sensitive info by mutating the node
|
|
563
|
+
if (node.text?.includes('SECRET_KEY')) {
|
|
564
|
+
node.text = node.text.replace(/SECRET_KEY: \w+/, 'SECRET_KEY: [REDACTED]');
|
|
565
|
+
}
|
|
566
|
+
|
|
567
|
+
// 3. Custom rendering for specific styles
|
|
568
|
+
if (node.metadata?.style === 'Callout') {
|
|
569
|
+
return `> [!INFO]\n> ${node.text}`;
|
|
570
|
+
}
|
|
571
|
+
|
|
572
|
+
// 4. Proceed with default rendering (implicitly returns void)
|
|
573
|
+
}
|
|
574
|
+
});
|
|
575
|
+
```
|
|
576
|
+
|
|
577
|
+
### Advanced Style Mapping (Semantic Translation)
|
|
578
|
+
The `styleMap` configuration is the primary way to define the "semantic meaning" of document styles. We recommend using **Structured Style Mappings** for full type safety and power.
|
|
579
|
+
|
|
580
|
+
#### 1. Structured Style Mappings (Recommended)
|
|
581
|
+
Use structured objects to match nodes based on type and attributes, and specify detailed output properties like classes and custom attributes.
|
|
582
|
+
|
|
583
|
+
```typescript
|
|
584
|
+
styleMap: [
|
|
585
|
+
{
|
|
586
|
+
selector: {
|
|
587
|
+
nodeType: 'paragraph',
|
|
588
|
+
attributes: { style: 'Heading 1' }
|
|
589
|
+
},
|
|
590
|
+
output: {
|
|
591
|
+
tag: 'h1',
|
|
592
|
+
classes: ['main-title'],
|
|
593
|
+
attributes: { id: 'top' }
|
|
594
|
+
}
|
|
595
|
+
},
|
|
596
|
+
{
|
|
597
|
+
// Use operators like '~=' for partial matches
|
|
598
|
+
selector: { attributes: { style: { value: 'Quote', operator: '~=' } } },
|
|
599
|
+
output: { tag: 'blockquote' }
|
|
600
|
+
}
|
|
601
|
+
]
|
|
602
|
+
```
|
|
603
|
+
|
|
604
|
+
#### 2. Legacy String DSL
|
|
605
|
+
The library also maintains support for a simple string-based DSL, highly compatible with `mammoth.js`.
|
|
606
|
+
|
|
607
|
+
- **Literal Matching**: `"p[style-name='Heading 1'] => h1"`
|
|
608
|
+
- **Regex-like Matching**: `"p[style~='Title'] => h2"`
|
|
609
|
+
- **Attribute Filters**: `"p[style-name='Quote'][lang='en'] => blockquote"`
|
|
610
|
+
|
|
611
|
+
## Chunking Configuration: ChunkingConfig
|
|
612
|
+
Specific options when using `format: 'chunks'`.
|
|
613
|
+
|
|
614
|
+
| Flag | DataType | Default | Explanation |
|
|
615
|
+
|------|----------|---------|-------------|
|
|
616
|
+
| `strategy` | string | `'fixed-size'`| The chunking strategy (`fixed-size`, `document-structure`, `semantic`). |
|
|
617
|
+
| `maxChunkSize` | number | `1000` | Maximum characters per chunk. |
|
|
618
|
+
| `chunkOverlap` | number | `200` | Overlap between consecutive chunks. |
|
|
619
|
+
| `similarityThreshold`| number | `0.5` | Threshold for semantic splitting (0.0 to 1.0). |
|
|
620
|
+
| `embedBatchSize` | number | `50` | Batch size for embedding requests. |
|
|
438
621
|
|
|
439
622
|
### OCR Scheduler & Resource Management
|
|
440
623
|
If your application uses OCR, `officeParser` utilizes an intelligent **Smart Worker Pool** to maintain a background worker pool and optimize repeated parse requests.
|
|
441
624
|
|
|
442
625
|
- **Dynamic Affinity**: Workers in the pool persist with their last used language affinity.
|
|
443
|
-
- **
|
|
626
|
+
- **LRU Re-allocation**: If a new language is requested and the pool is full, the manager identifies the **Least Recently Used (LRU)** idle worker and re-initializes it for the new language. This avoids the overhead of destroying and recreating workers.
|
|
444
627
|
- **Auto-Termination**: Workers are automatically cleaned up after 10 seconds of inactivity (configurable via `ocrConfig.autoTerminateTimeout`).
|
|
445
628
|
|
|
446
629
|
#### `OfficeParser.terminateOcr()`
|
|
@@ -462,7 +645,7 @@ async function runCleaner() {
|
|
|
462
645
|
```
|
|
463
646
|
|
|
464
647
|
> [!TIP]
|
|
465
|
-
> This is handled automatically in
|
|
648
|
+
> This is handled automatically in the built-in CLI (`npx officeparser ...`). You only need to call this manually if you are using the library in your own custom script and want a snappy exit.
|
|
466
649
|
|
|
467
650
|
```js
|
|
468
651
|
const config = {
|
|
@@ -509,36 +692,51 @@ The library provides two types of browser bundles in the `dist/` directory:
|
|
|
509
692
|
1. **`officeparser.browser.iife.js`**: Standard IIFE bundle for direct `<script>` tag usage. Exposes the global `officeParser` namespace.
|
|
510
693
|
2. **`officeparser.browser.mjs`**: Modern ESM bundle for use with `import` statements or modern bundlers.
|
|
511
694
|
|
|
695
|
+
### Usage (ESM)
|
|
696
|
+
If you are using a modern bundler like **Vite**, **Webpack**, or **Next.js**:
|
|
697
|
+
|
|
698
|
+
```javascript
|
|
699
|
+
import { OfficeParser } from 'officeparser';
|
|
700
|
+
|
|
701
|
+
const handleFile = async (event) => {
|
|
702
|
+
const file = event.target.files[0];
|
|
703
|
+
const buffer = await file.arrayBuffer();
|
|
704
|
+
|
|
705
|
+
try {
|
|
706
|
+
// Pass the Buffer or Uint8Array directly
|
|
707
|
+
const ast = await OfficeParser.parseOffice(new Uint8Array(buffer));
|
|
708
|
+
console.log(ast.toText());
|
|
709
|
+
} catch (err) {
|
|
710
|
+
console.error(err);
|
|
711
|
+
}
|
|
712
|
+
};
|
|
713
|
+
```
|
|
714
|
+
|
|
715
|
+
> [!NOTE]
|
|
716
|
+
> **Why `fs` fails in the browser**: Browsers do not have a built-in file system. If you try to pass a file path string in the browser, `officeParser` will throw a descriptive "Fail-Fast" error instead of crashing mysteriously:
|
|
717
|
+
> `[officeparser] Node.js 'fs' module is not available in the browser. Please pass a Buffer or Uint8Array instead.`
|
|
718
|
+
|
|
512
719
|
### Usage (Script Tag)
|
|
513
|
-
Include the IIFE bundle
|
|
720
|
+
Include the IIFE bundle available in the release assets or your `dist/` folder. This exposes the global `officeParser` object.
|
|
514
721
|
|
|
515
722
|
```html
|
|
516
723
|
<script src="dist/officeparser.browser.iife.js"></script>
|
|
517
724
|
<script>
|
|
518
|
-
async function handleFile(
|
|
519
|
-
|
|
725
|
+
async function handleFile(event) {
|
|
726
|
+
const file = event.target.files[0];
|
|
727
|
+
const buffer = await file.arrayBuffer();
|
|
728
|
+
|
|
520
729
|
try {
|
|
521
|
-
|
|
730
|
+
// Reconstruct as Uint8Array for the parser
|
|
731
|
+
const ast = await officeParser.parseOffice(new Uint8Array(buffer));
|
|
522
732
|
console.log(ast.toText());
|
|
523
733
|
} catch (error) {
|
|
524
|
-
console.error(error);
|
|
734
|
+
console.error("Parsing failed:", error);
|
|
525
735
|
}
|
|
526
736
|
}
|
|
527
737
|
</script>
|
|
528
738
|
```
|
|
529
739
|
|
|
530
|
-
### Usage (ESM)
|
|
531
|
-
If you are using a modern browser that supports modules or a dev server like Vite:
|
|
532
|
-
|
|
533
|
-
```html
|
|
534
|
-
<script type="module">
|
|
535
|
-
import { OfficeParser } from './dist/officeparser.browser.mjs';
|
|
536
|
-
|
|
537
|
-
const ast = await OfficeParser.parseOffice(fileBuffer);
|
|
538
|
-
console.log(ast.metadata);
|
|
539
|
-
</script>
|
|
540
|
-
```
|
|
541
|
-
|
|
542
740
|
### PDF Worker Configuration in Browser
|
|
543
741
|
When using `officeparser` in a browser environment to parse PDF files, you may provide the `pdfWorkerSrc` configuration option. If not provided, it defaults to a CDN link for `pdfjs-dist@5.6.205`.
|
|
544
742
|
|
|
@@ -550,7 +748,7 @@ const ast = await officeParser.parseOffice(file);
|
|
|
550
748
|
|
|
551
749
|
// Or override it with your own path or a different version:
|
|
552
750
|
const ast2 = await officeParser.parseOffice(file, {
|
|
553
|
-
pdfWorkerSrc: "https://
|
|
751
|
+
pdfWorkerSrc: "https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
|
|
554
752
|
});
|
|
555
753
|
```
|
|
556
754
|
|
|
@@ -579,7 +777,7 @@ For a comprehensive guide, visit our [Debugging & Troubleshooting Documentation]
|
|
|
579
777
|
|
|
580
778
|
## Contributing
|
|
581
779
|
|
|
582
|
-
|
|
780
|
+
Contributions are welcome! Please see [CONTRIBUTING.md](CONTRIBUTING.md) for details on how to get started.
|
|
583
781
|
|
|
584
782
|
## License
|
|
585
783
|
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { ConversionResult, OfficeConverterConfig, SupportedDestination, SupportedFileType } from './types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Utility type to infer the file type from a file path string literal.
|
|
4
|
+
*/
|
|
5
|
+
type InferFileTypeFromPath<T> = T extends `${string}.${infer E}` ? (Lowercase<E> extends SupportedFileType ? Lowercase<E> : SupportedFileType) : SupportedFileType;
|
|
6
|
+
/**
|
|
7
|
+
* Main converter class providing a streamlined one-step API for document conversion.
|
|
8
|
+
*
|
|
9
|
+
* This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
|
|
10
|
+
* documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
|
|
11
|
+
*/
|
|
12
|
+
export declare class OfficeConverter {
|
|
13
|
+
/**
|
|
14
|
+
* Converts an office document from its source format to a specified destination format.
|
|
15
|
+
*
|
|
16
|
+
* This method:
|
|
17
|
+
* 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
|
|
18
|
+
* 2. Automatically configures the parser based on the generator requirements (e.g., enabling
|
|
19
|
+
* attachment extraction if images are requested in the output).
|
|
20
|
+
* 3. Generates the destination document from the AST using `OfficeGenerator`.
|
|
21
|
+
*
|
|
22
|
+
* @template F The inferred type of the input file (path string or buffer).
|
|
23
|
+
* @template T The authoritative source file type (inferred from path or config).
|
|
24
|
+
*
|
|
25
|
+
* @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
|
|
26
|
+
* @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
|
|
27
|
+
* @param config - Optional unified configuration for both the parser and generator phases.
|
|
28
|
+
*
|
|
29
|
+
* @returns A promise resolving to the ConversionResult containing the value and messages.
|
|
30
|
+
* @throws {Error} If the source format is unsupported or parsing/generation fails.
|
|
31
|
+
*
|
|
32
|
+
* @example
|
|
33
|
+
* ```typescript
|
|
34
|
+
* // Convert Word to Markdown with a single call
|
|
35
|
+
* const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
|
|
36
|
+
*
|
|
37
|
+
* // Convert PDF to HTML with OCR enabled for images
|
|
38
|
+
* const { value: html } = await OfficeConverter.convert(buffer, 'html', {
|
|
39
|
+
* ocr: true,
|
|
40
|
+
* includeImages: true
|
|
41
|
+
* });
|
|
42
|
+
* ```
|
|
43
|
+
*/
|
|
44
|
+
static convert<F extends string | Buffer | ArrayBuffer, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
|
|
45
|
+
}
|
|
46
|
+
export {};
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.OfficeConverter = void 0;
|
|
4
|
+
const OfficeGenerator_js_1 = require("./OfficeGenerator.js");
|
|
5
|
+
const OfficeParser_js_1 = require("./OfficeParser.js");
|
|
6
|
+
/**
|
|
7
|
+
* Main converter class providing a streamlined one-step API for document conversion.
|
|
8
|
+
*
|
|
9
|
+
* This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
|
|
10
|
+
* documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
|
|
11
|
+
*/
|
|
12
|
+
class OfficeConverter {
|
|
13
|
+
/**
|
|
14
|
+
* Converts an office document from its source format to a specified destination format.
|
|
15
|
+
*
|
|
16
|
+
* This method:
|
|
17
|
+
* 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
|
|
18
|
+
* 2. Automatically configures the parser based on the generator requirements (e.g., enabling
|
|
19
|
+
* attachment extraction if images are requested in the output).
|
|
20
|
+
* 3. Generates the destination document from the AST using `OfficeGenerator`.
|
|
21
|
+
*
|
|
22
|
+
* @template F The inferred type of the input file (path string or buffer).
|
|
23
|
+
* @template T The authoritative source file type (inferred from path or config).
|
|
24
|
+
*
|
|
25
|
+
* @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
|
|
26
|
+
* @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
|
|
27
|
+
* @param config - Optional unified configuration for both the parser and generator phases.
|
|
28
|
+
*
|
|
29
|
+
* @returns A promise resolving to the ConversionResult containing the value and messages.
|
|
30
|
+
* @throws {Error} If the source format is unsupported or parsing/generation fails.
|
|
31
|
+
*
|
|
32
|
+
* @example
|
|
33
|
+
* ```typescript
|
|
34
|
+
* // Convert Word to Markdown with a single call
|
|
35
|
+
* const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
|
|
36
|
+
*
|
|
37
|
+
* // Convert PDF to HTML with OCR enabled for images
|
|
38
|
+
* const { value: html } = await OfficeConverter.convert(buffer, 'html', {
|
|
39
|
+
* ocr: true,
|
|
40
|
+
* includeImages: true
|
|
41
|
+
* });
|
|
42
|
+
* ```
|
|
43
|
+
*/
|
|
44
|
+
static async convert(file, destination, config) {
|
|
45
|
+
// 1. Prepare Parser Configuration
|
|
46
|
+
// We prioritize the top-level onWarning if provided.
|
|
47
|
+
const parserConfig = {
|
|
48
|
+
...config?.parseConfig,
|
|
49
|
+
onWarning: config?.onWarning || config?.parseConfig?.onWarning,
|
|
50
|
+
};
|
|
51
|
+
// Remove OCR settings for the streamlined converter as requested
|
|
52
|
+
parserConfig.ocr = false;
|
|
53
|
+
// Remove undefined keys to prevent overwriting defaults in resolveParserConfig
|
|
54
|
+
Object.keys(parserConfig).forEach((key) => parserConfig[key] === undefined && delete parserConfig[key]);
|
|
55
|
+
/**
|
|
56
|
+
* AUTOMATIC CONFIGURATION SYNC
|
|
57
|
+
* We sync extractAttachments from the generator configuration.
|
|
58
|
+
*/
|
|
59
|
+
parserConfig.extractAttachments = (config?.generatorConfig?.includeImages !== false) || (config?.generatorConfig?.includeCharts !== false);
|
|
60
|
+
// 2. Parse the source document into the universal AST
|
|
61
|
+
const ast = await OfficeParser_js_1.OfficeParser.parseOffice(file, parserConfig);
|
|
62
|
+
// 3. Generate the destination document from the AST
|
|
63
|
+
const generatorConfig = {
|
|
64
|
+
...config?.generatorConfig,
|
|
65
|
+
onWarning: config?.onWarning || config?.generatorConfig?.onWarning,
|
|
66
|
+
};
|
|
67
|
+
const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, destination, generatorConfig);
|
|
68
|
+
result.messages = [...(ast.warnings || []), ...result.messages];
|
|
69
|
+
return result;
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
exports.OfficeConverter = OfficeConverter;
|