officeparser 6.0.7 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +136 -52
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +44 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +117 -0
- package/dist/index.d.ts +4 -4
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +133 -3
- package/dist/officeparser.browser.iife.js +115 -0
- package/dist/officeparser.browser.mjs +114 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +76 -68
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +224 -159
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +188 -179
- package/dist/parsers/RtfParser.d.ts +21 -1
- package/dist/parsers/RtfParser.js +117 -48
- package/dist/parsers/WordParser.d.ts +2 -1
- package/dist/parsers/WordParser.js +214 -123
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +123 -3
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
package/README.md
CHANGED
|
@@ -3,6 +3,8 @@
|
|
|
3
3
|
A robust, strictly-typed Node.js and Browser library for parsing office files ([`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format)). It produces a clean, hierarchical Abstract Syntax Tree (AST) with rich metadata, text formatting, and full attachment support.
|
|
4
4
|
|
|
5
5
|
[](https://badge.fury.io/js/officeparser)
|
|
6
|
+
[](https://www.npmjs.com/package/officeparser)
|
|
7
|
+
[](https://www.npmjs.com/package/officeparser)
|
|
6
8
|
[](https://opensource.org/licenses/MIT)
|
|
7
9
|
|
|
8
10
|
---
|
|
@@ -21,37 +23,12 @@ A robust, strictly-typed Node.js and Browser library for parsing office files ([
|
|
|
21
23
|
---
|
|
22
24
|
|
|
23
25
|
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
- Attachment Handling: Extract images, charts, and embedded files as Base64.
|
|
31
|
-
- OCR Integration: Optional OCR for images using Tesseract.js.
|
|
32
|
-
- RTF Support: Added full support for Rich Text Format files.
|
|
33
|
-
- Improved Type Definitions: Full TypeScript support with detailed interfaces.
|
|
34
|
-
* 2024/11/12 - Added ArrayBuffer as a type of file input. Generating bundle files now which exposes namespace officeParser to be able to access parseOffice directly on the browser.
|
|
35
|
-
* 2024/10/21 - Replaced extracting zip files from decompress to yauzl. This means that we now extract files in memory and we no longer need to write them to disk. Removed config flags related to extracted files. Added flags for CLI execution.
|
|
36
|
-
* 2024/10/15 - Fixed erroring out while deleting temp files when multiple worker threads make parallel executions resulting in same file name for multiple files. Fixed erroring out when multiple executions are made without waiting for the previous execution to finish which resulted in deleting the file from other execution. Upgraded dependencies.
|
|
37
|
-
* 2024/10/13 - Fixed parsing text from xlsx files which contain no shared strings file and files which have inlineStr based strings.
|
|
38
|
-
* 2024/05/06 - Replaced pdf parsing support from pdf-parse library to natively building it using pdf.js library from Mozilla by analyzing its output. Added pdfjs-dist build as a local library.
|
|
39
|
-
* 2023/11/25 - Fixed error catching when an error occurs within the parsing of a file, especially after decompressing it. Also fixed the problem with parallel parsing of files as we were using only timestamp in file names.
|
|
40
|
-
* 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
|
|
41
|
-
* 2023/04/26 - Added support for file buffers as argument for filepath for parseOffice and parseOfficeAsync
|
|
42
|
-
* 2023/04/07 - Added typings to methods to help with Typescript projects.
|
|
43
|
-
* 2022/12/28 - Added command line method to use officeParser with or without installing it and instantly get parsed content on the console.
|
|
44
|
-
* 2022/12/10 - Fixed memory leak issues, bugs related to parsing open document files and improved error handling.
|
|
45
|
-
* 2021/11/21 - Added promise way to existing callback functions.
|
|
46
|
-
* 2020/06/01 - Added error handling and console.log enable/disable methods. Default is set at enabled. Everything backward compatible.
|
|
47
|
-
* 2019/06/17 - Added method to change location for decompressing office files in places with restricted write access.
|
|
48
|
-
* 2019/04/30 - Removed case sensitive file extension bug. File names with capital lettered extensions now supported.
|
|
49
|
-
* 2019/04/23 - Added support for open office files *.odt, *.odp, *.ods through parseOffice function. Created a new method parseOpenOffice for those who prefer targetted functions.
|
|
50
|
-
* 2019/04/23 - Added feature to delete the generated dist folder after function callback.
|
|
51
|
-
* 2019/04/22 - Added parseOffice method to avoid confusion between type of file and their extension.
|
|
52
|
-
* 2019/04/22 - Added file extension validations. Removed errors for excel files with no drawing elements.
|
|
53
|
-
* 2019/04/19 - Support added for *.xlsx files.
|
|
54
|
-
* 2019/04/18 - Support added for *.pptx files.
|
|
26
|
+
---
|
|
27
|
+
|
|
28
|
+
### 📝 [Changelog](CHANGELOG.md)
|
|
29
|
+
*Detailed release notes and the full history of updates are available in the project changelog.*
|
|
30
|
+
|
|
31
|
+
---
|
|
55
32
|
|
|
56
33
|
## Install via npm
|
|
57
34
|
|
|
@@ -82,6 +59,8 @@ npx officeparser /path/to/officeFile.docx --ignoreNotes=true --newlineDelimiter=
|
|
|
82
59
|
- `--extractAttachments=[true|false]` Flag to extract images/charts as Base64. Default is false.
|
|
83
60
|
- `--ocr=[true|false]` Flag to enable OCR for extracted images. Default is false.
|
|
84
61
|
- `--includeRawContent=[true|false]` Flag to include raw XML/RTF content in nodes. Default is false.
|
|
62
|
+
- `--includeBreakNodes=[true|false]` Flag to include break nodes. Currently only available for DOCX documents
|
|
63
|
+
- `--verbose=[true|false]` Show full error stack traces.
|
|
85
64
|
|
|
86
65
|
|
|
87
66
|
## Library Usage
|
|
@@ -121,7 +100,7 @@ console.log(text);
|
|
|
121
100
|
```
|
|
122
101
|
|
|
123
102
|
### Using Callbacks (Backward Compatibility Support)
|
|
124
|
-
|
|
103
|
+
Callbacks are still supported for those preferred, but the data returned is now the AST object.
|
|
125
104
|
```js
|
|
126
105
|
const officeParser = require('officeparser');
|
|
127
106
|
|
|
@@ -156,13 +135,13 @@ The `OfficeParserAST` provides a format-agnostic representation of your document
|
|
|
156
135
|
```text
|
|
157
136
|
OfficeParserAST
|
|
158
137
|
├── type: "docx" | "pptx" | "xlsx" | ...
|
|
159
|
-
├── metadata: { author, title, created, modified,
|
|
138
|
+
├── metadata: { author, title, created, modified, ..., customProperties }
|
|
160
139
|
├── content: [ OfficeContentNode ]
|
|
161
140
|
│ ├── type: "paragraph" | "heading" | "table" | "list" | ...
|
|
162
141
|
│ ├── text: "Concatenated text of this node and all children"
|
|
163
142
|
│ ├── children: [ OfficeContentNode ] (recursive)
|
|
164
143
|
│ ├── formatting: { bold, italic, color, size, font, ... }
|
|
165
|
-
│ ├── metadata: { level, listId, row, col, ... }
|
|
144
|
+
│ ├── metadata: { level, listId, paragraphIndentation, row, col, ... }
|
|
166
145
|
│ └── rawContent: "<xml>...</xml>" (if enabled)
|
|
167
146
|
├── attachments: [ OfficeAttachment ]
|
|
168
147
|
│ ├── type: "image" | "chart"
|
|
@@ -177,7 +156,7 @@ OfficeParserAST
|
|
|
177
156
|
```json
|
|
178
157
|
{
|
|
179
158
|
"type": "docx",
|
|
180
|
-
"metadata": { "author": "John Doe", "title": "Annual Report" },
|
|
159
|
+
"metadata": { "author": "John Doe", "title": "Annual Report", "customProperties": { "Department": "Finance" } },
|
|
181
160
|
"content": [
|
|
182
161
|
{
|
|
183
162
|
"type": "heading",
|
|
@@ -214,13 +193,15 @@ List Node
|
|
|
214
193
|
listId: "1",
|
|
215
194
|
listType: "ordered",
|
|
216
195
|
indentation: 0,
|
|
196
|
+
paragraphIndentation: { left: 720, hanging: 360 },
|
|
217
197
|
itemIndex: 0
|
|
218
198
|
}
|
|
219
199
|
└── children: [ Text Content... ]
|
|
220
200
|
```
|
|
221
201
|
|
|
222
202
|
- **`listId`**: A unique identifier for the list definition. Multiple items with the same `listId` belong to the same logical list.
|
|
223
|
-
- **`indentation`**: The nesting level (0-based).
|
|
203
|
+
- **`indentation`**: The structural nesting level (0-based).
|
|
204
|
+
- **`paragraphIndentation`**: The physical indentation formatting in twentieths of a point (twips) (e.g., `left`, `right`, `firstLine`, `hanging`).
|
|
224
205
|
- **`itemIndex`**: The sequential position within that list level.
|
|
225
206
|
- **`listType`**: Either `ordered` (numbered) or `unordered` (bulleted).
|
|
226
207
|
|
|
@@ -299,10 +280,38 @@ Formatting can be found at two levels:
|
|
|
299
280
|
1. **Node Level**: Applied directly to a text run or paragraph.
|
|
300
281
|
2. **Document Level**: Found in `ast.metadata.formatting` (defaults) or `ast.metadata.styleMap` (named styles).
|
|
301
282
|
|
|
302
|
-
### 6.
|
|
283
|
+
### 6. Breaks
|
|
284
|
+
Breaks are currently only supported when parsing DOCX-documents. Breaks are added as a node of type `break` and carry metadata of the type `BreakMetadata`. When `includeRawContent` is enabled, they also include the `rawContent` string from the original XML.
|
|
285
|
+
|
|
286
|
+
```text
|
|
287
|
+
Break Node
|
|
288
|
+
├── type: "break"
|
|
289
|
+
└── metadata: {
|
|
290
|
+
breakType: "textWrapping" | "page" | "column" | "lastRenderedPage" | "carriageReturn",
|
|
291
|
+
clear?: "all" | "left" | "none" | "right"
|
|
292
|
+
}
|
|
293
|
+
```
|
|
294
|
+
|
|
295
|
+
- `breakType`: Type of break. `textWrapping` (default) is a standard line break, `page` is a page break, `column` is a break to the next column, `lastRenderedPage` is a soft break inserted by Word, and `carriageReturn` is an explicit carriage return (`w:cr`).
|
|
296
|
+
- `clear`: Relevant for `textWrapping`. Indicates if text should wrap around floating objects.
|
|
297
|
+
|
|
298
|
+
> [!NOTE]
|
|
299
|
+
> Even though break nodes don't have a `text` property, the `ast.toText()` method will automatically convert them to newlines (`\n`) or the configured delimiter in the final string output.
|
|
300
|
+
|
|
301
|
+
### 7. Advanced Metadata
|
|
303
302
|
The `ast.metadata` object provides document-wide context:
|
|
304
303
|
- **`styleMap`**: A dictionary of style names to their `TextFormatting` definitions found in the document.
|
|
305
304
|
- **`formatting`**: Document-wide default settings (e.g., default font or font size).
|
|
305
|
+
- **`customProperties`**: A dictionary of user-defined metadata embedded in the document (OOXML `custom.xml`, ODF `meta:user-defined`, or PDF Info dictionary).
|
|
306
|
+
|
|
307
|
+
### 8. Custom Properties
|
|
308
|
+
You can access custom user-defined metadata that might be embedded in the document:
|
|
309
|
+
|
|
310
|
+
```javascript
|
|
311
|
+
const ast = await officeParser.parseOffice("contract.docx");
|
|
312
|
+
console.log("Custom Metadata:", ast.metadata.customProperties);
|
|
313
|
+
// Output: { "ProjectID": "ABC-123", "InternalReview": true }
|
|
314
|
+
```
|
|
306
315
|
|
|
307
316
|
### Advanced AST Usage
|
|
308
317
|
Beyond using `ast.toText()`, you can interact with the structural data directly:
|
|
@@ -403,10 +412,47 @@ Pass an optional config object as the second argument to `parseOffice`.
|
|
|
403
412
|
| `ignoreNotes` | boolean | `false` | Ignore notes in files like PowerPoint/ODP. |
|
|
404
413
|
| `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. (Note: Does not work for RTF. It is treated as true always.) |
|
|
405
414
|
| `extractAttachments` | boolean | `false` | Extract images and charts as Base64. |
|
|
406
|
-
| `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
|
|
407
|
-
| `ocrLanguage` | string | `eng` | Language for OCR (e.g., 'eng', 'fra'). Supports multiple languages with '+'. See [Language Codes](https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016). |
|
|
408
415
|
| `includeRawContent` | boolean | `false` | Include raw XML/RTF markup in the nodes. |
|
|
416
|
+
| `serializeRawContent` | boolean | `true` | When `includeRawContent` is true, re-serializes raw XML to clean strings. If false, extracts original raw substring. |
|
|
417
|
+
| `preserveXmlWhitespace` | boolean | `false` | When `serializeRawContent` is true, preserves original XML whitespace and line endings. |
|
|
418
|
+
| `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
|
|
419
|
+
| `ocrLanguage` | string | `eng` | **Deprecated**: Use `ocrConfig.language` instead. Language for OCR. |
|
|
409
420
|
| `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. Defaults to a CDN link if not provided. |
|
|
421
|
+
| `ocrConfig` | object | `{}` | **OCR Scheduler** configuration for fine-grained worker control. |
|
|
422
|
+
| `ocrConfig.language` | string | `eng` | Language(s) for OCR (e.g., 'eng', 'fra', 'eng+fra'). |
|
|
423
|
+
| `ocrConfig.autoTerminateTimeout` | number | `10000` | Inactivity timeout in milliseconds before workers are killed. |
|
|
424
|
+
| `ocrConfig.workerPath` | string | `undefined` | Path to Tesseract worker script (for offline use). |
|
|
425
|
+
| `ocrConfig.corePath` | string | `undefined` | Path to Tesseract core script (for offline use). |
|
|
426
|
+
| `ocrConfig.langPath` | string | `undefined` | Path for Tesseract language files (for offline use). |
|
|
427
|
+
| `includeBreakNodes` | boolean | `false` | Specifically targets Word documents (DOCX). When set to true, officeParser will also parse `w:br`, `w:cr` and `w:lastRenderedPageBreak` nodes.|
|
|
428
|
+
|
|
429
|
+
### OCR Scheduler & Resource Management
|
|
430
|
+
If your application uses OCR, `officeParser` utilizes an intelligent **Smart Worker Pool** to maintain a background worker pool and optimize repeated parse requests.
|
|
431
|
+
|
|
432
|
+
- **Dynamic Affinity**: Workers in the pool persist with their last used language affinity.
|
|
433
|
+
- **LRU Re-allocation**: If a new language is requested and the pool is full, the manager identifies the **Least Recently Used (LRU)** idle worker and re-initializes it for the new language. This avoids the overhead of destroying and recreating workers.
|
|
434
|
+
- **Auto-Termination**: Workers are automatically cleaned up after 10 seconds of inactivity (configurable via `ocrConfig.autoTerminateTimeout`).
|
|
435
|
+
|
|
436
|
+
#### `OfficeParser.terminateOcr()`
|
|
437
|
+
If you have used OCR (`{ ocr: true }`) in a short-lived script (like CLI tools or one-off automation), we recommend explicitly calling `terminateOcr()` after your processing is finished. This bypasses the 10-second idle timer and allows the process to return to the terminal prompt immediately.
|
|
438
|
+
|
|
439
|
+
> [!NOTE]
|
|
440
|
+
> If OCR was not used, this function is a no-op and does not need to be called.
|
|
441
|
+
|
|
442
|
+
```js
|
|
443
|
+
const officeParser = require('officeparser');
|
|
444
|
+
|
|
445
|
+
async function runCleaner() {
|
|
446
|
+
await officeParser.parseOffice("file.pdf", { ocr: true });
|
|
447
|
+
// ... process results ...
|
|
448
|
+
|
|
449
|
+
// Manually kill OCR workers for an immediate exit
|
|
450
|
+
await officeParser.terminateOcr();
|
|
451
|
+
}
|
|
452
|
+
```
|
|
453
|
+
|
|
454
|
+
> [!TIP]
|
|
455
|
+
> This is handled automatically in the built-in CLI (`npx officeparser ...`). You only need to call this manually if you are using the library in your own custom script and want a snappy exit.
|
|
410
456
|
|
|
411
457
|
```js
|
|
412
458
|
const config = {
|
|
@@ -449,29 +495,57 @@ officeParser.parseOffice("presentation.pptx", config).then(ast => {
|
|
|
449
495
|
```
|
|
450
496
|
|
|
451
497
|
## Browser Usage
|
|
452
|
-
The
|
|
498
|
+
The library provides two types of browser bundles in the `dist/` directory:
|
|
499
|
+
1. **`officeparser.browser.iife.js`**: Standard IIFE bundle for direct `<script>` tag usage. Exposes the global `officeParser` namespace.
|
|
500
|
+
2. **`officeparser.browser.mjs`**: Modern ESM bundle for use with `import` statements or modern bundlers.
|
|
501
|
+
|
|
502
|
+
### Usage (ESM)
|
|
503
|
+
If you are using a modern bundler like **Vite**, **Webpack**, or **Next.js**:
|
|
504
|
+
|
|
505
|
+
```javascript
|
|
506
|
+
import { OfficeParser } from 'officeparser';
|
|
507
|
+
|
|
508
|
+
const handleFile = async (event) => {
|
|
509
|
+
const file = event.target.files[0];
|
|
510
|
+
const buffer = await file.arrayBuffer();
|
|
511
|
+
|
|
512
|
+
try {
|
|
513
|
+
// Pass the Buffer or Uint8Array directly
|
|
514
|
+
const ast = await OfficeParser.parseOffice(new Uint8Array(buffer));
|
|
515
|
+
console.log(ast.toText());
|
|
516
|
+
} catch (err) {
|
|
517
|
+
console.error(err);
|
|
518
|
+
}
|
|
519
|
+
};
|
|
520
|
+
```
|
|
521
|
+
|
|
522
|
+
> [!NOTE]
|
|
523
|
+
> **Why `fs` fails in the browser**: Browsers do not have a built-in file system. If you try to pass a file path string in the browser, `officeParser` will throw a descriptive "Fail-Fast" error instead of crashing mysteriously:
|
|
524
|
+
> `[officeparser] Node.js 'fs' module is not available in the browser. Please pass a Buffer or Uint8Array instead.`
|
|
525
|
+
|
|
526
|
+
### Usage (Script Tag)
|
|
527
|
+
Include the IIFE bundle available in the release assets or your `dist/` folder. This exposes the global `officeParser` object.
|
|
453
528
|
|
|
454
529
|
```html
|
|
455
|
-
<script src="dist/officeparser.browser.js"></script>
|
|
530
|
+
<script src="dist/officeparser.browser.iife.js"></script>
|
|
456
531
|
<script>
|
|
457
|
-
async function handleFile(
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
// which contains the `OfficeParser` class.
|
|
532
|
+
async function handleFile(event) {
|
|
533
|
+
const file = event.target.files[0];
|
|
534
|
+
const buffer = await file.arrayBuffer();
|
|
461
535
|
|
|
462
536
|
try {
|
|
463
|
-
|
|
537
|
+
// Reconstruct as Uint8Array for the parser
|
|
538
|
+
const ast = await officeParser.parseOffice(new Uint8Array(buffer));
|
|
464
539
|
console.log(ast.toText());
|
|
465
|
-
console.log("Metadata:", ast.metadata);
|
|
466
540
|
} catch (error) {
|
|
467
|
-
console.error(error);
|
|
541
|
+
console.error("Parsing failed:", error);
|
|
468
542
|
}
|
|
469
543
|
}
|
|
470
544
|
</script>
|
|
471
545
|
```
|
|
472
546
|
|
|
473
547
|
### PDF Worker Configuration in Browser
|
|
474
|
-
When using `officeparser` in a browser environment to parse PDF files, you may provide the `pdfWorkerSrc` configuration option. If not provided, it defaults to a CDN link for `pdfjs-dist@5.
|
|
548
|
+
When using `officeparser` in a browser environment to parse PDF files, you may provide the `pdfWorkerSrc` configuration option. If not provided, it defaults to a CDN link for `pdfjs-dist@5.6.205`.
|
|
475
549
|
|
|
476
550
|
```javascript
|
|
477
551
|
const file = ...; // File object or ArrayBuffer
|
|
@@ -481,11 +555,21 @@ const ast = await officeParser.parseOffice(file);
|
|
|
481
555
|
|
|
482
556
|
// Or override it with your own path or a different version:
|
|
483
557
|
const ast2 = await officeParser.parseOffice(file, {
|
|
484
|
-
pdfWorkerSrc: "https://unpkg.com/pdfjs-dist@5.
|
|
558
|
+
pdfWorkerSrc: "https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
|
|
485
559
|
});
|
|
486
560
|
```
|
|
487
561
|
|
|
488
|
-
> **Note:** The version of `pdfjs-dist` in the worker source should match the version used by `officeparser` (currently `5.
|
|
562
|
+
> **Note:** The version of `pdfjs-dist` in the worker source should match the version used by `officeparser` (currently `5.6.205`).
|
|
563
|
+
|
|
564
|
+
## Troubleshooting & Common Issues
|
|
565
|
+
|
|
566
|
+
- **Node.js process stays alive after finishing**: If using OCR, the worker pool stays warm for 10s by default. Use `await terminateOcr()` at the end of your script for a snappy exit.
|
|
567
|
+
- **"Worker not found" in Browser**: Ensure `pdfWorkerSrc` is correctly pointed to the `pdf.worker.min.mjs` file matching version `5.6.205`.
|
|
568
|
+
- **OCR accuracy is low**: Verify your `ocrConfig.language` matches the document content. Note that OCR quality depends on image resolution.
|
|
569
|
+
- **Out of memory on large files**: For massive spreadsheets, consider using `ast.toText()` early and allowing the full AST object to be garbage-collected.
|
|
570
|
+
|
|
571
|
+
For a comprehensive guide, visit our [Debugging & Troubleshooting Documentation](https://harshankur.github.io/officeParser/#spec/debugging).
|
|
572
|
+
|
|
489
573
|
|
|
490
574
|
## Known Limitations
|
|
491
575
|
1. **ODT/ODS Charts**: Extraction may occasionally show inaccurate data when referencing external cell ranges or complex layout-based data.
|
|
@@ -500,7 +584,7 @@ const ast2 = await officeParser.parseOffice(file, {
|
|
|
500
584
|
|
|
501
585
|
## Contributing
|
|
502
586
|
|
|
503
|
-
|
|
587
|
+
Contributions are welcome! Please see [CONTRIBUTING.md](CONTRIBUTING.md) for details on how to get started.
|
|
504
588
|
|
|
505
589
|
## License
|
|
506
590
|
|
package/dist/OfficeParser.d.ts
CHANGED
|
@@ -32,7 +32,7 @@
|
|
|
32
32
|
*
|
|
33
33
|
* @module OfficeParser
|
|
34
34
|
*/
|
|
35
|
-
import { OfficeParserAST, OfficeParserConfig } from './types';
|
|
35
|
+
import { OfficeParserAST, OfficeParserConfig } from './types.js';
|
|
36
36
|
/**
|
|
37
37
|
* Main parser class providing office document parsing functionality.
|
|
38
38
|
*
|
|
@@ -86,4 +86,13 @@ export declare class OfficeParser {
|
|
|
86
86
|
* ```
|
|
87
87
|
*/
|
|
88
88
|
static parseOffice(file: string | Buffer | ArrayBuffer, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
89
|
+
/**
|
|
90
|
+
* Terminates all active OCR workers and cleans up resources.
|
|
91
|
+
*
|
|
92
|
+
* This should be called when the application is shutting down or when OCR
|
|
93
|
+
* is no longer needed to prevent memory leaks and orphaned worker processes.
|
|
94
|
+
*
|
|
95
|
+
* @returns A promise that resolves when all workers have been terminated
|
|
96
|
+
*/
|
|
97
|
+
static terminateOcr(): Promise<void>;
|
|
89
98
|
}
|
package/dist/OfficeParser.js
CHANGED
|
@@ -33,50 +33,18 @@
|
|
|
33
33
|
*
|
|
34
34
|
* @module OfficeParser
|
|
35
35
|
*/
|
|
36
|
-
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
37
|
-
if (k2 === undefined) k2 = k;
|
|
38
|
-
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
39
|
-
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
40
|
-
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
41
|
-
}
|
|
42
|
-
Object.defineProperty(o, k2, desc);
|
|
43
|
-
}) : (function(o, m, k, k2) {
|
|
44
|
-
if (k2 === undefined) k2 = k;
|
|
45
|
-
o[k2] = m[k];
|
|
46
|
-
}));
|
|
47
|
-
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
48
|
-
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
49
|
-
}) : function(o, v) {
|
|
50
|
-
o["default"] = v;
|
|
51
|
-
});
|
|
52
|
-
var __importStar = (this && this.__importStar) || (function () {
|
|
53
|
-
var ownKeys = function(o) {
|
|
54
|
-
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
55
|
-
var ar = [];
|
|
56
|
-
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
57
|
-
return ar;
|
|
58
|
-
};
|
|
59
|
-
return ownKeys(o);
|
|
60
|
-
};
|
|
61
|
-
return function (mod) {
|
|
62
|
-
if (mod && mod.__esModule) return mod;
|
|
63
|
-
var result = {};
|
|
64
|
-
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
65
|
-
__setModuleDefault(result, mod);
|
|
66
|
-
return result;
|
|
67
|
-
};
|
|
68
|
-
})();
|
|
69
36
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
70
37
|
exports.OfficeParser = void 0;
|
|
71
|
-
const
|
|
72
|
-
const
|
|
73
|
-
const
|
|
74
|
-
const
|
|
75
|
-
const
|
|
76
|
-
const
|
|
77
|
-
const
|
|
78
|
-
const
|
|
79
|
-
const
|
|
38
|
+
const envUtils_js_1 = require("./utils/envUtils.js");
|
|
39
|
+
const ExcelParser_js_1 = require("./parsers/ExcelParser.js");
|
|
40
|
+
const OpenOfficeParser_js_1 = require("./parsers/OpenOfficeParser.js");
|
|
41
|
+
const PdfParser_js_1 = require("./parsers/PdfParser.js");
|
|
42
|
+
const PowerPointParser_js_1 = require("./parsers/PowerPointParser.js");
|
|
43
|
+
const RtfParser_js_1 = require("./parsers/RtfParser.js");
|
|
44
|
+
const WordParser_js_1 = require("./parsers/WordParser.js");
|
|
45
|
+
const errorUtils_js_1 = require("./utils/errorUtils.js");
|
|
46
|
+
const moduleLoader_js_1 = require("./utils/moduleLoader.js");
|
|
47
|
+
const ocrUtils_js_1 = require("./utils/ocrUtils.js");
|
|
80
48
|
/**
|
|
81
49
|
* Main parser class providing office document parsing functionality.
|
|
82
50
|
*
|
|
@@ -148,7 +116,11 @@ class OfficeParser {
|
|
|
148
116
|
ocr: false,
|
|
149
117
|
ocrLanguage: 'eng',
|
|
150
118
|
includeRawContent: false,
|
|
119
|
+
serializeRawContent: true,
|
|
120
|
+
preserveXmlWhitespace: false,
|
|
151
121
|
pdfWorkerSrc: '',
|
|
122
|
+
ocrConfig: {},
|
|
123
|
+
includeBreakNodes: false,
|
|
152
124
|
...actualConfig
|
|
153
125
|
};
|
|
154
126
|
let buffer = Buffer.alloc(0);
|
|
@@ -156,7 +128,7 @@ class OfficeParser {
|
|
|
156
128
|
let filePath;
|
|
157
129
|
try {
|
|
158
130
|
if (!file) {
|
|
159
|
-
throw (0,
|
|
131
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
|
|
160
132
|
}
|
|
161
133
|
if (file instanceof ArrayBuffer) {
|
|
162
134
|
buffer = Buffer.from(file);
|
|
@@ -166,63 +138,79 @@ class OfficeParser {
|
|
|
166
138
|
}
|
|
167
139
|
else if (typeof file === 'string') {
|
|
168
140
|
filePath = file;
|
|
141
|
+
(0, envUtils_js_1.assertNode)('path-parsing');
|
|
142
|
+
// Safe to use dynamic import here as we've asserted we are in Node.
|
|
143
|
+
// Modern bundlers will still see this, but our browser builds
|
|
144
|
+
// shim 'fs' so it won't crash at build time.
|
|
145
|
+
const fs = await import('fs');
|
|
169
146
|
if (!fs.existsSync(file)) {
|
|
170
|
-
throw (0,
|
|
147
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
|
|
171
148
|
}
|
|
172
149
|
if (fs.lstatSync(file).isDirectory()) {
|
|
173
|
-
throw (0,
|
|
150
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
|
|
174
151
|
}
|
|
175
152
|
buffer = fs.readFileSync(file);
|
|
176
153
|
ext = file.split('.').pop()?.toLowerCase() || '';
|
|
177
154
|
}
|
|
178
155
|
else {
|
|
179
|
-
throw (0,
|
|
156
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
|
|
180
157
|
}
|
|
181
158
|
if (!ext) {
|
|
182
|
-
const { fileTypeFromBuffer } = await (0,
|
|
159
|
+
const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
|
|
183
160
|
const type = await fileTypeFromBuffer(buffer);
|
|
184
161
|
if (type) {
|
|
185
162
|
ext = type.ext.toLowerCase();
|
|
186
163
|
}
|
|
187
164
|
else {
|
|
188
|
-
throw (0,
|
|
165
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
|
|
189
166
|
}
|
|
190
167
|
}
|
|
191
168
|
let result;
|
|
192
169
|
switch (ext) {
|
|
193
170
|
case 'docx':
|
|
194
|
-
result = await (0,
|
|
171
|
+
result = await (0, WordParser_js_1.parseWord)(buffer, internalConfig);
|
|
195
172
|
break;
|
|
196
173
|
case 'pptx':
|
|
197
|
-
result = await (0,
|
|
174
|
+
result = await (0, PowerPointParser_js_1.parsePowerPoint)(buffer, internalConfig);
|
|
198
175
|
break;
|
|
199
176
|
case 'xlsx':
|
|
200
|
-
result = await (0,
|
|
177
|
+
result = await (0, ExcelParser_js_1.parseExcel)(buffer, internalConfig);
|
|
201
178
|
break;
|
|
202
179
|
case 'odt':
|
|
203
180
|
case 'odp':
|
|
204
181
|
case 'ods':
|
|
205
|
-
result = await (0,
|
|
182
|
+
result = await (0, OpenOfficeParser_js_1.parseOpenOffice)(buffer, internalConfig);
|
|
206
183
|
break;
|
|
207
184
|
case 'pdf':
|
|
208
|
-
result = await (0,
|
|
185
|
+
result = await (0, PdfParser_js_1.parsePdf)(buffer, internalConfig);
|
|
209
186
|
break;
|
|
210
187
|
case 'rtf':
|
|
211
|
-
result = await (0,
|
|
188
|
+
result = await (0, RtfParser_js_1.parseRtf)(buffer, internalConfig);
|
|
212
189
|
break;
|
|
213
190
|
default:
|
|
214
|
-
throw (0,
|
|
191
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
|
|
215
192
|
}
|
|
216
193
|
if (callback)
|
|
217
194
|
callback(result);
|
|
218
195
|
return result;
|
|
219
196
|
}
|
|
220
197
|
catch (error) {
|
|
221
|
-
const wrappedError = (0,
|
|
198
|
+
const wrappedError = (0, errorUtils_js_1.getWrappedError)(error, internalConfig, filePath);
|
|
222
199
|
if (callback)
|
|
223
200
|
callback(undefined, wrappedError);
|
|
224
201
|
throw wrappedError;
|
|
225
202
|
}
|
|
226
203
|
}
|
|
204
|
+
/**
|
|
205
|
+
* Terminates all active OCR workers and cleans up resources.
|
|
206
|
+
*
|
|
207
|
+
* This should be called when the application is shutting down or when OCR
|
|
208
|
+
* is no longer needed to prevent memory leaks and orphaned worker processes.
|
|
209
|
+
*
|
|
210
|
+
* @returns A promise that resolves when all workers have been terminated
|
|
211
|
+
*/
|
|
212
|
+
static async terminateOcr() {
|
|
213
|
+
await (0, ocrUtils_js_1.terminateOcr)();
|
|
214
|
+
}
|
|
227
215
|
}
|
|
228
216
|
exports.OfficeParser = OfficeParser;
|
package/dist/cli.d.ts
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* officeparser CLI
|
|
4
|
+
*
|
|
5
|
+
* Allows running officeparser from the command line:
|
|
6
|
+
* npx officeparser file.docx
|
|
7
|
+
* officeparser file.docx --toText=true
|
|
8
|
+
* officeparser file.docx --ocr=true --extractAttachments=true
|
|
9
|
+
*
|
|
10
|
+
* Options (--key=value):
|
|
11
|
+
* --toText=true Output plain text instead of JSON AST
|
|
12
|
+
* --ocr=true Enable OCR for images
|
|
13
|
+
* --ocrLanguage=eng OCR language (default: eng)
|
|
14
|
+
* --extractAttachments=true Extract embedded attachments
|
|
15
|
+
* --ignoreNotes=true Ignore footnotes/endnotes
|
|
16
|
+
* --putNotesAtLast=true Move notes to end of document
|
|
17
|
+
* --includeRawContent=true Include raw content in AST
|
|
18
|
+
* --outputErrorToConsole=true Log errors to console
|
|
19
|
+
*/
|
|
20
|
+
export {};
|
package/dist/cli.js
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
"use strict";
|
|
3
|
+
/**
|
|
4
|
+
* officeparser CLI
|
|
5
|
+
*
|
|
6
|
+
* Allows running officeparser from the command line:
|
|
7
|
+
* npx officeparser file.docx
|
|
8
|
+
* officeparser file.docx --toText=true
|
|
9
|
+
* officeparser file.docx --ocr=true --extractAttachments=true
|
|
10
|
+
*
|
|
11
|
+
* Options (--key=value):
|
|
12
|
+
* --toText=true Output plain text instead of JSON AST
|
|
13
|
+
* --ocr=true Enable OCR for images
|
|
14
|
+
* --ocrLanguage=eng OCR language (default: eng)
|
|
15
|
+
* --extractAttachments=true Extract embedded attachments
|
|
16
|
+
* --ignoreNotes=true Ignore footnotes/endnotes
|
|
17
|
+
* --putNotesAtLast=true Move notes to end of document
|
|
18
|
+
* --includeRawContent=true Include raw content in AST
|
|
19
|
+
* --outputErrorToConsole=true Log errors to console
|
|
20
|
+
*/
|
|
21
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
22
|
+
const OfficeParser_js_1 = require("./OfficeParser.js");
|
|
23
|
+
const args = process.argv.slice(2);
|
|
24
|
+
let fileArg;
|
|
25
|
+
let toText = false;
|
|
26
|
+
let verbose = false;
|
|
27
|
+
const configArgs = [];
|
|
28
|
+
function isConfigOption(arg) {
|
|
29
|
+
return arg.startsWith('--') && arg.includes('=');
|
|
30
|
+
}
|
|
31
|
+
args.forEach(arg => {
|
|
32
|
+
if (isConfigOption(arg)) {
|
|
33
|
+
configArgs.push(arg);
|
|
34
|
+
}
|
|
35
|
+
else if (!fileArg) {
|
|
36
|
+
fileArg = arg;
|
|
37
|
+
}
|
|
38
|
+
});
|
|
39
|
+
if (fileArg) {
|
|
40
|
+
const config = {};
|
|
41
|
+
configArgs.forEach(arg => {
|
|
42
|
+
const [key, value] = arg.split('=');
|
|
43
|
+
const cleanKey = key.replace('--', '');
|
|
44
|
+
const lowerValue = value.toLowerCase();
|
|
45
|
+
const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
|
|
46
|
+
if (cleanKey === 'toText') {
|
|
47
|
+
if (boolValue !== undefined)
|
|
48
|
+
toText = boolValue;
|
|
49
|
+
else
|
|
50
|
+
console.warn(`Invalid value for toText: ${value}`);
|
|
51
|
+
}
|
|
52
|
+
else if (cleanKey === 'verbose') {
|
|
53
|
+
if (boolValue !== undefined)
|
|
54
|
+
verbose = boolValue;
|
|
55
|
+
else
|
|
56
|
+
console.warn(`Invalid value for verbose: ${value}`);
|
|
57
|
+
}
|
|
58
|
+
else {
|
|
59
|
+
// @ts-ignore
|
|
60
|
+
if (boolValue !== undefined)
|
|
61
|
+
config[cleanKey] = boolValue;
|
|
62
|
+
// @ts-ignore
|
|
63
|
+
else
|
|
64
|
+
config[cleanKey] = value;
|
|
65
|
+
}
|
|
66
|
+
});
|
|
67
|
+
OfficeParser_js_1.OfficeParser.parseOffice(fileArg, config)
|
|
68
|
+
.then(async (ast) => {
|
|
69
|
+
if (toText) {
|
|
70
|
+
process.stdout.write(ast.toText() + '\n');
|
|
71
|
+
}
|
|
72
|
+
else {
|
|
73
|
+
process.stdout.write(JSON.stringify(ast, null, 2) + '\n');
|
|
74
|
+
}
|
|
75
|
+
// Ensure OCR workers are terminated for clean CLI exit
|
|
76
|
+
if (config.ocr) {
|
|
77
|
+
await OfficeParser_js_1.OfficeParser.terminateOcr();
|
|
78
|
+
}
|
|
79
|
+
})
|
|
80
|
+
.catch(async (err) => {
|
|
81
|
+
console.error(`Error parsing file "${fileArg}":`);
|
|
82
|
+
if (verbose) {
|
|
83
|
+
console.error(err);
|
|
84
|
+
}
|
|
85
|
+
else {
|
|
86
|
+
console.error(err.message || err);
|
|
87
|
+
console.error('Use --verbose=true for full stack trace.');
|
|
88
|
+
}
|
|
89
|
+
// Ensure OCR workers are terminated even on error
|
|
90
|
+
if (config.ocr) {
|
|
91
|
+
await OfficeParser_js_1.OfficeParser.terminateOcr();
|
|
92
|
+
}
|
|
93
|
+
process.exit(1);
|
|
94
|
+
});
|
|
95
|
+
}
|
|
96
|
+
else {
|
|
97
|
+
console.log('Usage: officeparser <file> [--option=value]');
|
|
98
|
+
console.log('');
|
|
99
|
+
console.log('Options:');
|
|
100
|
+
console.log(' --toText=true Output plain text instead of JSON AST');
|
|
101
|
+
console.log(' --ocr=true Enable OCR for images');
|
|
102
|
+
console.log(' --ocrLanguage=eng OCR language (default: eng)');
|
|
103
|
+
console.log(' --extractAttachments=true Extract embedded attachments');
|
|
104
|
+
console.log(' --ignoreNotes=true Ignore footnotes/endnotes');
|
|
105
|
+
console.log(' --putNotesAtLast=true Move notes to end of document');
|
|
106
|
+
console.log(' --includeRawContent=true Include raw content in AST');
|
|
107
|
+
console.log(' --serializeRawContent=true Serialize raw XML content (default: true)');
|
|
108
|
+
console.log(' --preserveXmlWhitespace=true Preserve whitespace in serialized XML (default: false)');
|
|
109
|
+
console.log(' --includeBreakNodes=false Include break nodes (DOCX only, default: false)');
|
|
110
|
+
console.log(' --verbose=true Show full error stack traces');
|
|
111
|
+
console.log('');
|
|
112
|
+
console.log('Examples:');
|
|
113
|
+
console.log(' officeparser document.docx');
|
|
114
|
+
console.log(' officeparser document.docx --toText=true');
|
|
115
|
+
console.log(' officeparser report.pdf --ocr=true --extractAttachments=true');
|
|
116
|
+
console.log(' officeparser complex.docx --serializeRawContent=false --includeRawContent=true');
|
|
117
|
+
}
|