officeparser 6.0.7 → 6.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +92 -13
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +43 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +116 -0
- package/dist/index.d.ts +3 -3
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +79 -1
- package/dist/officeparser.browser.iife.js +112 -0
- package/dist/officeparser.browser.mjs +111 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +71 -63
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +131 -114
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +85 -88
- package/dist/parsers/RtfParser.d.ts +1 -1
- package/dist/parsers/RtfParser.js +10 -6
- package/dist/parsers/WordParser.d.ts +1 -1
- package/dist/parsers/WordParser.js +109 -101
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +69 -1
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
package/README.md
CHANGED
|
@@ -3,6 +3,8 @@
|
|
|
3
3
|
A robust, strictly-typed Node.js and Browser library for parsing office files ([`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format)). It produces a clean, hierarchical Abstract Syntax Tree (AST) with rich metadata, text formatting, and full attachment support.
|
|
4
4
|
|
|
5
5
|
[](https://badge.fury.io/js/officeparser)
|
|
6
|
+
[](https://www.npmjs.com/package/officeparser)
|
|
7
|
+
[](https://www.npmjs.com/package/officeparser)
|
|
6
8
|
[](https://opensource.org/licenses/MIT)
|
|
7
9
|
|
|
8
10
|
---
|
|
@@ -22,6 +24,13 @@ A robust, strictly-typed Node.js and Browser library for parsing office files ([
|
|
|
22
24
|
|
|
23
25
|
|
|
24
26
|
#### Update
|
|
27
|
+
* 2026-04-14 - **v6.1.0 Release**: Major Infrastructure & Resource Stability. (Incremental since v6.0.0)
|
|
28
|
+
- **OCR Scheduler**: Intelligent worker pool that optimizes Tesseract lifecycle across parallel requests. **Note**: By default, Node.js processes stay active for 10s after OCR to keep workers warm (configurable via `ocrConfig.autoTerminateTimeout`); use `terminateOcr()` for immediate CLI/script exit.
|
|
29
|
+
- **Core Engine**: Replaced legacy zip extraction with `fflate` for significant performance gains and robust browser/edge compatibility.
|
|
30
|
+
- **Module System**: Full native ESM support with `Node16` resolution and verified browser bundles (Vite/Angular compatible).
|
|
31
|
+
- **Format Refinements**: Hierarchical PDF coordinate alignment and ODT/RTF list parsing stability.
|
|
32
|
+
- **Custom Properties**: Added support for extracting custom document metadata across OOXML, ODF, and PDF formats.
|
|
33
|
+
- **Sponsorship**: Integrated `funding.json` manifest and GitHub Sponsors support.
|
|
25
34
|
* 2025/12/29 - **v6.0.0 Release**: Major overhaul of the library. Transitioned from simple text extraction to a rich **Abstract Syntax Tree (AST)** output.
|
|
26
35
|
- Simplified API: Use `parseOffice` for all parsing needs (returns a Promise).
|
|
27
36
|
- Structured Output: Access hierarchical document structure (paragraphs, headings, tables, lists, etc.).
|
|
@@ -82,6 +91,7 @@ npx officeparser /path/to/officeFile.docx --ignoreNotes=true --newlineDelimiter=
|
|
|
82
91
|
- `--extractAttachments=[true|false]` Flag to extract images/charts as Base64. Default is false.
|
|
83
92
|
- `--ocr=[true|false]` Flag to enable OCR for extracted images. Default is false.
|
|
84
93
|
- `--includeRawContent=[true|false]` Flag to include raw XML/RTF content in nodes. Default is false.
|
|
94
|
+
- `--verbose=[true|false]` Show full error stack traces.
|
|
85
95
|
|
|
86
96
|
|
|
87
97
|
## Library Usage
|
|
@@ -156,7 +166,7 @@ The `OfficeParserAST` provides a format-agnostic representation of your document
|
|
|
156
166
|
```text
|
|
157
167
|
OfficeParserAST
|
|
158
168
|
├── type: "docx" | "pptx" | "xlsx" | ...
|
|
159
|
-
├── metadata: { author, title, created, modified,
|
|
169
|
+
├── metadata: { author, title, created, modified, ..., customProperties }
|
|
160
170
|
├── content: [ OfficeContentNode ]
|
|
161
171
|
│ ├── type: "paragraph" | "heading" | "table" | "list" | ...
|
|
162
172
|
│ ├── text: "Concatenated text of this node and all children"
|
|
@@ -177,7 +187,7 @@ OfficeParserAST
|
|
|
177
187
|
```json
|
|
178
188
|
{
|
|
179
189
|
"type": "docx",
|
|
180
|
-
"metadata": { "author": "John Doe", "title": "Annual Report" },
|
|
190
|
+
"metadata": { "author": "John Doe", "title": "Annual Report", "customProperties": { "Department": "Finance" } },
|
|
181
191
|
"content": [
|
|
182
192
|
{
|
|
183
193
|
"type": "heading",
|
|
@@ -303,6 +313,16 @@ Formatting can be found at two levels:
|
|
|
303
313
|
The `ast.metadata` object provides document-wide context:
|
|
304
314
|
- **`styleMap`**: A dictionary of style names to their `TextFormatting` definitions found in the document.
|
|
305
315
|
- **`formatting`**: Document-wide default settings (e.g., default font or font size).
|
|
316
|
+
- **`customProperties`**: A dictionary of user-defined metadata embedded in the document (OOXML `custom.xml`, ODF `meta:user-defined`, or PDF Info dictionary).
|
|
317
|
+
|
|
318
|
+
### 7. Custom Properties
|
|
319
|
+
You can access custom user-defined metadata that might be embedded in the document:
|
|
320
|
+
|
|
321
|
+
```javascript
|
|
322
|
+
const ast = await officeParser.parseOffice("contract.docx");
|
|
323
|
+
console.log("Custom Metadata:", ast.metadata.customProperties);
|
|
324
|
+
// Output: { "ProjectID": "ABC-123", "InternalReview": true }
|
|
325
|
+
```
|
|
306
326
|
|
|
307
327
|
### Advanced AST Usage
|
|
308
328
|
Beyond using `ast.toText()`, you can interact with the structural data directly:
|
|
@@ -403,10 +423,46 @@ Pass an optional config object as the second argument to `parseOffice`.
|
|
|
403
423
|
| `ignoreNotes` | boolean | `false` | Ignore notes in files like PowerPoint/ODP. |
|
|
404
424
|
| `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. (Note: Does not work for RTF. It is treated as true always.) |
|
|
405
425
|
| `extractAttachments` | boolean | `false` | Extract images and charts as Base64. |
|
|
406
|
-
| `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
|
|
407
|
-
| `ocrLanguage` | string | `eng` | Language for OCR (e.g., 'eng', 'fra'). Supports multiple languages with '+'. See [Language Codes](https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016). |
|
|
408
426
|
| `includeRawContent` | boolean | `false` | Include raw XML/RTF markup in the nodes. |
|
|
427
|
+
| `serializeRawContent` | boolean | `true` | When `includeRawContent` is true, re-serializes raw XML to clean strings. If false, extracts original raw substring. |
|
|
428
|
+
| `preserveXmlWhitespace` | boolean | `false` | When `serializeRawContent` is true, preserves original XML whitespace and line endings. |
|
|
429
|
+
| `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
|
|
430
|
+
| `ocrLanguage` | string | `eng` | **Deprecated**: Use `ocrConfig.language` instead. Language for OCR. |
|
|
409
431
|
| `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. Defaults to a CDN link if not provided. |
|
|
432
|
+
| `ocrConfig` | object | `{}` | **OCR Scheduler** configuration for fine-grained worker control. |
|
|
433
|
+
| `ocrConfig.language` | string | `eng` | Language(s) for OCR (e.g., 'eng', 'fra', 'eng+fra'). |
|
|
434
|
+
| `ocrConfig.autoTerminateTimeout` | number | `10000` | Inactivity timeout in milliseconds before workers are killed. |
|
|
435
|
+
| `ocrConfig.workerPath` | string | `undefined` | Path to Tesseract worker script (for offline use). |
|
|
436
|
+
| `ocrConfig.corePath` | string | `undefined` | Path to Tesseract core script (for offline use). |
|
|
437
|
+
| `ocrConfig.langPath` | string | `undefined` | Path for Tesseract language files (for offline use). |
|
|
438
|
+
|
|
439
|
+
### OCR Scheduler & Resource Management
|
|
440
|
+
If your application uses OCR, `officeParser` utilizes an intelligent **Smart Worker Pool** to maintain a background worker pool and optimize repeated parse requests.
|
|
441
|
+
|
|
442
|
+
- **Dynamic Affinity**: Workers in the pool persist with their last used language affinity.
|
|
443
|
+
- **Smart Re-initialization**: If a new language is requested and the pool is full, the manager identifies the **Least Recently Used (LRU)** idle worker and re-initializes it for the new language using the Tesseract.js v5 API. This avoids the overhead of destroying and recreating workers.
|
|
444
|
+
- **Auto-Termination**: Workers are automatically cleaned up after 10 seconds of inactivity (configurable via `ocrConfig.autoTerminateTimeout`).
|
|
445
|
+
|
|
446
|
+
#### `OfficeParser.terminateOcr()`
|
|
447
|
+
If you have used OCR (`{ ocr: true }`) in a short-lived script (like CLI tools or one-off automation), we recommend explicitly calling `terminateOcr()` after your processing is finished. This bypasses the 10-second idle timer and allows the process to return to the terminal prompt immediately.
|
|
448
|
+
|
|
449
|
+
> [!NOTE]
|
|
450
|
+
> If OCR was not used, this function is a no-op and does not need to be called.
|
|
451
|
+
|
|
452
|
+
```js
|
|
453
|
+
const officeParser = require('officeparser');
|
|
454
|
+
|
|
455
|
+
async function runCleaner() {
|
|
456
|
+
await officeParser.parseOffice("file.pdf", { ocr: true });
|
|
457
|
+
// ... process results ...
|
|
458
|
+
|
|
459
|
+
// Manually kill OCR workers for an immediate exit
|
|
460
|
+
await officeParser.terminateOcr();
|
|
461
|
+
}
|
|
462
|
+
```
|
|
463
|
+
|
|
464
|
+
> [!TIP]
|
|
465
|
+
> This is handled automatically in our own CLI (`npx officeparser ...`). You only need to call this manually if you are using the library in your own custom script and want a snappy exit.
|
|
410
466
|
|
|
411
467
|
```js
|
|
412
468
|
const config = {
|
|
@@ -449,20 +505,21 @@ officeParser.parseOffice("presentation.pptx", config).then(ast => {
|
|
|
449
505
|
```
|
|
450
506
|
|
|
451
507
|
## Browser Usage
|
|
452
|
-
The
|
|
508
|
+
The library provides two types of browser bundles in the `dist/` directory:
|
|
509
|
+
1. **`officeparser.browser.iife.js`**: Standard IIFE bundle for direct `<script>` tag usage. Exposes the global `officeParser` namespace.
|
|
510
|
+
2. **`officeparser.browser.mjs`**: Modern ESM bundle for use with `import` statements or modern bundlers.
|
|
511
|
+
|
|
512
|
+
### Usage (Script Tag)
|
|
513
|
+
Include the IIFE bundle file available in the release assets.
|
|
453
514
|
|
|
454
515
|
```html
|
|
455
|
-
<script src="dist/officeparser.browser.js"></script>
|
|
516
|
+
<script src="dist/officeparser.browser.iife.js"></script>
|
|
456
517
|
<script>
|
|
457
518
|
async function handleFile(file) {
|
|
458
519
|
// file can be a File object from an input element or an ArrayBuffer
|
|
459
|
-
// The browser bundle exposes the global variable `officeParser`
|
|
460
|
-
// which contains the `OfficeParser` class.
|
|
461
|
-
|
|
462
520
|
try {
|
|
463
521
|
const ast = await officeParser.parseOffice(file, { ocr: true });
|
|
464
522
|
console.log(ast.toText());
|
|
465
|
-
console.log("Metadata:", ast.metadata);
|
|
466
523
|
} catch (error) {
|
|
467
524
|
console.error(error);
|
|
468
525
|
}
|
|
@@ -470,8 +527,20 @@ The browser bundle exposes the `officeParser` namespace. Include the bundle file
|
|
|
470
527
|
</script>
|
|
471
528
|
```
|
|
472
529
|
|
|
530
|
+
### Usage (ESM)
|
|
531
|
+
If you are using a modern browser that supports modules or a dev server like Vite:
|
|
532
|
+
|
|
533
|
+
```html
|
|
534
|
+
<script type="module">
|
|
535
|
+
import { OfficeParser } from './dist/officeparser.browser.mjs';
|
|
536
|
+
|
|
537
|
+
const ast = await OfficeParser.parseOffice(fileBuffer);
|
|
538
|
+
console.log(ast.metadata);
|
|
539
|
+
</script>
|
|
540
|
+
```
|
|
541
|
+
|
|
473
542
|
### PDF Worker Configuration in Browser
|
|
474
|
-
When using `officeparser` in a browser environment to parse PDF files, you may provide the `pdfWorkerSrc` configuration option. If not provided, it defaults to a CDN link for `pdfjs-dist@5.
|
|
543
|
+
When using `officeparser` in a browser environment to parse PDF files, you may provide the `pdfWorkerSrc` configuration option. If not provided, it defaults to a CDN link for `pdfjs-dist@5.6.205`.
|
|
475
544
|
|
|
476
545
|
```javascript
|
|
477
546
|
const file = ...; // File object or ArrayBuffer
|
|
@@ -481,11 +550,21 @@ const ast = await officeParser.parseOffice(file);
|
|
|
481
550
|
|
|
482
551
|
// Or override it with your own path or a different version:
|
|
483
552
|
const ast2 = await officeParser.parseOffice(file, {
|
|
484
|
-
pdfWorkerSrc: "https://unpkg.com/pdfjs-dist@5.
|
|
553
|
+
pdfWorkerSrc: "https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
|
|
485
554
|
});
|
|
486
555
|
```
|
|
487
556
|
|
|
488
|
-
> **Note:** The version of `pdfjs-dist` in the worker source should match the version used by `officeparser` (currently `5.
|
|
557
|
+
> **Note:** The version of `pdfjs-dist` in the worker source should match the version used by `officeparser` (currently `5.6.205`).
|
|
558
|
+
|
|
559
|
+
## Troubleshooting & Common Issues
|
|
560
|
+
|
|
561
|
+
- **Node.js process stays alive after finishing**: If using OCR, the worker pool stays warm for 10s by default. Use `await terminateOcr()` at the end of your script for a snappy exit.
|
|
562
|
+
- **"Worker not found" in Browser**: Ensure `pdfWorkerSrc` is correctly pointed to the `pdf.worker.min.mjs` file matching version `5.6.205`.
|
|
563
|
+
- **OCR accuracy is low**: Verify your `ocrConfig.language` matches the document content. Note that OCR quality depends on image resolution.
|
|
564
|
+
- **Out of memory on large files**: For massive spreadsheets, consider using `ast.toText()` early and allowing the full AST object to be garbage-collected.
|
|
565
|
+
|
|
566
|
+
For a comprehensive guide, visit our [Debugging & Troubleshooting Documentation](https://harshankur.github.io/officeParser/#spec/debugging).
|
|
567
|
+
|
|
489
568
|
|
|
490
569
|
## Known Limitations
|
|
491
570
|
1. **ODT/ODS Charts**: Extraction may occasionally show inaccurate data when referencing external cell ranges or complex layout-based data.
|
package/dist/OfficeParser.d.ts
CHANGED
|
@@ -32,7 +32,7 @@
|
|
|
32
32
|
*
|
|
33
33
|
* @module OfficeParser
|
|
34
34
|
*/
|
|
35
|
-
import { OfficeParserAST, OfficeParserConfig } from './types';
|
|
35
|
+
import { OfficeParserAST, OfficeParserConfig } from './types.js';
|
|
36
36
|
/**
|
|
37
37
|
* Main parser class providing office document parsing functionality.
|
|
38
38
|
*
|
|
@@ -86,4 +86,13 @@ export declare class OfficeParser {
|
|
|
86
86
|
* ```
|
|
87
87
|
*/
|
|
88
88
|
static parseOffice(file: string | Buffer | ArrayBuffer, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
89
|
+
/**
|
|
90
|
+
* Terminates all active OCR workers and cleans up resources.
|
|
91
|
+
*
|
|
92
|
+
* This should be called when the application is shutting down or when OCR
|
|
93
|
+
* is no longer needed to prevent memory leaks and orphaned worker processes.
|
|
94
|
+
*
|
|
95
|
+
* @returns A promise that resolves when all workers have been terminated
|
|
96
|
+
*/
|
|
97
|
+
static terminateOcr(): Promise<void>;
|
|
89
98
|
}
|
package/dist/OfficeParser.js
CHANGED
|
@@ -33,50 +33,18 @@
|
|
|
33
33
|
*
|
|
34
34
|
* @module OfficeParser
|
|
35
35
|
*/
|
|
36
|
-
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
37
|
-
if (k2 === undefined) k2 = k;
|
|
38
|
-
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
39
|
-
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
40
|
-
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
41
|
-
}
|
|
42
|
-
Object.defineProperty(o, k2, desc);
|
|
43
|
-
}) : (function(o, m, k, k2) {
|
|
44
|
-
if (k2 === undefined) k2 = k;
|
|
45
|
-
o[k2] = m[k];
|
|
46
|
-
}));
|
|
47
|
-
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
48
|
-
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
49
|
-
}) : function(o, v) {
|
|
50
|
-
o["default"] = v;
|
|
51
|
-
});
|
|
52
|
-
var __importStar = (this && this.__importStar) || (function () {
|
|
53
|
-
var ownKeys = function(o) {
|
|
54
|
-
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
55
|
-
var ar = [];
|
|
56
|
-
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
57
|
-
return ar;
|
|
58
|
-
};
|
|
59
|
-
return ownKeys(o);
|
|
60
|
-
};
|
|
61
|
-
return function (mod) {
|
|
62
|
-
if (mod && mod.__esModule) return mod;
|
|
63
|
-
var result = {};
|
|
64
|
-
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
65
|
-
__setModuleDefault(result, mod);
|
|
66
|
-
return result;
|
|
67
|
-
};
|
|
68
|
-
})();
|
|
69
36
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
70
37
|
exports.OfficeParser = void 0;
|
|
71
|
-
const
|
|
72
|
-
const
|
|
73
|
-
const
|
|
74
|
-
const
|
|
75
|
-
const
|
|
76
|
-
const
|
|
77
|
-
const
|
|
78
|
-
const
|
|
79
|
-
const
|
|
38
|
+
const envUtils_js_1 = require("./utils/envUtils.js");
|
|
39
|
+
const ExcelParser_js_1 = require("./parsers/ExcelParser.js");
|
|
40
|
+
const OpenOfficeParser_js_1 = require("./parsers/OpenOfficeParser.js");
|
|
41
|
+
const PdfParser_js_1 = require("./parsers/PdfParser.js");
|
|
42
|
+
const PowerPointParser_js_1 = require("./parsers/PowerPointParser.js");
|
|
43
|
+
const RtfParser_js_1 = require("./parsers/RtfParser.js");
|
|
44
|
+
const WordParser_js_1 = require("./parsers/WordParser.js");
|
|
45
|
+
const errorUtils_js_1 = require("./utils/errorUtils.js");
|
|
46
|
+
const moduleLoader_js_1 = require("./utils/moduleLoader.js");
|
|
47
|
+
const ocrUtils_js_1 = require("./utils/ocrUtils.js");
|
|
80
48
|
/**
|
|
81
49
|
* Main parser class providing office document parsing functionality.
|
|
82
50
|
*
|
|
@@ -148,7 +116,10 @@ class OfficeParser {
|
|
|
148
116
|
ocr: false,
|
|
149
117
|
ocrLanguage: 'eng',
|
|
150
118
|
includeRawContent: false,
|
|
119
|
+
serializeRawContent: true,
|
|
120
|
+
preserveXmlWhitespace: false,
|
|
151
121
|
pdfWorkerSrc: '',
|
|
122
|
+
ocrConfig: {},
|
|
152
123
|
...actualConfig
|
|
153
124
|
};
|
|
154
125
|
let buffer = Buffer.alloc(0);
|
|
@@ -156,7 +127,7 @@ class OfficeParser {
|
|
|
156
127
|
let filePath;
|
|
157
128
|
try {
|
|
158
129
|
if (!file) {
|
|
159
|
-
throw (0,
|
|
130
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
|
|
160
131
|
}
|
|
161
132
|
if (file instanceof ArrayBuffer) {
|
|
162
133
|
buffer = Buffer.from(file);
|
|
@@ -166,63 +137,79 @@ class OfficeParser {
|
|
|
166
137
|
}
|
|
167
138
|
else if (typeof file === 'string') {
|
|
168
139
|
filePath = file;
|
|
140
|
+
(0, envUtils_js_1.assertNode)('path-parsing');
|
|
141
|
+
// Safe to use dynamic import here as we've asserted we are in Node.
|
|
142
|
+
// Modern bundlers will still see this, but our browser builds
|
|
143
|
+
// shim 'fs' so it won't crash at build time.
|
|
144
|
+
const fs = await import('fs');
|
|
169
145
|
if (!fs.existsSync(file)) {
|
|
170
|
-
throw (0,
|
|
146
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
|
|
171
147
|
}
|
|
172
148
|
if (fs.lstatSync(file).isDirectory()) {
|
|
173
|
-
throw (0,
|
|
149
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
|
|
174
150
|
}
|
|
175
151
|
buffer = fs.readFileSync(file);
|
|
176
152
|
ext = file.split('.').pop()?.toLowerCase() || '';
|
|
177
153
|
}
|
|
178
154
|
else {
|
|
179
|
-
throw (0,
|
|
155
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
|
|
180
156
|
}
|
|
181
157
|
if (!ext) {
|
|
182
|
-
const { fileTypeFromBuffer } = await (0,
|
|
158
|
+
const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
|
|
183
159
|
const type = await fileTypeFromBuffer(buffer);
|
|
184
160
|
if (type) {
|
|
185
161
|
ext = type.ext.toLowerCase();
|
|
186
162
|
}
|
|
187
163
|
else {
|
|
188
|
-
throw (0,
|
|
164
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
|
|
189
165
|
}
|
|
190
166
|
}
|
|
191
167
|
let result;
|
|
192
168
|
switch (ext) {
|
|
193
169
|
case 'docx':
|
|
194
|
-
result = await (0,
|
|
170
|
+
result = await (0, WordParser_js_1.parseWord)(buffer, internalConfig);
|
|
195
171
|
break;
|
|
196
172
|
case 'pptx':
|
|
197
|
-
result = await (0,
|
|
173
|
+
result = await (0, PowerPointParser_js_1.parsePowerPoint)(buffer, internalConfig);
|
|
198
174
|
break;
|
|
199
175
|
case 'xlsx':
|
|
200
|
-
result = await (0,
|
|
176
|
+
result = await (0, ExcelParser_js_1.parseExcel)(buffer, internalConfig);
|
|
201
177
|
break;
|
|
202
178
|
case 'odt':
|
|
203
179
|
case 'odp':
|
|
204
180
|
case 'ods':
|
|
205
|
-
result = await (0,
|
|
181
|
+
result = await (0, OpenOfficeParser_js_1.parseOpenOffice)(buffer, internalConfig);
|
|
206
182
|
break;
|
|
207
183
|
case 'pdf':
|
|
208
|
-
result = await (0,
|
|
184
|
+
result = await (0, PdfParser_js_1.parsePdf)(buffer, internalConfig);
|
|
209
185
|
break;
|
|
210
186
|
case 'rtf':
|
|
211
|
-
result = await (0,
|
|
187
|
+
result = await (0, RtfParser_js_1.parseRtf)(buffer, internalConfig);
|
|
212
188
|
break;
|
|
213
189
|
default:
|
|
214
|
-
throw (0,
|
|
190
|
+
throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
|
|
215
191
|
}
|
|
216
192
|
if (callback)
|
|
217
193
|
callback(result);
|
|
218
194
|
return result;
|
|
219
195
|
}
|
|
220
196
|
catch (error) {
|
|
221
|
-
const wrappedError = (0,
|
|
197
|
+
const wrappedError = (0, errorUtils_js_1.getWrappedError)(error, internalConfig, filePath);
|
|
222
198
|
if (callback)
|
|
223
199
|
callback(undefined, wrappedError);
|
|
224
200
|
throw wrappedError;
|
|
225
201
|
}
|
|
226
202
|
}
|
|
203
|
+
/**
|
|
204
|
+
* Terminates all active OCR workers and cleans up resources.
|
|
205
|
+
*
|
|
206
|
+
* This should be called when the application is shutting down or when OCR
|
|
207
|
+
* is no longer needed to prevent memory leaks and orphaned worker processes.
|
|
208
|
+
*
|
|
209
|
+
* @returns A promise that resolves when all workers have been terminated
|
|
210
|
+
*/
|
|
211
|
+
static async terminateOcr() {
|
|
212
|
+
await (0, ocrUtils_js_1.terminateOcr)();
|
|
213
|
+
}
|
|
227
214
|
}
|
|
228
215
|
exports.OfficeParser = OfficeParser;
|
package/dist/cli.d.ts
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* officeparser CLI
|
|
4
|
+
*
|
|
5
|
+
* Allows running officeparser from the command line:
|
|
6
|
+
* npx officeparser file.docx
|
|
7
|
+
* officeparser file.docx --toText=true
|
|
8
|
+
* officeparser file.docx --ocr=true --extractAttachments=true
|
|
9
|
+
*
|
|
10
|
+
* Options (--key=value):
|
|
11
|
+
* --toText=true Output plain text instead of JSON AST
|
|
12
|
+
* --ocr=true Enable OCR for images
|
|
13
|
+
* --ocrLanguage=eng OCR language (default: eng)
|
|
14
|
+
* --extractAttachments=true Extract embedded attachments
|
|
15
|
+
* --ignoreNotes=true Ignore footnotes/endnotes
|
|
16
|
+
* --putNotesAtLast=true Move notes to end of document
|
|
17
|
+
* --includeRawContent=true Include raw content in AST
|
|
18
|
+
* --outputErrorToConsole=true Log errors to console
|
|
19
|
+
*/
|
|
20
|
+
export {};
|
package/dist/cli.js
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
"use strict";
|
|
3
|
+
/**
|
|
4
|
+
* officeparser CLI
|
|
5
|
+
*
|
|
6
|
+
* Allows running officeparser from the command line:
|
|
7
|
+
* npx officeparser file.docx
|
|
8
|
+
* officeparser file.docx --toText=true
|
|
9
|
+
* officeparser file.docx --ocr=true --extractAttachments=true
|
|
10
|
+
*
|
|
11
|
+
* Options (--key=value):
|
|
12
|
+
* --toText=true Output plain text instead of JSON AST
|
|
13
|
+
* --ocr=true Enable OCR for images
|
|
14
|
+
* --ocrLanguage=eng OCR language (default: eng)
|
|
15
|
+
* --extractAttachments=true Extract embedded attachments
|
|
16
|
+
* --ignoreNotes=true Ignore footnotes/endnotes
|
|
17
|
+
* --putNotesAtLast=true Move notes to end of document
|
|
18
|
+
* --includeRawContent=true Include raw content in AST
|
|
19
|
+
* --outputErrorToConsole=true Log errors to console
|
|
20
|
+
*/
|
|
21
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
22
|
+
const OfficeParser_js_1 = require("./OfficeParser.js");
|
|
23
|
+
const args = process.argv.slice(2);
|
|
24
|
+
let fileArg;
|
|
25
|
+
let toText = false;
|
|
26
|
+
let verbose = false;
|
|
27
|
+
const configArgs = [];
|
|
28
|
+
function isConfigOption(arg) {
|
|
29
|
+
return arg.startsWith('--') && arg.includes('=');
|
|
30
|
+
}
|
|
31
|
+
args.forEach(arg => {
|
|
32
|
+
if (isConfigOption(arg)) {
|
|
33
|
+
configArgs.push(arg);
|
|
34
|
+
}
|
|
35
|
+
else if (!fileArg) {
|
|
36
|
+
fileArg = arg;
|
|
37
|
+
}
|
|
38
|
+
});
|
|
39
|
+
if (fileArg) {
|
|
40
|
+
const config = {};
|
|
41
|
+
configArgs.forEach(arg => {
|
|
42
|
+
const [key, value] = arg.split('=');
|
|
43
|
+
const cleanKey = key.replace('--', '');
|
|
44
|
+
const lowerValue = value.toLowerCase();
|
|
45
|
+
const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
|
|
46
|
+
if (cleanKey === 'toText') {
|
|
47
|
+
if (boolValue !== undefined)
|
|
48
|
+
toText = boolValue;
|
|
49
|
+
else
|
|
50
|
+
console.warn(`Invalid value for toText: ${value}`);
|
|
51
|
+
}
|
|
52
|
+
else if (cleanKey === 'verbose') {
|
|
53
|
+
if (boolValue !== undefined)
|
|
54
|
+
verbose = boolValue;
|
|
55
|
+
else
|
|
56
|
+
console.warn(`Invalid value for verbose: ${value}`);
|
|
57
|
+
}
|
|
58
|
+
else {
|
|
59
|
+
// @ts-ignore
|
|
60
|
+
if (boolValue !== undefined)
|
|
61
|
+
config[cleanKey] = boolValue;
|
|
62
|
+
// @ts-ignore
|
|
63
|
+
else
|
|
64
|
+
config[cleanKey] = value;
|
|
65
|
+
}
|
|
66
|
+
});
|
|
67
|
+
OfficeParser_js_1.OfficeParser.parseOffice(fileArg, config)
|
|
68
|
+
.then(async (ast) => {
|
|
69
|
+
if (toText) {
|
|
70
|
+
process.stdout.write(ast.toText() + '\n');
|
|
71
|
+
}
|
|
72
|
+
else {
|
|
73
|
+
process.stdout.write(JSON.stringify(ast, null, 2) + '\n');
|
|
74
|
+
}
|
|
75
|
+
// Ensure OCR workers are terminated for clean CLI exit
|
|
76
|
+
if (config.ocr) {
|
|
77
|
+
await OfficeParser_js_1.OfficeParser.terminateOcr();
|
|
78
|
+
}
|
|
79
|
+
})
|
|
80
|
+
.catch(async (err) => {
|
|
81
|
+
console.error(`Error parsing file "${fileArg}":`);
|
|
82
|
+
if (verbose) {
|
|
83
|
+
console.error(err);
|
|
84
|
+
}
|
|
85
|
+
else {
|
|
86
|
+
console.error(err.message || err);
|
|
87
|
+
console.error('Use --verbose=true for full stack trace.');
|
|
88
|
+
}
|
|
89
|
+
// Ensure OCR workers are terminated even on error
|
|
90
|
+
if (config.ocr) {
|
|
91
|
+
await OfficeParser_js_1.OfficeParser.terminateOcr();
|
|
92
|
+
}
|
|
93
|
+
process.exit(1);
|
|
94
|
+
});
|
|
95
|
+
}
|
|
96
|
+
else {
|
|
97
|
+
console.log('Usage: officeparser <file> [--option=value]');
|
|
98
|
+
console.log('');
|
|
99
|
+
console.log('Options:');
|
|
100
|
+
console.log(' --toText=true Output plain text instead of JSON AST');
|
|
101
|
+
console.log(' --ocr=true Enable OCR for images');
|
|
102
|
+
console.log(' --ocrLanguage=eng OCR language (default: eng)');
|
|
103
|
+
console.log(' --extractAttachments=true Extract embedded attachments');
|
|
104
|
+
console.log(' --ignoreNotes=true Ignore footnotes/endnotes');
|
|
105
|
+
console.log(' --putNotesAtLast=true Move notes to end of document');
|
|
106
|
+
console.log(' --includeRawContent=true Include raw content in AST');
|
|
107
|
+
console.log(' --serializeRawContent=true Serialize raw XML content (default: true)');
|
|
108
|
+
console.log(' --preserveXmlWhitespace=true Preserve whitespace in serialized XML (default: false)');
|
|
109
|
+
console.log(' --verbose=true Show full error stack traces');
|
|
110
|
+
console.log('');
|
|
111
|
+
console.log('Examples:');
|
|
112
|
+
console.log(' officeparser document.docx');
|
|
113
|
+
console.log(' officeparser document.docx --toText=true');
|
|
114
|
+
console.log(' officeparser report.pdf --ocr=true --extractAttachments=true');
|
|
115
|
+
console.log(' officeparser complex.docx --serializeRawContent=false --includeRawContent=true');
|
|
116
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -1,4 +1,3 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
1
|
/**
|
|
3
2
|
* officeparser - Universal Office Document Parser
|
|
4
3
|
*
|
|
@@ -44,8 +43,9 @@
|
|
|
44
43
|
* @packageDocumentation
|
|
45
44
|
* @module officeparser
|
|
46
45
|
*/
|
|
47
|
-
import { OfficeParser } from './OfficeParser';
|
|
46
|
+
import { OfficeParser } from './OfficeParser.js';
|
|
48
47
|
import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata } from './types';
|
|
49
48
|
declare const parseOffice: typeof OfficeParser.parseOffice;
|
|
50
|
-
|
|
49
|
+
declare const terminateOcr: typeof OfficeParser.terminateOcr;
|
|
50
|
+
export { OfficeParser, parseOffice, terminateOcr, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata };
|
|
51
51
|
export default OfficeParser;
|
package/dist/index.js
CHANGED
|
@@ -1,4 +1,3 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
1
|
"use strict";
|
|
3
2
|
/**
|
|
4
3
|
* officeparser - Universal Office Document Parser
|
|
@@ -46,63 +45,12 @@
|
|
|
46
45
|
* @module officeparser
|
|
47
46
|
*/
|
|
48
47
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
49
|
-
exports.parseOffice = exports.OfficeParser = void 0;
|
|
50
|
-
const
|
|
51
|
-
Object.defineProperty(exports, "OfficeParser", { enumerable: true, get: function () { return
|
|
52
|
-
const parseOffice =
|
|
48
|
+
exports.terminateOcr = exports.parseOffice = exports.OfficeParser = void 0;
|
|
49
|
+
const OfficeParser_js_1 = require("./OfficeParser.js");
|
|
50
|
+
Object.defineProperty(exports, "OfficeParser", { enumerable: true, get: function () { return OfficeParser_js_1.OfficeParser; } });
|
|
51
|
+
const parseOffice = OfficeParser_js_1.OfficeParser.parseOffice;
|
|
53
52
|
exports.parseOffice = parseOffice;
|
|
53
|
+
const terminateOcr = OfficeParser_js_1.OfficeParser.terminateOcr;
|
|
54
|
+
exports.terminateOcr = terminateOcr;
|
|
54
55
|
// Default export for backward compatibility
|
|
55
|
-
exports.default =
|
|
56
|
-
// CLI handling - allows running as: node index.js file.docx
|
|
57
|
-
if (typeof require !== 'undefined' && typeof module !== 'undefined' && require.main === module) {
|
|
58
|
-
const args = process.argv.slice(2);
|
|
59
|
-
let fileArg;
|
|
60
|
-
let toText = false;
|
|
61
|
-
const configArgs = [];
|
|
62
|
-
function isConfigOption(arg) {
|
|
63
|
-
return arg.startsWith('--') && arg.includes('=');
|
|
64
|
-
}
|
|
65
|
-
args.forEach(arg => {
|
|
66
|
-
if (isConfigOption(arg)) {
|
|
67
|
-
configArgs.push(arg);
|
|
68
|
-
}
|
|
69
|
-
else if (!fileArg) {
|
|
70
|
-
fileArg = arg;
|
|
71
|
-
}
|
|
72
|
-
});
|
|
73
|
-
if (fileArg) {
|
|
74
|
-
const config = {};
|
|
75
|
-
configArgs.forEach(arg => {
|
|
76
|
-
const [key, value] = arg.split('=');
|
|
77
|
-
const cleanKey = key.replace('--', '');
|
|
78
|
-
if (cleanKey === 'toText') {
|
|
79
|
-
if (value.toLowerCase() === 'true')
|
|
80
|
-
toText = true;
|
|
81
|
-
else if (value.toLowerCase() === 'false')
|
|
82
|
-
toText = false;
|
|
83
|
-
else
|
|
84
|
-
console.log(`Invalid value for toText: ${value}`);
|
|
85
|
-
}
|
|
86
|
-
// @ts-ignore
|
|
87
|
-
else if (value.toLowerCase() === 'true')
|
|
88
|
-
config[cleanKey] = true;
|
|
89
|
-
// @ts-ignore
|
|
90
|
-
else if (value.toLowerCase() === 'false')
|
|
91
|
-
config[cleanKey] = false;
|
|
92
|
-
// @ts-ignore
|
|
93
|
-
else
|
|
94
|
-
config[cleanKey] = value;
|
|
95
|
-
});
|
|
96
|
-
OfficeParser_1.OfficeParser.parseOffice(fileArg, config)
|
|
97
|
-
.then((ast) => {
|
|
98
|
-
if (toText)
|
|
99
|
-
console.log(ast.toText());
|
|
100
|
-
else
|
|
101
|
-
console.log(JSON.stringify(ast, null, 2));
|
|
102
|
-
})
|
|
103
|
-
.catch(console.error);
|
|
104
|
-
}
|
|
105
|
-
else {
|
|
106
|
-
console.log("Usage: node officeparser [file] [--option=value]");
|
|
107
|
-
}
|
|
108
|
-
}
|
|
56
|
+
exports.default = OfficeParser_js_1.OfficeParser;
|
package/dist/index.mjs
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ESM wrapper for officeparser
|
|
3
|
+
*
|
|
4
|
+
* AUTO-GENERATED — do not edit manually.
|
|
5
|
+
* Generated by scripts/generate-esm-wrapper.js during build.
|
|
6
|
+
*
|
|
7
|
+
* This file re-exports from the CJS build (dist/index.js) to provide
|
|
8
|
+
* proper ESM named exports without duplicating the source code.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import _module from './index.js';
|
|
12
|
+
|
|
13
|
+
// Named exports
|
|
14
|
+
const { OfficeParser, parseOffice, terminateOcr } = _module;
|
|
15
|
+
export { OfficeParser, parseOffice, terminateOcr };
|
|
16
|
+
|
|
17
|
+
// Default export
|
|
18
|
+
export default _module.default ?? _module;
|