officeparser 6.0.6 → 6.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +92 -13
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +43 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +116 -0
  6. package/dist/index.d.ts +3 -3
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +756 -0
  10. package/dist/officeparser.browser.iife.js +112 -0
  11. package/dist/officeparser.browser.mjs +111 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +71 -63
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +131 -114
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +85 -88
  20. package/dist/parsers/RtfParser.d.ts +1 -1
  21. package/dist/parsers/RtfParser.js +10 -6
  22. package/dist/parsers/WordParser.d.ts +1 -1
  23. package/dist/parsers/WordParser.js +109 -101
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +69 -1
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +39 -18
  39. package/dist/officeparser.browser.js +0 -153
  40. package/dist/officeparser.browser.js.map +0 -7
package/README.md CHANGED
@@ -3,6 +3,8 @@
3
3
  A robust, strictly-typed Node.js and Browser library for parsing office files ([`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format)). It produces a clean, hierarchical Abstract Syntax Tree (AST) with rich metadata, text formatting, and full attachment support.
4
4
 
5
5
  [![npm version](https://badge.fury.io/js/officeparser.svg)](https://badge.fury.io/js/officeparser)
6
+ [![Total Downloads](https://img.shields.io/npm/dt/officeparser.svg)](https://www.npmjs.com/package/officeparser)
7
+ [![Weekly Downloads](https://img.shields.io/npm/dw/officeparser.svg)](https://www.npmjs.com/package/officeparser)
6
8
  [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
7
9
 
8
10
  ---
@@ -22,6 +24,13 @@ A robust, strictly-typed Node.js and Browser library for parsing office files ([
22
24
 
23
25
 
24
26
  #### Update
27
+ * 2026-04-14 - **v6.1.0 Release**: Major Infrastructure & Resource Stability. (Incremental since v6.0.0)
28
+ - **OCR Scheduler**: Intelligent worker pool that optimizes Tesseract lifecycle across parallel requests. **Note**: By default, Node.js processes stay active for 10s after OCR to keep workers warm (configurable via `ocrConfig.autoTerminateTimeout`); use `terminateOcr()` for immediate CLI/script exit.
29
+ - **Core Engine**: Replaced legacy zip extraction with `fflate` for significant performance gains and robust browser/edge compatibility.
30
+ - **Module System**: Full native ESM support with `Node16` resolution and verified browser bundles (Vite/Angular compatible).
31
+ - **Format Refinements**: Hierarchical PDF coordinate alignment and ODT/RTF list parsing stability.
32
+ - **Custom Properties**: Added support for extracting custom document metadata across OOXML, ODF, and PDF formats.
33
+ - **Sponsorship**: Integrated `funding.json` manifest and GitHub Sponsors support.
25
34
  * 2025/12/29 - **v6.0.0 Release**: Major overhaul of the library. Transitioned from simple text extraction to a rich **Abstract Syntax Tree (AST)** output.
26
35
  - Simplified API: Use `parseOffice` for all parsing needs (returns a Promise).
27
36
  - Structured Output: Access hierarchical document structure (paragraphs, headings, tables, lists, etc.).
@@ -82,6 +91,7 @@ npx officeparser /path/to/officeFile.docx --ignoreNotes=true --newlineDelimiter=
82
91
  - `--extractAttachments=[true|false]` Flag to extract images/charts as Base64. Default is false.
83
92
  - `--ocr=[true|false]` Flag to enable OCR for extracted images. Default is false.
84
93
  - `--includeRawContent=[true|false]` Flag to include raw XML/RTF content in nodes. Default is false.
94
+ - `--verbose=[true|false]` Show full error stack traces.
85
95
 
86
96
 
87
97
  ## Library Usage
@@ -156,7 +166,7 @@ The `OfficeParserAST` provides a format-agnostic representation of your document
156
166
  ```text
157
167
  OfficeParserAST
158
168
  ├── type: "docx" | "pptx" | "xlsx" | ...
159
- ├── metadata: { author, title, created, modified, ... }
169
+ ├── metadata: { author, title, created, modified, ..., customProperties }
160
170
  ├── content: [ OfficeContentNode ]
161
171
  │ ├── type: "paragraph" | "heading" | "table" | "list" | ...
162
172
  │ ├── text: "Concatenated text of this node and all children"
@@ -177,7 +187,7 @@ OfficeParserAST
177
187
  ```json
178
188
  {
179
189
  "type": "docx",
180
- "metadata": { "author": "John Doe", "title": "Annual Report" },
190
+ "metadata": { "author": "John Doe", "title": "Annual Report", "customProperties": { "Department": "Finance" } },
181
191
  "content": [
182
192
  {
183
193
  "type": "heading",
@@ -303,6 +313,16 @@ Formatting can be found at two levels:
303
313
  The `ast.metadata` object provides document-wide context:
304
314
  - **`styleMap`**: A dictionary of style names to their `TextFormatting` definitions found in the document.
305
315
  - **`formatting`**: Document-wide default settings (e.g., default font or font size).
316
+ - **`customProperties`**: A dictionary of user-defined metadata embedded in the document (OOXML `custom.xml`, ODF `meta:user-defined`, or PDF Info dictionary).
317
+
318
+ ### 7. Custom Properties
319
+ You can access custom user-defined metadata that might be embedded in the document:
320
+
321
+ ```javascript
322
+ const ast = await officeParser.parseOffice("contract.docx");
323
+ console.log("Custom Metadata:", ast.metadata.customProperties);
324
+ // Output: { "ProjectID": "ABC-123", "InternalReview": true }
325
+ ```
306
326
 
307
327
  ### Advanced AST Usage
308
328
  Beyond using `ast.toText()`, you can interact with the structural data directly:
@@ -403,10 +423,46 @@ Pass an optional config object as the second argument to `parseOffice`.
403
423
  | `ignoreNotes` | boolean | `false` | Ignore notes in files like PowerPoint/ODP. |
404
424
  | `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. (Note: Does not work for RTF. It is treated as true always.) |
405
425
  | `extractAttachments` | boolean | `false` | Extract images and charts as Base64. |
406
- | `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
407
- | `ocrLanguage` | string | `eng` | Language for OCR (e.g., 'eng', 'fra'). Supports multiple languages with '+'. See [Language Codes](https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016). |
408
426
  | `includeRawContent` | boolean | `false` | Include raw XML/RTF markup in the nodes. |
427
+ | `serializeRawContent` | boolean | `true` | When `includeRawContent` is true, re-serializes raw XML to clean strings. If false, extracts original raw substring. |
428
+ | `preserveXmlWhitespace` | boolean | `false` | When `serializeRawContent` is true, preserves original XML whitespace and line endings. |
429
+ | `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
430
+ | `ocrLanguage` | string | `eng` | **Deprecated**: Use `ocrConfig.language` instead. Language for OCR. |
409
431
  | `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. Defaults to a CDN link if not provided. |
432
+ | `ocrConfig` | object | `{}` | **OCR Scheduler** configuration for fine-grained worker control. |
433
+ | `ocrConfig.language` | string | `eng` | Language(s) for OCR (e.g., 'eng', 'fra', 'eng+fra'). |
434
+ | `ocrConfig.autoTerminateTimeout` | number | `10000` | Inactivity timeout in milliseconds before workers are killed. |
435
+ | `ocrConfig.workerPath` | string | `undefined` | Path to Tesseract worker script (for offline use). |
436
+ | `ocrConfig.corePath` | string | `undefined` | Path to Tesseract core script (for offline use). |
437
+ | `ocrConfig.langPath` | string | `undefined` | Path for Tesseract language files (for offline use). |
438
+
439
+ ### OCR Scheduler & Resource Management
440
+ If your application uses OCR, `officeParser` utilizes an intelligent **Smart Worker Pool** to maintain a background worker pool and optimize repeated parse requests.
441
+
442
+ - **Dynamic Affinity**: Workers in the pool persist with their last used language affinity.
443
+ - **Smart Re-initialization**: If a new language is requested and the pool is full, the manager identifies the **Least Recently Used (LRU)** idle worker and re-initializes it for the new language using the Tesseract.js v5 API. This avoids the overhead of destroying and recreating workers.
444
+ - **Auto-Termination**: Workers are automatically cleaned up after 10 seconds of inactivity (configurable via `ocrConfig.autoTerminateTimeout`).
445
+
446
+ #### `OfficeParser.terminateOcr()`
447
+ If you have used OCR (`{ ocr: true }`) in a short-lived script (like CLI tools or one-off automation), we recommend explicitly calling `terminateOcr()` after your processing is finished. This bypasses the 10-second idle timer and allows the process to return to the terminal prompt immediately.
448
+
449
+ > [!NOTE]
450
+ > If OCR was not used, this function is a no-op and does not need to be called.
451
+
452
+ ```js
453
+ const officeParser = require('officeparser');
454
+
455
+ async function runCleaner() {
456
+ await officeParser.parseOffice("file.pdf", { ocr: true });
457
+ // ... process results ...
458
+
459
+ // Manually kill OCR workers for an immediate exit
460
+ await officeParser.terminateOcr();
461
+ }
462
+ ```
463
+
464
+ > [!TIP]
465
+ > This is handled automatically in our own CLI (`npx officeparser ...`). You only need to call this manually if you are using the library in your own custom script and want a snappy exit.
410
466
 
411
467
  ```js
412
468
  const config = {
@@ -449,20 +505,21 @@ officeParser.parseOffice("presentation.pptx", config).then(ast => {
449
505
  ```
450
506
 
451
507
  ## Browser Usage
452
- The browser bundle exposes the `officeParser` namespace. Include the bundle file available in the release assets.
508
+ The library provides two types of browser bundles in the `dist/` directory:
509
+ 1. **`officeparser.browser.iife.js`**: Standard IIFE bundle for direct `<script>` tag usage. Exposes the global `officeParser` namespace.
510
+ 2. **`officeparser.browser.mjs`**: Modern ESM bundle for use with `import` statements or modern bundlers.
511
+
512
+ ### Usage (Script Tag)
513
+ Include the IIFE bundle file available in the release assets.
453
514
 
454
515
  ```html
455
- <script src="dist/officeparser.browser.js"></script>
516
+ <script src="dist/officeparser.browser.iife.js"></script>
456
517
  <script>
457
518
  async function handleFile(file) {
458
519
  // file can be a File object from an input element or an ArrayBuffer
459
- // The browser bundle exposes the global variable `officeParser`
460
- // which contains the `OfficeParser` class.
461
-
462
520
  try {
463
521
  const ast = await officeParser.parseOffice(file, { ocr: true });
464
522
  console.log(ast.toText());
465
- console.log("Metadata:", ast.metadata);
466
523
  } catch (error) {
467
524
  console.error(error);
468
525
  }
@@ -470,8 +527,20 @@ The browser bundle exposes the `officeParser` namespace. Include the bundle file
470
527
  </script>
471
528
  ```
472
529
 
530
+ ### Usage (ESM)
531
+ If you are using a modern browser that supports modules or a dev server like Vite:
532
+
533
+ ```html
534
+ <script type="module">
535
+ import { OfficeParser } from './dist/officeparser.browser.mjs';
536
+
537
+ const ast = await OfficeParser.parseOffice(fileBuffer);
538
+ console.log(ast.metadata);
539
+ </script>
540
+ ```
541
+
473
542
  ### PDF Worker Configuration in Browser
474
- When using `officeparser` in a browser environment to parse PDF files, you may provide the `pdfWorkerSrc` configuration option. If not provided, it defaults to a CDN link for `pdfjs-dist@5.5.207`.
543
+ When using `officeparser` in a browser environment to parse PDF files, you may provide the `pdfWorkerSrc` configuration option. If not provided, it defaults to a CDN link for `pdfjs-dist@5.6.205`.
475
544
 
476
545
  ```javascript
477
546
  const file = ...; // File object or ArrayBuffer
@@ -481,11 +550,21 @@ const ast = await officeParser.parseOffice(file);
481
550
 
482
551
  // Or override it with your own path or a different version:
483
552
  const ast2 = await officeParser.parseOffice(file, {
484
- pdfWorkerSrc: "https://unpkg.com/pdfjs-dist@5.5.207/build/pdf.worker.min.mjs"
553
+ pdfWorkerSrc: "https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
485
554
  });
486
555
  ```
487
556
 
488
- > **Note:** The version of `pdfjs-dist` in the worker source should match the version used by `officeparser` (currently `5.4.530`).
557
+ > **Note:** The version of `pdfjs-dist` in the worker source should match the version used by `officeparser` (currently `5.6.205`).
558
+
559
+ ## Troubleshooting & Common Issues
560
+
561
+ - **Node.js process stays alive after finishing**: If using OCR, the worker pool stays warm for 10s by default. Use `await terminateOcr()` at the end of your script for a snappy exit.
562
+ - **"Worker not found" in Browser**: Ensure `pdfWorkerSrc` is correctly pointed to the `pdf.worker.min.mjs` file matching version `5.6.205`.
563
+ - **OCR accuracy is low**: Verify your `ocrConfig.language` matches the document content. Note that OCR quality depends on image resolution.
564
+ - **Out of memory on large files**: For massive spreadsheets, consider using `ast.toText()` early and allowing the full AST object to be garbage-collected.
565
+
566
+ For a comprehensive guide, visit our [Debugging & Troubleshooting Documentation](https://harshankur.github.io/officeParser/#spec/debugging).
567
+
489
568
 
490
569
  ## Known Limitations
491
570
  1. **ODT/ODS Charts**: Extraction may occasionally show inaccurate data when referencing external cell ranges or complex layout-based data.
@@ -32,7 +32,7 @@
32
32
  *
33
33
  * @module OfficeParser
34
34
  */
35
- import { OfficeParserAST, OfficeParserConfig } from './types';
35
+ import { OfficeParserAST, OfficeParserConfig } from './types.js';
36
36
  /**
37
37
  * Main parser class providing office document parsing functionality.
38
38
  *
@@ -86,4 +86,13 @@ export declare class OfficeParser {
86
86
  * ```
87
87
  */
88
88
  static parseOffice(file: string | Buffer | ArrayBuffer, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
89
+ /**
90
+ * Terminates all active OCR workers and cleans up resources.
91
+ *
92
+ * This should be called when the application is shutting down or when OCR
93
+ * is no longer needed to prevent memory leaks and orphaned worker processes.
94
+ *
95
+ * @returns A promise that resolves when all workers have been terminated
96
+ */
97
+ static terminateOcr(): Promise<void>;
89
98
  }
@@ -33,50 +33,18 @@
33
33
  *
34
34
  * @module OfficeParser
35
35
  */
36
- var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
37
- if (k2 === undefined) k2 = k;
38
- var desc = Object.getOwnPropertyDescriptor(m, k);
39
- if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
40
- desc = { enumerable: true, get: function() { return m[k]; } };
41
- }
42
- Object.defineProperty(o, k2, desc);
43
- }) : (function(o, m, k, k2) {
44
- if (k2 === undefined) k2 = k;
45
- o[k2] = m[k];
46
- }));
47
- var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
48
- Object.defineProperty(o, "default", { enumerable: true, value: v });
49
- }) : function(o, v) {
50
- o["default"] = v;
51
- });
52
- var __importStar = (this && this.__importStar) || (function () {
53
- var ownKeys = function(o) {
54
- ownKeys = Object.getOwnPropertyNames || function (o) {
55
- var ar = [];
56
- for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
57
- return ar;
58
- };
59
- return ownKeys(o);
60
- };
61
- return function (mod) {
62
- if (mod && mod.__esModule) return mod;
63
- var result = {};
64
- if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
65
- __setModuleDefault(result, mod);
66
- return result;
67
- };
68
- })();
69
36
  Object.defineProperty(exports, "__esModule", { value: true });
70
37
  exports.OfficeParser = void 0;
71
- const fs = __importStar(require("fs"));
72
- const ExcelParser_1 = require("./parsers/ExcelParser");
73
- const OpenOfficeParser_1 = require("./parsers/OpenOfficeParser");
74
- const PdfParser_1 = require("./parsers/PdfParser");
75
- const PowerPointParser_1 = require("./parsers/PowerPointParser");
76
- const RtfParser_1 = require("./parsers/RtfParser");
77
- const WordParser_1 = require("./parsers/WordParser");
78
- const errorUtils_1 = require("./utils/errorUtils");
79
- const moduleLoader_1 = require("./utils/moduleLoader");
38
+ const envUtils_js_1 = require("./utils/envUtils.js");
39
+ const ExcelParser_js_1 = require("./parsers/ExcelParser.js");
40
+ const OpenOfficeParser_js_1 = require("./parsers/OpenOfficeParser.js");
41
+ const PdfParser_js_1 = require("./parsers/PdfParser.js");
42
+ const PowerPointParser_js_1 = require("./parsers/PowerPointParser.js");
43
+ const RtfParser_js_1 = require("./parsers/RtfParser.js");
44
+ const WordParser_js_1 = require("./parsers/WordParser.js");
45
+ const errorUtils_js_1 = require("./utils/errorUtils.js");
46
+ const moduleLoader_js_1 = require("./utils/moduleLoader.js");
47
+ const ocrUtils_js_1 = require("./utils/ocrUtils.js");
80
48
  /**
81
49
  * Main parser class providing office document parsing functionality.
82
50
  *
@@ -148,7 +116,10 @@ class OfficeParser {
148
116
  ocr: false,
149
117
  ocrLanguage: 'eng',
150
118
  includeRawContent: false,
119
+ serializeRawContent: true,
120
+ preserveXmlWhitespace: false,
151
121
  pdfWorkerSrc: '',
122
+ ocrConfig: {},
152
123
  ...actualConfig
153
124
  };
154
125
  let buffer = Buffer.alloc(0);
@@ -156,7 +127,7 @@ class OfficeParser {
156
127
  let filePath;
157
128
  try {
158
129
  if (!file) {
159
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
130
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
160
131
  }
161
132
  if (file instanceof ArrayBuffer) {
162
133
  buffer = Buffer.from(file);
@@ -166,63 +137,79 @@ class OfficeParser {
166
137
  }
167
138
  else if (typeof file === 'string') {
168
139
  filePath = file;
140
+ (0, envUtils_js_1.assertNode)('path-parsing');
141
+ // Safe to use dynamic import here as we've asserted we are in Node.
142
+ // Modern bundlers will still see this, but our browser builds
143
+ // shim 'fs' so it won't crash at build time.
144
+ const fs = await import('fs');
169
145
  if (!fs.existsSync(file)) {
170
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
146
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
171
147
  }
172
148
  if (fs.lstatSync(file).isDirectory()) {
173
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
149
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
174
150
  }
175
151
  buffer = fs.readFileSync(file);
176
152
  ext = file.split('.').pop()?.toLowerCase() || '';
177
153
  }
178
154
  else {
179
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.INVALID_INPUT, internalConfig);
155
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
180
156
  }
181
157
  if (!ext) {
182
- const { fileTypeFromBuffer } = await (0, moduleLoader_1.loadFileType)();
158
+ const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
183
159
  const type = await fileTypeFromBuffer(buffer);
184
160
  if (type) {
185
161
  ext = type.ext.toLowerCase();
186
162
  }
187
163
  else {
188
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
164
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
189
165
  }
190
166
  }
191
167
  let result;
192
168
  switch (ext) {
193
169
  case 'docx':
194
- result = await (0, WordParser_1.parseWord)(buffer, internalConfig);
170
+ result = await (0, WordParser_js_1.parseWord)(buffer, internalConfig);
195
171
  break;
196
172
  case 'pptx':
197
- result = await (0, PowerPointParser_1.parsePowerPoint)(buffer, internalConfig);
173
+ result = await (0, PowerPointParser_js_1.parsePowerPoint)(buffer, internalConfig);
198
174
  break;
199
175
  case 'xlsx':
200
- result = await (0, ExcelParser_1.parseExcel)(buffer, internalConfig);
176
+ result = await (0, ExcelParser_js_1.parseExcel)(buffer, internalConfig);
201
177
  break;
202
178
  case 'odt':
203
179
  case 'odp':
204
180
  case 'ods':
205
- result = await (0, OpenOfficeParser_1.parseOpenOffice)(buffer, internalConfig);
181
+ result = await (0, OpenOfficeParser_js_1.parseOpenOffice)(buffer, internalConfig);
206
182
  break;
207
183
  case 'pdf':
208
- result = await (0, PdfParser_1.parsePdf)(buffer, internalConfig);
184
+ result = await (0, PdfParser_js_1.parsePdf)(buffer, internalConfig);
209
185
  break;
210
186
  case 'rtf':
211
- result = await (0, RtfParser_1.parseRtf)(buffer, internalConfig);
187
+ result = await (0, RtfParser_js_1.parseRtf)(buffer, internalConfig);
212
188
  break;
213
189
  default:
214
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
190
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
215
191
  }
216
192
  if (callback)
217
193
  callback(result);
218
194
  return result;
219
195
  }
220
196
  catch (error) {
221
- const wrappedError = (0, errorUtils_1.getWrappedError)(error, internalConfig, filePath);
197
+ const wrappedError = (0, errorUtils_js_1.getWrappedError)(error, internalConfig, filePath);
222
198
  if (callback)
223
199
  callback(undefined, wrappedError);
224
200
  throw wrappedError;
225
201
  }
226
202
  }
203
+ /**
204
+ * Terminates all active OCR workers and cleans up resources.
205
+ *
206
+ * This should be called when the application is shutting down or when OCR
207
+ * is no longer needed to prevent memory leaks and orphaned worker processes.
208
+ *
209
+ * @returns A promise that resolves when all workers have been terminated
210
+ */
211
+ static async terminateOcr() {
212
+ await (0, ocrUtils_js_1.terminateOcr)();
213
+ }
227
214
  }
228
215
  exports.OfficeParser = OfficeParser;
package/dist/cli.d.ts ADDED
@@ -0,0 +1,20 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * officeparser CLI
4
+ *
5
+ * Allows running officeparser from the command line:
6
+ * npx officeparser file.docx
7
+ * officeparser file.docx --toText=true
8
+ * officeparser file.docx --ocr=true --extractAttachments=true
9
+ *
10
+ * Options (--key=value):
11
+ * --toText=true Output plain text instead of JSON AST
12
+ * --ocr=true Enable OCR for images
13
+ * --ocrLanguage=eng OCR language (default: eng)
14
+ * --extractAttachments=true Extract embedded attachments
15
+ * --ignoreNotes=true Ignore footnotes/endnotes
16
+ * --putNotesAtLast=true Move notes to end of document
17
+ * --includeRawContent=true Include raw content in AST
18
+ * --outputErrorToConsole=true Log errors to console
19
+ */
20
+ export {};
package/dist/cli.js ADDED
@@ -0,0 +1,116 @@
1
+ #!/usr/bin/env node
2
+ "use strict";
3
+ /**
4
+ * officeparser CLI
5
+ *
6
+ * Allows running officeparser from the command line:
7
+ * npx officeparser file.docx
8
+ * officeparser file.docx --toText=true
9
+ * officeparser file.docx --ocr=true --extractAttachments=true
10
+ *
11
+ * Options (--key=value):
12
+ * --toText=true Output plain text instead of JSON AST
13
+ * --ocr=true Enable OCR for images
14
+ * --ocrLanguage=eng OCR language (default: eng)
15
+ * --extractAttachments=true Extract embedded attachments
16
+ * --ignoreNotes=true Ignore footnotes/endnotes
17
+ * --putNotesAtLast=true Move notes to end of document
18
+ * --includeRawContent=true Include raw content in AST
19
+ * --outputErrorToConsole=true Log errors to console
20
+ */
21
+ Object.defineProperty(exports, "__esModule", { value: true });
22
+ const OfficeParser_js_1 = require("./OfficeParser.js");
23
+ const args = process.argv.slice(2);
24
+ let fileArg;
25
+ let toText = false;
26
+ let verbose = false;
27
+ const configArgs = [];
28
+ function isConfigOption(arg) {
29
+ return arg.startsWith('--') && arg.includes('=');
30
+ }
31
+ args.forEach(arg => {
32
+ if (isConfigOption(arg)) {
33
+ configArgs.push(arg);
34
+ }
35
+ else if (!fileArg) {
36
+ fileArg = arg;
37
+ }
38
+ });
39
+ if (fileArg) {
40
+ const config = {};
41
+ configArgs.forEach(arg => {
42
+ const [key, value] = arg.split('=');
43
+ const cleanKey = key.replace('--', '');
44
+ const lowerValue = value.toLowerCase();
45
+ const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
46
+ if (cleanKey === 'toText') {
47
+ if (boolValue !== undefined)
48
+ toText = boolValue;
49
+ else
50
+ console.warn(`Invalid value for toText: ${value}`);
51
+ }
52
+ else if (cleanKey === 'verbose') {
53
+ if (boolValue !== undefined)
54
+ verbose = boolValue;
55
+ else
56
+ console.warn(`Invalid value for verbose: ${value}`);
57
+ }
58
+ else {
59
+ // @ts-ignore
60
+ if (boolValue !== undefined)
61
+ config[cleanKey] = boolValue;
62
+ // @ts-ignore
63
+ else
64
+ config[cleanKey] = value;
65
+ }
66
+ });
67
+ OfficeParser_js_1.OfficeParser.parseOffice(fileArg, config)
68
+ .then(async (ast) => {
69
+ if (toText) {
70
+ process.stdout.write(ast.toText() + '\n');
71
+ }
72
+ else {
73
+ process.stdout.write(JSON.stringify(ast, null, 2) + '\n');
74
+ }
75
+ // Ensure OCR workers are terminated for clean CLI exit
76
+ if (config.ocr) {
77
+ await OfficeParser_js_1.OfficeParser.terminateOcr();
78
+ }
79
+ })
80
+ .catch(async (err) => {
81
+ console.error(`Error parsing file "${fileArg}":`);
82
+ if (verbose) {
83
+ console.error(err);
84
+ }
85
+ else {
86
+ console.error(err.message || err);
87
+ console.error('Use --verbose=true for full stack trace.');
88
+ }
89
+ // Ensure OCR workers are terminated even on error
90
+ if (config.ocr) {
91
+ await OfficeParser_js_1.OfficeParser.terminateOcr();
92
+ }
93
+ process.exit(1);
94
+ });
95
+ }
96
+ else {
97
+ console.log('Usage: officeparser <file> [--option=value]');
98
+ console.log('');
99
+ console.log('Options:');
100
+ console.log(' --toText=true Output plain text instead of JSON AST');
101
+ console.log(' --ocr=true Enable OCR for images');
102
+ console.log(' --ocrLanguage=eng OCR language (default: eng)');
103
+ console.log(' --extractAttachments=true Extract embedded attachments');
104
+ console.log(' --ignoreNotes=true Ignore footnotes/endnotes');
105
+ console.log(' --putNotesAtLast=true Move notes to end of document');
106
+ console.log(' --includeRawContent=true Include raw content in AST');
107
+ console.log(' --serializeRawContent=true Serialize raw XML content (default: true)');
108
+ console.log(' --preserveXmlWhitespace=true Preserve whitespace in serialized XML (default: false)');
109
+ console.log(' --verbose=true Show full error stack traces');
110
+ console.log('');
111
+ console.log('Examples:');
112
+ console.log(' officeparser document.docx');
113
+ console.log(' officeparser document.docx --toText=true');
114
+ console.log(' officeparser report.pdf --ocr=true --extractAttachments=true');
115
+ console.log(' officeparser complex.docx --serializeRawContent=false --includeRawContent=true');
116
+ }
package/dist/index.d.ts CHANGED
@@ -1,4 +1,3 @@
1
- #!/usr/bin/env node
2
1
  /**
3
2
  * officeparser - Universal Office Document Parser
4
3
  *
@@ -44,8 +43,9 @@
44
43
  * @packageDocumentation
45
44
  * @module officeparser
46
45
  */
47
- import { OfficeParser } from './OfficeParser';
46
+ import { OfficeParser } from './OfficeParser.js';
48
47
  import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata } from './types';
49
48
  declare const parseOffice: typeof OfficeParser.parseOffice;
50
- export { OfficeParser, parseOffice, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata };
49
+ declare const terminateOcr: typeof OfficeParser.terminateOcr;
50
+ export { OfficeParser, parseOffice, terminateOcr, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata };
51
51
  export default OfficeParser;
package/dist/index.js CHANGED
@@ -1,4 +1,3 @@
1
- #!/usr/bin/env node
2
1
  "use strict";
3
2
  /**
4
3
  * officeparser - Universal Office Document Parser
@@ -46,63 +45,12 @@
46
45
  * @module officeparser
47
46
  */
48
47
  Object.defineProperty(exports, "__esModule", { value: true });
49
- exports.parseOffice = exports.OfficeParser = void 0;
50
- const OfficeParser_1 = require("./OfficeParser");
51
- Object.defineProperty(exports, "OfficeParser", { enumerable: true, get: function () { return OfficeParser_1.OfficeParser; } });
52
- const parseOffice = OfficeParser_1.OfficeParser.parseOffice;
48
+ exports.terminateOcr = exports.parseOffice = exports.OfficeParser = void 0;
49
+ const OfficeParser_js_1 = require("./OfficeParser.js");
50
+ Object.defineProperty(exports, "OfficeParser", { enumerable: true, get: function () { return OfficeParser_js_1.OfficeParser; } });
51
+ const parseOffice = OfficeParser_js_1.OfficeParser.parseOffice;
53
52
  exports.parseOffice = parseOffice;
53
+ const terminateOcr = OfficeParser_js_1.OfficeParser.terminateOcr;
54
+ exports.terminateOcr = terminateOcr;
54
55
  // Default export for backward compatibility
55
- exports.default = OfficeParser_1.OfficeParser;
56
- // CLI handling - allows running as: node index.js file.docx
57
- if (typeof require !== 'undefined' && typeof module !== 'undefined' && require.main === module) {
58
- const args = process.argv.slice(2);
59
- let fileArg;
60
- let toText = false;
61
- const configArgs = [];
62
- function isConfigOption(arg) {
63
- return arg.startsWith('--') && arg.includes('=');
64
- }
65
- args.forEach(arg => {
66
- if (isConfigOption(arg)) {
67
- configArgs.push(arg);
68
- }
69
- else if (!fileArg) {
70
- fileArg = arg;
71
- }
72
- });
73
- if (fileArg) {
74
- const config = {};
75
- configArgs.forEach(arg => {
76
- const [key, value] = arg.split('=');
77
- const cleanKey = key.replace('--', '');
78
- if (cleanKey === 'toText') {
79
- if (value.toLowerCase() === 'true')
80
- toText = true;
81
- else if (value.toLowerCase() === 'false')
82
- toText = false;
83
- else
84
- console.log(`Invalid value for toText: ${value}`);
85
- }
86
- // @ts-ignore
87
- else if (value.toLowerCase() === 'true')
88
- config[cleanKey] = true;
89
- // @ts-ignore
90
- else if (value.toLowerCase() === 'false')
91
- config[cleanKey] = false;
92
- // @ts-ignore
93
- else
94
- config[cleanKey] = value;
95
- });
96
- OfficeParser_1.OfficeParser.parseOffice(fileArg, config)
97
- .then((ast) => {
98
- if (toText)
99
- console.log(ast.toText());
100
- else
101
- console.log(JSON.stringify(ast, null, 2));
102
- })
103
- .catch(console.error);
104
- }
105
- else {
106
- console.log("Usage: node officeparser [file] [--option=value]");
107
- }
108
- }
56
+ exports.default = OfficeParser_js_1.OfficeParser;
package/dist/index.mjs ADDED
@@ -0,0 +1,18 @@
1
+ /**
2
+ * ESM wrapper for officeparser
3
+ *
4
+ * AUTO-GENERATED — do not edit manually.
5
+ * Generated by scripts/generate-esm-wrapper.js during build.
6
+ *
7
+ * This file re-exports from the CJS build (dist/index.js) to provide
8
+ * proper ESM named exports without duplicating the source code.
9
+ */
10
+
11
+ import _module from './index.js';
12
+
13
+ // Named exports
14
+ const { OfficeParser, parseOffice, terminateOcr } = _module;
15
+ export { OfficeParser, parseOffice, terminateOcr };
16
+
17
+ // Default export
18
+ export default _module.default ?? _module;