officeparser 6.0.7 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +136 -52
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +44 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +117 -0
  6. package/dist/index.d.ts +4 -4
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +133 -3
  10. package/dist/officeparser.browser.iife.js +115 -0
  11. package/dist/officeparser.browser.mjs +114 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +76 -68
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +224 -159
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +188 -179
  20. package/dist/parsers/RtfParser.d.ts +21 -1
  21. package/dist/parsers/RtfParser.js +117 -48
  22. package/dist/parsers/WordParser.d.ts +2 -1
  23. package/dist/parsers/WordParser.js +214 -123
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +123 -3
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +31 -16
  39. package/dist/officeParserBundle@6.0.7.js +0 -154
  40. package/dist/officeparser.browser.js +0 -154
package/README.md CHANGED
@@ -3,6 +3,8 @@
3
3
  A robust, strictly-typed Node.js and Browser library for parsing office files ([`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format)). It produces a clean, hierarchical Abstract Syntax Tree (AST) with rich metadata, text formatting, and full attachment support.
4
4
 
5
5
  [![npm version](https://badge.fury.io/js/officeparser.svg)](https://badge.fury.io/js/officeparser)
6
+ [![Total Downloads](https://img.shields.io/npm/dt/officeparser.svg)](https://www.npmjs.com/package/officeparser)
7
+ [![Weekly Downloads](https://img.shields.io/npm/dw/officeparser.svg)](https://www.npmjs.com/package/officeparser)
6
8
  [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
7
9
 
8
10
  ---
@@ -21,37 +23,12 @@ A robust, strictly-typed Node.js and Browser library for parsing office files ([
21
23
  ---
22
24
 
23
25
 
24
- #### Update
25
- * 2025/12/29 - **v6.0.0 Release**: Major overhaul of the library. Transitioned from simple text extraction to a rich **Abstract Syntax Tree (AST)** output.
26
- - Simplified API: Use `parseOffice` for all parsing needs (returns a Promise).
27
- - Structured Output: Access hierarchical document structure (paragraphs, headings, tables, lists, etc.).
28
- - Rich Metadata: Extracted document properties (author, title, creation date).
29
- - Enhanced Formatting: Support for bold, italic, colors, fonts, alignment, etc.
30
- - Attachment Handling: Extract images, charts, and embedded files as Base64.
31
- - OCR Integration: Optional OCR for images using Tesseract.js.
32
- - RTF Support: Added full support for Rich Text Format files.
33
- - Improved Type Definitions: Full TypeScript support with detailed interfaces.
34
- * 2024/11/12 - Added ArrayBuffer as a type of file input. Generating bundle files now which exposes namespace officeParser to be able to access parseOffice directly on the browser.
35
- * 2024/10/21 - Replaced extracting zip files from decompress to yauzl. This means that we now extract files in memory and we no longer need to write them to disk. Removed config flags related to extracted files. Added flags for CLI execution.
36
- * 2024/10/15 - Fixed erroring out while deleting temp files when multiple worker threads make parallel executions resulting in same file name for multiple files. Fixed erroring out when multiple executions are made without waiting for the previous execution to finish which resulted in deleting the file from other execution. Upgraded dependencies.
37
- * 2024/10/13 - Fixed parsing text from xlsx files which contain no shared strings file and files which have inlineStr based strings.
38
- * 2024/05/06 - Replaced pdf parsing support from pdf-parse library to natively building it using pdf.js library from Mozilla by analyzing its output. Added pdfjs-dist build as a local library.
39
- * 2023/11/25 - Fixed error catching when an error occurs within the parsing of a file, especially after decompressing it. Also fixed the problem with parallel parsing of files as we were using only timestamp in file names.
40
- * 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
41
- * 2023/04/26 - Added support for file buffers as argument for filepath for parseOffice and parseOfficeAsync
42
- * 2023/04/07 - Added typings to methods to help with Typescript projects.
43
- * 2022/12/28 - Added command line method to use officeParser with or without installing it and instantly get parsed content on the console.
44
- * 2022/12/10 - Fixed memory leak issues, bugs related to parsing open document files and improved error handling.
45
- * 2021/11/21 - Added promise way to existing callback functions.
46
- * 2020/06/01 - Added error handling and console.log enable/disable methods. Default is set at enabled. Everything backward compatible.
47
- * 2019/06/17 - Added method to change location for decompressing office files in places with restricted write access.
48
- * 2019/04/30 - Removed case sensitive file extension bug. File names with capital lettered extensions now supported.
49
- * 2019/04/23 - Added support for open office files *.odt, *.odp, *.ods through parseOffice function. Created a new method parseOpenOffice for those who prefer targetted functions.
50
- * 2019/04/23 - Added feature to delete the generated dist folder after function callback.
51
- * 2019/04/22 - Added parseOffice method to avoid confusion between type of file and their extension.
52
- * 2019/04/22 - Added file extension validations. Removed errors for excel files with no drawing elements.
53
- * 2019/04/19 - Support added for *.xlsx files.
54
- * 2019/04/18 - Support added for *.pptx files.
26
+ ---
27
+
28
+ ### 📝 [Changelog](CHANGELOG.md)
29
+ *Detailed release notes and the full history of updates are available in the project changelog.*
30
+
31
+ ---
55
32
 
56
33
  ## Install via npm
57
34
 
@@ -82,6 +59,8 @@ npx officeparser /path/to/officeFile.docx --ignoreNotes=true --newlineDelimiter=
82
59
  - `--extractAttachments=[true|false]` Flag to extract images/charts as Base64. Default is false.
83
60
  - `--ocr=[true|false]` Flag to enable OCR for extracted images. Default is false.
84
61
  - `--includeRawContent=[true|false]` Flag to include raw XML/RTF content in nodes. Default is false.
62
+ - `--includeBreakNodes=[true|false]` Flag to include break nodes. Currently only available for DOCX documents
63
+ - `--verbose=[true|false]` Show full error stack traces.
85
64
 
86
65
 
87
66
  ## Library Usage
@@ -121,7 +100,7 @@ console.log(text);
121
100
  ```
122
101
 
123
102
  ### Using Callbacks (Backward Compatibility Support)
124
- We still support callbacks, but the data returned is now the AST object.
103
+ Callbacks are still supported for those preferred, but the data returned is now the AST object.
125
104
  ```js
126
105
  const officeParser = require('officeparser');
127
106
 
@@ -156,13 +135,13 @@ The `OfficeParserAST` provides a format-agnostic representation of your document
156
135
  ```text
157
136
  OfficeParserAST
158
137
  ├── type: "docx" | "pptx" | "xlsx" | ...
159
- ├── metadata: { author, title, created, modified, ... }
138
+ ├── metadata: { author, title, created, modified, ..., customProperties }
160
139
  ├── content: [ OfficeContentNode ]
161
140
  │ ├── type: "paragraph" | "heading" | "table" | "list" | ...
162
141
  │ ├── text: "Concatenated text of this node and all children"
163
142
  │ ├── children: [ OfficeContentNode ] (recursive)
164
143
  │ ├── formatting: { bold, italic, color, size, font, ... }
165
- │ ├── metadata: { level, listId, row, col, ... }
144
+ │ ├── metadata: { level, listId, paragraphIndentation, row, col, ... }
166
145
  │ └── rawContent: "<xml>...</xml>" (if enabled)
167
146
  ├── attachments: [ OfficeAttachment ]
168
147
  │ ├── type: "image" | "chart"
@@ -177,7 +156,7 @@ OfficeParserAST
177
156
  ```json
178
157
  {
179
158
  "type": "docx",
180
- "metadata": { "author": "John Doe", "title": "Annual Report" },
159
+ "metadata": { "author": "John Doe", "title": "Annual Report", "customProperties": { "Department": "Finance" } },
181
160
  "content": [
182
161
  {
183
162
  "type": "heading",
@@ -214,13 +193,15 @@ List Node
214
193
  listId: "1",
215
194
  listType: "ordered",
216
195
  indentation: 0,
196
+ paragraphIndentation: { left: 720, hanging: 360 },
217
197
  itemIndex: 0
218
198
  }
219
199
  └── children: [ Text Content... ]
220
200
  ```
221
201
 
222
202
  - **`listId`**: A unique identifier for the list definition. Multiple items with the same `listId` belong to the same logical list.
223
- - **`indentation`**: The nesting level (0-based).
203
+ - **`indentation`**: The structural nesting level (0-based).
204
+ - **`paragraphIndentation`**: The physical indentation formatting in twentieths of a point (twips) (e.g., `left`, `right`, `firstLine`, `hanging`).
224
205
  - **`itemIndex`**: The sequential position within that list level.
225
206
  - **`listType`**: Either `ordered` (numbered) or `unordered` (bulleted).
226
207
 
@@ -299,10 +280,38 @@ Formatting can be found at two levels:
299
280
  1. **Node Level**: Applied directly to a text run or paragraph.
300
281
  2. **Document Level**: Found in `ast.metadata.formatting` (defaults) or `ast.metadata.styleMap` (named styles).
301
282
 
302
- ### 6. Advanced Metadata
283
+ ### 6. Breaks
284
+ Breaks are currently only supported when parsing DOCX-documents. Breaks are added as a node of type `break` and carry metadata of the type `BreakMetadata`. When `includeRawContent` is enabled, they also include the `rawContent` string from the original XML.
285
+
286
+ ```text
287
+ Break Node
288
+ ├── type: "break"
289
+ └── metadata: {
290
+ breakType: "textWrapping" | "page" | "column" | "lastRenderedPage" | "carriageReturn",
291
+ clear?: "all" | "left" | "none" | "right"
292
+ }
293
+ ```
294
+
295
+ - `breakType`: Type of break. `textWrapping` (default) is a standard line break, `page` is a page break, `column` is a break to the next column, `lastRenderedPage` is a soft break inserted by Word, and `carriageReturn` is an explicit carriage return (`w:cr`).
296
+ - `clear`: Relevant for `textWrapping`. Indicates if text should wrap around floating objects.
297
+
298
+ > [!NOTE]
299
+ > Even though break nodes don't have a `text` property, the `ast.toText()` method will automatically convert them to newlines (`\n`) or the configured delimiter in the final string output.
300
+
301
+ ### 7. Advanced Metadata
303
302
  The `ast.metadata` object provides document-wide context:
304
303
  - **`styleMap`**: A dictionary of style names to their `TextFormatting` definitions found in the document.
305
304
  - **`formatting`**: Document-wide default settings (e.g., default font or font size).
305
+ - **`customProperties`**: A dictionary of user-defined metadata embedded in the document (OOXML `custom.xml`, ODF `meta:user-defined`, or PDF Info dictionary).
306
+
307
+ ### 8. Custom Properties
308
+ You can access custom user-defined metadata that might be embedded in the document:
309
+
310
+ ```javascript
311
+ const ast = await officeParser.parseOffice("contract.docx");
312
+ console.log("Custom Metadata:", ast.metadata.customProperties);
313
+ // Output: { "ProjectID": "ABC-123", "InternalReview": true }
314
+ ```
306
315
 
307
316
  ### Advanced AST Usage
308
317
  Beyond using `ast.toText()`, you can interact with the structural data directly:
@@ -403,10 +412,47 @@ Pass an optional config object as the second argument to `parseOffice`.
403
412
  | `ignoreNotes` | boolean | `false` | Ignore notes in files like PowerPoint/ODP. |
404
413
  | `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. (Note: Does not work for RTF. It is treated as true always.) |
405
414
  | `extractAttachments` | boolean | `false` | Extract images and charts as Base64. |
406
- | `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
407
- | `ocrLanguage` | string | `eng` | Language for OCR (e.g., 'eng', 'fra'). Supports multiple languages with '+'. See [Language Codes](https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016). |
408
415
  | `includeRawContent` | boolean | `false` | Include raw XML/RTF markup in the nodes. |
416
+ | `serializeRawContent` | boolean | `true` | When `includeRawContent` is true, re-serializes raw XML to clean strings. If false, extracts original raw substring. |
417
+ | `preserveXmlWhitespace` | boolean | `false` | When `serializeRawContent` is true, preserves original XML whitespace and line endings. |
418
+ | `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
419
+ | `ocrLanguage` | string | `eng` | **Deprecated**: Use `ocrConfig.language` instead. Language for OCR. |
409
420
  | `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. Defaults to a CDN link if not provided. |
421
+ | `ocrConfig` | object | `{}` | **OCR Scheduler** configuration for fine-grained worker control. |
422
+ | `ocrConfig.language` | string | `eng` | Language(s) for OCR (e.g., 'eng', 'fra', 'eng+fra'). |
423
+ | `ocrConfig.autoTerminateTimeout` | number | `10000` | Inactivity timeout in milliseconds before workers are killed. |
424
+ | `ocrConfig.workerPath` | string | `undefined` | Path to Tesseract worker script (for offline use). |
425
+ | `ocrConfig.corePath` | string | `undefined` | Path to Tesseract core script (for offline use). |
426
+ | `ocrConfig.langPath` | string | `undefined` | Path for Tesseract language files (for offline use). |
427
+ | `includeBreakNodes` | boolean | `false` | Specifically targets Word documents (DOCX). When set to true, officeParser will also parse `w:br`, `w:cr` and `w:lastRenderedPageBreak` nodes.|
428
+
429
+ ### OCR Scheduler & Resource Management
430
+ If your application uses OCR, `officeParser` utilizes an intelligent **Smart Worker Pool** to maintain a background worker pool and optimize repeated parse requests.
431
+
432
+ - **Dynamic Affinity**: Workers in the pool persist with their last used language affinity.
433
+ - **LRU Re-allocation**: If a new language is requested and the pool is full, the manager identifies the **Least Recently Used (LRU)** idle worker and re-initializes it for the new language. This avoids the overhead of destroying and recreating workers.
434
+ - **Auto-Termination**: Workers are automatically cleaned up after 10 seconds of inactivity (configurable via `ocrConfig.autoTerminateTimeout`).
435
+
436
+ #### `OfficeParser.terminateOcr()`
437
+ If you have used OCR (`{ ocr: true }`) in a short-lived script (like CLI tools or one-off automation), we recommend explicitly calling `terminateOcr()` after your processing is finished. This bypasses the 10-second idle timer and allows the process to return to the terminal prompt immediately.
438
+
439
+ > [!NOTE]
440
+ > If OCR was not used, this function is a no-op and does not need to be called.
441
+
442
+ ```js
443
+ const officeParser = require('officeparser');
444
+
445
+ async function runCleaner() {
446
+ await officeParser.parseOffice("file.pdf", { ocr: true });
447
+ // ... process results ...
448
+
449
+ // Manually kill OCR workers for an immediate exit
450
+ await officeParser.terminateOcr();
451
+ }
452
+ ```
453
+
454
+ > [!TIP]
455
+ > This is handled automatically in the built-in CLI (`npx officeparser ...`). You only need to call this manually if you are using the library in your own custom script and want a snappy exit.
410
456
 
411
457
  ```js
412
458
  const config = {
@@ -449,29 +495,57 @@ officeParser.parseOffice("presentation.pptx", config).then(ast => {
449
495
  ```
450
496
 
451
497
  ## Browser Usage
452
- The browser bundle exposes the `officeParser` namespace. Include the bundle file available in the release assets.
498
+ The library provides two types of browser bundles in the `dist/` directory:
499
+ 1. **`officeparser.browser.iife.js`**: Standard IIFE bundle for direct `<script>` tag usage. Exposes the global `officeParser` namespace.
500
+ 2. **`officeparser.browser.mjs`**: Modern ESM bundle for use with `import` statements or modern bundlers.
501
+
502
+ ### Usage (ESM)
503
+ If you are using a modern bundler like **Vite**, **Webpack**, or **Next.js**:
504
+
505
+ ```javascript
506
+ import { OfficeParser } from 'officeparser';
507
+
508
+ const handleFile = async (event) => {
509
+ const file = event.target.files[0];
510
+ const buffer = await file.arrayBuffer();
511
+
512
+ try {
513
+ // Pass the Buffer or Uint8Array directly
514
+ const ast = await OfficeParser.parseOffice(new Uint8Array(buffer));
515
+ console.log(ast.toText());
516
+ } catch (err) {
517
+ console.error(err);
518
+ }
519
+ };
520
+ ```
521
+
522
+ > [!NOTE]
523
+ > **Why `fs` fails in the browser**: Browsers do not have a built-in file system. If you try to pass a file path string in the browser, `officeParser` will throw a descriptive "Fail-Fast" error instead of crashing mysteriously:
524
+ > `[officeparser] Node.js 'fs' module is not available in the browser. Please pass a Buffer or Uint8Array instead.`
525
+
526
+ ### Usage (Script Tag)
527
+ Include the IIFE bundle available in the release assets or your `dist/` folder. This exposes the global `officeParser` object.
453
528
 
454
529
  ```html
455
- <script src="dist/officeparser.browser.js"></script>
530
+ <script src="dist/officeparser.browser.iife.js"></script>
456
531
  <script>
457
- async function handleFile(file) {
458
- // file can be a File object from an input element or an ArrayBuffer
459
- // The browser bundle exposes the global variable `officeParser`
460
- // which contains the `OfficeParser` class.
532
+ async function handleFile(event) {
533
+ const file = event.target.files[0];
534
+ const buffer = await file.arrayBuffer();
461
535
 
462
536
  try {
463
- const ast = await officeParser.parseOffice(file, { ocr: true });
537
+ // Reconstruct as Uint8Array for the parser
538
+ const ast = await officeParser.parseOffice(new Uint8Array(buffer));
464
539
  console.log(ast.toText());
465
- console.log("Metadata:", ast.metadata);
466
540
  } catch (error) {
467
- console.error(error);
541
+ console.error("Parsing failed:", error);
468
542
  }
469
543
  }
470
544
  </script>
471
545
  ```
472
546
 
473
547
  ### PDF Worker Configuration in Browser
474
- When using `officeparser` in a browser environment to parse PDF files, you may provide the `pdfWorkerSrc` configuration option. If not provided, it defaults to a CDN link for `pdfjs-dist@5.5.207`.
548
+ When using `officeparser` in a browser environment to parse PDF files, you may provide the `pdfWorkerSrc` configuration option. If not provided, it defaults to a CDN link for `pdfjs-dist@5.6.205`.
475
549
 
476
550
  ```javascript
477
551
  const file = ...; // File object or ArrayBuffer
@@ -481,11 +555,21 @@ const ast = await officeParser.parseOffice(file);
481
555
 
482
556
  // Or override it with your own path or a different version:
483
557
  const ast2 = await officeParser.parseOffice(file, {
484
- pdfWorkerSrc: "https://unpkg.com/pdfjs-dist@5.5.207/build/pdf.worker.min.mjs"
558
+ pdfWorkerSrc: "https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
485
559
  });
486
560
  ```
487
561
 
488
- > **Note:** The version of `pdfjs-dist` in the worker source should match the version used by `officeparser` (currently `5.4.530`).
562
+ > **Note:** The version of `pdfjs-dist` in the worker source should match the version used by `officeparser` (currently `5.6.205`).
563
+
564
+ ## Troubleshooting & Common Issues
565
+
566
+ - **Node.js process stays alive after finishing**: If using OCR, the worker pool stays warm for 10s by default. Use `await terminateOcr()` at the end of your script for a snappy exit.
567
+ - **"Worker not found" in Browser**: Ensure `pdfWorkerSrc` is correctly pointed to the `pdf.worker.min.mjs` file matching version `5.6.205`.
568
+ - **OCR accuracy is low**: Verify your `ocrConfig.language` matches the document content. Note that OCR quality depends on image resolution.
569
+ - **Out of memory on large files**: For massive spreadsheets, consider using `ast.toText()` early and allowing the full AST object to be garbage-collected.
570
+
571
+ For a comprehensive guide, visit our [Debugging & Troubleshooting Documentation](https://harshankur.github.io/officeParser/#spec/debugging).
572
+
489
573
 
490
574
  ## Known Limitations
491
575
  1. **ODT/ODS Charts**: Extraction may occasionally show inaccurate data when referencing external cell ranges or complex layout-based data.
@@ -500,7 +584,7 @@ const ast2 = await officeParser.parseOffice(file, {
500
584
 
501
585
  ## Contributing
502
586
 
503
- We welcome contributions! Please see [CONTRIBUTING.md](CONTRIBUTING.md) for details on how to get started.
587
+ Contributions are welcome! Please see [CONTRIBUTING.md](CONTRIBUTING.md) for details on how to get started.
504
588
 
505
589
  ## License
506
590
 
@@ -32,7 +32,7 @@
32
32
  *
33
33
  * @module OfficeParser
34
34
  */
35
- import { OfficeParserAST, OfficeParserConfig } from './types';
35
+ import { OfficeParserAST, OfficeParserConfig } from './types.js';
36
36
  /**
37
37
  * Main parser class providing office document parsing functionality.
38
38
  *
@@ -86,4 +86,13 @@ export declare class OfficeParser {
86
86
  * ```
87
87
  */
88
88
  static parseOffice(file: string | Buffer | ArrayBuffer, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
89
+ /**
90
+ * Terminates all active OCR workers and cleans up resources.
91
+ *
92
+ * This should be called when the application is shutting down or when OCR
93
+ * is no longer needed to prevent memory leaks and orphaned worker processes.
94
+ *
95
+ * @returns A promise that resolves when all workers have been terminated
96
+ */
97
+ static terminateOcr(): Promise<void>;
89
98
  }
@@ -33,50 +33,18 @@
33
33
  *
34
34
  * @module OfficeParser
35
35
  */
36
- var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
37
- if (k2 === undefined) k2 = k;
38
- var desc = Object.getOwnPropertyDescriptor(m, k);
39
- if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
40
- desc = { enumerable: true, get: function() { return m[k]; } };
41
- }
42
- Object.defineProperty(o, k2, desc);
43
- }) : (function(o, m, k, k2) {
44
- if (k2 === undefined) k2 = k;
45
- o[k2] = m[k];
46
- }));
47
- var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
48
- Object.defineProperty(o, "default", { enumerable: true, value: v });
49
- }) : function(o, v) {
50
- o["default"] = v;
51
- });
52
- var __importStar = (this && this.__importStar) || (function () {
53
- var ownKeys = function(o) {
54
- ownKeys = Object.getOwnPropertyNames || function (o) {
55
- var ar = [];
56
- for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
57
- return ar;
58
- };
59
- return ownKeys(o);
60
- };
61
- return function (mod) {
62
- if (mod && mod.__esModule) return mod;
63
- var result = {};
64
- if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
65
- __setModuleDefault(result, mod);
66
- return result;
67
- };
68
- })();
69
36
  Object.defineProperty(exports, "__esModule", { value: true });
70
37
  exports.OfficeParser = void 0;
71
- const fs = __importStar(require("fs"));
72
- const ExcelParser_1 = require("./parsers/ExcelParser");
73
- const OpenOfficeParser_1 = require("./parsers/OpenOfficeParser");
74
- const PdfParser_1 = require("./parsers/PdfParser");
75
- const PowerPointParser_1 = require("./parsers/PowerPointParser");
76
- const RtfParser_1 = require("./parsers/RtfParser");
77
- const WordParser_1 = require("./parsers/WordParser");
78
- const errorUtils_1 = require("./utils/errorUtils");
79
- const moduleLoader_1 = require("./utils/moduleLoader");
38
+ const envUtils_js_1 = require("./utils/envUtils.js");
39
+ const ExcelParser_js_1 = require("./parsers/ExcelParser.js");
40
+ const OpenOfficeParser_js_1 = require("./parsers/OpenOfficeParser.js");
41
+ const PdfParser_js_1 = require("./parsers/PdfParser.js");
42
+ const PowerPointParser_js_1 = require("./parsers/PowerPointParser.js");
43
+ const RtfParser_js_1 = require("./parsers/RtfParser.js");
44
+ const WordParser_js_1 = require("./parsers/WordParser.js");
45
+ const errorUtils_js_1 = require("./utils/errorUtils.js");
46
+ const moduleLoader_js_1 = require("./utils/moduleLoader.js");
47
+ const ocrUtils_js_1 = require("./utils/ocrUtils.js");
80
48
  /**
81
49
  * Main parser class providing office document parsing functionality.
82
50
  *
@@ -148,7 +116,11 @@ class OfficeParser {
148
116
  ocr: false,
149
117
  ocrLanguage: 'eng',
150
118
  includeRawContent: false,
119
+ serializeRawContent: true,
120
+ preserveXmlWhitespace: false,
151
121
  pdfWorkerSrc: '',
122
+ ocrConfig: {},
123
+ includeBreakNodes: false,
152
124
  ...actualConfig
153
125
  };
154
126
  let buffer = Buffer.alloc(0);
@@ -156,7 +128,7 @@ class OfficeParser {
156
128
  let filePath;
157
129
  try {
158
130
  if (!file) {
159
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
131
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.IMPROPER_ARGUMENTS, internalConfig);
160
132
  }
161
133
  if (file instanceof ArrayBuffer) {
162
134
  buffer = Buffer.from(file);
@@ -166,63 +138,79 @@ class OfficeParser {
166
138
  }
167
139
  else if (typeof file === 'string') {
168
140
  filePath = file;
141
+ (0, envUtils_js_1.assertNode)('path-parsing');
142
+ // Safe to use dynamic import here as we've asserted we are in Node.
143
+ // Modern bundlers will still see this, but our browser builds
144
+ // shim 'fs' so it won't crash at build time.
145
+ const fs = await import('fs');
169
146
  if (!fs.existsSync(file)) {
170
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
147
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST, internalConfig, file);
171
148
  }
172
149
  if (fs.lstatSync(file).isDirectory()) {
173
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
150
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.LOCATION_NOT_FOUND, internalConfig, file);
174
151
  }
175
152
  buffer = fs.readFileSync(file);
176
153
  ext = file.split('.').pop()?.toLowerCase() || '';
177
154
  }
178
155
  else {
179
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.INVALID_INPUT, internalConfig);
156
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.INVALID_INPUT, internalConfig);
180
157
  }
181
158
  if (!ext) {
182
- const { fileTypeFromBuffer } = await (0, moduleLoader_1.loadFileType)();
159
+ const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
183
160
  const type = await fileTypeFromBuffer(buffer);
184
161
  if (type) {
185
162
  ext = type.ext.toLowerCase();
186
163
  }
187
164
  else {
188
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
165
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.IMPROPER_BUFFERS, internalConfig);
189
166
  }
190
167
  }
191
168
  let result;
192
169
  switch (ext) {
193
170
  case 'docx':
194
- result = await (0, WordParser_1.parseWord)(buffer, internalConfig);
171
+ result = await (0, WordParser_js_1.parseWord)(buffer, internalConfig);
195
172
  break;
196
173
  case 'pptx':
197
- result = await (0, PowerPointParser_1.parsePowerPoint)(buffer, internalConfig);
174
+ result = await (0, PowerPointParser_js_1.parsePowerPoint)(buffer, internalConfig);
198
175
  break;
199
176
  case 'xlsx':
200
- result = await (0, ExcelParser_1.parseExcel)(buffer, internalConfig);
177
+ result = await (0, ExcelParser_js_1.parseExcel)(buffer, internalConfig);
201
178
  break;
202
179
  case 'odt':
203
180
  case 'odp':
204
181
  case 'ods':
205
- result = await (0, OpenOfficeParser_1.parseOpenOffice)(buffer, internalConfig);
182
+ result = await (0, OpenOfficeParser_js_1.parseOpenOffice)(buffer, internalConfig);
206
183
  break;
207
184
  case 'pdf':
208
- result = await (0, PdfParser_1.parsePdf)(buffer, internalConfig);
185
+ result = await (0, PdfParser_js_1.parsePdf)(buffer, internalConfig);
209
186
  break;
210
187
  case 'rtf':
211
- result = await (0, RtfParser_1.parseRtf)(buffer, internalConfig);
188
+ result = await (0, RtfParser_js_1.parseRtf)(buffer, internalConfig);
212
189
  break;
213
190
  default:
214
- throw (0, errorUtils_1.getOfficeError)(errorUtils_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
191
+ throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, internalConfig, ext);
215
192
  }
216
193
  if (callback)
217
194
  callback(result);
218
195
  return result;
219
196
  }
220
197
  catch (error) {
221
- const wrappedError = (0, errorUtils_1.getWrappedError)(error, internalConfig, filePath);
198
+ const wrappedError = (0, errorUtils_js_1.getWrappedError)(error, internalConfig, filePath);
222
199
  if (callback)
223
200
  callback(undefined, wrappedError);
224
201
  throw wrappedError;
225
202
  }
226
203
  }
204
+ /**
205
+ * Terminates all active OCR workers and cleans up resources.
206
+ *
207
+ * This should be called when the application is shutting down or when OCR
208
+ * is no longer needed to prevent memory leaks and orphaned worker processes.
209
+ *
210
+ * @returns A promise that resolves when all workers have been terminated
211
+ */
212
+ static async terminateOcr() {
213
+ await (0, ocrUtils_js_1.terminateOcr)();
214
+ }
227
215
  }
228
216
  exports.OfficeParser = OfficeParser;
package/dist/cli.d.ts ADDED
@@ -0,0 +1,20 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * officeparser CLI
4
+ *
5
+ * Allows running officeparser from the command line:
6
+ * npx officeparser file.docx
7
+ * officeparser file.docx --toText=true
8
+ * officeparser file.docx --ocr=true --extractAttachments=true
9
+ *
10
+ * Options (--key=value):
11
+ * --toText=true Output plain text instead of JSON AST
12
+ * --ocr=true Enable OCR for images
13
+ * --ocrLanguage=eng OCR language (default: eng)
14
+ * --extractAttachments=true Extract embedded attachments
15
+ * --ignoreNotes=true Ignore footnotes/endnotes
16
+ * --putNotesAtLast=true Move notes to end of document
17
+ * --includeRawContent=true Include raw content in AST
18
+ * --outputErrorToConsole=true Log errors to console
19
+ */
20
+ export {};
package/dist/cli.js ADDED
@@ -0,0 +1,117 @@
1
+ #!/usr/bin/env node
2
+ "use strict";
3
+ /**
4
+ * officeparser CLI
5
+ *
6
+ * Allows running officeparser from the command line:
7
+ * npx officeparser file.docx
8
+ * officeparser file.docx --toText=true
9
+ * officeparser file.docx --ocr=true --extractAttachments=true
10
+ *
11
+ * Options (--key=value):
12
+ * --toText=true Output plain text instead of JSON AST
13
+ * --ocr=true Enable OCR for images
14
+ * --ocrLanguage=eng OCR language (default: eng)
15
+ * --extractAttachments=true Extract embedded attachments
16
+ * --ignoreNotes=true Ignore footnotes/endnotes
17
+ * --putNotesAtLast=true Move notes to end of document
18
+ * --includeRawContent=true Include raw content in AST
19
+ * --outputErrorToConsole=true Log errors to console
20
+ */
21
+ Object.defineProperty(exports, "__esModule", { value: true });
22
+ const OfficeParser_js_1 = require("./OfficeParser.js");
23
+ const args = process.argv.slice(2);
24
+ let fileArg;
25
+ let toText = false;
26
+ let verbose = false;
27
+ const configArgs = [];
28
+ function isConfigOption(arg) {
29
+ return arg.startsWith('--') && arg.includes('=');
30
+ }
31
+ args.forEach(arg => {
32
+ if (isConfigOption(arg)) {
33
+ configArgs.push(arg);
34
+ }
35
+ else if (!fileArg) {
36
+ fileArg = arg;
37
+ }
38
+ });
39
+ if (fileArg) {
40
+ const config = {};
41
+ configArgs.forEach(arg => {
42
+ const [key, value] = arg.split('=');
43
+ const cleanKey = key.replace('--', '');
44
+ const lowerValue = value.toLowerCase();
45
+ const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
46
+ if (cleanKey === 'toText') {
47
+ if (boolValue !== undefined)
48
+ toText = boolValue;
49
+ else
50
+ console.warn(`Invalid value for toText: ${value}`);
51
+ }
52
+ else if (cleanKey === 'verbose') {
53
+ if (boolValue !== undefined)
54
+ verbose = boolValue;
55
+ else
56
+ console.warn(`Invalid value for verbose: ${value}`);
57
+ }
58
+ else {
59
+ // @ts-ignore
60
+ if (boolValue !== undefined)
61
+ config[cleanKey] = boolValue;
62
+ // @ts-ignore
63
+ else
64
+ config[cleanKey] = value;
65
+ }
66
+ });
67
+ OfficeParser_js_1.OfficeParser.parseOffice(fileArg, config)
68
+ .then(async (ast) => {
69
+ if (toText) {
70
+ process.stdout.write(ast.toText() + '\n');
71
+ }
72
+ else {
73
+ process.stdout.write(JSON.stringify(ast, null, 2) + '\n');
74
+ }
75
+ // Ensure OCR workers are terminated for clean CLI exit
76
+ if (config.ocr) {
77
+ await OfficeParser_js_1.OfficeParser.terminateOcr();
78
+ }
79
+ })
80
+ .catch(async (err) => {
81
+ console.error(`Error parsing file "${fileArg}":`);
82
+ if (verbose) {
83
+ console.error(err);
84
+ }
85
+ else {
86
+ console.error(err.message || err);
87
+ console.error('Use --verbose=true for full stack trace.');
88
+ }
89
+ // Ensure OCR workers are terminated even on error
90
+ if (config.ocr) {
91
+ await OfficeParser_js_1.OfficeParser.terminateOcr();
92
+ }
93
+ process.exit(1);
94
+ });
95
+ }
96
+ else {
97
+ console.log('Usage: officeparser <file> [--option=value]');
98
+ console.log('');
99
+ console.log('Options:');
100
+ console.log(' --toText=true Output plain text instead of JSON AST');
101
+ console.log(' --ocr=true Enable OCR for images');
102
+ console.log(' --ocrLanguage=eng OCR language (default: eng)');
103
+ console.log(' --extractAttachments=true Extract embedded attachments');
104
+ console.log(' --ignoreNotes=true Ignore footnotes/endnotes');
105
+ console.log(' --putNotesAtLast=true Move notes to end of document');
106
+ console.log(' --includeRawContent=true Include raw content in AST');
107
+ console.log(' --serializeRawContent=true Serialize raw XML content (default: true)');
108
+ console.log(' --preserveXmlWhitespace=true Preserve whitespace in serialized XML (default: false)');
109
+ console.log(' --includeBreakNodes=false Include break nodes (DOCX only, default: false)');
110
+ console.log(' --verbose=true Show full error stack traces');
111
+ console.log('');
112
+ console.log('Examples:');
113
+ console.log(' officeparser document.docx');
114
+ console.log(' officeparser document.docx --toText=true');
115
+ console.log(' officeparser report.pdf --ocr=true --extractAttachments=true');
116
+ console.log(' officeparser complex.docx --serializeRawContent=false --includeRawContent=true');
117
+ }