officeparser 6.1.0 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +284 -86
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -28
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +107 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +878 -5
  32. package/dist/officeparser.browser.iife.js +703 -49
  33. package/dist/officeparser.browser.mjs +703 -49
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +237 -128
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +132 -123
  48. package/dist/parsers/RtfParser.d.ts +22 -2
  49. package/dist/parsers/RtfParser.js +1398 -1282
  50. package/dist/parsers/WordParser.d.ts +3 -2
  51. package/dist/parsers/WordParser.js +333 -115
  52. package/dist/sbom.cdx.json +103 -103
  53. package/dist/types.d.ts +833 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +28 -9
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
- # officeParser 📄🚀
1
+ # officeParser 📄🚀 - The Most Versatile Office Parser & Generator
2
2
 
3
- A robust, strictly-typed Node.js and Browser library for parsing office files ([`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format)). It produces a clean, hierarchical Abstract Syntax Tree (AST) with rich metadata, text formatting, and full attachment support.
3
+ A robust, strictly-typed Node.js and Browser library for parsing and generating office files. It not only extracts content from [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format), [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values), [`md`](https://en.wikipedia.org/wiki/Markdown), and [`html`](https://en.wikipedia.org/wiki/HTML) into a rich Abstract Syntax Tree (AST), but also provides a powerful generation engine to convert that AST into formats like **Markdown**, **HTML**, **CSV**, **RTF**, **Text**, **PDF**, and **JSON**, including native **RAG-focused chunking** support.
4
4
 
5
5
  [![npm version](https://badge.fury.io/js/officeparser.svg)](https://badge.fury.io/js/officeparser)
6
6
  [![Total Downloads](https://img.shields.io/npm/dt/officeparser.svg)](https://www.npmjs.com/package/officeparser)
@@ -18,49 +18,15 @@ A robust, strictly-typed Node.js and Browser library for parsing office files ([
18
18
  - **Debugging**: Use the visualizer to debug parsing issues by inspecting exactly how nodes are interpreted.
19
19
  - **Format Specs**: Read detailed specifications for the AST structure and configuration options.
20
20
 
21
- *(Legacy Visualizer: If you prefer the [old simple visualizer](https://harshankur.github.io/officeParser/visualizer_old.html), it is still available.)*
21
+ ---
22
+
22
23
 
23
24
  ---
24
25
 
26
+ ### 📝 [Changelog](CHANGELOG.md)
27
+ *Detailed release notes and the full history of updates are available in the project changelog.*
25
28
 
26
- #### Update
27
- * 2026-04-14 - **v6.1.0 Release**: Major Infrastructure & Resource Stability. (Incremental since v6.0.0)
28
- - **OCR Scheduler**: Intelligent worker pool that optimizes Tesseract lifecycle across parallel requests. **Note**: By default, Node.js processes stay active for 10s after OCR to keep workers warm (configurable via `ocrConfig.autoTerminateTimeout`); use `terminateOcr()` for immediate CLI/script exit.
29
- - **Core Engine**: Replaced legacy zip extraction with `fflate` for significant performance gains and robust browser/edge compatibility.
30
- - **Module System**: Full native ESM support with `Node16` resolution and verified browser bundles (Vite/Angular compatible).
31
- - **Format Refinements**: Hierarchical PDF coordinate alignment and ODT/RTF list parsing stability.
32
- - **Custom Properties**: Added support for extracting custom document metadata across OOXML, ODF, and PDF formats.
33
- - **Sponsorship**: Integrated `funding.json` manifest and GitHub Sponsors support.
34
- * 2025/12/29 - **v6.0.0 Release**: Major overhaul of the library. Transitioned from simple text extraction to a rich **Abstract Syntax Tree (AST)** output.
35
- - Simplified API: Use `parseOffice` for all parsing needs (returns a Promise).
36
- - Structured Output: Access hierarchical document structure (paragraphs, headings, tables, lists, etc.).
37
- - Rich Metadata: Extracted document properties (author, title, creation date).
38
- - Enhanced Formatting: Support for bold, italic, colors, fonts, alignment, etc.
39
- - Attachment Handling: Extract images, charts, and embedded files as Base64.
40
- - OCR Integration: Optional OCR for images using Tesseract.js.
41
- - RTF Support: Added full support for Rich Text Format files.
42
- - Improved Type Definitions: Full TypeScript support with detailed interfaces.
43
- * 2024/11/12 - Added ArrayBuffer as a type of file input. Generating bundle files now which exposes namespace officeParser to be able to access parseOffice directly on the browser.
44
- * 2024/10/21 - Replaced extracting zip files from decompress to yauzl. This means that we now extract files in memory and we no longer need to write them to disk. Removed config flags related to extracted files. Added flags for CLI execution.
45
- * 2024/10/15 - Fixed erroring out while deleting temp files when multiple worker threads make parallel executions resulting in same file name for multiple files. Fixed erroring out when multiple executions are made without waiting for the previous execution to finish which resulted in deleting the file from other execution. Upgraded dependencies.
46
- * 2024/10/13 - Fixed parsing text from xlsx files which contain no shared strings file and files which have inlineStr based strings.
47
- * 2024/05/06 - Replaced pdf parsing support from pdf-parse library to natively building it using pdf.js library from Mozilla by analyzing its output. Added pdfjs-dist build as a local library.
48
- * 2023/11/25 - Fixed error catching when an error occurs within the parsing of a file, especially after decompressing it. Also fixed the problem with parallel parsing of files as we were using only timestamp in file names.
49
- * 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
50
- * 2023/04/26 - Added support for file buffers as argument for filepath for parseOffice and parseOfficeAsync
51
- * 2023/04/07 - Added typings to methods to help with Typescript projects.
52
- * 2022/12/28 - Added command line method to use officeParser with or without installing it and instantly get parsed content on the console.
53
- * 2022/12/10 - Fixed memory leak issues, bugs related to parsing open document files and improved error handling.
54
- * 2021/11/21 - Added promise way to existing callback functions.
55
- * 2020/06/01 - Added error handling and console.log enable/disable methods. Default is set at enabled. Everything backward compatible.
56
- * 2019/06/17 - Added method to change location for decompressing office files in places with restricted write access.
57
- * 2019/04/30 - Removed case sensitive file extension bug. File names with capital lettered extensions now supported.
58
- * 2019/04/23 - Added support for open office files *.odt, *.odp, *.ods through parseOffice function. Created a new method parseOpenOffice for those who prefer targetted functions.
59
- * 2019/04/23 - Added feature to delete the generated dist folder after function callback.
60
- * 2019/04/22 - Added parseOffice method to avoid confusion between type of file and their extension.
61
- * 2019/04/22 - Added file extension validations. Removed errors for excel files with no drawing elements.
62
- * 2019/04/19 - Support added for *.xlsx files.
63
- * 2019/04/18 - Support added for *.pptx files.
29
+ ---
64
30
 
65
31
  ## Install via npm
66
32
 
@@ -69,7 +35,7 @@ npm i officeparser
69
35
  ```
70
36
 
71
37
  ## Command Line usage
72
- You can use `officeparser` directly from the terminal to get either the full AST (as JSON) or plain text.
38
+ You can use `officeparser` directly from the terminal to extract content as JSON AST, plain text, or generate new formats like Markdown and HTML.
73
39
 
74
40
  ```bash
75
41
  # Get full AST as JSON (default)
@@ -78,24 +44,33 @@ npx officeparser /path/to/officeFile.docx
78
44
  # Get plain text only
79
45
  npx officeparser /path/to/officeFile.docx --toText=true
80
46
 
81
- # Use configuration options
82
- npx officeparser /path/to/officeFile.docx --ignoreNotes=true --newlineDelimiter=" "
47
+ # Generate Markdown file
48
+ npx officeparser report.docx --format=md --output=report.md
49
+
50
+ # Generate HTML file with specific output
51
+ npx officeparser presentation.pptx --format=html --output=preview.html
52
+
53
+ # Convert spreadsheet to CSV
54
+ npx officeparser data.xlsx --format=csv
83
55
  ```
84
56
 
85
57
  ### Config Options:
86
- - `--toText=[true|false]` Flag to output only plain text instead of JSON AST.
58
+ - `--format=[json|text|md|html|csv|rtf|pdf|chunks]` The output format. Default is `json`.
59
+ - `--output=[path]` Optional file path to write the output to.
60
+ - `--toText=[true|false]` Legacy flag to output only plain text. Use `--format=text` instead.
87
61
  - `--ignoreNotes=[true|false]` Flag to ignore notes from files like PowerPoint. Default is false.
88
62
  - `--newlineDelimiter=[delimiter]` The delimiter to use for new lines. Default is `\n`.
89
63
  - `--putNotesAtLast=[true|false]` Flag to collect notes at the end of files like PowerPoint. Default is false.
90
- - `--outputErrorToConsole=[true|false]` Flag to output errors to the console. Default is false.
64
+ - `--outputErrorToConsole=[true|false]` **(Deprecated)** Flag to output errors to the console. Use `onWarning` callback in library usage.
91
65
  - `--extractAttachments=[true|false]` Flag to extract images/charts as Base64. Default is false.
92
66
  - `--ocr=[true|false]` Flag to enable OCR for extracted images. Default is false.
93
67
  - `--includeRawContent=[true|false]` Flag to include raw XML/RTF content in nodes. Default is false.
68
+ - `--includeBreakNodes=[true|false]` Flag to include break nodes. Currently only available for DOCX documents.
94
69
  - `--verbose=[true|false]` Show full error stack traces.
95
70
 
96
71
 
97
72
  ## Library Usage
98
- In **v6.0.0**, the library has moved to a structured AST output. While this is a change for those expecting a string directly, it provides significantly more power and flexibility.
73
+ In **v7.0.0**, the library has evolved into a dual-purpose **Parser** and **Generator**. You can first parse any office file into a structured AST and then use the `OfficeGenerator` to transform that AST into various formats or chunks.
99
74
 
100
75
  ### Getting Started (Async/Await)
101
76
  ```js
@@ -130,8 +105,103 @@ const text = await getText("/path/to/officeFile.docx");
130
105
  console.log(text);
131
106
  ```
132
107
 
108
+ ## Using the OfficeGenerator
109
+ The `OfficeGenerator` is a powerful tool to convert your AST into human-readable formats or structured data.
110
+
111
+ ```typescript
112
+ import { OfficeParser, OfficeGenerator } from 'officeparser';
113
+
114
+ const ast = await OfficeParser.parseOffice('report.docx');
115
+
116
+ // 1. Convert to Markdown
117
+ const md = await OfficeGenerator.generate(ast, 'md');
118
+ console.log(md.value);
119
+
120
+ // 2. Convert to HTML with structured style mapping (Recommended)
121
+ const html = await OfficeGenerator.generate(ast, 'html', {
122
+ includeFormatting: true,
123
+ styleMap: [
124
+ {
125
+ selector: { nodeType: 'paragraph', attributes: { style: 'Heading 1' } },
126
+ output: { tag: 'h1', classes: ['main-title'] }
127
+ }
128
+ ]
129
+ });
130
+ console.log(html.value);
131
+
132
+ // 3. Convert to CSV (for spreadsheets)
133
+ const csv = await OfficeGenerator.generate(ast, 'csv');
134
+ console.log(csv.value);
135
+ ```
136
+
137
+ ## The New "One-Step" API: `OfficeConverter`
138
+ In **v7.0.0**, we introduced the `OfficeConverter.convert` method. This is the new high-level API designed for one-step transformations where you don't need to manually interact with the AST. It automatically handles parser and generator configuration synchronization.
139
+
140
+ ```typescript
141
+ import { OfficeConverter } from 'officeparser';
142
+
143
+ // One-step conversion from DOCX to Markdown
144
+ const result = await OfficeConverter.convert('report.docx', 'md');
145
+ console.log(result.value); // The generated Markdown string
146
+ console.log(result.messages); // Array of warnings/info (e.g., "Skipped unsupported drawing")
147
+
148
+ // Complex conversion with nested configuration
149
+ const htmlResult = await OfficeConverter.convert('data.xlsx', 'html', {
150
+ parseConfig: {
151
+ ignoreNotes: true
152
+ },
153
+ generatorConfig: {
154
+ includeFormatting: true,
155
+ styleMap: [
156
+ {
157
+ selector: { attributes: { style: { value: 'Header', operator: '~=' } } },
158
+ output: { tag: 'h2', classes: ['data-header'] }
159
+ }
160
+ ]
161
+ },
162
+ onWarning: (msg) => console.warn("Conversion Warning:", msg)
163
+ });
164
+ ```
165
+
166
+ ## Native RAG Chunking
167
+ `officeParser` provides native support for document chunking, specifically designed for Retrieval-Augmented Generation (RAG) workflows. It offers three distinct strategies to split your documents while maintaining context and metadata.
168
+
169
+ ### 1. Fixed-Size Strategy (Recursive)
170
+ Splits text into chunks based on character count with a specified overlap. It uses smart boundary detection to avoid cutting in the middle of sentences or paragraphs.
171
+
172
+ ### 2. Document Structure Strategy
173
+ Splits the document at natural structural boundaries like pages (PDF/Word), slides (PPTX), or high-level headings. This preserves the logical flow of the document.
174
+
175
+ ### 3. Semantic Strategy
176
+ Uses cosine similarity between sentence embeddings to identify coherent topic boundaries. This ensures that each chunk contains semantically related content (requires an embedding function).
177
+
178
+ ### The `OfficeChunk` Interface
179
+ Every chunk produced contains not just text, but rich metadata to help your RAG pipeline:
180
+ ```typescript
181
+ {
182
+ text: string; // The chunk content
183
+ metadata: {
184
+ sourceType: string; // e.g., "docx", "pdf"
185
+ pageNumber?: number; // Current page
186
+ slideNumber?: number; // Current slide
187
+ closestHeading?: string; // The heading this chunk belongs to
188
+ chunkIndex: number; // Sequential index
189
+ }
190
+ }
191
+ ```
192
+
193
+ #### Example: Generating Chunks
194
+ ```typescript
195
+ const chunks = await OfficeGenerator.generate(ast, 'chunks', {
196
+ strategy: 'fixed-size',
197
+ maxChunkSize: 1000,
198
+ chunkOverlap: 200
199
+ });
200
+ console.log(`Generated ${chunks.value.length} chunks`);
201
+ ```
202
+
133
203
  ### Using Callbacks (Backward Compatibility Support)
134
- We still support callbacks, but the data returned is now the AST object.
204
+ Callbacks are still supported for those preferred, but the data returned is now the AST object.
135
205
  ```js
136
206
  const officeParser = require('officeparser');
137
207
 
@@ -165,14 +235,14 @@ The `OfficeParserAST` provides a format-agnostic representation of your document
165
235
 
166
236
  ```text
167
237
  OfficeParserAST
168
- ├── type: "docx" | "pptx" | "xlsx" | ...
238
+ ├── type: "docx" | "pdf" | "xlsx" | "csv" | "md" | ... (11 formats supported)
169
239
  ├── metadata: { author, title, created, modified, ..., customProperties }
170
240
  ├── content: [ OfficeContentNode ]
171
241
  │ ├── type: "paragraph" | "heading" | "table" | "list" | ...
172
242
  │ ├── text: "Concatenated text of this node and all children"
173
243
  │ ├── children: [ OfficeContentNode ] (recursive)
174
244
  │ ├── formatting: { bold, italic, color, size, font, ... }
175
- │ ├── metadata: { level, listId, row, col, ... }
245
+ │ ├── metadata: { level, listId, paragraphIndentation, row, col, ... }
176
246
  │ └── rawContent: "<xml>...</xml>" (if enabled)
177
247
  ├── attachments: [ OfficeAttachment ]
178
248
  │ ├── type: "image" | "chart"
@@ -224,13 +294,15 @@ List Node
224
294
  listId: "1",
225
295
  listType: "ordered",
226
296
  indentation: 0,
297
+ paragraphIndentation: { left: 720, hanging: 360 },
227
298
  itemIndex: 0
228
299
  }
229
300
  └── children: [ Text Content... ]
230
301
  ```
231
302
 
232
303
  - **`listId`**: A unique identifier for the list definition. Multiple items with the same `listId` belong to the same logical list.
233
- - **`indentation`**: The nesting level (0-based).
304
+ - **`indentation`**: The structural nesting level (0-based).
305
+ - **`paragraphIndentation`**: The physical indentation formatting in twentieths of a point (twips) (e.g., `left`, `right`, `firstLine`, `hanging`).
234
306
  - **`itemIndex`**: The sequential position within that list level.
235
307
  - **`listType`**: Either `ordered` (numbered) or `unordered` (bulleted).
236
308
 
@@ -309,13 +381,31 @@ Formatting can be found at two levels:
309
381
  1. **Node Level**: Applied directly to a text run or paragraph.
310
382
  2. **Document Level**: Found in `ast.metadata.formatting` (defaults) or `ast.metadata.styleMap` (named styles).
311
383
 
312
- ### 6. Advanced Metadata
384
+ ### 6. Breaks
385
+ Breaks are currently only supported when parsing DOCX-documents. Breaks are added as a node of type `break` and carry metadata of the type `BreakMetadata`. When `includeRawContent` is enabled, they also include the `rawContent` string from the original XML.
386
+
387
+ ```text
388
+ Break Node
389
+ ├── type: "break"
390
+ └── metadata: {
391
+ breakType: "textWrapping" | "page" | "column" | "lastRenderedPage" | "carriageReturn",
392
+ clear?: "all" | "left" | "none" | "right"
393
+ }
394
+ ```
395
+
396
+ - `breakType`: Type of break. `textWrapping` (default) is a standard line break, `page` is a page break, `column` is a break to the next column, `lastRenderedPage` is a soft break inserted by Word, and `carriageReturn` is an explicit carriage return (`w:cr`).
397
+ - `clear`: Relevant for `textWrapping`. Indicates if text should wrap around floating objects.
398
+
399
+ > [!NOTE]
400
+ > Even though break nodes don't have a `text` property, the `ast.toText()` method will automatically convert them to newlines (`\n`) or the configured delimiter in the final string output.
401
+
402
+ ### 7. Advanced Metadata
313
403
  The `ast.metadata` object provides document-wide context:
314
404
  - **`styleMap`**: A dictionary of style names to their `TextFormatting` definitions found in the document.
315
405
  - **`formatting`**: Document-wide default settings (e.g., default font or font size).
316
406
  - **`customProperties`**: A dictionary of user-defined metadata embedded in the document (OOXML `custom.xml`, ODF `meta:user-defined`, or PDF Info dictionary).
317
407
 
318
- ### 7. Custom Properties
408
+ ### 8. Custom Properties
319
409
  You can access custom user-defined metadata that might be embedded in the document:
320
410
 
321
411
  ```javascript
@@ -324,6 +414,14 @@ console.log("Custom Metadata:", ast.metadata.customProperties);
324
414
  // Output: { "ProjectID": "ABC-123", "InternalReview": true }
325
415
  ```
326
416
 
417
+ ## Performance & Fidelity Highlights (v7.0.0)
418
+ The v7.0.0 release brings significant internal optimizations and fidelity improvements:
419
+ - **OpenOffice Speedups**: Up to **23x faster** parsing for ODP presentations thanks to optimized XML caching.
420
+ - **Excel Memory Efficiency**: Resolved $O(n)$ memory overhead issues for large spreadsheets (#91) by switching to iterative stream-based parsing.
421
+ - **RTF Performance**: Rewritten core loop to resolve $O(n^2)$ bottlenecks during string accumulation.
422
+ - **Advanced Table Fidelity**: Native support for **vertical cell merging** (`vMerge`) and **horizontal spanning** (`gridSpan`) in DOCX, ensuring complex tables look exactly as they do in Word.
423
+ - **Parser Extensions**: You can now parse `CSV`, `Markdown`, and `HTML` files *into* the unified Office AST, allowing you to use the `OfficeGenerator` on them just like any other format.
424
+
327
425
  ### Advanced AST Usage
328
426
  Beyond using `ast.toText()`, you can interact with the structural data directly:
329
427
 
@@ -418,29 +516,114 @@ Pass an optional config object as the second argument to `parseOffice`.
418
516
 
419
517
  | Flag | DataType | Default | Explanation |
420
518
  |------|----------|---------|-------------|
421
- | `outputErrorToConsole` | boolean | `false` | Show logs to console in case of an error. |
519
+ | `outputErrorToConsole` | boolean | `false` | **Deprecated**: Use `onWarning` instead. Show logs to console in case of an error. |
422
520
  | `newlineDelimiter` | string | `\n` | Delimiter for new lines in text output. |
423
521
  | `ignoreNotes` | boolean | `false` | Ignore notes in files like PowerPoint/ODP. |
424
- | `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. (Note: Does not work for RTF. It is treated as true always.) |
522
+ | `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. |
425
523
  | `extractAttachments` | boolean | `false` | Extract images and charts as Base64. |
426
524
  | `includeRawContent` | boolean | `false` | Include raw XML/RTF markup in the nodes. |
427
- | `serializeRawContent` | boolean | `true` | When `includeRawContent` is true, re-serializes raw XML to clean strings. If false, extracts original raw substring. |
428
- | `preserveXmlWhitespace` | boolean | `false` | When `serializeRawContent` is true, preserves original XML whitespace and line endings. |
525
+ | `serializeRawContent` | boolean | `true` | Re-serializes raw XML to clean strings. |
526
+ | `preserveXmlWhitespace` | boolean | `false` | Preserves original XML whitespace. |
429
527
  | `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
430
- | `ocrLanguage` | string | `eng` | **Deprecated**: Use `ocrConfig.language` instead. Language for OCR. |
431
- | `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. Defaults to a CDN link if not provided. |
432
- | `ocrConfig` | object | `{}` | **OCR Scheduler** configuration for fine-grained worker control. |
433
- | `ocrConfig.language` | string | `eng` | Language(s) for OCR (e.g., 'eng', 'fra', 'eng+fra'). |
434
- | `ocrConfig.autoTerminateTimeout` | number | `10000` | Inactivity timeout in milliseconds before workers are killed. |
435
- | `ocrConfig.workerPath` | string | `undefined` | Path to Tesseract worker script (for offline use). |
436
- | `ocrConfig.corePath` | string | `undefined` | Path to Tesseract core script (for offline use). |
437
- | `ocrConfig.langPath` | string | `undefined` | Path for Tesseract language files (for offline use). |
528
+ | `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. |
529
+ | `ocrConfig` | object | `{}` | OCR Scheduler configuration. |
530
+ | `includeBreakNodes` | boolean | `false` | Include `w:br`, `w:cr` nodes (DOCX only).|
531
+ | `ignoreInternalLinks` | boolean | `false` | Remove all bookmarks and internal jumps. |
532
+ | `csvDelimiter` | string | `,` | Custom delimiter for parsing CSV files. |
533
+ | `fileType` | string | `null` | Manual format override (authoritative). |
534
+
535
+ ## Generator Configuration: GeneratorConfig
536
+ Configuration options for `OfficeGenerator.generate`.
537
+
538
+ | Flag | DataType | Default | Explanation |
539
+ |------|----------|---------|-------------|
540
+ | `includeFormatting` | boolean | `false` | Whether to include semantic styles (bold, italic) in output. |
541
+ | `styleMap` | string[] \| array | `[]` | Array of style mappings (DSL strings or structured objects). |
542
+ | `ignoreDefaultStyleMap`| boolean | `false` | Ignore the library's default style mappings. |
543
+ | `includeMetadata` | boolean | `false` | Include document metadata in the output (e.g., as frontmatter). |
544
+ | `onNode` | function | `undefined` | Callback to intercept/modify any node during generation. |
545
+
546
+ ### 🛠️ Advanced Node Manipulation (Pro Users)
547
+ The `onNode` callback is a powerful tool that gives you complete control over the generation process. It is called for **every single node** in the AST before it is rendered.
548
+
549
+ #### Callback Capabilities:
550
+ 1. **Filter/Remove Nodes**: Return `false` to skip a node and all its children.
551
+ 2. **Override Rendering**: Return a `string` to use that exact text as the output, bypassing default logic and recursion.
552
+ 3. **Mutate Nodes**: Modify the `node` object directly (e.g., changing `node.text`) and return `void` to let the generator proceed with your changes.
553
+ 4. **Async Support**: The callback can be `async`, allowing you to fetch external data or perform complex logic during generation.
554
+
555
+ #### Pro Example:
556
+ ```typescript
557
+ const result = await ast.to('md', {
558
+ onNode: async (node) => {
559
+ // 1. Skip all images
560
+ if (node.type === 'image') return false;
561
+
562
+ // 2. Redact sensitive info by mutating the node
563
+ if (node.text?.includes('SECRET_KEY')) {
564
+ node.text = node.text.replace(/SECRET_KEY: \w+/, 'SECRET_KEY: [REDACTED]');
565
+ }
566
+
567
+ // 3. Custom rendering for specific styles
568
+ if (node.metadata?.style === 'Callout') {
569
+ return `> [!INFO]\n> ${node.text}`;
570
+ }
571
+
572
+ // 4. Proceed with default rendering (implicitly returns void)
573
+ }
574
+ });
575
+ ```
576
+
577
+ ### Advanced Style Mapping (Semantic Translation)
578
+ The `styleMap` configuration is the primary way to define the "semantic meaning" of document styles. We recommend using **Structured Style Mappings** for full type safety and power.
579
+
580
+ #### 1. Structured Style Mappings (Recommended)
581
+ Use structured objects to match nodes based on type and attributes, and specify detailed output properties like classes and custom attributes.
582
+
583
+ ```typescript
584
+ styleMap: [
585
+ {
586
+ selector: {
587
+ nodeType: 'paragraph',
588
+ attributes: { style: 'Heading 1' }
589
+ },
590
+ output: {
591
+ tag: 'h1',
592
+ classes: ['main-title'],
593
+ attributes: { id: 'top' }
594
+ }
595
+ },
596
+ {
597
+ // Use operators like '~=' for partial matches
598
+ selector: { attributes: { style: { value: 'Quote', operator: '~=' } } },
599
+ output: { tag: 'blockquote' }
600
+ }
601
+ ]
602
+ ```
603
+
604
+ #### 2. Legacy String DSL
605
+ The library also maintains support for a simple string-based DSL, highly compatible with `mammoth.js`.
606
+
607
+ - **Literal Matching**: `"p[style-name='Heading 1'] => h1"`
608
+ - **Regex-like Matching**: `"p[style~='Title'] => h2"`
609
+ - **Attribute Filters**: `"p[style-name='Quote'][lang='en'] => blockquote"`
610
+
611
+ ## Chunking Configuration: ChunkingConfig
612
+ Specific options when using `format: 'chunks'`.
613
+
614
+ | Flag | DataType | Default | Explanation |
615
+ |------|----------|---------|-------------|
616
+ | `strategy` | string | `'fixed-size'`| The chunking strategy (`fixed-size`, `document-structure`, `semantic`). |
617
+ | `maxChunkSize` | number | `1000` | Maximum characters per chunk. |
618
+ | `chunkOverlap` | number | `200` | Overlap between consecutive chunks. |
619
+ | `similarityThreshold`| number | `0.5` | Threshold for semantic splitting (0.0 to 1.0). |
620
+ | `embedBatchSize` | number | `50` | Batch size for embedding requests. |
438
621
 
439
622
  ### OCR Scheduler & Resource Management
440
623
  If your application uses OCR, `officeParser` utilizes an intelligent **Smart Worker Pool** to maintain a background worker pool and optimize repeated parse requests.
441
624
 
442
625
  - **Dynamic Affinity**: Workers in the pool persist with their last used language affinity.
443
- - **Smart Re-initialization**: If a new language is requested and the pool is full, the manager identifies the **Least Recently Used (LRU)** idle worker and re-initializes it for the new language using the Tesseract.js v5 API. This avoids the overhead of destroying and recreating workers.
626
+ - **LRU Re-allocation**: If a new language is requested and the pool is full, the manager identifies the **Least Recently Used (LRU)** idle worker and re-initializes it for the new language. This avoids the overhead of destroying and recreating workers.
444
627
  - **Auto-Termination**: Workers are automatically cleaned up after 10 seconds of inactivity (configurable via `ocrConfig.autoTerminateTimeout`).
445
628
 
446
629
  #### `OfficeParser.terminateOcr()`
@@ -462,7 +645,7 @@ async function runCleaner() {
462
645
  ```
463
646
 
464
647
  > [!TIP]
465
- > This is handled automatically in our own CLI (`npx officeparser ...`). You only need to call this manually if you are using the library in your own custom script and want a snappy exit.
648
+ > This is handled automatically in the built-in CLI (`npx officeparser ...`). You only need to call this manually if you are using the library in your own custom script and want a snappy exit.
466
649
 
467
650
  ```js
468
651
  const config = {
@@ -509,36 +692,51 @@ The library provides two types of browser bundles in the `dist/` directory:
509
692
  1. **`officeparser.browser.iife.js`**: Standard IIFE bundle for direct `<script>` tag usage. Exposes the global `officeParser` namespace.
510
693
  2. **`officeparser.browser.mjs`**: Modern ESM bundle for use with `import` statements or modern bundlers.
511
694
 
695
+ ### Usage (ESM)
696
+ If you are using a modern bundler like **Vite**, **Webpack**, or **Next.js**:
697
+
698
+ ```javascript
699
+ import { OfficeParser } from 'officeparser';
700
+
701
+ const handleFile = async (event) => {
702
+ const file = event.target.files[0];
703
+ const buffer = await file.arrayBuffer();
704
+
705
+ try {
706
+ // Pass the Buffer or Uint8Array directly
707
+ const ast = await OfficeParser.parseOffice(new Uint8Array(buffer));
708
+ console.log(ast.toText());
709
+ } catch (err) {
710
+ console.error(err);
711
+ }
712
+ };
713
+ ```
714
+
715
+ > [!NOTE]
716
+ > **Why `fs` fails in the browser**: Browsers do not have a built-in file system. If you try to pass a file path string in the browser, `officeParser` will throw a descriptive "Fail-Fast" error instead of crashing mysteriously:
717
+ > `[officeparser] Node.js 'fs' module is not available in the browser. Please pass a Buffer or Uint8Array instead.`
718
+
512
719
  ### Usage (Script Tag)
513
- Include the IIFE bundle file available in the release assets.
720
+ Include the IIFE bundle available in the release assets or your `dist/` folder. This exposes the global `officeParser` object.
514
721
 
515
722
  ```html
516
723
  <script src="dist/officeparser.browser.iife.js"></script>
517
724
  <script>
518
- async function handleFile(file) {
519
- // file can be a File object from an input element or an ArrayBuffer
725
+ async function handleFile(event) {
726
+ const file = event.target.files[0];
727
+ const buffer = await file.arrayBuffer();
728
+
520
729
  try {
521
- const ast = await officeParser.parseOffice(file, { ocr: true });
730
+ // Reconstruct as Uint8Array for the parser
731
+ const ast = await officeParser.parseOffice(new Uint8Array(buffer));
522
732
  console.log(ast.toText());
523
733
  } catch (error) {
524
- console.error(error);
734
+ console.error("Parsing failed:", error);
525
735
  }
526
736
  }
527
737
  </script>
528
738
  ```
529
739
 
530
- ### Usage (ESM)
531
- If you are using a modern browser that supports modules or a dev server like Vite:
532
-
533
- ```html
534
- <script type="module">
535
- import { OfficeParser } from './dist/officeparser.browser.mjs';
536
-
537
- const ast = await OfficeParser.parseOffice(fileBuffer);
538
- console.log(ast.metadata);
539
- </script>
540
- ```
541
-
542
740
  ### PDF Worker Configuration in Browser
543
741
  When using `officeparser` in a browser environment to parse PDF files, you may provide the `pdfWorkerSrc` configuration option. If not provided, it defaults to a CDN link for `pdfjs-dist@5.6.205`.
544
742
 
@@ -550,7 +748,7 @@ const ast = await officeParser.parseOffice(file);
550
748
 
551
749
  // Or override it with your own path or a different version:
552
750
  const ast2 = await officeParser.parseOffice(file, {
553
- pdfWorkerSrc: "https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
751
+ pdfWorkerSrc: "https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
554
752
  });
555
753
  ```
556
754
 
@@ -579,7 +777,7 @@ For a comprehensive guide, visit our [Debugging & Troubleshooting Documentation]
579
777
 
580
778
  ## Contributing
581
779
 
582
- We welcome contributions! Please see [CONTRIBUTING.md](CONTRIBUTING.md) for details on how to get started.
780
+ Contributions are welcome! Please see [CONTRIBUTING.md](CONTRIBUTING.md) for details on how to get started.
583
781
 
584
782
  ## License
585
783
 
@@ -0,0 +1,46 @@
1
+ import { ConversionResult, OfficeConverterConfig, SupportedDestination, SupportedFileType } from './types.js';
2
+ /**
3
+ * Utility type to infer the file type from a file path string literal.
4
+ */
5
+ type InferFileTypeFromPath<T> = T extends `${string}.${infer E}` ? (Lowercase<E> extends SupportedFileType ? Lowercase<E> : SupportedFileType) : SupportedFileType;
6
+ /**
7
+ * Main converter class providing a streamlined one-step API for document conversion.
8
+ *
9
+ * This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
10
+ * documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
11
+ */
12
+ export declare class OfficeConverter {
13
+ /**
14
+ * Converts an office document from its source format to a specified destination format.
15
+ *
16
+ * This method:
17
+ * 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
18
+ * 2. Automatically configures the parser based on the generator requirements (e.g., enabling
19
+ * attachment extraction if images are requested in the output).
20
+ * 3. Generates the destination document from the AST using `OfficeGenerator`.
21
+ *
22
+ * @template F The inferred type of the input file (path string or buffer).
23
+ * @template T The authoritative source file type (inferred from path or config).
24
+ *
25
+ * @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
26
+ * @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
27
+ * @param config - Optional unified configuration for both the parser and generator phases.
28
+ *
29
+ * @returns A promise resolving to the ConversionResult containing the value and messages.
30
+ * @throws {Error} If the source format is unsupported or parsing/generation fails.
31
+ *
32
+ * @example
33
+ * ```typescript
34
+ * // Convert Word to Markdown with a single call
35
+ * const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
36
+ *
37
+ * // Convert PDF to HTML with OCR enabled for images
38
+ * const { value: html } = await OfficeConverter.convert(buffer, 'html', {
39
+ * ocr: true,
40
+ * includeImages: true
41
+ * });
42
+ * ```
43
+ */
44
+ static convert<F extends string | Buffer | ArrayBuffer, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
45
+ }
46
+ export {};
@@ -0,0 +1,72 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.OfficeConverter = void 0;
4
+ const OfficeGenerator_js_1 = require("./OfficeGenerator.js");
5
+ const OfficeParser_js_1 = require("./OfficeParser.js");
6
+ /**
7
+ * Main converter class providing a streamlined one-step API for document conversion.
8
+ *
9
+ * This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
10
+ * documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
11
+ */
12
+ class OfficeConverter {
13
+ /**
14
+ * Converts an office document from its source format to a specified destination format.
15
+ *
16
+ * This method:
17
+ * 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
18
+ * 2. Automatically configures the parser based on the generator requirements (e.g., enabling
19
+ * attachment extraction if images are requested in the output).
20
+ * 3. Generates the destination document from the AST using `OfficeGenerator`.
21
+ *
22
+ * @template F The inferred type of the input file (path string or buffer).
23
+ * @template T The authoritative source file type (inferred from path or config).
24
+ *
25
+ * @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
26
+ * @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
27
+ * @param config - Optional unified configuration for both the parser and generator phases.
28
+ *
29
+ * @returns A promise resolving to the ConversionResult containing the value and messages.
30
+ * @throws {Error} If the source format is unsupported or parsing/generation fails.
31
+ *
32
+ * @example
33
+ * ```typescript
34
+ * // Convert Word to Markdown with a single call
35
+ * const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
36
+ *
37
+ * // Convert PDF to HTML with OCR enabled for images
38
+ * const { value: html } = await OfficeConverter.convert(buffer, 'html', {
39
+ * ocr: true,
40
+ * includeImages: true
41
+ * });
42
+ * ```
43
+ */
44
+ static async convert(file, destination, config) {
45
+ // 1. Prepare Parser Configuration
46
+ // We prioritize the top-level onWarning if provided.
47
+ const parserConfig = {
48
+ ...config?.parseConfig,
49
+ onWarning: config?.onWarning || config?.parseConfig?.onWarning,
50
+ };
51
+ // Remove OCR settings for the streamlined converter as requested
52
+ parserConfig.ocr = false;
53
+ // Remove undefined keys to prevent overwriting defaults in resolveParserConfig
54
+ Object.keys(parserConfig).forEach((key) => parserConfig[key] === undefined && delete parserConfig[key]);
55
+ /**
56
+ * AUTOMATIC CONFIGURATION SYNC
57
+ * We sync extractAttachments from the generator configuration.
58
+ */
59
+ parserConfig.extractAttachments = (config?.generatorConfig?.includeImages !== false) || (config?.generatorConfig?.includeCharts !== false);
60
+ // 2. Parse the source document into the universal AST
61
+ const ast = await OfficeParser_js_1.OfficeParser.parseOffice(file, parserConfig);
62
+ // 3. Generate the destination document from the AST
63
+ const generatorConfig = {
64
+ ...config?.generatorConfig,
65
+ onWarning: config?.onWarning || config?.generatorConfig?.onWarning,
66
+ };
67
+ const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, destination, generatorConfig);
68
+ result.messages = [...(ast.warnings || []), ...result.messages];
69
+ return result;
70
+ }
71
+ }
72
+ exports.OfficeConverter = OfficeConverter;