officeparser 6.1.1 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +219 -26
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -29
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +106 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +826 -5
  32. package/dist/officeparser.browser.iife.js +703 -52
  33. package/dist/officeparser.browser.mjs +703 -52
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +140 -79
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +20 -23
  48. package/dist/parsers/RtfParser.d.ts +2 -2
  49. package/dist/parsers/RtfParser.js +1291 -1240
  50. package/dist/parsers/WordParser.d.ts +2 -2
  51. package/dist/parsers/WordParser.js +232 -97
  52. package/dist/sbom.cdx.json +99 -99
  53. package/dist/types.d.ts +781 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +27 -8
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
- # officeParser 📄🚀
1
+ # officeParser 📄🚀 - The Most Versatile Office Parser & Generator
2
2
 
3
- A robust, strictly-typed Node.js and Browser library for parsing office files ([`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format)). It produces a clean, hierarchical Abstract Syntax Tree (AST) with rich metadata, text formatting, and full attachment support.
3
+ A robust, strictly-typed Node.js and Browser library for parsing and generating office files. It not only extracts content from [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format), [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values), [`md`](https://en.wikipedia.org/wiki/Markdown), and [`html`](https://en.wikipedia.org/wiki/HTML) into a rich Abstract Syntax Tree (AST), but also provides a powerful generation engine to convert that AST into formats like **Markdown**, **HTML**, **CSV**, **RTF**, **Text**, **PDF**, and **JSON**, including native **RAG-focused chunking** support.
4
4
 
5
5
  [![npm version](https://badge.fury.io/js/officeparser.svg)](https://badge.fury.io/js/officeparser)
6
6
  [![Total Downloads](https://img.shields.io/npm/dt/officeparser.svg)](https://www.npmjs.com/package/officeparser)
@@ -18,8 +18,6 @@ A robust, strictly-typed Node.js and Browser library for parsing office files ([
18
18
  - **Debugging**: Use the visualizer to debug parsing issues by inspecting exactly how nodes are interpreted.
19
19
  - **Format Specs**: Read detailed specifications for the AST structure and configuration options.
20
20
 
21
- *(Legacy Visualizer: If you prefer the [old simple visualizer](https://harshankur.github.io/officeParser/visualizer_old.html), it is still available.)*
22
-
23
21
  ---
24
22
 
25
23
 
@@ -37,7 +35,7 @@ npm i officeparser
37
35
  ```
38
36
 
39
37
  ## Command Line usage
40
- You can use `officeparser` directly from the terminal to get either the full AST (as JSON) or plain text.
38
+ You can use `officeparser` directly from the terminal to extract content as JSON AST, plain text, or generate new formats like Markdown and HTML.
41
39
 
42
40
  ```bash
43
41
  # Get full AST as JSON (default)
@@ -46,25 +44,33 @@ npx officeparser /path/to/officeFile.docx
46
44
  # Get plain text only
47
45
  npx officeparser /path/to/officeFile.docx --toText=true
48
46
 
49
- # Use configuration options
50
- npx officeparser /path/to/officeFile.docx --ignoreNotes=true --newlineDelimiter=" "
47
+ # Generate Markdown file
48
+ npx officeparser report.docx --format=md --output=report.md
49
+
50
+ # Generate HTML file with specific output
51
+ npx officeparser presentation.pptx --format=html --output=preview.html
52
+
53
+ # Convert spreadsheet to CSV
54
+ npx officeparser data.xlsx --format=csv
51
55
  ```
52
56
 
53
57
  ### Config Options:
54
- - `--toText=[true|false]` Flag to output only plain text instead of JSON AST.
58
+ - `--format=[json|text|md|html|csv|rtf|pdf|chunks]` The output format. Default is `json`.
59
+ - `--output=[path]` Optional file path to write the output to.
60
+ - `--toText=[true|false]` Legacy flag to output only plain text. Use `--format=text` instead.
55
61
  - `--ignoreNotes=[true|false]` Flag to ignore notes from files like PowerPoint. Default is false.
56
62
  - `--newlineDelimiter=[delimiter]` The delimiter to use for new lines. Default is `\n`.
57
63
  - `--putNotesAtLast=[true|false]` Flag to collect notes at the end of files like PowerPoint. Default is false.
58
- - `--outputErrorToConsole=[true|false]` Flag to output errors to the console. Default is false.
64
+ - `--outputErrorToConsole=[true|false]` **(Deprecated)** Flag to output errors to the console. Use `onWarning` callback in library usage.
59
65
  - `--extractAttachments=[true|false]` Flag to extract images/charts as Base64. Default is false.
60
66
  - `--ocr=[true|false]` Flag to enable OCR for extracted images. Default is false.
61
67
  - `--includeRawContent=[true|false]` Flag to include raw XML/RTF content in nodes. Default is false.
62
- - `--includeBreakNodes=[true|false]` Flag to include break nodes. Currently only available for DOCX documents
68
+ - `--includeBreakNodes=[true|false]` Flag to include break nodes. Currently only available for DOCX documents.
63
69
  - `--verbose=[true|false]` Show full error stack traces.
64
70
 
65
71
 
66
72
  ## Library Usage
67
- In **v6.0.0**, the library has moved to a structured AST output. While this is a change for those expecting a string directly, it provides significantly more power and flexibility.
73
+ In **v7.0.0**, the library has evolved into a dual-purpose **Parser** and **Generator**. You can first parse any office file into a structured AST and then use the `OfficeGenerator` to transform that AST into various formats or chunks.
68
74
 
69
75
  ### Getting Started (Async/Await)
70
76
  ```js
@@ -99,6 +105,101 @@ const text = await getText("/path/to/officeFile.docx");
99
105
  console.log(text);
100
106
  ```
101
107
 
108
+ ## Using the OfficeGenerator
109
+ The `OfficeGenerator` is a powerful tool to convert your AST into human-readable formats or structured data.
110
+
111
+ ```typescript
112
+ import { OfficeParser, OfficeGenerator } from 'officeparser';
113
+
114
+ const ast = await OfficeParser.parseOffice('report.docx');
115
+
116
+ // 1. Convert to Markdown
117
+ const md = await OfficeGenerator.generate(ast, 'md');
118
+ console.log(md.value);
119
+
120
+ // 2. Convert to HTML with structured style mapping (Recommended)
121
+ const html = await OfficeGenerator.generate(ast, 'html', {
122
+ includeFormatting: true,
123
+ styleMap: [
124
+ {
125
+ selector: { nodeType: 'paragraph', attributes: { style: 'Heading 1' } },
126
+ output: { tag: 'h1', classes: ['main-title'] }
127
+ }
128
+ ]
129
+ });
130
+ console.log(html.value);
131
+
132
+ // 3. Convert to CSV (for spreadsheets)
133
+ const csv = await OfficeGenerator.generate(ast, 'csv');
134
+ console.log(csv.value);
135
+ ```
136
+
137
+ ## The New "One-Step" API: `OfficeConverter`
138
+ In **v7.0.0**, we introduced the `OfficeConverter.convert` method. This is the new high-level API designed for one-step transformations where you don't need to manually interact with the AST. It automatically handles parser and generator configuration synchronization.
139
+
140
+ ```typescript
141
+ import { OfficeConverter } from 'officeparser';
142
+
143
+ // One-step conversion from DOCX to Markdown
144
+ const result = await OfficeConverter.convert('report.docx', 'md');
145
+ console.log(result.value); // The generated Markdown string
146
+ console.log(result.messages); // Array of warnings/info (e.g., "Skipped unsupported drawing")
147
+
148
+ // Complex conversion with nested configuration
149
+ const htmlResult = await OfficeConverter.convert('data.xlsx', 'html', {
150
+ parseConfig: {
151
+ ignoreNotes: true
152
+ },
153
+ generatorConfig: {
154
+ includeFormatting: true,
155
+ styleMap: [
156
+ {
157
+ selector: { attributes: { style: { value: 'Header', operator: '~=' } } },
158
+ output: { tag: 'h2', classes: ['data-header'] }
159
+ }
160
+ ]
161
+ },
162
+ onWarning: (msg) => console.warn("Conversion Warning:", msg)
163
+ });
164
+ ```
165
+
166
+ ## Native RAG Chunking
167
+ `officeParser` provides native support for document chunking, specifically designed for Retrieval-Augmented Generation (RAG) workflows. It offers three distinct strategies to split your documents while maintaining context and metadata.
168
+
169
+ ### 1. Fixed-Size Strategy (Recursive)
170
+ Splits text into chunks based on character count with a specified overlap. It uses smart boundary detection to avoid cutting in the middle of sentences or paragraphs.
171
+
172
+ ### 2. Document Structure Strategy
173
+ Splits the document at natural structural boundaries like pages (PDF/Word), slides (PPTX), or high-level headings. This preserves the logical flow of the document.
174
+
175
+ ### 3. Semantic Strategy
176
+ Uses cosine similarity between sentence embeddings to identify coherent topic boundaries. This ensures that each chunk contains semantically related content (requires an embedding function).
177
+
178
+ ### The `OfficeChunk` Interface
179
+ Every chunk produced contains not just text, but rich metadata to help your RAG pipeline:
180
+ ```typescript
181
+ {
182
+ text: string; // The chunk content
183
+ metadata: {
184
+ sourceType: string; // e.g., "docx", "pdf"
185
+ pageNumber?: number; // Current page
186
+ slideNumber?: number; // Current slide
187
+ closestHeading?: string; // The heading this chunk belongs to
188
+ chunkIndex: number; // Sequential index
189
+ }
190
+ }
191
+ ```
192
+
193
+ #### Example: Generating Chunks
194
+ ```typescript
195
+ const chunks = await OfficeGenerator.generate(ast, 'chunks', {
196
+ strategy: 'fixed-size',
197
+ maxChunkSize: 1000,
198
+ chunkOverlap: 200
199
+ });
200
+ console.log(`Generated ${chunks.value.length} chunks`);
201
+ ```
202
+
102
203
  ### Using Callbacks (Backward Compatibility Support)
103
204
  Callbacks are still supported for those preferred, but the data returned is now the AST object.
104
205
  ```js
@@ -134,7 +235,7 @@ The `OfficeParserAST` provides a format-agnostic representation of your document
134
235
 
135
236
  ```text
136
237
  OfficeParserAST
137
- ├── type: "docx" | "pptx" | "xlsx" | ...
238
+ ├── type: "docx" | "pdf" | "xlsx" | "csv" | "md" | ... (11 formats supported)
138
239
  ├── metadata: { author, title, created, modified, ..., customProperties }
139
240
  ├── content: [ OfficeContentNode ]
140
241
  │ ├── type: "paragraph" | "heading" | "table" | "list" | ...
@@ -313,6 +414,14 @@ console.log("Custom Metadata:", ast.metadata.customProperties);
313
414
  // Output: { "ProjectID": "ABC-123", "InternalReview": true }
314
415
  ```
315
416
 
417
+ ## Performance & Fidelity Highlights (v7.0.0)
418
+ The v7.0.0 release brings significant internal optimizations and fidelity improvements:
419
+ - **OpenOffice Speedups**: Up to **23x faster** parsing for ODP presentations thanks to optimized XML caching.
420
+ - **Excel Memory Efficiency**: Resolved $O(n)$ memory overhead issues for large spreadsheets (#91) by switching to iterative stream-based parsing.
421
+ - **RTF Performance**: Rewritten core loop to resolve $O(n^2)$ bottlenecks during string accumulation.
422
+ - **Advanced Table Fidelity**: Native support for **vertical cell merging** (`vMerge`) and **horizontal spanning** (`gridSpan`) in DOCX, ensuring complex tables look exactly as they do in Word.
423
+ - **Parser Extensions**: You can now parse `CSV`, `Markdown`, and `HTML` files *into* the unified Office AST, allowing you to use the `OfficeGenerator` on them just like any other format.
424
+
316
425
  ### Advanced AST Usage
317
426
  Beyond using `ast.toText()`, you can interact with the structural data directly:
318
427
 
@@ -407,24 +516,108 @@ Pass an optional config object as the second argument to `parseOffice`.
407
516
 
408
517
  | Flag | DataType | Default | Explanation |
409
518
  |------|----------|---------|-------------|
410
- | `outputErrorToConsole` | boolean | `false` | Show logs to console in case of an error. |
519
+ | `outputErrorToConsole` | boolean | `false` | **Deprecated**: Use `onWarning` instead. Show logs to console in case of an error. |
411
520
  | `newlineDelimiter` | string | `\n` | Delimiter for new lines in text output. |
412
521
  | `ignoreNotes` | boolean | `false` | Ignore notes in files like PowerPoint/ODP. |
413
- | `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. (Note: Does not work for RTF. It is treated as true always.) |
522
+ | `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. |
414
523
  | `extractAttachments` | boolean | `false` | Extract images and charts as Base64. |
415
524
  | `includeRawContent` | boolean | `false` | Include raw XML/RTF markup in the nodes. |
416
- | `serializeRawContent` | boolean | `true` | When `includeRawContent` is true, re-serializes raw XML to clean strings. If false, extracts original raw substring. |
417
- | `preserveXmlWhitespace` | boolean | `false` | When `serializeRawContent` is true, preserves original XML whitespace and line endings. |
525
+ | `serializeRawContent` | boolean | `true` | Re-serializes raw XML to clean strings. |
526
+ | `preserveXmlWhitespace` | boolean | `false` | Preserves original XML whitespace. |
418
527
  | `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
419
- | `ocrLanguage` | string | `eng` | **Deprecated**: Use `ocrConfig.language` instead. Language for OCR. |
420
- | `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. Defaults to a CDN link if not provided. |
421
- | `ocrConfig` | object | `{}` | **OCR Scheduler** configuration for fine-grained worker control. |
422
- | `ocrConfig.language` | string | `eng` | Language(s) for OCR (e.g., 'eng', 'fra', 'eng+fra'). |
423
- | `ocrConfig.autoTerminateTimeout` | number | `10000` | Inactivity timeout in milliseconds before workers are killed. |
424
- | `ocrConfig.workerPath` | string | `undefined` | Path to Tesseract worker script (for offline use). |
425
- | `ocrConfig.corePath` | string | `undefined` | Path to Tesseract core script (for offline use). |
426
- | `ocrConfig.langPath` | string | `undefined` | Path for Tesseract language files (for offline use). |
427
- | `includeBreakNodes` | boolean | `false` | Specifically targets Word documents (DOCX). When set to true, officeParser will also parse `w:br`, `w:cr` and `w:lastRenderedPageBreak` nodes.|
528
+ | `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. |
529
+ | `ocrConfig` | object | `{}` | OCR Scheduler configuration. |
530
+ | `includeBreakNodes` | boolean | `false` | Include `w:br`, `w:cr` nodes (DOCX only).|
531
+ | `ignoreInternalLinks` | boolean | `false` | Remove all bookmarks and internal jumps. |
532
+ | `csvDelimiter` | string | `,` | Custom delimiter for parsing CSV files. |
533
+ | `fileType` | string | `null` | Manual format override (authoritative). |
534
+
535
+ ## Generator Configuration: GeneratorConfig
536
+ Configuration options for `OfficeGenerator.generate`.
537
+
538
+ | Flag | DataType | Default | Explanation |
539
+ |------|----------|---------|-------------|
540
+ | `includeFormatting` | boolean | `false` | Whether to include semantic styles (bold, italic) in output. |
541
+ | `styleMap` | string[] \| array | `[]` | Array of style mappings (DSL strings or structured objects). |
542
+ | `ignoreDefaultStyleMap`| boolean | `false` | Ignore the library's default style mappings. |
543
+ | `includeMetadata` | boolean | `false` | Include document metadata in the output (e.g., as frontmatter). |
544
+ | `onNode` | function | `undefined` | Callback to intercept/modify any node during generation. |
545
+
546
+ ### 🛠️ Advanced Node Manipulation (Pro Users)
547
+ The `onNode` callback is a powerful tool that gives you complete control over the generation process. It is called for **every single node** in the AST before it is rendered.
548
+
549
+ #### Callback Capabilities:
550
+ 1. **Filter/Remove Nodes**: Return `false` to skip a node and all its children.
551
+ 2. **Override Rendering**: Return a `string` to use that exact text as the output, bypassing default logic and recursion.
552
+ 3. **Mutate Nodes**: Modify the `node` object directly (e.g., changing `node.text`) and return `void` to let the generator proceed with your changes.
553
+ 4. **Async Support**: The callback can be `async`, allowing you to fetch external data or perform complex logic during generation.
554
+
555
+ #### Pro Example:
556
+ ```typescript
557
+ const result = await ast.to('md', {
558
+ onNode: async (node) => {
559
+ // 1. Skip all images
560
+ if (node.type === 'image') return false;
561
+
562
+ // 2. Redact sensitive info by mutating the node
563
+ if (node.text?.includes('SECRET_KEY')) {
564
+ node.text = node.text.replace(/SECRET_KEY: \w+/, 'SECRET_KEY: [REDACTED]');
565
+ }
566
+
567
+ // 3. Custom rendering for specific styles
568
+ if (node.metadata?.style === 'Callout') {
569
+ return `> [!INFO]\n> ${node.text}`;
570
+ }
571
+
572
+ // 4. Proceed with default rendering (implicitly returns void)
573
+ }
574
+ });
575
+ ```
576
+
577
+ ### Advanced Style Mapping (Semantic Translation)
578
+ The `styleMap` configuration is the primary way to define the "semantic meaning" of document styles. We recommend using **Structured Style Mappings** for full type safety and power.
579
+
580
+ #### 1. Structured Style Mappings (Recommended)
581
+ Use structured objects to match nodes based on type and attributes, and specify detailed output properties like classes and custom attributes.
582
+
583
+ ```typescript
584
+ styleMap: [
585
+ {
586
+ selector: {
587
+ nodeType: 'paragraph',
588
+ attributes: { style: 'Heading 1' }
589
+ },
590
+ output: {
591
+ tag: 'h1',
592
+ classes: ['main-title'],
593
+ attributes: { id: 'top' }
594
+ }
595
+ },
596
+ {
597
+ // Use operators like '~=' for partial matches
598
+ selector: { attributes: { style: { value: 'Quote', operator: '~=' } } },
599
+ output: { tag: 'blockquote' }
600
+ }
601
+ ]
602
+ ```
603
+
604
+ #### 2. Legacy String DSL
605
+ The library also maintains support for a simple string-based DSL, highly compatible with `mammoth.js`.
606
+
607
+ - **Literal Matching**: `"p[style-name='Heading 1'] => h1"`
608
+ - **Regex-like Matching**: `"p[style~='Title'] => h2"`
609
+ - **Attribute Filters**: `"p[style-name='Quote'][lang='en'] => blockquote"`
610
+
611
+ ## Chunking Configuration: ChunkingConfig
612
+ Specific options when using `format: 'chunks'`.
613
+
614
+ | Flag | DataType | Default | Explanation |
615
+ |------|----------|---------|-------------|
616
+ | `strategy` | string | `'fixed-size'`| The chunking strategy (`fixed-size`, `document-structure`, `semantic`). |
617
+ | `maxChunkSize` | number | `1000` | Maximum characters per chunk. |
618
+ | `chunkOverlap` | number | `200` | Overlap between consecutive chunks. |
619
+ | `similarityThreshold`| number | `0.5` | Threshold for semantic splitting (0.0 to 1.0). |
620
+ | `embedBatchSize` | number | `50` | Batch size for embedding requests. |
428
621
 
429
622
  ### OCR Scheduler & Resource Management
430
623
  If your application uses OCR, `officeParser` utilizes an intelligent **Smart Worker Pool** to maintain a background worker pool and optimize repeated parse requests.
@@ -555,7 +748,7 @@ const ast = await officeParser.parseOffice(file);
555
748
 
556
749
  // Or override it with your own path or a different version:
557
750
  const ast2 = await officeParser.parseOffice(file, {
558
- pdfWorkerSrc: "https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
751
+ pdfWorkerSrc: "https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
559
752
  });
560
753
  ```
561
754
 
@@ -0,0 +1,46 @@
1
+ import { ConversionResult, OfficeConverterConfig, SupportedDestination, SupportedFileType } from './types.js';
2
+ /**
3
+ * Utility type to infer the file type from a file path string literal.
4
+ */
5
+ type InferFileTypeFromPath<T> = T extends `${string}.${infer E}` ? (Lowercase<E> extends SupportedFileType ? Lowercase<E> : SupportedFileType) : SupportedFileType;
6
+ /**
7
+ * Main converter class providing a streamlined one-step API for document conversion.
8
+ *
9
+ * This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
10
+ * documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
11
+ */
12
+ export declare class OfficeConverter {
13
+ /**
14
+ * Converts an office document from its source format to a specified destination format.
15
+ *
16
+ * This method:
17
+ * 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
18
+ * 2. Automatically configures the parser based on the generator requirements (e.g., enabling
19
+ * attachment extraction if images are requested in the output).
20
+ * 3. Generates the destination document from the AST using `OfficeGenerator`.
21
+ *
22
+ * @template F The inferred type of the input file (path string or buffer).
23
+ * @template T The authoritative source file type (inferred from path or config).
24
+ *
25
+ * @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
26
+ * @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
27
+ * @param config - Optional unified configuration for both the parser and generator phases.
28
+ *
29
+ * @returns A promise resolving to the ConversionResult containing the value and messages.
30
+ * @throws {Error} If the source format is unsupported or parsing/generation fails.
31
+ *
32
+ * @example
33
+ * ```typescript
34
+ * // Convert Word to Markdown with a single call
35
+ * const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
36
+ *
37
+ * // Convert PDF to HTML with OCR enabled for images
38
+ * const { value: html } = await OfficeConverter.convert(buffer, 'html', {
39
+ * ocr: true,
40
+ * includeImages: true
41
+ * });
42
+ * ```
43
+ */
44
+ static convert<F extends string | Buffer | ArrayBuffer, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
45
+ }
46
+ export {};
@@ -0,0 +1,72 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.OfficeConverter = void 0;
4
+ const OfficeGenerator_js_1 = require("./OfficeGenerator.js");
5
+ const OfficeParser_js_1 = require("./OfficeParser.js");
6
+ /**
7
+ * Main converter class providing a streamlined one-step API for document conversion.
8
+ *
9
+ * This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
10
+ * documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
11
+ */
12
+ class OfficeConverter {
13
+ /**
14
+ * Converts an office document from its source format to a specified destination format.
15
+ *
16
+ * This method:
17
+ * 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
18
+ * 2. Automatically configures the parser based on the generator requirements (e.g., enabling
19
+ * attachment extraction if images are requested in the output).
20
+ * 3. Generates the destination document from the AST using `OfficeGenerator`.
21
+ *
22
+ * @template F The inferred type of the input file (path string or buffer).
23
+ * @template T The authoritative source file type (inferred from path or config).
24
+ *
25
+ * @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
26
+ * @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
27
+ * @param config - Optional unified configuration for both the parser and generator phases.
28
+ *
29
+ * @returns A promise resolving to the ConversionResult containing the value and messages.
30
+ * @throws {Error} If the source format is unsupported or parsing/generation fails.
31
+ *
32
+ * @example
33
+ * ```typescript
34
+ * // Convert Word to Markdown with a single call
35
+ * const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
36
+ *
37
+ * // Convert PDF to HTML with OCR enabled for images
38
+ * const { value: html } = await OfficeConverter.convert(buffer, 'html', {
39
+ * ocr: true,
40
+ * includeImages: true
41
+ * });
42
+ * ```
43
+ */
44
+ static async convert(file, destination, config) {
45
+ // 1. Prepare Parser Configuration
46
+ // We prioritize the top-level onWarning if provided.
47
+ const parserConfig = {
48
+ ...config?.parseConfig,
49
+ onWarning: config?.onWarning || config?.parseConfig?.onWarning,
50
+ };
51
+ // Remove OCR settings for the streamlined converter as requested
52
+ parserConfig.ocr = false;
53
+ // Remove undefined keys to prevent overwriting defaults in resolveParserConfig
54
+ Object.keys(parserConfig).forEach((key) => parserConfig[key] === undefined && delete parserConfig[key]);
55
+ /**
56
+ * AUTOMATIC CONFIGURATION SYNC
57
+ * We sync extractAttachments from the generator configuration.
58
+ */
59
+ parserConfig.extractAttachments = (config?.generatorConfig?.includeImages !== false) || (config?.generatorConfig?.includeCharts !== false);
60
+ // 2. Parse the source document into the universal AST
61
+ const ast = await OfficeParser_js_1.OfficeParser.parseOffice(file, parserConfig);
62
+ // 3. Generate the destination document from the AST
63
+ const generatorConfig = {
64
+ ...config?.generatorConfig,
65
+ onWarning: config?.onWarning || config?.generatorConfig?.onWarning,
66
+ };
67
+ const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, destination, generatorConfig);
68
+ result.messages = [...(ast.warnings || []), ...result.messages];
69
+ return result;
70
+ }
71
+ }
72
+ exports.OfficeConverter = OfficeConverter;
@@ -0,0 +1,19 @@
1
+ import { ConversionResult, GeneratorConfig, OfficeParserAST, SupportedDestination, SupportedFileType } from './types.js';
2
+ /**
3
+ * Main generator class providing document conversion functionality.
4
+ */
5
+ export declare class OfficeGenerator {
6
+ /**
7
+ * Generates a file of the specified type from an AST.
8
+ * This is the single source of truth for generation logic.
9
+ *
10
+ * @param ast - The OfficeParserAST to generate from
11
+ * @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
12
+ * @param config - Optional configuration for the generator
13
+ * @returns A promise resolving to the ConversionResult containing the value and messages
14
+ * @throws {Error} If the destination format is unsupported
15
+ */
16
+ static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
17
+ type: T;
18
+ }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
19
+ }
@@ -0,0 +1,48 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.OfficeGenerator = void 0;
4
+ const ChunkingGenerator_js_1 = require("./generators/ChunkingGenerator.js");
5
+ const CsvGenerator_js_1 = require("./generators/CsvGenerator.js");
6
+ const HtmlGenerator_js_1 = require("./generators/HtmlGenerator.js");
7
+ const MarkdownGenerator_js_1 = require("./generators/MarkdownGenerator.js");
8
+ const PdfGenerator_js_1 = require("./generators/PdfGenerator.js");
9
+ const RtfGenerator_js_1 = require("./generators/RtfGenerator.js");
10
+ const TextGenerator_js_1 = require("./generators/TextGenerator.js");
11
+ const types_js_1 = require("./types.js");
12
+ const errorUtils_js_1 = require("./utils/errorUtils.js");
13
+ /**
14
+ * Main generator class providing document conversion functionality.
15
+ */
16
+ class OfficeGenerator {
17
+ /**
18
+ * Generates a file of the specified type from an AST.
19
+ * This is the single source of truth for generation logic.
20
+ *
21
+ * @param ast - The OfficeParserAST to generate from
22
+ * @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
23
+ * @param config - Optional configuration for the generator
24
+ * @returns A promise resolving to the ConversionResult containing the value and messages
25
+ * @throws {Error} If the destination format is unsupported
26
+ */
27
+ static async generate(ast, destination, config) {
28
+ switch (destination.toLowerCase()) {
29
+ case 'text':
30
+ return new TextGenerator_js_1.TextGenerator(ast, config).generate();
31
+ case 'md':
32
+ return new MarkdownGenerator_js_1.MarkdownGenerator(ast, config).generate();
33
+ case 'html':
34
+ return new HtmlGenerator_js_1.HtmlGenerator(ast, config).generate();
35
+ case 'pdf':
36
+ return new PdfGenerator_js_1.PdfGenerator(ast, config).generate();
37
+ case 'csv':
38
+ return new CsvGenerator_js_1.CsvGenerator(ast, config).generate();
39
+ case 'rtf':
40
+ return new RtfGenerator_js_1.RtfGenerator(ast, config).generate();
41
+ case 'chunks':
42
+ return new ChunkingGenerator_js_1.ChunkingGenerator(ast, config).generate();
43
+ default:
44
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, undefined, destination);
45
+ }
46
+ }
47
+ }
48
+ exports.OfficeGenerator = OfficeGenerator;
@@ -11,6 +11,9 @@
11
11
  * - ODT, ODP, ODS (OpenDocument formats)
12
12
  * - PDF (Portable Document Format)
13
13
  * - RTF (Rich Text Format)
14
+ * - CSV (Comma-Separated Values)
15
+ * - MD (Markdown)
16
+ * - HTML (HyperText Markup Language)
14
17
  *
15
18
  * **Usage:**
16
19
  * ```typescript
@@ -60,6 +63,9 @@ export declare class OfficeParser {
60
63
  * - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
61
64
  * - `.pdf` → PdfParser (PDF.js)
62
65
  * - `.rtf` → RtfParser (custom RTF parser)
66
+ * - `.csv` → CsvParser
67
+ * - `.md` → MarkdownParser
68
+ * - `.html` → HtmlParser
63
69
  *
64
70
  * @param file - File path (string), Buffer, or ArrayBuffer containing the document
65
71
  * @param config - Optional configuration object (defaults applied for all omitted options)