officeparser 6.1.1 → 7.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +219 -26
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +55 -29
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +106 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +826 -5
- package/dist/officeparser.browser.iife.js +703 -52
- package/dist/officeparser.browser.mjs +703 -52
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +140 -79
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +20 -23
- package/dist/parsers/RtfParser.d.ts +2 -2
- package/dist/parsers/RtfParser.js +1291 -1240
- package/dist/parsers/WordParser.d.ts +2 -2
- package/dist/parsers/WordParser.js +232 -97
- package/dist/sbom.cdx.json +99 -99
- package/dist/types.d.ts +781 -5
- package/dist/types.js +71 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.js +56 -2
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +109 -52
- package/dist/utils/moduleLoader.js +15 -9
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +27 -8
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
# officeParser 📄🚀
|
|
1
|
+
# officeParser 📄🚀 - The Most Versatile Office Parser & Generator
|
|
2
2
|
|
|
3
|
-
A robust, strictly-typed Node.js and Browser library for parsing office files
|
|
3
|
+
A robust, strictly-typed Node.js and Browser library for parsing and generating office files. It not only extracts content from [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format), [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values), [`md`](https://en.wikipedia.org/wiki/Markdown), and [`html`](https://en.wikipedia.org/wiki/HTML) into a rich Abstract Syntax Tree (AST), but also provides a powerful generation engine to convert that AST into formats like **Markdown**, **HTML**, **CSV**, **RTF**, **Text**, **PDF**, and **JSON**, including native **RAG-focused chunking** support.
|
|
4
4
|
|
|
5
5
|
[](https://badge.fury.io/js/officeparser)
|
|
6
6
|
[](https://www.npmjs.com/package/officeparser)
|
|
@@ -18,8 +18,6 @@ A robust, strictly-typed Node.js and Browser library for parsing office files ([
|
|
|
18
18
|
- **Debugging**: Use the visualizer to debug parsing issues by inspecting exactly how nodes are interpreted.
|
|
19
19
|
- **Format Specs**: Read detailed specifications for the AST structure and configuration options.
|
|
20
20
|
|
|
21
|
-
*(Legacy Visualizer: If you prefer the [old simple visualizer](https://harshankur.github.io/officeParser/visualizer_old.html), it is still available.)*
|
|
22
|
-
|
|
23
21
|
---
|
|
24
22
|
|
|
25
23
|
|
|
@@ -37,7 +35,7 @@ npm i officeparser
|
|
|
37
35
|
```
|
|
38
36
|
|
|
39
37
|
## Command Line usage
|
|
40
|
-
You can use `officeparser` directly from the terminal to
|
|
38
|
+
You can use `officeparser` directly from the terminal to extract content as JSON AST, plain text, or generate new formats like Markdown and HTML.
|
|
41
39
|
|
|
42
40
|
```bash
|
|
43
41
|
# Get full AST as JSON (default)
|
|
@@ -46,25 +44,33 @@ npx officeparser /path/to/officeFile.docx
|
|
|
46
44
|
# Get plain text only
|
|
47
45
|
npx officeparser /path/to/officeFile.docx --toText=true
|
|
48
46
|
|
|
49
|
-
#
|
|
50
|
-
npx officeparser
|
|
47
|
+
# Generate Markdown file
|
|
48
|
+
npx officeparser report.docx --format=md --output=report.md
|
|
49
|
+
|
|
50
|
+
# Generate HTML file with specific output
|
|
51
|
+
npx officeparser presentation.pptx --format=html --output=preview.html
|
|
52
|
+
|
|
53
|
+
# Convert spreadsheet to CSV
|
|
54
|
+
npx officeparser data.xlsx --format=csv
|
|
51
55
|
```
|
|
52
56
|
|
|
53
57
|
### Config Options:
|
|
54
|
-
- `--
|
|
58
|
+
- `--format=[json|text|md|html|csv|rtf|pdf|chunks]` The output format. Default is `json`.
|
|
59
|
+
- `--output=[path]` Optional file path to write the output to.
|
|
60
|
+
- `--toText=[true|false]` Legacy flag to output only plain text. Use `--format=text` instead.
|
|
55
61
|
- `--ignoreNotes=[true|false]` Flag to ignore notes from files like PowerPoint. Default is false.
|
|
56
62
|
- `--newlineDelimiter=[delimiter]` The delimiter to use for new lines. Default is `\n`.
|
|
57
63
|
- `--putNotesAtLast=[true|false]` Flag to collect notes at the end of files like PowerPoint. Default is false.
|
|
58
|
-
- `--outputErrorToConsole=[true|false]` Flag to output errors to the console.
|
|
64
|
+
- `--outputErrorToConsole=[true|false]` **(Deprecated)** Flag to output errors to the console. Use `onWarning` callback in library usage.
|
|
59
65
|
- `--extractAttachments=[true|false]` Flag to extract images/charts as Base64. Default is false.
|
|
60
66
|
- `--ocr=[true|false]` Flag to enable OCR for extracted images. Default is false.
|
|
61
67
|
- `--includeRawContent=[true|false]` Flag to include raw XML/RTF content in nodes. Default is false.
|
|
62
|
-
- `--includeBreakNodes=[true|false]` Flag to include break nodes. Currently only available for DOCX documents
|
|
68
|
+
- `--includeBreakNodes=[true|false]` Flag to include break nodes. Currently only available for DOCX documents.
|
|
63
69
|
- `--verbose=[true|false]` Show full error stack traces.
|
|
64
70
|
|
|
65
71
|
|
|
66
72
|
## Library Usage
|
|
67
|
-
In **
|
|
73
|
+
In **v7.0.0**, the library has evolved into a dual-purpose **Parser** and **Generator**. You can first parse any office file into a structured AST and then use the `OfficeGenerator` to transform that AST into various formats or chunks.
|
|
68
74
|
|
|
69
75
|
### Getting Started (Async/Await)
|
|
70
76
|
```js
|
|
@@ -99,6 +105,101 @@ const text = await getText("/path/to/officeFile.docx");
|
|
|
99
105
|
console.log(text);
|
|
100
106
|
```
|
|
101
107
|
|
|
108
|
+
## Using the OfficeGenerator
|
|
109
|
+
The `OfficeGenerator` is a powerful tool to convert your AST into human-readable formats or structured data.
|
|
110
|
+
|
|
111
|
+
```typescript
|
|
112
|
+
import { OfficeParser, OfficeGenerator } from 'officeparser';
|
|
113
|
+
|
|
114
|
+
const ast = await OfficeParser.parseOffice('report.docx');
|
|
115
|
+
|
|
116
|
+
// 1. Convert to Markdown
|
|
117
|
+
const md = await OfficeGenerator.generate(ast, 'md');
|
|
118
|
+
console.log(md.value);
|
|
119
|
+
|
|
120
|
+
// 2. Convert to HTML with structured style mapping (Recommended)
|
|
121
|
+
const html = await OfficeGenerator.generate(ast, 'html', {
|
|
122
|
+
includeFormatting: true,
|
|
123
|
+
styleMap: [
|
|
124
|
+
{
|
|
125
|
+
selector: { nodeType: 'paragraph', attributes: { style: 'Heading 1' } },
|
|
126
|
+
output: { tag: 'h1', classes: ['main-title'] }
|
|
127
|
+
}
|
|
128
|
+
]
|
|
129
|
+
});
|
|
130
|
+
console.log(html.value);
|
|
131
|
+
|
|
132
|
+
// 3. Convert to CSV (for spreadsheets)
|
|
133
|
+
const csv = await OfficeGenerator.generate(ast, 'csv');
|
|
134
|
+
console.log(csv.value);
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
## The New "One-Step" API: `OfficeConverter`
|
|
138
|
+
In **v7.0.0**, we introduced the `OfficeConverter.convert` method. This is the new high-level API designed for one-step transformations where you don't need to manually interact with the AST. It automatically handles parser and generator configuration synchronization.
|
|
139
|
+
|
|
140
|
+
```typescript
|
|
141
|
+
import { OfficeConverter } from 'officeparser';
|
|
142
|
+
|
|
143
|
+
// One-step conversion from DOCX to Markdown
|
|
144
|
+
const result = await OfficeConverter.convert('report.docx', 'md');
|
|
145
|
+
console.log(result.value); // The generated Markdown string
|
|
146
|
+
console.log(result.messages); // Array of warnings/info (e.g., "Skipped unsupported drawing")
|
|
147
|
+
|
|
148
|
+
// Complex conversion with nested configuration
|
|
149
|
+
const htmlResult = await OfficeConverter.convert('data.xlsx', 'html', {
|
|
150
|
+
parseConfig: {
|
|
151
|
+
ignoreNotes: true
|
|
152
|
+
},
|
|
153
|
+
generatorConfig: {
|
|
154
|
+
includeFormatting: true,
|
|
155
|
+
styleMap: [
|
|
156
|
+
{
|
|
157
|
+
selector: { attributes: { style: { value: 'Header', operator: '~=' } } },
|
|
158
|
+
output: { tag: 'h2', classes: ['data-header'] }
|
|
159
|
+
}
|
|
160
|
+
]
|
|
161
|
+
},
|
|
162
|
+
onWarning: (msg) => console.warn("Conversion Warning:", msg)
|
|
163
|
+
});
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
## Native RAG Chunking
|
|
167
|
+
`officeParser` provides native support for document chunking, specifically designed for Retrieval-Augmented Generation (RAG) workflows. It offers three distinct strategies to split your documents while maintaining context and metadata.
|
|
168
|
+
|
|
169
|
+
### 1. Fixed-Size Strategy (Recursive)
|
|
170
|
+
Splits text into chunks based on character count with a specified overlap. It uses smart boundary detection to avoid cutting in the middle of sentences or paragraphs.
|
|
171
|
+
|
|
172
|
+
### 2. Document Structure Strategy
|
|
173
|
+
Splits the document at natural structural boundaries like pages (PDF/Word), slides (PPTX), or high-level headings. This preserves the logical flow of the document.
|
|
174
|
+
|
|
175
|
+
### 3. Semantic Strategy
|
|
176
|
+
Uses cosine similarity between sentence embeddings to identify coherent topic boundaries. This ensures that each chunk contains semantically related content (requires an embedding function).
|
|
177
|
+
|
|
178
|
+
### The `OfficeChunk` Interface
|
|
179
|
+
Every chunk produced contains not just text, but rich metadata to help your RAG pipeline:
|
|
180
|
+
```typescript
|
|
181
|
+
{
|
|
182
|
+
text: string; // The chunk content
|
|
183
|
+
metadata: {
|
|
184
|
+
sourceType: string; // e.g., "docx", "pdf"
|
|
185
|
+
pageNumber?: number; // Current page
|
|
186
|
+
slideNumber?: number; // Current slide
|
|
187
|
+
closestHeading?: string; // The heading this chunk belongs to
|
|
188
|
+
chunkIndex: number; // Sequential index
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
#### Example: Generating Chunks
|
|
194
|
+
```typescript
|
|
195
|
+
const chunks = await OfficeGenerator.generate(ast, 'chunks', {
|
|
196
|
+
strategy: 'fixed-size',
|
|
197
|
+
maxChunkSize: 1000,
|
|
198
|
+
chunkOverlap: 200
|
|
199
|
+
});
|
|
200
|
+
console.log(`Generated ${chunks.value.length} chunks`);
|
|
201
|
+
```
|
|
202
|
+
|
|
102
203
|
### Using Callbacks (Backward Compatibility Support)
|
|
103
204
|
Callbacks are still supported for those preferred, but the data returned is now the AST object.
|
|
104
205
|
```js
|
|
@@ -134,7 +235,7 @@ The `OfficeParserAST` provides a format-agnostic representation of your document
|
|
|
134
235
|
|
|
135
236
|
```text
|
|
136
237
|
OfficeParserAST
|
|
137
|
-
├── type: "docx" | "
|
|
238
|
+
├── type: "docx" | "pdf" | "xlsx" | "csv" | "md" | ... (11 formats supported)
|
|
138
239
|
├── metadata: { author, title, created, modified, ..., customProperties }
|
|
139
240
|
├── content: [ OfficeContentNode ]
|
|
140
241
|
│ ├── type: "paragraph" | "heading" | "table" | "list" | ...
|
|
@@ -313,6 +414,14 @@ console.log("Custom Metadata:", ast.metadata.customProperties);
|
|
|
313
414
|
// Output: { "ProjectID": "ABC-123", "InternalReview": true }
|
|
314
415
|
```
|
|
315
416
|
|
|
417
|
+
## Performance & Fidelity Highlights (v7.0.0)
|
|
418
|
+
The v7.0.0 release brings significant internal optimizations and fidelity improvements:
|
|
419
|
+
- **OpenOffice Speedups**: Up to **23x faster** parsing for ODP presentations thanks to optimized XML caching.
|
|
420
|
+
- **Excel Memory Efficiency**: Resolved $O(n)$ memory overhead issues for large spreadsheets (#91) by switching to iterative stream-based parsing.
|
|
421
|
+
- **RTF Performance**: Rewritten core loop to resolve $O(n^2)$ bottlenecks during string accumulation.
|
|
422
|
+
- **Advanced Table Fidelity**: Native support for **vertical cell merging** (`vMerge`) and **horizontal spanning** (`gridSpan`) in DOCX, ensuring complex tables look exactly as they do in Word.
|
|
423
|
+
- **Parser Extensions**: You can now parse `CSV`, `Markdown`, and `HTML` files *into* the unified Office AST, allowing you to use the `OfficeGenerator` on them just like any other format.
|
|
424
|
+
|
|
316
425
|
### Advanced AST Usage
|
|
317
426
|
Beyond using `ast.toText()`, you can interact with the structural data directly:
|
|
318
427
|
|
|
@@ -407,24 +516,108 @@ Pass an optional config object as the second argument to `parseOffice`.
|
|
|
407
516
|
|
|
408
517
|
| Flag | DataType | Default | Explanation |
|
|
409
518
|
|------|----------|---------|-------------|
|
|
410
|
-
| `outputErrorToConsole` | boolean | `false` | Show logs to console in case of an error. |
|
|
519
|
+
| `outputErrorToConsole` | boolean | `false` | **Deprecated**: Use `onWarning` instead. Show logs to console in case of an error. |
|
|
411
520
|
| `newlineDelimiter` | string | `\n` | Delimiter for new lines in text output. |
|
|
412
521
|
| `ignoreNotes` | boolean | `false` | Ignore notes in files like PowerPoint/ODP. |
|
|
413
|
-
| `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document.
|
|
522
|
+
| `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. |
|
|
414
523
|
| `extractAttachments` | boolean | `false` | Extract images and charts as Base64. |
|
|
415
524
|
| `includeRawContent` | boolean | `false` | Include raw XML/RTF markup in the nodes. |
|
|
416
|
-
| `serializeRawContent` | boolean | `true` |
|
|
417
|
-
| `preserveXmlWhitespace` | boolean | `false` |
|
|
525
|
+
| `serializeRawContent` | boolean | `true` | Re-serializes raw XML to clean strings. |
|
|
526
|
+
| `preserveXmlWhitespace` | boolean | `false` | Preserves original XML whitespace. |
|
|
418
527
|
| `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
|
|
419
|
-
| `
|
|
420
|
-
| `
|
|
421
|
-
| `
|
|
422
|
-
| `
|
|
423
|
-
| `
|
|
424
|
-
| `
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
528
|
+
| `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. |
|
|
529
|
+
| `ocrConfig` | object | `{}` | OCR Scheduler configuration. |
|
|
530
|
+
| `includeBreakNodes` | boolean | `false` | Include `w:br`, `w:cr` nodes (DOCX only).|
|
|
531
|
+
| `ignoreInternalLinks` | boolean | `false` | Remove all bookmarks and internal jumps. |
|
|
532
|
+
| `csvDelimiter` | string | `,` | Custom delimiter for parsing CSV files. |
|
|
533
|
+
| `fileType` | string | `null` | Manual format override (authoritative). |
|
|
534
|
+
|
|
535
|
+
## Generator Configuration: GeneratorConfig
|
|
536
|
+
Configuration options for `OfficeGenerator.generate`.
|
|
537
|
+
|
|
538
|
+
| Flag | DataType | Default | Explanation |
|
|
539
|
+
|------|----------|---------|-------------|
|
|
540
|
+
| `includeFormatting` | boolean | `false` | Whether to include semantic styles (bold, italic) in output. |
|
|
541
|
+
| `styleMap` | string[] \| array | `[]` | Array of style mappings (DSL strings or structured objects). |
|
|
542
|
+
| `ignoreDefaultStyleMap`| boolean | `false` | Ignore the library's default style mappings. |
|
|
543
|
+
| `includeMetadata` | boolean | `false` | Include document metadata in the output (e.g., as frontmatter). |
|
|
544
|
+
| `onNode` | function | `undefined` | Callback to intercept/modify any node during generation. |
|
|
545
|
+
|
|
546
|
+
### 🛠️ Advanced Node Manipulation (Pro Users)
|
|
547
|
+
The `onNode` callback is a powerful tool that gives you complete control over the generation process. It is called for **every single node** in the AST before it is rendered.
|
|
548
|
+
|
|
549
|
+
#### Callback Capabilities:
|
|
550
|
+
1. **Filter/Remove Nodes**: Return `false` to skip a node and all its children.
|
|
551
|
+
2. **Override Rendering**: Return a `string` to use that exact text as the output, bypassing default logic and recursion.
|
|
552
|
+
3. **Mutate Nodes**: Modify the `node` object directly (e.g., changing `node.text`) and return `void` to let the generator proceed with your changes.
|
|
553
|
+
4. **Async Support**: The callback can be `async`, allowing you to fetch external data or perform complex logic during generation.
|
|
554
|
+
|
|
555
|
+
#### Pro Example:
|
|
556
|
+
```typescript
|
|
557
|
+
const result = await ast.to('md', {
|
|
558
|
+
onNode: async (node) => {
|
|
559
|
+
// 1. Skip all images
|
|
560
|
+
if (node.type === 'image') return false;
|
|
561
|
+
|
|
562
|
+
// 2. Redact sensitive info by mutating the node
|
|
563
|
+
if (node.text?.includes('SECRET_KEY')) {
|
|
564
|
+
node.text = node.text.replace(/SECRET_KEY: \w+/, 'SECRET_KEY: [REDACTED]');
|
|
565
|
+
}
|
|
566
|
+
|
|
567
|
+
// 3. Custom rendering for specific styles
|
|
568
|
+
if (node.metadata?.style === 'Callout') {
|
|
569
|
+
return `> [!INFO]\n> ${node.text}`;
|
|
570
|
+
}
|
|
571
|
+
|
|
572
|
+
// 4. Proceed with default rendering (implicitly returns void)
|
|
573
|
+
}
|
|
574
|
+
});
|
|
575
|
+
```
|
|
576
|
+
|
|
577
|
+
### Advanced Style Mapping (Semantic Translation)
|
|
578
|
+
The `styleMap` configuration is the primary way to define the "semantic meaning" of document styles. We recommend using **Structured Style Mappings** for full type safety and power.
|
|
579
|
+
|
|
580
|
+
#### 1. Structured Style Mappings (Recommended)
|
|
581
|
+
Use structured objects to match nodes based on type and attributes, and specify detailed output properties like classes and custom attributes.
|
|
582
|
+
|
|
583
|
+
```typescript
|
|
584
|
+
styleMap: [
|
|
585
|
+
{
|
|
586
|
+
selector: {
|
|
587
|
+
nodeType: 'paragraph',
|
|
588
|
+
attributes: { style: 'Heading 1' }
|
|
589
|
+
},
|
|
590
|
+
output: {
|
|
591
|
+
tag: 'h1',
|
|
592
|
+
classes: ['main-title'],
|
|
593
|
+
attributes: { id: 'top' }
|
|
594
|
+
}
|
|
595
|
+
},
|
|
596
|
+
{
|
|
597
|
+
// Use operators like '~=' for partial matches
|
|
598
|
+
selector: { attributes: { style: { value: 'Quote', operator: '~=' } } },
|
|
599
|
+
output: { tag: 'blockquote' }
|
|
600
|
+
}
|
|
601
|
+
]
|
|
602
|
+
```
|
|
603
|
+
|
|
604
|
+
#### 2. Legacy String DSL
|
|
605
|
+
The library also maintains support for a simple string-based DSL, highly compatible with `mammoth.js`.
|
|
606
|
+
|
|
607
|
+
- **Literal Matching**: `"p[style-name='Heading 1'] => h1"`
|
|
608
|
+
- **Regex-like Matching**: `"p[style~='Title'] => h2"`
|
|
609
|
+
- **Attribute Filters**: `"p[style-name='Quote'][lang='en'] => blockquote"`
|
|
610
|
+
|
|
611
|
+
## Chunking Configuration: ChunkingConfig
|
|
612
|
+
Specific options when using `format: 'chunks'`.
|
|
613
|
+
|
|
614
|
+
| Flag | DataType | Default | Explanation |
|
|
615
|
+
|------|----------|---------|-------------|
|
|
616
|
+
| `strategy` | string | `'fixed-size'`| The chunking strategy (`fixed-size`, `document-structure`, `semantic`). |
|
|
617
|
+
| `maxChunkSize` | number | `1000` | Maximum characters per chunk. |
|
|
618
|
+
| `chunkOverlap` | number | `200` | Overlap between consecutive chunks. |
|
|
619
|
+
| `similarityThreshold`| number | `0.5` | Threshold for semantic splitting (0.0 to 1.0). |
|
|
620
|
+
| `embedBatchSize` | number | `50` | Batch size for embedding requests. |
|
|
428
621
|
|
|
429
622
|
### OCR Scheduler & Resource Management
|
|
430
623
|
If your application uses OCR, `officeParser` utilizes an intelligent **Smart Worker Pool** to maintain a background worker pool and optimize repeated parse requests.
|
|
@@ -555,7 +748,7 @@ const ast = await officeParser.parseOffice(file);
|
|
|
555
748
|
|
|
556
749
|
// Or override it with your own path or a different version:
|
|
557
750
|
const ast2 = await officeParser.parseOffice(file, {
|
|
558
|
-
pdfWorkerSrc: "https://
|
|
751
|
+
pdfWorkerSrc: "https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
|
|
559
752
|
});
|
|
560
753
|
```
|
|
561
754
|
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { ConversionResult, OfficeConverterConfig, SupportedDestination, SupportedFileType } from './types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Utility type to infer the file type from a file path string literal.
|
|
4
|
+
*/
|
|
5
|
+
type InferFileTypeFromPath<T> = T extends `${string}.${infer E}` ? (Lowercase<E> extends SupportedFileType ? Lowercase<E> : SupportedFileType) : SupportedFileType;
|
|
6
|
+
/**
|
|
7
|
+
* Main converter class providing a streamlined one-step API for document conversion.
|
|
8
|
+
*
|
|
9
|
+
* This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
|
|
10
|
+
* documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
|
|
11
|
+
*/
|
|
12
|
+
export declare class OfficeConverter {
|
|
13
|
+
/**
|
|
14
|
+
* Converts an office document from its source format to a specified destination format.
|
|
15
|
+
*
|
|
16
|
+
* This method:
|
|
17
|
+
* 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
|
|
18
|
+
* 2. Automatically configures the parser based on the generator requirements (e.g., enabling
|
|
19
|
+
* attachment extraction if images are requested in the output).
|
|
20
|
+
* 3. Generates the destination document from the AST using `OfficeGenerator`.
|
|
21
|
+
*
|
|
22
|
+
* @template F The inferred type of the input file (path string or buffer).
|
|
23
|
+
* @template T The authoritative source file type (inferred from path or config).
|
|
24
|
+
*
|
|
25
|
+
* @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
|
|
26
|
+
* @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
|
|
27
|
+
* @param config - Optional unified configuration for both the parser and generator phases.
|
|
28
|
+
*
|
|
29
|
+
* @returns A promise resolving to the ConversionResult containing the value and messages.
|
|
30
|
+
* @throws {Error} If the source format is unsupported or parsing/generation fails.
|
|
31
|
+
*
|
|
32
|
+
* @example
|
|
33
|
+
* ```typescript
|
|
34
|
+
* // Convert Word to Markdown with a single call
|
|
35
|
+
* const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
|
|
36
|
+
*
|
|
37
|
+
* // Convert PDF to HTML with OCR enabled for images
|
|
38
|
+
* const { value: html } = await OfficeConverter.convert(buffer, 'html', {
|
|
39
|
+
* ocr: true,
|
|
40
|
+
* includeImages: true
|
|
41
|
+
* });
|
|
42
|
+
* ```
|
|
43
|
+
*/
|
|
44
|
+
static convert<F extends string | Buffer | ArrayBuffer, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
|
|
45
|
+
}
|
|
46
|
+
export {};
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.OfficeConverter = void 0;
|
|
4
|
+
const OfficeGenerator_js_1 = require("./OfficeGenerator.js");
|
|
5
|
+
const OfficeParser_js_1 = require("./OfficeParser.js");
|
|
6
|
+
/**
|
|
7
|
+
* Main converter class providing a streamlined one-step API for document conversion.
|
|
8
|
+
*
|
|
9
|
+
* This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
|
|
10
|
+
* documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
|
|
11
|
+
*/
|
|
12
|
+
class OfficeConverter {
|
|
13
|
+
/**
|
|
14
|
+
* Converts an office document from its source format to a specified destination format.
|
|
15
|
+
*
|
|
16
|
+
* This method:
|
|
17
|
+
* 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
|
|
18
|
+
* 2. Automatically configures the parser based on the generator requirements (e.g., enabling
|
|
19
|
+
* attachment extraction if images are requested in the output).
|
|
20
|
+
* 3. Generates the destination document from the AST using `OfficeGenerator`.
|
|
21
|
+
*
|
|
22
|
+
* @template F The inferred type of the input file (path string or buffer).
|
|
23
|
+
* @template T The authoritative source file type (inferred from path or config).
|
|
24
|
+
*
|
|
25
|
+
* @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
|
|
26
|
+
* @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
|
|
27
|
+
* @param config - Optional unified configuration for both the parser and generator phases.
|
|
28
|
+
*
|
|
29
|
+
* @returns A promise resolving to the ConversionResult containing the value and messages.
|
|
30
|
+
* @throws {Error} If the source format is unsupported or parsing/generation fails.
|
|
31
|
+
*
|
|
32
|
+
* @example
|
|
33
|
+
* ```typescript
|
|
34
|
+
* // Convert Word to Markdown with a single call
|
|
35
|
+
* const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
|
|
36
|
+
*
|
|
37
|
+
* // Convert PDF to HTML with OCR enabled for images
|
|
38
|
+
* const { value: html } = await OfficeConverter.convert(buffer, 'html', {
|
|
39
|
+
* ocr: true,
|
|
40
|
+
* includeImages: true
|
|
41
|
+
* });
|
|
42
|
+
* ```
|
|
43
|
+
*/
|
|
44
|
+
static async convert(file, destination, config) {
|
|
45
|
+
// 1. Prepare Parser Configuration
|
|
46
|
+
// We prioritize the top-level onWarning if provided.
|
|
47
|
+
const parserConfig = {
|
|
48
|
+
...config?.parseConfig,
|
|
49
|
+
onWarning: config?.onWarning || config?.parseConfig?.onWarning,
|
|
50
|
+
};
|
|
51
|
+
// Remove OCR settings for the streamlined converter as requested
|
|
52
|
+
parserConfig.ocr = false;
|
|
53
|
+
// Remove undefined keys to prevent overwriting defaults in resolveParserConfig
|
|
54
|
+
Object.keys(parserConfig).forEach((key) => parserConfig[key] === undefined && delete parserConfig[key]);
|
|
55
|
+
/**
|
|
56
|
+
* AUTOMATIC CONFIGURATION SYNC
|
|
57
|
+
* We sync extractAttachments from the generator configuration.
|
|
58
|
+
*/
|
|
59
|
+
parserConfig.extractAttachments = (config?.generatorConfig?.includeImages !== false) || (config?.generatorConfig?.includeCharts !== false);
|
|
60
|
+
// 2. Parse the source document into the universal AST
|
|
61
|
+
const ast = await OfficeParser_js_1.OfficeParser.parseOffice(file, parserConfig);
|
|
62
|
+
// 3. Generate the destination document from the AST
|
|
63
|
+
const generatorConfig = {
|
|
64
|
+
...config?.generatorConfig,
|
|
65
|
+
onWarning: config?.onWarning || config?.generatorConfig?.onWarning,
|
|
66
|
+
};
|
|
67
|
+
const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, destination, generatorConfig);
|
|
68
|
+
result.messages = [...(ast.warnings || []), ...result.messages];
|
|
69
|
+
return result;
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
exports.OfficeConverter = OfficeConverter;
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
import { ConversionResult, GeneratorConfig, OfficeParserAST, SupportedDestination, SupportedFileType } from './types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Main generator class providing document conversion functionality.
|
|
4
|
+
*/
|
|
5
|
+
export declare class OfficeGenerator {
|
|
6
|
+
/**
|
|
7
|
+
* Generates a file of the specified type from an AST.
|
|
8
|
+
* This is the single source of truth for generation logic.
|
|
9
|
+
*
|
|
10
|
+
* @param ast - The OfficeParserAST to generate from
|
|
11
|
+
* @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
|
|
12
|
+
* @param config - Optional configuration for the generator
|
|
13
|
+
* @returns A promise resolving to the ConversionResult containing the value and messages
|
|
14
|
+
* @throws {Error} If the destination format is unsupported
|
|
15
|
+
*/
|
|
16
|
+
static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
|
|
17
|
+
type: T;
|
|
18
|
+
}, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
|
|
19
|
+
}
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.OfficeGenerator = void 0;
|
|
4
|
+
const ChunkingGenerator_js_1 = require("./generators/ChunkingGenerator.js");
|
|
5
|
+
const CsvGenerator_js_1 = require("./generators/CsvGenerator.js");
|
|
6
|
+
const HtmlGenerator_js_1 = require("./generators/HtmlGenerator.js");
|
|
7
|
+
const MarkdownGenerator_js_1 = require("./generators/MarkdownGenerator.js");
|
|
8
|
+
const PdfGenerator_js_1 = require("./generators/PdfGenerator.js");
|
|
9
|
+
const RtfGenerator_js_1 = require("./generators/RtfGenerator.js");
|
|
10
|
+
const TextGenerator_js_1 = require("./generators/TextGenerator.js");
|
|
11
|
+
const types_js_1 = require("./types.js");
|
|
12
|
+
const errorUtils_js_1 = require("./utils/errorUtils.js");
|
|
13
|
+
/**
|
|
14
|
+
* Main generator class providing document conversion functionality.
|
|
15
|
+
*/
|
|
16
|
+
class OfficeGenerator {
|
|
17
|
+
/**
|
|
18
|
+
* Generates a file of the specified type from an AST.
|
|
19
|
+
* This is the single source of truth for generation logic.
|
|
20
|
+
*
|
|
21
|
+
* @param ast - The OfficeParserAST to generate from
|
|
22
|
+
* @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
|
|
23
|
+
* @param config - Optional configuration for the generator
|
|
24
|
+
* @returns A promise resolving to the ConversionResult containing the value and messages
|
|
25
|
+
* @throws {Error} If the destination format is unsupported
|
|
26
|
+
*/
|
|
27
|
+
static async generate(ast, destination, config) {
|
|
28
|
+
switch (destination.toLowerCase()) {
|
|
29
|
+
case 'text':
|
|
30
|
+
return new TextGenerator_js_1.TextGenerator(ast, config).generate();
|
|
31
|
+
case 'md':
|
|
32
|
+
return new MarkdownGenerator_js_1.MarkdownGenerator(ast, config).generate();
|
|
33
|
+
case 'html':
|
|
34
|
+
return new HtmlGenerator_js_1.HtmlGenerator(ast, config).generate();
|
|
35
|
+
case 'pdf':
|
|
36
|
+
return new PdfGenerator_js_1.PdfGenerator(ast, config).generate();
|
|
37
|
+
case 'csv':
|
|
38
|
+
return new CsvGenerator_js_1.CsvGenerator(ast, config).generate();
|
|
39
|
+
case 'rtf':
|
|
40
|
+
return new RtfGenerator_js_1.RtfGenerator(ast, config).generate();
|
|
41
|
+
case 'chunks':
|
|
42
|
+
return new ChunkingGenerator_js_1.ChunkingGenerator(ast, config).generate();
|
|
43
|
+
default:
|
|
44
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, undefined, destination);
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
exports.OfficeGenerator = OfficeGenerator;
|
package/dist/OfficeParser.d.ts
CHANGED
|
@@ -11,6 +11,9 @@
|
|
|
11
11
|
* - ODT, ODP, ODS (OpenDocument formats)
|
|
12
12
|
* - PDF (Portable Document Format)
|
|
13
13
|
* - RTF (Rich Text Format)
|
|
14
|
+
* - CSV (Comma-Separated Values)
|
|
15
|
+
* - MD (Markdown)
|
|
16
|
+
* - HTML (HyperText Markup Language)
|
|
14
17
|
*
|
|
15
18
|
* **Usage:**
|
|
16
19
|
* ```typescript
|
|
@@ -60,6 +63,9 @@ export declare class OfficeParser {
|
|
|
60
63
|
* - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
|
|
61
64
|
* - `.pdf` → PdfParser (PDF.js)
|
|
62
65
|
* - `.rtf` → RtfParser (custom RTF parser)
|
|
66
|
+
* - `.csv` → CsvParser
|
|
67
|
+
* - `.md` → MarkdownParser
|
|
68
|
+
* - `.html` → HtmlParser
|
|
63
69
|
*
|
|
64
70
|
* @param file - File path (string), Buffer, or ArrayBuffer containing the document
|
|
65
71
|
* @param config - Optional configuration object (defaults applied for all omitted options)
|