officeparser 7.0.3 → 7.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +152 -18
- package/dist/OfficeGenerator.d.ts +1 -1
- package/dist/OfficeGenerator.js +16 -7
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +4 -0
- package/dist/cli.js +12 -3
- package/dist/defaults.js +27 -1
- package/dist/generators/BaseGenerator.d.ts +3 -3
- package/dist/generators/ChunkingGenerator.js +31 -4
- package/dist/generators/CsvGenerator.d.ts +1 -1
- package/dist/generators/HtmlGenerator.d.ts +2 -1
- package/dist/generators/HtmlGenerator.js +462 -40
- package/dist/generators/MarkdownGenerator.d.ts +1 -1
- package/dist/generators/MarkdownGenerator.js +3 -1
- package/dist/generators/PdfGenerator.d.ts +1 -1
- package/dist/generators/PdfGenerator.js +51 -10
- package/dist/generators/RtfGenerator.d.ts +2 -1
- package/dist/generators/RtfGenerator.js +43 -6
- package/dist/generators/TextGenerator.d.ts +1 -1
- package/dist/officeparser.browser.d.ts +377 -53
- package/dist/officeparser.browser.iife.js +380 -93
- package/dist/officeparser.browser.mjs +380 -93
- package/dist/parsers/CsvParser.js +6 -1
- package/dist/parsers/ExcelParser.js +69 -21
- package/dist/parsers/HtmlParser.js +15 -1
- package/dist/parsers/MarkdownParser.js +18 -10
- package/dist/parsers/OpenOfficeParser.js +61 -34
- package/dist/parsers/PdfParser.js +26 -1
- package/dist/parsers/PowerPointParser.js +168 -40
- package/dist/parsers/RtfParser.js +30 -24
- package/dist/parsers/WordParser.js +158 -11
- package/dist/sbom.cdx.json +100 -100
- package/dist/types.d.ts +383 -53
- package/dist/types.js +4 -0
- package/dist/utils/astUtils.d.ts +2 -2
- package/dist/utils/astUtils.js +2 -1
- package/dist/utils/configUtils.d.ts +5 -0
- package/dist/utils/configUtils.js +69 -2
- package/dist/utils/errorUtils.d.ts +20 -0
- package/dist/utils/errorUtils.js +39 -3
- package/dist/utils/moduleLoader.js +3 -3
- package/dist/utils/ocrUtils.js +271 -66
- package/dist/utils/xmlUtils.d.ts +17 -0
- package/dist/utils/xmlUtils.js +85 -1
- package/package.json +3 -2
package/README.md
CHANGED
|
@@ -100,7 +100,7 @@ npx officeparser document.pdf --format=chunks
|
|
|
100
100
|
|------|--------|---------|-------------|
|
|
101
101
|
| `--format` | `json\|text\|md\|html\|csv\|rtf\|pdf\|chunks` | `json` | Output format |
|
|
102
102
|
| `--output` | path | — | Write output to a file |
|
|
103
|
-
|
|
|
103
|
+
| ~~`--toText`~~ | `true\|false` | `false` | **Deprecated.** Use `--format=text` |
|
|
104
104
|
| `--ignoreNotes` | `true\|false` | `false` | Ignore speaker notes (PPTX/ODP) |
|
|
105
105
|
| `--putNotesAtLast` | `true\|false` | `false` | Collect notes at end of output |
|
|
106
106
|
| `--newlineDelimiter` | string | `\n` | Delimiter between lines |
|
|
@@ -108,7 +108,7 @@ npx officeparser document.pdf --format=chunks
|
|
|
108
108
|
| `--ocr` | `true\|false` | `false` | Enable OCR for images |
|
|
109
109
|
| `--includeRawContent` | `true\|false` | `false` | Include raw XML/RTF in nodes |
|
|
110
110
|
| `--includeBreakNodes` | `true\|false` | `false` | Include break nodes (DOCX only) |
|
|
111
|
-
|
|
|
111
|
+
| ~~`--outputErrorToConsole`~~ | `true\|false` | `false` | **Deprecated.** Use `onWarning` callback |
|
|
112
112
|
| `--verbose` | `true\|false` | `false` | Show full error stack traces |
|
|
113
113
|
|
|
114
114
|
---
|
|
@@ -178,6 +178,60 @@ const ast = await officeParser.parseOffice(buffer);
|
|
|
178
178
|
> const ast = await officeParser.parseOffice(markdownBuffer, { fileType: 'md' });
|
|
179
179
|
> ```
|
|
180
180
|
|
|
181
|
+
### Cancellation with AbortSignal
|
|
182
|
+
|
|
183
|
+
You can pass a standard `AbortSignal` (e.g. from an `AbortController`) to cancel an active parse operation. This is especially useful for setting request-level timeouts or canceling long-running parses (like large PDFs with OCR).
|
|
184
|
+
|
|
185
|
+
```js
|
|
186
|
+
const controller = new AbortController();
|
|
187
|
+
|
|
188
|
+
// Cancel parsing if it takes longer than 5 seconds
|
|
189
|
+
setTimeout(() => controller.abort(), 5000);
|
|
190
|
+
|
|
191
|
+
try {
|
|
192
|
+
const ast = await officeParser.parseOffice('large_scanned_file.pdf', {
|
|
193
|
+
abortSignal: controller.signal,
|
|
194
|
+
ocr: true
|
|
195
|
+
});
|
|
196
|
+
} catch (err) {
|
|
197
|
+
if (err.name === 'AbortError') {
|
|
198
|
+
console.log('Parsing was cancelled.');
|
|
199
|
+
} else {
|
|
200
|
+
console.error('Parsing failed:', err);
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
> [!IMPORTANT]
|
|
206
|
+
> **AbortError Propagation**
|
|
207
|
+
> When parsing is cancelled via `AbortSignal`, the parser rejects with a standard `AbortError` (a `DOMException` or an Error with `name: 'AbortError'`).
|
|
208
|
+
> This error is *not* wrapped in standard OfficeParser error types so that you can reliably detect cancellation using `error.name === 'AbortError'`.
|
|
209
|
+
|
|
210
|
+
> [!NOTE]
|
|
211
|
+
> **Worker Cleanup on Abort**
|
|
212
|
+
> If an OCR job is actively running in the background when the signal is aborted, `officeParser` automatically terminates the Tesseract worker process immediately and removes it from the pool to prevent thread/memory leaks.
|
|
213
|
+
|
|
214
|
+
### Custom OCR Timeouts
|
|
215
|
+
|
|
216
|
+
To prevent the parser from hanging indefinitely due to slow network connections (when downloading Tesseract language datasets) or complex image processing, you can configure granular timeouts under `ocrConfig.timeout`.
|
|
217
|
+
|
|
218
|
+
```js
|
|
219
|
+
const ast = await officeParser.parseOffice('scanned_document.pdf', {
|
|
220
|
+
ocr: true,
|
|
221
|
+
ocrConfig: {
|
|
222
|
+
timeout: {
|
|
223
|
+
workerLoad: 30000, // 30s max to load worker & download language training files
|
|
224
|
+
recognition: 15000, // 15s max per image text recognition
|
|
225
|
+
autoTerminate: 10000 // 10s of inactivity before terminating idle workers
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
});
|
|
229
|
+
```
|
|
230
|
+
|
|
231
|
+
> [!TIP]
|
|
232
|
+
> **Non-Fatal Timeout Recovery**
|
|
233
|
+
> If `workerLoad` or `recognition` timeouts are exceeded, the parser will log a warning in `ast.warnings` and **continue parsing the rest of the document**. The overall promise resolves successfully with the text extracted from the document layers (rather than failing the entire parse).
|
|
234
|
+
|
|
181
235
|
### `ast.to()` — Generate from AST
|
|
182
236
|
|
|
183
237
|
The preferred way to convert a parsed AST to another format. Returns a `ConversionResult`.
|
|
@@ -366,13 +420,19 @@ interface OfficeChunk {
|
|
|
366
420
|
```text
|
|
367
421
|
OfficeParserAST
|
|
368
422
|
├── type: 'docx' | 'pdf' | 'xlsx' | 'csv' | 'md' | ... (11 formats)
|
|
369
|
-
├── metadata: { author, title, created, modified, customProperties, styleMap, ... }
|
|
423
|
+
├── metadata: { author, title, created, modified, keywords, customProperties, nativeProperties, styleMap, ... }
|
|
370
424
|
├── content: [ OfficeContentNode ]
|
|
371
|
-
│ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | ...
|
|
425
|
+
│ ├── type: 'paragraph' | 'heading' | 'table' | 'list' | 'image' | 'chart' | 'comment' | ...
|
|
372
426
|
│ ├── text: string (concatenated text of node + all descendants)
|
|
373
|
-
│ ├── children: [ OfficeContentNode ] (recursive)
|
|
427
|
+
│ ├── children: [ OfficeContentNode ] (recursive structural children)
|
|
428
|
+
│ ├── notes: [ OfficeContentNode ] (footnotes/endnotes/slide notes attached to this node)
|
|
429
|
+
│ ├── comments: [ OfficeContentNode ] (inline comments attached to this node)
|
|
374
430
|
│ ├── formatting: { bold, italic, underline, color, size, font, alignment, ... }
|
|
375
|
-
│ └── metadata: { level, listId, row, col, rowSpan, colSpan, style, ... }
|
|
431
|
+
│ └── metadata: { level, listId, row, col, rowSpan, colSpan, backgroundColor, style, ... }
|
|
432
|
+
├── auxiliary?: OfficeAuxiliaryContent (out-of-band layout elements)
|
|
433
|
+
│ ├── headers?: OfficeContentNode[] (DOCX headers)
|
|
434
|
+
│ ├── footers?: OfficeContentNode[] (DOCX footers)
|
|
435
|
+
│ └── slideMasters?: OfficeContentNode[] (PPTX slide masters)
|
|
376
436
|
├── attachments: [ OfficeAttachment ] (populated when extractAttachments: true)
|
|
377
437
|
│ ├── type: 'image' | 'chart'
|
|
378
438
|
│ ├── name: string
|
|
@@ -382,7 +442,7 @@ OfficeParserAST
|
|
|
382
442
|
│ └── chartData?: { title, dataSets, labels }
|
|
383
443
|
├── warnings: OfficeIssue[] (non-fatal issues from the parsing phase)
|
|
384
444
|
├── to(format, config?) (format: 'html'|'md'|'text'|'csv'|'rtf'|'pdf'|'chunks', returns { value, messages })
|
|
385
|
-
└── toText() (Deprecated: use .to('text') instead)
|
|
445
|
+
└── ~~toText()~~ (Deprecated: use .to('text') instead)
|
|
386
446
|
```
|
|
387
447
|
|
|
388
448
|
### `OfficeIssue` — Warning / Error Object
|
|
@@ -498,17 +558,21 @@ ast.metadata = {
|
|
|
498
558
|
created?: Date
|
|
499
559
|
modified?: Date
|
|
500
560
|
description?: string
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
561
|
+
keywords?: string // NEW: Keywords from document properties
|
|
562
|
+
customProperties?: Record<string, any> // User-defined metadata from the document
|
|
563
|
+
nativeProperties?: Record<string, any> // NEW: All format-specific raw metadata
|
|
564
|
+
styleMap?: Record<string, TextFormatting> // Named styles → formatting definitions
|
|
565
|
+
formatting?: TextFormatting // Document-wide defaults
|
|
504
566
|
}
|
|
505
567
|
```
|
|
506
568
|
|
|
507
|
-
**Accessing
|
|
569
|
+
**Accessing native properties (format-specific metadata):**
|
|
508
570
|
```js
|
|
509
571
|
const ast = await officeParser.parseOffice('contract.docx');
|
|
510
|
-
console.log(ast.metadata.
|
|
511
|
-
// {
|
|
572
|
+
console.log(ast.metadata.nativeProperties);
|
|
573
|
+
// DOCX: { Pages: 5, Application: 'Microsoft Word' }
|
|
574
|
+
// HTML: { description: 'My page', 'og:title': 'Title' }
|
|
575
|
+
// PDF: { Title: 'Report', XMP: { ... } }
|
|
512
576
|
```
|
|
513
577
|
|
|
514
578
|
---
|
|
@@ -532,6 +596,61 @@ const headings = ast.content.filter(n => n.type === 'heading' && n.metadata?.lev
|
|
|
532
596
|
console.log(headings.map(h => h.text));
|
|
533
597
|
```
|
|
534
598
|
|
|
599
|
+
### Extract comments
|
|
600
|
+
```ts
|
|
601
|
+
// Comments can be attached to any nested node, so we must traverse recursively
|
|
602
|
+
const printComments = (nodes: OfficeContentNode[]) => {
|
|
603
|
+
nodes.forEach(node => {
|
|
604
|
+
if (node.comments) {
|
|
605
|
+
node.comments.forEach(c => {
|
|
606
|
+
console.log(`Comment by ${c.metadata?.author}: ${c.text}`);
|
|
607
|
+
});
|
|
608
|
+
}
|
|
609
|
+
if (node.children) {
|
|
610
|
+
printComments(node.children);
|
|
611
|
+
}
|
|
612
|
+
});
|
|
613
|
+
};
|
|
614
|
+
|
|
615
|
+
printComments(ast.content);
|
|
616
|
+
```
|
|
617
|
+
|
|
618
|
+
Set `ignoreComments: true` to skip extraction.
|
|
619
|
+
|
|
620
|
+
### Extract footnotes, endnotes & slide notes
|
|
621
|
+
```ts
|
|
622
|
+
// Slide speaker notes (PPTX) live on the slide node itself
|
|
623
|
+
const slide = ast.content.find(n => n.type === 'slide');
|
|
624
|
+
console.log(slide?.notes?.map(n => n.text));
|
|
625
|
+
|
|
626
|
+
// Footnotes and endnotes (DOCX/RTF) can be deeply nested, so we traverse recursively:
|
|
627
|
+
const printNotes = (nodes: OfficeContentNode[]) => {
|
|
628
|
+
nodes.forEach(node => {
|
|
629
|
+
if (node.notes) {
|
|
630
|
+
node.notes.forEach(note => console.log(note.text));
|
|
631
|
+
}
|
|
632
|
+
if (node.children) {
|
|
633
|
+
printNotes(node.children);
|
|
634
|
+
}
|
|
635
|
+
});
|
|
636
|
+
};
|
|
637
|
+
|
|
638
|
+
printNotes(ast.content);
|
|
639
|
+
```
|
|
640
|
+
|
|
641
|
+
> [!IMPORTANT]
|
|
642
|
+
> `putNotesAtLast` is **deprecated**. Notes are always attached via `node.notes` — this flag has no effect and will be removed in a future major version.
|
|
643
|
+
|
|
644
|
+
### Access headers, footers & slide masters
|
|
645
|
+
```ts
|
|
646
|
+
// These are NOT in ast.content — use ast.auxiliary
|
|
647
|
+
console.log(ast.auxiliary?.headers?.map(h => h.text)); // DOCX headers
|
|
648
|
+
console.log(ast.auxiliary?.footers?.map(f => f.text)); // DOCX footers
|
|
649
|
+
console.log(ast.auxiliary?.slideMasters?.length); // PPTX slide masters
|
|
650
|
+
```
|
|
651
|
+
|
|
652
|
+
Set `ignoreHeadersAndFooters: true` or `ignoreSlideMasters: true` to skip extraction.
|
|
653
|
+
|
|
535
654
|
### Extract images with OCR text
|
|
536
655
|
```js
|
|
537
656
|
const ast = await officeParser.parseOffice('report.docx', { extractAttachments: true, ocr: true });
|
|
@@ -596,8 +715,11 @@ Pass as the second argument to `parseOffice(file, config)`.
|
|
|
596
715
|
| Option | Type | Default | Description |
|
|
597
716
|
|--------|------|---------|-------------|
|
|
598
717
|
| `newlineDelimiter` | `string` | `'\n'` | Delimiter inserted between lines in text output |
|
|
599
|
-
| `ignoreNotes` | `boolean` | `false` | Ignore speaker notes (PPTX/ODP) |
|
|
600
|
-
| `
|
|
718
|
+
| `ignoreNotes` | `boolean` | `false` | Ignore footnotes/endnotes (DOCX, RTF) and speaker notes (PPTX/ODP) |
|
|
719
|
+
| `ignoreComments` | `boolean` | `false` | **New**: Ignore inline comments/annotations (DOCX, XLSX, PPTX) — by default attached via `node.comments[]` |
|
|
720
|
+
| `ignoreHeadersAndFooters` | `boolean` | `false` | **New**: Skip DOCX headers & footers (populated in `ast.auxiliary.headers/footers` by default) |
|
|
721
|
+
| `ignoreSlideMasters` | `boolean` | `false` | **New**: Skip PPTX slide masters (populated in `ast.auxiliary.slideMasters` by default) |
|
|
722
|
+
| ~~`putNotesAtLast`~~ | `boolean` | `false` | **Deprecated**: Notes are now attached via `node.notes[]`. This flag has no effect |
|
|
601
723
|
| `extractAttachments` | `boolean` | `false` | Populate `ast.attachments` with Base64 images/charts |
|
|
602
724
|
| `ocr` | `boolean` | `false` | Run Tesseract OCR on images (requires `extractAttachments: true`) |
|
|
603
725
|
| `ocrConfig` | `OcrConfig` | `{}` | OCR worker pool settings — see [OCR section](#ocr-scheduler--resource-management) |
|
|
@@ -610,7 +732,8 @@ Pass as the second argument to `parseOffice(file, config)`.
|
|
|
610
732
|
| `csvDelimiter` | `string` | `','` | Input delimiter when parsing CSV files |
|
|
611
733
|
| `pdfWorkerSrc` | `string` | CDN (jsDelivr) | Path/URL to `pdf.worker.min.mjs` (required in browser) |
|
|
612
734
|
| `onWarning` | `(issue: OfficeIssue) => void` | — | Callback for non-fatal parsing issues |
|
|
613
|
-
| `
|
|
735
|
+
| `abortSignal` | `AbortSignal \| null` | `null` | Optional signal to cancel parsing (rejects with AbortError) |
|
|
736
|
+
| ~~`outputErrorToConsole`~~ | `boolean` | `false` | **Deprecated.** Use `onWarning` instead |
|
|
614
737
|
|
|
615
738
|
---
|
|
616
739
|
|
|
@@ -630,6 +753,7 @@ Options shared by all generator formats. Pass to `OfficeGenerator.generate(ast,
|
|
|
630
753
|
| `styleMap` | `string[] \| StructuredStyleMapping[]` | `[]` | Custom semantic style mappings |
|
|
631
754
|
| `onNode` | `(node) => string \| false \| void` | — | Per-node callback for filtering, overriding, or mutating |
|
|
632
755
|
| `onWarning` | `(issue: OfficeIssue) => void` | — | Callback for non-fatal generation issues |
|
|
756
|
+
| `abortSignal` | `AbortSignal \| null` | `null` | Optional signal to cancel the generation operation (rejects with AbortError) |
|
|
633
757
|
|
|
634
758
|
---
|
|
635
759
|
|
|
@@ -708,6 +832,12 @@ Pass as `htmlConfig` inside `GeneratorConfig`.
|
|
|
708
832
|
|--------|------|---------|-------------|
|
|
709
833
|
| `standalone` | `boolean` | `true` | Wrap output in a full `<html>` document with CSS |
|
|
710
834
|
| `chartJsSrc` | `string` | jsDelivr CDN | URL for the Chart.js library |
|
|
835
|
+
| `containerWidth` | `string \| number` | `'auto'` | Max width of the content container. Positive number (px), CSS length string (`'900px'`, `'100%'`, `'60vw'`), or `'auto'`. Invalid values fall back to `'auto'` with an `INVALID_CONTAINER_WIDTH` warning |
|
|
836
|
+
| `customCss` | `string` | `''` | Raw CSS injected into the `<style>` block — use to override built-in styles |
|
|
837
|
+
| `injections.headStart` | `string` | `''` | Raw HTML injected after `<head>` |
|
|
838
|
+
| `injections.headEnd` | `string` | `''` | Raw HTML injected before `</head>` |
|
|
839
|
+
| `injections.bodyStart` | `string` | `''` | Raw HTML injected after `<body>` |
|
|
840
|
+
| `injections.bodyEnd` | `string` | `''` | Raw HTML injected before `</body>` |
|
|
711
841
|
|
|
712
842
|
### MdGeneratorConfig
|
|
713
843
|
|
|
@@ -724,6 +854,8 @@ Pass as `pdfConfig` inside `GeneratorConfig`. Requires the optional `puppeteer`
|
|
|
724
854
|
| Option | Type | Default | Description |
|
|
725
855
|
|--------|------|---------|-------------|
|
|
726
856
|
| `format` | `string` | `'A4'` | Paper format (`'A4'`, `'Letter'`, `'Legal'`, etc.) |
|
|
857
|
+
| `width` | `string \| number` | `''` | Paper width (e.g., `'5in'`, `'3cm'`) or pixels |
|
|
858
|
+
| `height` | `string \| number` | `''` | Paper height (e.g., `'5in'`, `'3cm'`) or pixels |
|
|
727
859
|
| `landscape` | `boolean` | `false` | Landscape page orientation |
|
|
728
860
|
| `printBackground` | `boolean` | `true` | Print background graphics |
|
|
729
861
|
| `margin` | `object` | `{0,0,0,0}` | Page margins (`top`, `right`, `bottom`, `left`) |
|
|
@@ -732,6 +864,7 @@ Pass as `pdfConfig` inside `GeneratorConfig`. Requires the optional `puppeteer`
|
|
|
732
864
|
| `footerTemplate` | `string` | `''` | HTML template for the print footer |
|
|
733
865
|
| `scale` | `number` | `1` | Rendering scale factor |
|
|
734
866
|
| `launchOptions` | `object` | headless defaults | Puppeteer launch options (e.g., `executablePath`) |
|
|
867
|
+
| `timeout` | `number` | `30000` | PDF rendering timeout in milliseconds. Set to `0` to disable. |
|
|
735
868
|
|
|
736
869
|
### CsvGeneratorConfig
|
|
737
870
|
|
|
@@ -807,6 +940,7 @@ Configuration for `OfficeConverter.convert(file, format, config)`.
|
|
|
807
940
|
| `maxChunkSize` | `number` | `2000` | Max characters even if similarity stays high |
|
|
808
941
|
| `bufferSize` | `number` | `1` | Surrounding sentences used when computing similarity |
|
|
809
942
|
| `embeddingBatchSize` | `number` | `50` | Sentences per embedding API batch |
|
|
943
|
+
| `timeout` | `number` | `10000` | Timeout in milliseconds for individual embedding API calls. Set to `0` to disable. |
|
|
810
944
|
|
|
811
945
|
---
|
|
812
946
|
|
|
@@ -826,7 +960,8 @@ When `ocr: true` is set, `officeParser` maintains an intelligent **Smart Worker
|
|
|
826
960
|
| `workerPath` | `string` | `''` | Custom path to Tesseract worker script |
|
|
827
961
|
| `corePath` | `string` | `''` | Custom path to Tesseract core script |
|
|
828
962
|
| `langPath` | `string` | `''` | Custom path for language data files |
|
|
829
|
-
| `
|
|
963
|
+
| `timeout` | `OcrTimeoutConfig` | `{}` | Consolidated timeouts: `autoTerminate`, `workerLoad`, `recognition` |
|
|
964
|
+
| ~~`autoTerminateTimeout`~~ | `number` | `10000` | **Deprecated.** Use `timeout.autoTerminate` instead |
|
|
830
965
|
|
|
831
966
|
See all language codes at [tesseract-ocr.github.io](https://tesseract-ocr.github.io/tessdoc/Data-Files).
|
|
832
967
|
|
|
@@ -927,7 +1062,6 @@ For a full debugging guide, visit the [Live Documentation](https://harshankur.gi
|
|
|
927
1062
|
|
|
928
1063
|
1. **ODT/ODS Charts**: May show inaccurate data when the chart references external cell ranges or uses complex layout-based data.
|
|
929
1064
|
2. **PDF Images (Browser)**: Extracted as BMP files for cross-platform compatibility. Conversion is automatic.
|
|
930
|
-
3. **RTF Notes**: `putNotesAtLast` has no effect for RTF files; footnotes and endnotes are always appended at the end.
|
|
931
1065
|
|
|
932
1066
|
---
|
|
933
1067
|
|
|
@@ -15,5 +15,5 @@ export declare class OfficeGenerator {
|
|
|
15
15
|
*/
|
|
16
16
|
static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
|
|
17
17
|
type: T;
|
|
18
|
-
}, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult
|
|
18
|
+
}, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
|
|
19
19
|
}
|
package/dist/OfficeGenerator.js
CHANGED
|
@@ -25,24 +25,33 @@ class OfficeGenerator {
|
|
|
25
25
|
* @throws {Error} If the destination format is unsupported
|
|
26
26
|
*/
|
|
27
27
|
static async generate(ast, destination, config) {
|
|
28
|
+
let generator;
|
|
28
29
|
switch (destination.toLowerCase()) {
|
|
29
30
|
case 'text':
|
|
30
|
-
|
|
31
|
+
generator = new TextGenerator_js_1.TextGenerator(ast, config);
|
|
32
|
+
break;
|
|
31
33
|
case 'md':
|
|
32
|
-
|
|
34
|
+
generator = new MarkdownGenerator_js_1.MarkdownGenerator(ast, config);
|
|
35
|
+
break;
|
|
33
36
|
case 'html':
|
|
34
|
-
|
|
37
|
+
generator = new HtmlGenerator_js_1.HtmlGenerator(ast, config);
|
|
38
|
+
break;
|
|
35
39
|
case 'pdf':
|
|
36
|
-
|
|
40
|
+
generator = new PdfGenerator_js_1.PdfGenerator(ast, config);
|
|
41
|
+
break;
|
|
37
42
|
case 'csv':
|
|
38
|
-
|
|
43
|
+
generator = new CsvGenerator_js_1.CsvGenerator(ast, config);
|
|
44
|
+
break;
|
|
39
45
|
case 'rtf':
|
|
40
|
-
|
|
46
|
+
generator = new RtfGenerator_js_1.RtfGenerator(ast, config);
|
|
47
|
+
break;
|
|
41
48
|
case 'chunks':
|
|
42
|
-
|
|
49
|
+
generator = new ChunkingGenerator_js_1.ChunkingGenerator(ast, config);
|
|
50
|
+
break;
|
|
43
51
|
default:
|
|
44
52
|
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, undefined, destination);
|
|
45
53
|
}
|
|
54
|
+
return generator.generate();
|
|
46
55
|
}
|
|
47
56
|
}
|
|
48
57
|
exports.OfficeGenerator = OfficeGenerator;
|
package/dist/OfficeParser.js
CHANGED
|
@@ -241,6 +241,12 @@ class OfficeParser {
|
|
|
241
241
|
return result;
|
|
242
242
|
}
|
|
243
243
|
catch (error) {
|
|
244
|
+
// AbortError must pass through untouched so callers can distinguish a
|
|
245
|
+
// deliberate cancellation (err.name === 'AbortError') from a real parse failure.
|
|
246
|
+
// getWrappedError always creates a plain new Error(), which would strip the
|
|
247
|
+
// AbortError identity and break any instanceof / name checks on the caller side.
|
|
248
|
+
if (error?.name === 'AbortError')
|
|
249
|
+
throw error;
|
|
244
250
|
const wrappedError = (0, errorUtils_js_1.getWrappedError)(error, internalConfig, filePath);
|
|
245
251
|
if (callback)
|
|
246
252
|
callback(undefined, wrappedError);
|
package/dist/cli.d.ts
CHANGED
|
@@ -15,6 +15,10 @@
|
|
|
15
15
|
* --ocrLanguage=eng OCR language (default: eng)
|
|
16
16
|
* --extractAttachments=true Extract embedded attachments
|
|
17
17
|
* --ignoreNotes=true Ignore footnotes/endnotes
|
|
18
|
+
* --ignoreComments=true Ignore inline comments
|
|
19
|
+
* --ignoreHeadersAndFooters=true Ignore headers and footers
|
|
20
|
+
* --ignoreSlideMasters=true Ignore slide masters
|
|
21
|
+
* --ignoreInternalLinks=true Ignore internal links
|
|
18
22
|
* --putNotesAtLast=true Move notes to end of document
|
|
19
23
|
* --includeRawContent=true Include raw content in AST
|
|
20
24
|
* --outputErrorToConsole=true Log errors to console
|
package/dist/cli.js
CHANGED
|
@@ -16,6 +16,10 @@
|
|
|
16
16
|
* --ocrLanguage=eng OCR language (default: eng)
|
|
17
17
|
* --extractAttachments=true Extract embedded attachments
|
|
18
18
|
* --ignoreNotes=true Ignore footnotes/endnotes
|
|
19
|
+
* --ignoreComments=true Ignore inline comments
|
|
20
|
+
* --ignoreHeadersAndFooters=true Ignore headers and footers
|
|
21
|
+
* --ignoreSlideMasters=true Ignore slide masters
|
|
22
|
+
* --ignoreInternalLinks=true Ignore internal links
|
|
19
23
|
* --putNotesAtLast=true Move notes to end of document
|
|
20
24
|
* --includeRawContent=true Include raw content in AST
|
|
21
25
|
* --outputErrorToConsole=true Log errors to console
|
|
@@ -83,9 +87,10 @@ if (fileArg) {
|
|
|
83
87
|
const lowerValue = value.toLowerCase();
|
|
84
88
|
const boolValue = lowerValue === 'true' ? true : (lowerValue === 'false' ? false : undefined);
|
|
85
89
|
const knownBooleans = new Set([
|
|
86
|
-
'toText', 'ocr', 'extractAttachments', 'ignoreNotes', '
|
|
87
|
-
'
|
|
88
|
-
'
|
|
90
|
+
'toText', 'ocr', 'extractAttachments', 'ignoreNotes', 'ignoreComments',
|
|
91
|
+
'ignoreHeadersAndFooters', 'ignoreSlideMasters', 'ignoreInternalLinks',
|
|
92
|
+
'putNotesAtLast', 'includeRawContent', 'outputErrorToConsole',
|
|
93
|
+
'serializeRawContent', 'preserveXmlWhitespace', 'includeBreakNodes', 'verbose'
|
|
89
94
|
]);
|
|
90
95
|
if (cleanKey === 'format') {
|
|
91
96
|
outputFormat = value;
|
|
@@ -184,6 +189,10 @@ else {
|
|
|
184
189
|
console.log(' --ocrLanguage=eng OCR language (default: eng)');
|
|
185
190
|
console.log(' --extractAttachments=true Extract embedded attachments');
|
|
186
191
|
console.log(' --ignoreNotes=true Ignore footnotes/endnotes');
|
|
192
|
+
console.log(' --ignoreComments=true Ignore inline comments');
|
|
193
|
+
console.log(' --ignoreHeadersAndFooters=true Ignore headers and footers');
|
|
194
|
+
console.log(' --ignoreSlideMasters=true Ignore slide masters');
|
|
195
|
+
console.log(' --ignoreInternalLinks=true Ignore internal links');
|
|
187
196
|
console.log(' --putNotesAtLast=true Move notes to end of document');
|
|
188
197
|
console.log(' --includeRawContent=true Include raw content in AST');
|
|
189
198
|
console.log(' --serializeRawContent=true Serialize raw XML content (default: true)');
|
package/dist/defaults.js
CHANGED
|
@@ -13,6 +13,12 @@ exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = /[.!?。!?]/;
|
|
|
13
13
|
* Common abbreviations that should not trigger a sentence split when followed by a period.
|
|
14
14
|
*/
|
|
15
15
|
exports.DEFAULT_ABBREVIATIONS = ['Mr', 'Dr', 'Ms', 'Inc', 'Ltd', 'Prof', 'Sr', 'Jr', 'vs', 'etc'];
|
|
16
|
+
/** Default timeout values for OCR */
|
|
17
|
+
const DEFAULT_OCR_TIMEOUT = {
|
|
18
|
+
autoTerminate: 10000,
|
|
19
|
+
workerLoad: 60000,
|
|
20
|
+
recognition: 30000,
|
|
21
|
+
};
|
|
16
22
|
/**
|
|
17
23
|
* Default configuration for OCR.
|
|
18
24
|
*/
|
|
@@ -21,7 +27,12 @@ const DEFAULT_OCR_CONFIG = {
|
|
|
21
27
|
workerPath: '',
|
|
22
28
|
corePath: '',
|
|
23
29
|
langPath: '',
|
|
24
|
-
|
|
30
|
+
// Preferred: consolidated timeout object. New code should always read from here.
|
|
31
|
+
timeout: DEFAULT_OCR_TIMEOUT,
|
|
32
|
+
// Kept for backward compatibility. When timeout.autoTerminate is set (as above),
|
|
33
|
+
// the ocrUtils resolution logic will prefer timeout.autoTerminate over this flat field.
|
|
34
|
+
autoTerminateTimeout: DEFAULT_OCR_TIMEOUT.autoTerminate,
|
|
35
|
+
abortSignal: null,
|
|
25
36
|
};
|
|
26
37
|
/**
|
|
27
38
|
* Default configuration for the OfficeParser.
|
|
@@ -31,12 +42,16 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
|
|
|
31
42
|
onWarning: () => { },
|
|
32
43
|
newlineDelimiter: '\n',
|
|
33
44
|
ignoreNotes: false,
|
|
45
|
+
ignoreComments: false,
|
|
46
|
+
ignoreHeadersAndFooters: false,
|
|
47
|
+
ignoreSlideMasters: false,
|
|
34
48
|
putNotesAtLast: false,
|
|
35
49
|
extractAttachments: false,
|
|
36
50
|
includeRawContent: false,
|
|
37
51
|
ocr: false,
|
|
38
52
|
ocrLanguage: 'eng',
|
|
39
53
|
ocrConfig: DEFAULT_OCR_CONFIG,
|
|
54
|
+
abortSignal: null,
|
|
40
55
|
serializeRawContent: true,
|
|
41
56
|
preserveXmlWhitespace: false,
|
|
42
57
|
pdfWorkerSrc: DEFAULT_PDF_WORKER_SRC,
|
|
@@ -51,6 +66,14 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
|
|
|
51
66
|
const DEFAULT_HTML_GENERATOR_CONFIG = {
|
|
52
67
|
standalone: true,
|
|
53
68
|
chartJsSrc: 'https://cdn.jsdelivr.net/npm/chart.js',
|
|
69
|
+
containerWidth: 'auto',
|
|
70
|
+
customCss: '',
|
|
71
|
+
injections: {
|
|
72
|
+
headStart: '',
|
|
73
|
+
headEnd: '',
|
|
74
|
+
bodyStart: '',
|
|
75
|
+
bodyEnd: '',
|
|
76
|
+
}
|
|
54
77
|
};
|
|
55
78
|
/**
|
|
56
79
|
* Default configuration for PDF generation.
|
|
@@ -75,6 +98,7 @@ const DEFAULT_PDF_GENERATOR_CONFIG = {
|
|
|
75
98
|
headless: true,
|
|
76
99
|
args: ['--no-sandbox', '--disable-setuid-sandbox']
|
|
77
100
|
},
|
|
101
|
+
timeout: 30000,
|
|
78
102
|
};
|
|
79
103
|
/**
|
|
80
104
|
* Default configuration for CSV generation.
|
|
@@ -143,6 +167,7 @@ exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = {
|
|
|
143
167
|
lengthFunction: (text) => text.length,
|
|
144
168
|
sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
|
|
145
169
|
abbreviations: exports.DEFAULT_ABBREVIATIONS,
|
|
170
|
+
timeout: 10000,
|
|
146
171
|
};
|
|
147
172
|
/**
|
|
148
173
|
* The resolved default chunking config (uses document-structure as default strategy).
|
|
@@ -162,6 +187,7 @@ exports.DEFAULT_GENERATOR_CONFIG = {
|
|
|
162
187
|
includeImages: true,
|
|
163
188
|
includeCharts: true,
|
|
164
189
|
ignoreInternalLinks: false,
|
|
190
|
+
abortSignal: null,
|
|
165
191
|
htmlConfig: DEFAULT_HTML_GENERATOR_CONFIG,
|
|
166
192
|
mdConfig: DEFAULT_MD_GENERATOR_CONFIG,
|
|
167
193
|
pdfConfig: DEFAULT_PDF_GENERATOR_CONFIG,
|
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
import { OfficeIssue, ConversionResult, FullGeneratorConfig, GeneratorConfig, OfficeContentNode, OfficeParserAST, OfficeWarningType } from '../types.js';
|
|
1
|
+
import { OfficeIssue, ConversionResult, FullGeneratorConfig, GeneratorConfig, OfficeContentNode, OfficeParserAST, OfficeWarningType, UniversalGeneratorFormat } from '../types.js';
|
|
2
2
|
import { StyleMapper } from '../utils/styleMapper.js';
|
|
3
3
|
/**
|
|
4
4
|
* Base class for all document generators.
|
|
5
5
|
* Provides common traversal logic and configuration handling.
|
|
6
6
|
*/
|
|
7
|
-
export declare abstract class BaseGenerator<D extends
|
|
7
|
+
export declare abstract class BaseGenerator<D extends UniversalGeneratorFormat = UniversalGeneratorFormat> {
|
|
8
8
|
protected destination: D;
|
|
9
9
|
protected config: FullGeneratorConfig;
|
|
10
10
|
protected ast: OfficeParserAST;
|
|
@@ -24,7 +24,7 @@ export declare abstract class BaseGenerator<D extends string = string> {
|
|
|
24
24
|
/**
|
|
25
25
|
* Entry point for generation.
|
|
26
26
|
*/
|
|
27
|
-
abstract generate(): Promise<ConversionResult
|
|
27
|
+
abstract generate(): Promise<ConversionResult<D>>;
|
|
28
28
|
/**
|
|
29
29
|
* Centralized logic for handling the onNode callback.
|
|
30
30
|
* Evaluates the callback and returns a result that tells the generator how to proceed.
|
|
@@ -39,6 +39,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
39
39
|
* Note: ConversionResult.value is a JSON string of OfficeChunk[] for the 'chunks' destination.
|
|
40
40
|
*/
|
|
41
41
|
async generate() {
|
|
42
|
+
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
42
43
|
let chunks;
|
|
43
44
|
switch (this.chunkConfig.strategy) {
|
|
44
45
|
case 'fixed-size':
|
|
@@ -186,6 +187,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
186
187
|
return this.finalizeChunks(chunks, config);
|
|
187
188
|
}
|
|
188
189
|
async processNodeForStructure(node, config, splitBy, maxChunkSize, measure, chunks, contextStack) {
|
|
190
|
+
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
189
191
|
// Check for node override or skip
|
|
190
192
|
const override = await this.handleOnNode(node);
|
|
191
193
|
if (override === false)
|
|
@@ -391,7 +393,14 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
391
393
|
renderedRows.push(row.text ?? '');
|
|
392
394
|
continue;
|
|
393
395
|
}
|
|
394
|
-
const
|
|
396
|
+
const getCellText = (cell) => {
|
|
397
|
+
if (cell.text)
|
|
398
|
+
return cell.text;
|
|
399
|
+
if (!cell.children || cell.children.length === 0)
|
|
400
|
+
return '';
|
|
401
|
+
return cell.children.map(c => getCellText(c)).join(' ');
|
|
402
|
+
};
|
|
403
|
+
const cells = row.children.map(cell => getCellText(cell).replace(/\n/g, ' ').trim());
|
|
395
404
|
renderedRows.push(`| ${cells.join(' | ')} |`);
|
|
396
405
|
}
|
|
397
406
|
return renderedRows.join('\n');
|
|
@@ -412,7 +421,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
412
421
|
if (sentences.length === 0)
|
|
413
422
|
return [];
|
|
414
423
|
// Embed all sentences in batches to avoid rate limiting
|
|
415
|
-
const embeddings = await this.batchEmbeddings(sentences, config.embeddingFunction, batchSize);
|
|
424
|
+
const embeddings = await this.batchEmbeddings(sentences, config.embeddingFunction, batchSize, config.timeout);
|
|
416
425
|
// Calculate cosine similarity between adjacent sentence windows
|
|
417
426
|
const chunks = [];
|
|
418
427
|
let currentSentences = [];
|
|
@@ -479,6 +488,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
479
488
|
let currentPage;
|
|
480
489
|
let currentSheet;
|
|
481
490
|
const walk = async (node) => {
|
|
491
|
+
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
482
492
|
const override = await this.handleOnNode(node);
|
|
483
493
|
if (override === false)
|
|
484
494
|
return;
|
|
@@ -533,6 +543,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
533
543
|
let currentPage;
|
|
534
544
|
let currentSheet;
|
|
535
545
|
const walk = async (node) => {
|
|
546
|
+
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
536
547
|
const override = await this.handleOnNode(node);
|
|
537
548
|
if (override === false)
|
|
538
549
|
return;
|
|
@@ -602,11 +613,27 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
602
613
|
/**
|
|
603
614
|
* Helper to process embeddings in sequential batches to avoid API rate limits and memory issues.
|
|
604
615
|
*/
|
|
605
|
-
async batchEmbeddings(sentences, embedFn, batchSize = 50) {
|
|
616
|
+
async batchEmbeddings(sentences, embedFn, batchSize = 50, timeoutMs) {
|
|
606
617
|
const results = [];
|
|
607
618
|
for (let i = 0; i < sentences.length; i += batchSize) {
|
|
619
|
+
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
608
620
|
const batch = sentences.slice(i, i + batchSize);
|
|
609
|
-
const
|
|
621
|
+
const batchPromises = batch.map(s => {
|
|
622
|
+
const call = embedFn(s.text);
|
|
623
|
+
if (timeoutMs !== undefined && timeoutMs > 0) {
|
|
624
|
+
let timerId;
|
|
625
|
+
const timeoutPromise = new Promise((_, reject) => {
|
|
626
|
+
timerId = setTimeout(() => {
|
|
627
|
+
reject(new Error(`Embedding call timed out after ${timeoutMs}ms`));
|
|
628
|
+
}, timeoutMs);
|
|
629
|
+
});
|
|
630
|
+
return Promise.race([call, timeoutPromise]).finally(() => {
|
|
631
|
+
clearTimeout(timerId);
|
|
632
|
+
});
|
|
633
|
+
}
|
|
634
|
+
return call;
|
|
635
|
+
});
|
|
636
|
+
const batchResults = await Promise.all(batchPromises);
|
|
610
637
|
results.push(...batchResults);
|
|
611
638
|
}
|
|
612
639
|
return results;
|
|
@@ -10,7 +10,7 @@ export declare class CsvGenerator extends BaseGenerator<'csv'> {
|
|
|
10
10
|
*
|
|
11
11
|
* @returns A CSV string or a ZIP archive containing multiple CSVs
|
|
12
12
|
*/
|
|
13
|
-
generate(): Promise<ConversionResult
|
|
13
|
+
generate(): Promise<ConversionResult<'csv'>>;
|
|
14
14
|
/**
|
|
15
15
|
* Recursively finds all nodes that can be treated as sheets (sheet or table).
|
|
16
16
|
*/
|
|
@@ -12,7 +12,7 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
|
|
|
12
12
|
*
|
|
13
13
|
* @returns An HTML string
|
|
14
14
|
*/
|
|
15
|
-
generate(): Promise<ConversionResult
|
|
15
|
+
generate(): Promise<ConversionResult<'html'>>;
|
|
16
16
|
private renderMetaTags;
|
|
17
17
|
private renderMetadataSummary;
|
|
18
18
|
/**
|
|
@@ -33,5 +33,6 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
|
|
|
33
33
|
private getInlineStyles;
|
|
34
34
|
private getPremiumStyles;
|
|
35
35
|
protected slugify(text: string): string;
|
|
36
|
+
private getColumnLetter;
|
|
36
37
|
private escape;
|
|
37
38
|
}
|